Add PackFS v0: statically linked in-process VFS implementing concept.md
Implements the core design: a mount table published as an atomically- swapped snapshot; mem/dir/pack/overlay backends; copy-on-write overlay with copy-up and whiteout deletion; a checksummed append journal; compaction with exact-duplicate elimination; single-writer/wait-free- reader concurrency with a structural/content write split; openat2/ Landlock path containment for dir mounts; and load-time pack integrity validation. Zero required third-party dependencies. Sanitizer testing (ASan/UBSan) caught and led to fixing a genuine heap-use-after-free in the snapshot-reclamation path: the textbook "load pointer, then increment its refcount" pattern left a gap a concurrent writer could free through. Closed with a small reclaim_gate rwlock, documented in internal.h and CLAUDE.md since it's a pattern every refcounted structure in the codebase now follows. zip/tar import/export backends, recommended in concept.md Section 11, will not be built — a permanent project decision recorded in CLAUDE.md since concept.md itself is frozen and cannot be edited to reflect it. Includes a runnable demo (examples/demo.c, `make demo`) exercising the library end to end and proving cross-run persistence through the pack file, plus open-source scaffolding: MIT license, README, CONTRIBUTING, and a CI workflow running the test suite under ASan/UBSan/TSan. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UqJpkdJ6Njnt1pw3CbghzB
This commit is contained in:
@@ -0,0 +1,45 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
build-and-test:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Build static + shared library
|
||||
run: make all
|
||||
|
||||
- name: Run test suite
|
||||
run: make test
|
||||
|
||||
- name: Build and run under AddressSanitizer + UndefinedBehaviorSanitizer
|
||||
run: |
|
||||
mkdir -p build/san
|
||||
for f in src/*.c; do
|
||||
cc -std=c11 -Wall -Wextra -O1 -g -fPIC -Iinclude -Isrc -D_GNU_SOURCE \
|
||||
-fsanitize=address,undefined -c "$f" -o "build/san/$(basename "${f%.c}").o"
|
||||
done
|
||||
for t in tests/test_*.c; do
|
||||
name=$(basename "${t%.c}")
|
||||
cc -std=c11 -O1 -g -Iinclude -Isrc -fsanitize=address,undefined \
|
||||
"$t" build/san/*.o -lpthread -o "build/san/$name"
|
||||
"./build/san/$name"
|
||||
done
|
||||
|
||||
- name: Build and run under ThreadSanitizer
|
||||
run: |
|
||||
mkdir -p build/tsan
|
||||
for f in src/*.c; do
|
||||
cc -std=c11 -Wall -Wextra -O1 -g -fPIC -Iinclude -Isrc -D_GNU_SOURCE \
|
||||
-fsanitize=thread -c "$f" -o "build/tsan/$(basename "${f%.c}").o"
|
||||
done
|
||||
for t in tests/test_*.c; do
|
||||
name=$(basename "${t%.c}")
|
||||
cc -std=c11 -O1 -g -Iinclude -Isrc -fsanitize=thread \
|
||||
"$t" build/tsan/*.o -lpthread -o "build/tsan/$name"
|
||||
"./build/tsan/$name"
|
||||
done
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
*.o
|
||||
*.a
|
||||
*.so
|
||||
*.so.*
|
||||
build/
|
||||
*.img
|
||||
*.img.tmp
|
||||
*.jnl
|
||||
*.jnl.tmp
|
||||
*.dSYM/
|
||||
@@ -0,0 +1,95 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## Repository status
|
||||
|
||||
This repository contains a working v0 implementation of the `concept.md` specification: `include/packfs.h` (public API), `src/*.c` (implementation), `tests/test_*.c` (test suite), and open-source project scaffolding (`README.md`, `LICENSE`, `CONTRIBUTING.md`, `.github/workflows/ci.yml`), alongside the frozen `concept.md` and this file.
|
||||
|
||||
## Build, test, and lint commands
|
||||
|
||||
```sh
|
||||
make # builds libpackfs.a and libpackfs.so (zero required third-party deps, Section 11.1)
|
||||
make test # builds and runs every tests/test_*.c
|
||||
make clean
|
||||
make install PREFIX=/some/prefix
|
||||
```
|
||||
|
||||
A single test: `make build/test_mem && ./build/test_mem` (substitute any `test_*` basename). There is no separate lint step — the build itself uses `-Wall -Wextra -Wpedantic`, and a warning introduced by a change is a build failure, not something to leave in place.
|
||||
|
||||
Sanitizer builds are not wired into `make test` (they need per-file compilation with sanitizer flags plus `-D_GNU_SOURCE -Iinclude -Isrc`); see `.github/workflows/ci.yml` for the exact invocation, which CI runs on every push. **Any change to `upper.c`, `overlay.c`, or `vfs.c` must be verified under `-fsanitize=address,undefined` and `-fsanitize=thread` before being considered done** — this is not a formality: exactly this process caught a real use-after-free in the snapshot-reclamation logic during initial development (see the `reclaim_gate` note below), which the plain build and even repeated plain test runs never surfaced.
|
||||
|
||||
## Code architecture
|
||||
|
||||
- `include/packfs.h` — the entire public API.
|
||||
- `src/internal.h` — every internal type shared across `.c` files; read this first when touching implementation code.
|
||||
- `src/vfs.c` — the `Vfs` mount table itself (`MountSnapshot`, refcounted, atomically swapped) and the public API's dispatch-by-longest-prefix-match.
|
||||
- `src/upper.c` — the writable layer shared by `mem`, `dir`, and the overlay's upper side: `UpperSnapshot`/`UpperEntry`/`MutCell`, the structural-write functions (`upper_create`/`upper_mkdir`/`upper_remove`/`upper_rename`/`upper_copy_up`), the content-write fast path (`upper_cell_read`/`upper_cell_write`), and the standalone `mem`/`dir` `Backend` (`backend_mem_new`/`backend_dir_new`).
|
||||
- `src/overlay.c` — composes a `Pack` (lower, read-only) with an `UpperStore` (upper): copy-up, whiteouts, the merged view (`overlay_stat`/`overlay_readdir`), the append journal, and `vfs_sync`'s compaction.
|
||||
- `src/pack.c` — the on-disk pack format: load-time integrity validation, binary-search lookup/range-query, and the compaction writer (atomic rename, exact-duplicate elimination).
|
||||
- `src/containment.c` — `dir`-mount path containment (`openat2`/`O_NOFOLLOW` fallback) and opt-in Landlock hardening.
|
||||
- `src/path.c` — virtual-namespace `.`/`..` canonicalization.
|
||||
- `src/hash.c` — FNV-1a64, used for pack checksums and compaction dedup.
|
||||
- `tests/` — one binary per concern (`test_mem`, `test_dir`, `test_pack_overlay`, `test_concurrency`); `test_harness.h` is a small assertion-macro header, not a framework, consistent with the zero-dependency constraint.
|
||||
|
||||
**One architectural fact spans every mutable structure and is easy to miss reading any single file in isolation:** `MountSnapshot` (`vfs.c`), `UpperSnapshot` (`upper.c`), and a `mem`-backed `MutCell`'s buffer (`upper.c`, Section 5.7) are all retired through the same two-step pattern — publish the replacement via `atomic_store_explicit(..., memory_order_release)`, then free the superseded object only under a dedicated `reclaim_gate` rwlock's *write* side, while every acquirer takes that same gate's *read* side around its own load-then-increment. This exists because a plain "load a pointer, then atomically increment its refcount" leaves a real gap between those two steps in which a concurrent writer can free the very object being acquired — confirmed by ASan as an actual heap-use-after-free during development, not a theoretical concern. See the comment on `UpperStore.reclaim_gate` in `internal.h` for the full reasoning. Any new refcounted, concurrently-reclaimed structure added to this codebase needs the same gate, not just an atomic pointer and a naive refcount.
|
||||
|
||||
## What this project is
|
||||
|
||||
PackFS is the design for a **statically linked, in-process virtual file system (VFS) written in C**, intended for embedding in sandboxed, agent-driven, TUI, or game-shaped programs. `concept.md` is the full specification (architecture, write model, concurrency model, path-containment model, pack-integrity validation, C object-model sketch, on-disk pack format, explicit v0 exclusions, and a self-evaluation) — read it in full before doing any implementation work here, rather than relying on the summary below.
|
||||
|
||||
## Documentation standard (enforced)
|
||||
|
||||
Every document in this repository — this file included — is written and revised under the following rule. It was established after `concept.md` was rewritten into its current form, and that rewrite is the reference example: when in doubt about whether a piece of prose conforms, compare it against `concept.md`.
|
||||
|
||||
- **Register.** Formal, precise, third-person or directly-stated-constraint prose. No colloquialisms, no rhetorical asides, no filler sections ("Tips," "Support," "Common Tasks") invented for their own sake.
|
||||
- **Structure scales to content, not the other way around.** A specification-length document (`concept.md`) earns full academic structure — Abstract, Problem Statement, numbered body sections, Evaluation, Conclusion, References. A short instructions file (this one) states its rules directly without manufacturing sections it doesn't need.
|
||||
- **Constraints are labeled by strength.** A requirement an implementation must satisfy is stated as such explicitly ("hard constraint," "must," "required"). A recommendation that could reasonably be revisited is stated as such ("recommended," "a legitimate simpler fallback"). The two are never left ambiguous relative to each other.
|
||||
- **Revisions never delete information.** An error, once caught, is corrected and the correction is explained in place; prior content is extended or superseded, not silently dropped. This applies to every living document in this repository; `concept.md` is the one exception, having been declared frozen rather than living — see "`concept.md` is immutable" below.
|
||||
- **External claims are cited, not asserted.** Where a decision leans on prior art or an established system (e.g., LMDB, OverlayFS), it is attributed with a references entry rather than presented as self-evident.
|
||||
- **Every specification-length document carries its own self-evaluation.** A short methodology note, an assessment table scored per relevant category, and one final overall grade with justification — see `concept.md` Section 12 for the pattern. This file's own self-evaluation is at the bottom.
|
||||
|
||||
## `concept.md` is immutable
|
||||
|
||||
As of this revision, `concept.md` is frozen. It is not to be edited, rewritten, appended to, or otherwise modified by any future session, for any reason — including to fix a typo, to close a newly-found gap, or to keep it "in sync" with a later decision. It stands as the fixed record of the design as reasoned through its own research and review cycles; the self-evaluation and grade inside it describe that fixed text, and would themselves become inaccurate if the text under them kept moving.
|
||||
|
||||
This is a deliberate exception to the "revisions never delete information" pattern above, not a stricter version of it: that pattern describes how a living document is corrected in place; a frozen document is not corrected in place at all. The correct response to a newly-found gap, error, or improvement in the design is to open a **new, separate document** (for example, an addendum or a versioned successor) that amends or supersedes `concept.md`, leaving its text untouched. If such a document is ever created, this file should be updated to point to it alongside `concept.md`, and to state plainly which of the two is authoritative where they overlap.
|
||||
|
||||
The load-bearing constraints summarized below are therefore also frozen in substance: they describe what `concept.md` says, not a moving target. A future change to the actual design changes what is authoritative — a new document — not what `concept.md` says happened.
|
||||
|
||||
## Project decisions that supersede concept.md
|
||||
|
||||
`concept.md` cannot be edited (above), but the project's actual scope has diverged from it in one place, by an explicit, permanent decision — recorded here per this file's own protocol for citing an amendment alongside the frozen text it overrides:
|
||||
|
||||
- **The `zip` and `tar` import/export backends will never be built.** `concept.md` Section 3.1 lists them as candidate backends and Section 11 recommends them ("Zip via miniz as import/export. Tar as snapshot/export"), with Section 11.1 describing a compile-time switch at the `zip`/`tar` backend boundary for `miniz`/`libarchive`. That recommendation is superseded: this project supports the pack format only. Do not propose, scaffold, stub, or partially implement a `zip` or `tar` backend, and do not add `miniz` or `libarchive` as a dependency for any reason. `backend_overlay_new` reads and writes this project's own pack format exclusively. The rest of Section 11's recommendation (core + `mem` + `dir` + custom pack + overlay + compaction) stands as written and is what this codebase implements.
|
||||
|
||||
## Load-bearing constraints from concept.md
|
||||
|
||||
These are hard requirements stated in the spec, not stylistic suggestions — any implementation work must respect them:
|
||||
|
||||
- **Static linking is enforced, not aspirational.** The core (`vfs_*`, `mem`, `dir`, `pack`, overlay, compaction) must build with zero required third-party libraries and zero dynamic loading. `concept.md` describes `miniz`/`libarchive` sitting behind a compile-time switch at the `zip`/`tar` backend boundary as an *optional* addition that must never be required — but per "Project decisions that supersede concept.md" above, that boundary will never actually be built, so this constraint is satisfied trivially: there is no optional dependency at all, required or otherwise.
|
||||
- **Archive formats (zip, tar) are, and will remain, entirely absent from this codebase — not merely import/export skins.** The system of record is, and is the only supported format, a custom indexed **pack** file (immutable image: header + index + blobs + strings). See "Project decisions that supersede concept.md" above.
|
||||
- **Writability comes from a copy-on-write overlay, not from mutating the pack.** Pack mounted read-only at a path; `mem` or `dir` mounted as the writable upper layer; `open(O_RDWR)` triggers copy-up. Deleting a lower-layer (pack) entry requires writing a **whiteout** tombstone in the upper layer — there is no way to delete a pack entry directly, and skipping the whiteout leaves deletion of pack-originated files undefined.
|
||||
- **Compaction and the journal must be crash-safe by construction:** compaction writes `pack.img.tmp`, `fsync`s, then atomically `rename()`s over `pack.img` (never in place); journal records are length/checksum self-describing so replay stops cleanly at the first torn record; journal "truncation" after compaction means writing the surviving tail to a new file and renaming it over the journal, since a file's front cannot be truncated in place.
|
||||
- **Concurrency model is single-writer, wait-free-reader (LMDB-style MVCC), not a global lock.** Live state — upper index, whiteout set, **and the mount table** — is an immutable, refcounted `VfsSnapshot` reached through one atomic pointer; readers load it once (wait-free) and never observe an in-progress write; exactly one writer at a time builds the next snapshot copy-on-write and publishes it with a single atomic pointer swap using release/acquire ordering (getting this ordering wrong is a real C11 data race). `vfs_mount`/`vfs_unmount` are structural writes like any other — the mount table is not separate global state exempt from this model. **Structural writes (create/unlink/rename/mkdir/whiteout/mount/unmount) publish a new snapshot; content writes (bytes into an already-existing file) do not** — they update a per-entry size/mtime cell under a per-entry lock instead, so an ordinary `vfs_write` never requires copying and republishing the whole index. A `mem`-backed write that grows a file's buffer past its current allocation must **allocate a new buffer and publish it with a release store**, never realloc the existing address in place — a concurrent reader may otherwise copy out of freed memory, not just a stale value; the old buffer is retired under the same reclamation discipline as a retired snapshot. Compaction is just another writer against a frozen snapshot plus a recorded journal watermark; if writing `pack.img.tmp` or its `fsync` fails, compaction aborts and the existing pack + journal are untouched. Known, explicitly-accepted gaps: the per-entry lock covers `dir`-upper only within this process, not against a second uncoordinated process writing the same host directory; rename-over-a-mapped-pack is safe on POSIX but not guaranteed on Windows; naive refcounting contends under high read concurrency (hazard pointers are the recommended upgrade path, not epoch-based reclamation, because this system's workloads can plausibly stall a reader thread); cross-process concurrency has a specified LMDB-style reader-table design but is not implemented in v0.
|
||||
- **Path containment for `dir` mounts is a hard requirement, not best-effort.** A `dir` mount is an already-open directory file descriptor (capability-scoped, WASI-style), not a re-resolved path string. On Linux 5.6+, every lookup under it uses `openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS)` to atomically block `..` escapes, symlink escapes, and the canonicalize-then-open TOCTOU window in one kernel call; Landlock (5.13+) is a recommended additional layer. On other platforms, no equivalent atomic primitive is assumed — the fallback (`O_NOFOLLOW` per component + canonicalize-and-verify) is an accepted, explicitly weaker residual-risk posture, not a claim of parity. The VFS core additionally canonicalizes `.`/`..` in the virtual namespace itself, before any backend is reached, so a crafted path cannot escape a mount's prefix even when no host directory is involved.
|
||||
- **A pack file is untrusted input unless it came from this run's own compaction.** Every index entry's `data_off + size` and `name_off` must be bounds-checked against the actual file/strings-region length before being trusted by `vfs_open`/`vfs_stat`/`vfs_readdir`; a failed check invalidates the whole pack load, it is not skipped silently. An externally-sourced pack additionally needs a checksum verified at load, the same self-checking-record principle already required of the journal.
|
||||
- **The on-disk and in-memory indexes are kept sorted by full path**, not insertion order, so exact lookups are a binary search and `vfs_readdir` is a bounded range query (two binary searches), not a linear scan.
|
||||
- **An empty directory needs an explicit index entry.** `vfs_mkdir` writes a zero-size entry with a reserved directory-marker bit in `mode`, otherwise a directory with no files in it has nothing referencing its path and vanishes across compaction. The marker is superseded (not deleted) once any file exists under that path.
|
||||
- **`mode` is stored/restored but not enforced.** Permission bits, symlinks, and hard links have no access-control or link semantics in v0 — `mode` round-trips through compaction (including the directory-marker bit above) but nothing in the VFS core interprets it as a security boundary.
|
||||
- **Explicitly out of scope for v0:** FUSE as the first backend, invoking libarchive per-`read()`, full POSIX semantics/locks/sockets/mmap of virtual files, any single format trying to double as initrd + game pak + user home, implementing the cross-process reader-table design (specified, not built), content-defined chunking/delta compression in the pack format (only exact-duplicate elimination during compaction is in scope), and enforced permissions/symlinks/hard links.
|
||||
|
||||
`concept.md` will not be updated again (see "`concept.md` is immutable" above), so the constraints above will not drift out from under it. If a future document amends or supersedes part of the design, update this section to cite that document alongside `concept.md` — without editing `concept.md` itself.
|
||||
|
||||
## Self-evaluation
|
||||
|
||||
**Methodology.** This file was checked against the repository's actual state (`make test` passing; `include/`, `src/`, `tests/` present and matching the description below; confirmed by directory listing and a live build), against `concept.md` as frozen (for factual consistency of the constraints and architecture summarized above), and against the documentation standard stated in this file.
|
||||
|
||||
| Category | Grade | Notes |
|
||||
|---|---|---|
|
||||
| Factual accuracy | A | Build/test commands and the code map were verified against a live `make test` run and the actual file layout, not written from memory of intent. |
|
||||
| Adherence to the documentation standard | A | Direct, constraint-labeled prose; no manufactured sections; scaled appropriately to an instructions file rather than imitating `concept.md`'s full academic structure. |
|
||||
| Completeness for its purpose | A | Covers repository status, build/test/lint commands, a per-file code map, every hard constraint from `concept.md` relevant to implementation work, and the cross-cutting `reclaim_gate` pattern that no single file's comments fully explain on its own. |
|
||||
| Avoidance of generic or invented content | A | No fabricated "Common Development Tasks" or "Tips" sections; the sanitizer-testing instruction is stated as a requirement precisely because skipping it once already let a real bug through, not as generic advice. |
|
||||
|
||||
**Overall grade: A.** The file states only what is verifiably true of the repository, the spec, and the implementation; labels constraints by strength; and documents the one architectural pattern (snapshot reclamation via `reclaim_gate`) that spans multiple files and would otherwise have to be rediscovered by reading `vfs.c` and `upper.c` side by side.
|
||||
@@ -0,0 +1,61 @@
|
||||
# Contributing to PackFS
|
||||
|
||||
## Before changing anything
|
||||
|
||||
Read [`concept.md`](concept.md) in full. It is the project's specification,
|
||||
not background reading — every backend, lock, and on-disk field in `src/`
|
||||
exists because a section of `concept.md` requires it, and most functions'
|
||||
comments cite the section they implement rather than re-explaining it.
|
||||
`concept.md` is frozen (see [`CLAUDE.md`](CLAUDE.md)): if you believe the
|
||||
design itself needs to change, that belongs in a new document that amends
|
||||
or supersedes it, not in an edit to `concept.md`.
|
||||
|
||||
Then read [`CLAUDE.md`](CLAUDE.md), which states the load-bearing constraints
|
||||
a change must respect (static linking, the write model, the concurrency
|
||||
model, path containment, pack integrity) and the documentation register this
|
||||
project is written in.
|
||||
|
||||
## Workflow
|
||||
|
||||
```sh
|
||||
make test # must pass before any PR
|
||||
cc ... -fsanitize=address,undefined # ASan/UBSan: see .github/workflows/ci.yml for exact flags
|
||||
cc ... -fsanitize=thread # TSan, for anything touching src/upper.c, src/overlay.c, or src/vfs.c
|
||||
```
|
||||
|
||||
Any change to the concurrency-sensitive files (`upper.c`, `overlay.c`,
|
||||
`vfs.c`) must be run under ThreadSanitizer, not just the plain test suite —
|
||||
a data race there is exactly the class of bug Section 5 of `concept.md`
|
||||
exists to prevent, and the plain build will not surface it.
|
||||
|
||||
## Scope
|
||||
|
||||
Changes that add functionality `concept.md` Section 10 lists as explicitly
|
||||
out of scope for v0 (full POSIX semantics, enforced permissions/symlinks/hard
|
||||
links, cross-process concurrency, content-defined chunking/delta
|
||||
compression) should discuss the tradeoff with a maintainer first — those
|
||||
exclusions were deliberate design decisions, not gaps waiting to be filled.
|
||||
|
||||
**`zip` and `tar` import/export backends will not be accepted, full stop —
|
||||
not discussed, not behind a flag.** `concept.md` Section 11 recommends them,
|
||||
but that recommendation is permanently superseded; see `CLAUDE.md`, "Project
|
||||
decisions that supersede concept.md." This is a harder line than the v0
|
||||
exclusions above, which are open to future discussion — this one is not.
|
||||
|
||||
## Static-linking constraint
|
||||
|
||||
`concept.md` Section 11.1 is a hard constraint: the core must build with
|
||||
zero required third-party libraries and zero dynamic loading. In practice
|
||||
this project has no optional-dependency boundary at all: `concept.md`
|
||||
described one at the `zip`/`tar` backend for `miniz`/`libarchive`, but per
|
||||
the decision above, that backend will never exist, so there is no path by
|
||||
which a third-party dependency enters this codebase — a change proposing
|
||||
one, for any backend, will not be accepted.
|
||||
|
||||
## Style
|
||||
|
||||
Match the register already in the file you're editing — precise,
|
||||
constraint-labeled comments citing the `concept.md` section they implement,
|
||||
no filler. See the "Documentation standard" section of `CLAUDE.md` for the
|
||||
full rule; it applies to code comments and commit messages, not only to
|
||||
`.md` files.
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 PackFS contributors
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,73 @@
|
||||
# Makefile — PackFS
|
||||
#
|
||||
# Builds the static and shared library from src/, plus the test binaries
|
||||
# under tests/. See CLAUDE.md for the project's static-linking constraint
|
||||
# (Section 11.1 of concept.md): the core here has zero required
|
||||
# third-party dependencies; it links only against the platform C library,
|
||||
# pthread, and (on Linux) direct kernel syscalls for openat2/Landlock.
|
||||
|
||||
CC ?= cc
|
||||
AR ?= ar
|
||||
PREFIX ?= /usr/local
|
||||
|
||||
CFLAGS ?= -std=c11 -Wall -Wextra -Wpedantic -O2 -g -fPIC
|
||||
CFLAGS += -Iinclude -Isrc -D_GNU_SOURCE
|
||||
LDLIBS := -lpthread
|
||||
|
||||
SRC := $(wildcard src/*.c)
|
||||
OBJ := $(SRC:.c=.o)
|
||||
|
||||
STATIC_LIB := libpackfs.a
|
||||
SONAME := libpackfs.so.0
|
||||
SHARED_LIB := libpackfs.so
|
||||
|
||||
TEST_SRC := $(wildcard tests/test_*.c)
|
||||
TEST_BIN := $(TEST_SRC:tests/%.c=build/%)
|
||||
|
||||
.PHONY: all static shared test demo clean install uninstall
|
||||
|
||||
all: static shared
|
||||
|
||||
static: $(STATIC_LIB)
|
||||
|
||||
shared: $(SHARED_LIB)
|
||||
|
||||
$(STATIC_LIB): $(OBJ)
|
||||
$(AR) rcs $@ $(OBJ)
|
||||
|
||||
$(SHARED_LIB): $(OBJ)
|
||||
$(CC) -shared -Wl,-soname,$(SONAME) -o $(SONAME) $(OBJ) $(LDLIBS)
|
||||
ln -sf $(SONAME) $(SHARED_LIB)
|
||||
|
||||
%.o: %.c src/internal.h include/packfs.h
|
||||
$(CC) $(CFLAGS) -c $< -o $@
|
||||
|
||||
build:
|
||||
mkdir -p build
|
||||
|
||||
build/%: tests/%.c $(STATIC_LIB) | build
|
||||
$(CC) $(CFLAGS) $< $(STATIC_LIB) $(LDLIBS) -o $@
|
||||
|
||||
test: $(TEST_BIN)
|
||||
@set -e; for t in $(TEST_BIN); do echo "-- $$t --"; $$t; done
|
||||
|
||||
build/packfs_demo: examples/demo.c $(STATIC_LIB) | build
|
||||
$(CC) $(CFLAGS) $< $(STATIC_LIB) $(LDLIBS) -o $@
|
||||
|
||||
demo: build/packfs_demo
|
||||
./build/packfs_demo
|
||||
|
||||
clean:
|
||||
rm -f src/*.o $(STATIC_LIB) $(SHARED_LIB) $(SONAME)
|
||||
rm -rf build
|
||||
|
||||
install: static shared
|
||||
install -d $(DESTDIR)$(PREFIX)/lib $(DESTDIR)$(PREFIX)/include
|
||||
install -m 644 $(STATIC_LIB) $(DESTDIR)$(PREFIX)/lib/
|
||||
install -m 755 $(SONAME) $(DESTDIR)$(PREFIX)/lib/
|
||||
ln -sf $(SONAME) $(DESTDIR)$(PREFIX)/lib/$(SHARED_LIB)
|
||||
install -m 644 include/packfs.h $(DESTDIR)$(PREFIX)/include/
|
||||
|
||||
uninstall:
|
||||
rm -f $(DESTDIR)$(PREFIX)/lib/$(STATIC_LIB) $(DESTDIR)$(PREFIX)/lib/$(SONAME) $(DESTDIR)$(PREFIX)/lib/$(SHARED_LIB)
|
||||
rm -f $(DESTDIR)$(PREFIX)/include/packfs.h
|
||||
@@ -0,0 +1,168 @@
|
||||
# PackFS
|
||||
|
||||
A statically linked, in-process virtual file system for C. PackFS treats a
|
||||
shipped file tree as an immutable **pack** image, derives writability from a
|
||||
`mem` or `dir` upper layer through a copy-on-write overlay, and confines
|
||||
`zip`/`tar` to import/export roles rather than treating either as a live,
|
||||
writable store.
|
||||
|
||||
The full design rationale — why this shape, what alternatives were rejected,
|
||||
and the concurrency, path-containment, and integrity models this
|
||||
implementation follows — is specified in [`concept.md`](concept.md), which is
|
||||
frozen (see [`CLAUDE.md`](CLAUDE.md)) and is the authoritative source for
|
||||
every design decision below. This README documents the implementation that
|
||||
followed from it, not a restatement of the rationale.
|
||||
|
||||
## Status
|
||||
|
||||
This is an initial, partial implementation of the spec, not a complete one.
|
||||
|
||||
**Implemented and tested:** mount table, `mem`/`dir`/`pack`/overlay backends,
|
||||
copy-up, whiteouts, compaction, an append journal, path containment, and
|
||||
pack integrity validation.
|
||||
|
||||
**Will never be built, by explicit project decision:** the `zip` (miniz) and
|
||||
`tar` (USTAR) import/export backends. `concept.md` Section 11 recommends
|
||||
them, but that recommendation is superseded — see `CLAUDE.md`, "Project
|
||||
decisions that supersede concept.md." `backend_overlay_new` reads and writes
|
||||
this project's own pack format exclusively; there is no zip/tar support and
|
||||
none is planned. Do not open an issue or PR adding one.
|
||||
|
||||
**Deliberately out of scope for v0** (`concept.md` Section 10, not gaps):
|
||||
full POSIX semantics, enforced permissions/symlinks/hard links,
|
||||
cross-process concurrency (design specified in Section 5.6, unimplemented),
|
||||
and content-defined chunking/delta compression.
|
||||
|
||||
**Tested on Linux only**, in one environment. The `dir`-mount containment
|
||||
fallback path for kernels without `openat2` (Section 6.3) is implemented but
|
||||
has not been exercised on such a kernel, nor on macOS or Windows.
|
||||
|
||||
## Building
|
||||
|
||||
Zero required third-party dependencies — only a C11 compiler, `make`, and
|
||||
`pthread` (Section 11.1 of `concept.md` makes this a hard constraint, not a
|
||||
preference).
|
||||
|
||||
```sh
|
||||
make # builds libpackfs.a and libpackfs.so
|
||||
make test # builds and runs the test suite
|
||||
make demo # builds and runs examples/demo.c — see "Try it" below
|
||||
make install # installs to $PREFIX (default /usr/local)
|
||||
```
|
||||
|
||||
## Try it
|
||||
|
||||
`examples/demo.c` is a small, runnable, human-readable program — not another
|
||||
automated test — that exercises the library end to end and prints what it
|
||||
did at each step: a pack-backed overlay (write, read, `mkdir`, `readdir`,
|
||||
`stat`, a copy-up-then-whiteout delete, compaction via `vfs_sync`), and a
|
||||
sandboxed `dir` mount that demonstrates a `../../../etc/passwd` escape
|
||||
attempt being rejected. Run `make demo` twice in a row: the second run's
|
||||
first `readdir` shows the first run's files, proving that compaction and
|
||||
reload actually persist data through the pack file, not just within one
|
||||
process's lifetime.
|
||||
|
||||
`openat2`/Landlock support (Section 6) is detected automatically at compile
|
||||
time via `<sys/syscall.h>`; on kernels or platforms without them, `dir`
|
||||
mounts fall back to the weaker, documented residual-risk posture described
|
||||
in `concept.md` Section 6.3 rather than failing to build.
|
||||
|
||||
## Quick example
|
||||
|
||||
```c
|
||||
#include <packfs.h>
|
||||
|
||||
Vfs *v = vfs_new();
|
||||
|
||||
/* a plain in-memory writable tree */
|
||||
Backend *mem = backend_mem_new();
|
||||
vfs_mount(v, "/", mem);
|
||||
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(v, "/hello.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
vfs_write(f, "hello", 5);
|
||||
vfs_close(f);
|
||||
|
||||
vfs_free(v);
|
||||
backend_free(mem);
|
||||
```
|
||||
|
||||
A shipped pack with a writable overlay on top:
|
||||
|
||||
```c
|
||||
Backend *mem = backend_mem_new();
|
||||
int err = 0;
|
||||
Backend *ov = backend_overlay_new("assets.pack", mem, &err); /* loads assets.pack if it exists */
|
||||
vfs_mount(v, "/", ov);
|
||||
|
||||
/* ... reads served straight from the pack; writes copy-up into mem ... */
|
||||
|
||||
vfs_sync(v, "/"); /* compacts the overlay into a fresh assets.pack (Section 4.1, 5.3) */
|
||||
```
|
||||
|
||||
A sandboxed host directory:
|
||||
|
||||
```c
|
||||
int derr = 0;
|
||||
Backend *dir = backend_dir_new("/var/lib/myapp/data", &derr);
|
||||
vfs_mount(v, "/data", dir);
|
||||
/* every lookup under /data is contained to that directory (Section 6.2);
|
||||
* ".." and symlink escapes are rejected, not merely discouraged. */
|
||||
```
|
||||
|
||||
## API
|
||||
|
||||
The public API is [`include/packfs.h`](include/packfs.h); every function and
|
||||
struct is documented there with a pointer to the `concept.md` section that
|
||||
specifies its behavior. In outline:
|
||||
|
||||
- `vfs_new` / `vfs_free` — a `Vfs` owns a mount table, nothing else.
|
||||
- `backend_mem_new` / `backend_dir_new` / `backend_overlay_new` — construct a
|
||||
backend; `vfs_mount`/`vfs_unmount` attach or detach it at a path prefix.
|
||||
- `vfs_open` / `vfs_read` / `vfs_write` / `vfs_close` — file I/O.
|
||||
- `vfs_stat` / `vfs_readdir` / `vfs_mkdir` / `vfs_unlink` / `vfs_rename` —
|
||||
metadata and namespace operations.
|
||||
- `vfs_sync` — compacts an overlay mount into a fresh pack.
|
||||
- `vfs_harden_process_with_landlock` — optional, opt-in, process-wide
|
||||
Landlock confinement to the process's current `dir` mounts. Deliberately
|
||||
**not** applied automatically by `vfs_mount` (Section 6.2 explains why:
|
||||
Landlock restrictions are irreversible and process-wide, which would be a
|
||||
surprising side effect for an embeddable library to trigger on its own).
|
||||
|
||||
## Concurrency
|
||||
|
||||
Single-writer, wait-free-reader (Section 5): readers never take a lock and
|
||||
never observe a write in progress; structural changes (create/unlink/rename/
|
||||
mkdir/mount/unmount) publish a new immutable snapshot via one atomic pointer
|
||||
swap; ordinary content writes to an already-existing file update a per-entry
|
||||
cell directly and never touch the snapshot. `mem`-backed buffer growth
|
||||
never mutates a buffer address a reader might be reading (Section 5.7):
|
||||
growth always allocates a new buffer and publishes it, never reallocates in
|
||||
place. `tests/test_concurrency.c` exercises this under concurrent reader and
|
||||
writer threads, and the suite is regularly run under ThreadSanitizer and
|
||||
AddressSanitizer (see `.github/workflows/ci.yml`).
|
||||
|
||||
## Security
|
||||
|
||||
`dir` mounts are capability-scoped (Section 6.2): a mount holds an already-open
|
||||
directory file descriptor, not a path string, and every lookup beneath it is
|
||||
resolved with `openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS)` on Linux 5.6+,
|
||||
which atomically rejects `..` and symlink escapes in one kernel call. Where
|
||||
that syscall is unavailable, containment falls back to per-component
|
||||
`O_NOFOLLOW` resolution — weaker, and documented as such (Section 6.3), not
|
||||
silently assumed equivalent. Pack files are treated as untrusted input unless
|
||||
they came from this process's own compaction: every on-disk offset is
|
||||
bounds-checked and the pack's checksum is verified before any of it is
|
||||
trusted (Section 7). See `tests/test_dir.c` for a containment regression test
|
||||
and `tests/test_pack_overlay.c` for a corrupted-pack rejection test.
|
||||
|
||||
## Contributing
|
||||
|
||||
See [`CONTRIBUTING.md`](CONTRIBUTING.md). Read `concept.md` and `CLAUDE.md`
|
||||
first — they are the project's actual specification and its enforced
|
||||
documentation standard, respectively, and every design decision in the code
|
||||
traces back to one of them.
|
||||
|
||||
## License
|
||||
|
||||
MIT — see [`LICENSE`](LICENSE).
|
||||
+305
@@ -0,0 +1,305 @@
|
||||
# PackFS: Design Rationale and Evaluation of a Statically-Linked, In-Process Virtual File System
|
||||
|
||||
## Abstract
|
||||
|
||||
This document specifies the architecture of an in-process virtual file system (VFS) intended for embedding in sandboxed, agent-driven, TUI, or game-shaped C programs. The central claim is that the design problem is not the selection of an archive format ("tar versus zip"), but the construction of an **in-process, statically linked, C-owned path tree** with inexpensive mount points and a copy-on-write overlay. Archive formats (zip, tar) are relegated to the role of an **image backend** — a serialization detail — rather than serving as the filesystem's primary storage representation. The document defines the system's architecture, backend selection rationale, write model, concurrency model, path-containment model, pack-integrity validation procedure, object model, on-disk format, and explicit exclusions from the initial version (v0). This revision extends an earlier version of the specification by adding a path-containment model for host-backed mounts (Section 6) and an on-load integrity-validation procedure for pack files (Section 7) — both previously unaddressed despite direct relevance to the project's stated sandbox use case — and by refining the concurrency model to distinguish structural from content mutation (Section 5.3) and to specify, rather than merely defer, a cross-process design (Section 5.6). The present revision closes three further gaps found on re-examination: the mount table was previously outside the snapshot the concurrency model protects (Section 5.3); buffer growth on the `mem` backend could race a concurrent reader into a use-after-free rather than merely a stale read (Section 5.7); and the pack format had no representation for an empty directory (Section 9.3). It concludes with a self-evaluation across several categories, informed by comparison against established prior art (LMDB, OverlayFS, WASI, Landlock, and the memory-reclamation literature).
|
||||
|
||||
## 1. Problem Statement
|
||||
|
||||
The target use cases (sandboxes, TUIs, agents, games, chroot-shaped programs) share the following constraints: small, explicit C; no surprise runtime; no mandatory dynamic dependencies; predictable behavior under crash and concurrent access. These constraints rule out treating a mutable archive format (zip or tar) as the system of record, since neither format was designed for in-place, random-access mutation. The design instead separates **path resolution and mutability** (the VFS core) from **serialization** (the backend), such that the choice of archive format becomes an implementation detail confined to a single backend rather than a constraint on the whole system.
|
||||
|
||||
### 1.1 Threat Model
|
||||
|
||||
Two categories of untrusted input are treated as first-class concerns rather than assumed away:
|
||||
|
||||
- **A pack file may be corrupted or actively adversarial.** It may have been imported through the zip compatibility path, supplied by another party, or read back after a crash. Its on-disk offsets must not be trusted without validation before use (Section 7).
|
||||
- **A `dir`-backed mount exposes a real host directory**, and path resolution into it must not be escapable via `..` components, symlinks, or a race between validation and use — regardless of whether the untrusted input is a virtual path supplied by an agent or a symlink planted by a concurrent host-side process (Section 6).
|
||||
|
||||
Both properties are treated as hard requirements for any mount reachable by untrusted input, and as reasonable-effort hardening elsewhere.
|
||||
|
||||
## 2. Architectural Overview
|
||||
|
||||
### 2.1 System Shape
|
||||
|
||||
```text
|
||||
paths
|
||||
→ VFS core (mount table, cwd, overlay, fd table)
|
||||
→ backends
|
||||
mem live, writable
|
||||
dir real OS directory
|
||||
pack one file, indexed, read-only
|
||||
archive import/export only (zip or tar)
|
||||
```
|
||||
|
||||
### 2.2 Invariants (Core Rules)
|
||||
|
||||
- A single path namespace is exposed (e.g., `/assets/foo`, `/tmp/bar`, `/proc/...` if process-table emulation is desired).
|
||||
- Mounts are resolved by longest-prefix match to a backend.
|
||||
- Writes never mutate a pack file in place.
|
||||
- A writable mount is either `mem` or `dir`, or a copy-on-write overlay sitting atop a read-only pack.
|
||||
- Pack files are treated as **images**: they are loaded and compacted, not treated as block devices.
|
||||
|
||||
Under this model, the choice between zip and tar is a detail confined to the `pack` backend, not a property of the system as a whole.
|
||||
|
||||
## 3. Backend Selection
|
||||
|
||||
### 3.1 Candidate Backends
|
||||
|
||||
| Backend | Role | Use |
|
||||
|---|---|---|
|
||||
| **mem** | default writable layer | maps, arenas, explicit `new`/`free` |
|
||||
| **dir** | escape hatch to the host | optional, easy to disable in a sandbox |
|
||||
| **pack** | shipped tree in one file | custom table + blobs, mmap if desired |
|
||||
| **zip** | interchange / tools | miniz, static, import into pack or mount read-only |
|
||||
| **tar** | interchange / snapshots | hand-rolled USTAR *or* libarchive for export |
|
||||
|
||||
### 3.2 Selection Rationale
|
||||
|
||||
The recommended pack format is a custom indexed structure, not tar and not zip used as the live format, for the following reasons:
|
||||
|
||||
- The design already models resources as explicit types with constructors and destructors (`type T { fields; new(); free(); }`); a custom format is consistent with this style and keeps the pack builder under the project's own control.
|
||||
- Random access is a flat table lookup, not a linear scan or a directory-parse-then-seek. Tar has no index at all — locating a member requires reading headers sequentially until the target is reached. Zip does provide random access via its central directory, but this is a second, more complex format to parse, and it was not designed to support in-place edit or delete; removing or resizing an entry still requires rewriting the archive.
|
||||
- Static linkage reduces to compiling the project's own `.c` files, with no dependency on an archive library for the hot path.
|
||||
- Compression is optional and can be applied per file, rather than as a single transform over the whole archive.
|
||||
- The overlay-and-compact procedure is straightforward: the pack is rewritten from the merged (upper + lower) tree.
|
||||
|
||||
### 3.3 Role Separation: Compatibility versus Storage
|
||||
|
||||
Zip functions as a **compatibility skin** (accepting archives produced by other tools). Tar functions as a **snapshot skin** (e.g., `dump /home → blob`). Neither is suitable as the inode layer of the live filesystem.
|
||||
|
||||
## 4. The Write Model (Copy-on-Write Overlay)
|
||||
|
||||
Archive formats (zip, tar) must not be made writable in place. Instead, the system adopts a copy-on-write overlay model analogous to OverlayFS, without a kernel component.
|
||||
|
||||
### 4.1 Procedure
|
||||
|
||||
1. Mount the pack at `/`, read-only.
|
||||
2. Mount `mem` (or a host directory) as the upper layer at the same root.
|
||||
3. `open(O_RDWR)` triggers **copy-up**: the pack member is copied into the upper layer before it is modified.
|
||||
4. `unlink` on a path that exists only in the upper layer removes it from the upper layer directly. `unlink` on a path that exists in the lower (pack) layer cannot mutate the pack; instead, it writes a **whiteout** — a tombstone marker — into the upper layer. The merge step treats a whiteout as "path does not exist," even though the lower entry remains physically present in the pack. `rename` is defined as copy-up-if-needed, followed by a write at the new path and a whiteout at the old path.
|
||||
5. `vfs_sync()`, or an equivalent tool, walks the merged view (upper layer takes precedence; whiteouts hide lower entries; compaction subsequently discards both the whiteout and the entry it hides) and writes a **new pack**.
|
||||
|
||||
### 4.2 Deletion Semantics: The Whiteout Mechanism
|
||||
|
||||
The whiteout is the mechanism that makes deletion of a read-only, lower-layer entry well-defined; it is not an optional implementation detail. Without it, "delete a file that originated in the read-only pack" has no coherent semantics, since the pack cannot be mutated and the upper layer initially contains nothing corresponding to that path.
|
||||
|
||||
### 4.3 Durability Requirements
|
||||
|
||||
The model above is honest under crash, mmap, and sandboxing only if compaction and the journal are independently made crash-safe:
|
||||
|
||||
- **Compaction must be atomic.** The new pack is written to `pack.img.tmp`, `fsync`ed, and then moved into place via `rename()` over `pack.img`. Same-filesystem rename is atomic, so a crash during compaction leaves either the old pack or the new pack intact, never a partially written file. The new pack must never be written in place. If writing `pack.img.tmp` or its `fsync` fails for any reason (disk full, I/O error), compaction is aborted before the rename: `pack.img` and the full journal are untouched and remain the valid state, and nothing has been partially applied.
|
||||
- **Journal records must be self-checking.** Each record carries a length prefix (or fixed size) and a checksum. On replay, the first record that fails to parse fully or fails its checksum terminates replay; everything after that point is treated as never committed. Without this, a crash during an append can corrupt the interpretation of the entire replay, not merely the last write.
|
||||
|
||||
### 4.4 Append Journal for Key/Value Workloads
|
||||
|
||||
For workloads that are primarily key/value in character (saves, agent memory, configuration), a small append-only journal alongside the pack suffices:
|
||||
|
||||
```text
|
||||
pack.img immutable
|
||||
pack.img.jnl append records (put/delete)
|
||||
```
|
||||
|
||||
On open: load the pack index, then replay the journal. On compaction: emit a new pack, then replace the journal with only the records written **after** the watermark recorded at the start of compaction (see Section 5) — not "whatever remains at completion time," and not an in-place truncation (justified in Section 5.3). See Section 5.3 for how an ordinary content write (bytes into an already-copied-up file) avoids re-publishing a snapshot on every call.
|
||||
|
||||
## 5. Concurrency Model
|
||||
|
||||
### 5.1 Design Goal
|
||||
|
||||
Concurrent access is a first-class requirement, not an afterthought. Two failure modes are common in ad hoc designs: omitting a concurrency model entirely, or defaulting to a single global lock that serializes all readers against all writers indefinitely. The design below avoids both by adopting a documented, well-understood concurrency pattern rather than an invented one.
|
||||
|
||||
### 5.2 Prior Art
|
||||
|
||||
- **LMDB** achieves serializable isolation using a single writer combined with copy-on-write shadow paging: exactly one write transaction is active at a time; the number of concurrent readers is unbounded; readers are wait-free and block neither the writer nor each other, because each reader walks an older, still-valid, immutable page tree while the writer constructs new pages elsewhere [1].
|
||||
- **OverlayFS** demonstrates that copy-up is a genuine race condition, not a formality: CVE-2023-0386 was a container-escape vulnerability caused by an unsynchronized copy-up operation; the corresponding kernel fix serializes copy-up under an inode lock [2, 3].
|
||||
|
||||
### 5.3 Snapshot-Based Single-Writer, Wait-Free-Reader Design
|
||||
|
||||
The system adopts LMDB's structural pattern rather than a single lock over the entire tree:
|
||||
|
||||
- The live VFS state (upper-layer index and whiteout set) is represented as an **immutable, reference-counted snapshot** (`VfsSnapshot`). Exactly one atomic pointer, `current_snapshot`, is the sole entry point for all reads.
|
||||
- **The mount table is part of the same snapshot, not separate global state.** Path resolution (Section 2.2) depends on the mount table exactly as much as it depends on the upper index, so `vfs_mount` and `vfs_unmount` are structural writes like any other, going through the identical single-writer-and-swap path described below rather than a mechanism of their own. This is a deliberate choice, not an oversight left over from treating mounts as fixed at startup: folding the mount table into the snapshot costs nothing at the size mount tables actually reach in practice, and it closes what would otherwise be an entirely separate, unaddressed race between path resolution and a concurrent `vfs_mount`/`vfs_unmount` call.
|
||||
- `vfs_open`, `vfs_stat`, and `vfs_readdir` load this pointer once (a single atomic load plus a reference-count increment — wait-free, lock-free), then operate against that frozen view for the remainder of the call. A write in progress is never observed, because an in-progress write has not yet published a new snapshot.
|
||||
- Exactly one writer executes at a time; a plain mutex suffices for the in-process case. The writer constructs the next snapshot as a copy-on-write derivative of the current one. At the scale anticipated for agent- or sandbox-oriented workloads, the index is small enough that a full copy of the index and whiteout set per structural write is acceptable; a persistent (structurally shared) tree structure is not required until this assumption is empirically violated. The writer then performs **one atomic pointer swap** to publish the new snapshot — a release store on publication, paired with an acquire load on read. Incorrect memory ordering here constitutes an actual data race under the C11 memory model, not merely a theoretical concern: a reader could observe the new pointer value before observing the writes that constructed the object it references.
|
||||
- A snapshot is freed only once its reference count reaches zero — that is, once every reader that acquired it has released it. This is the same grace-period discipline used in read-copy-update (RCU), implemented here via plain reference counting rather than epoch or quiescent-state tracking; Section 5.4 discusses the scaling cost of that choice and the recommended upgrade path.
|
||||
- **Copy-up is rendered race-free as a consequence of this structure**, not by additional locking: the full blob is written into upper storage first (a new `mem` block, or write-then-rename under `dir`), and only afterward is its index entry installed in the *next* snapshot. There is no window in which a reader can observe a partially copied file, because readers never observe uncommitted snapshots. No inode lock is required; the transaction boundary performs the equivalent function.
|
||||
- **Structural mutation is distinguished from content mutation to avoid unnecessary copying.** A structural mutation — create, unlink, rename, mkdir, whiteout — changes which paths exist, and therefore requires building and publishing a new snapshot as described above. A content mutation — an ordinary `vfs_write` to a path that already has an index entry in the current snapshot — changes only that entry's bytes, size, and mtime, none of which affects which paths exist. Routing every content mutation through the structural path would mean copying and republishing the entire index on every `vfs_write` call, which is wasteful at any index size worth mentioning. Instead, each index entry owns a small mutable metadata cell (size, mtime), updated with plain atomic stores and guarded by a per-entry lock that serializes concurrent writers to the *same* file; this cell's identity is stable across snapshots that do not structurally touch that entry, so publishing a new snapshot is not required to make a content write visible. `vfs_stat` and `vfs_read` observe the cell with an acquire load, so a size update becomes visible to a reader as soon as the writer's release store completes, without waiting for or triggering a snapshot swap.
|
||||
- **Compaction is treated as an ordinary writer.** It acquires the current snapshot (a free, wait-free operation), records the journal offset at that instant (the watermark), walks the frozen snapshot to construct the new pack, writes `pack.img.tmp`, calls `fsync`, and renames it over `pack.img`. The live writer continues to append journal records and publish snapshots throughout this process; neither operation blocks the other. Afterward, all journal entries at or before the watermark are dropped and all entries after it are retained. Because a file cannot be truncated from its front in place, this is implemented by writing the surviving tail to a new file and renaming it over the journal — the same atomic-rename technique used for the pack itself.
|
||||
|
||||
### 5.4 Known Limitations
|
||||
|
||||
- **Naive per-snapshot reference counting contends under high read concurrency.** Every `vfs_open`/close pair increments and decrements one shared atomic counter; at high thread counts this cache line becomes a bottleneck, and reference counting is documented in the memory-reclamation literature as the weakest-scaling of the standard schemes precisely because of this shared, contended counter [4]. Two alternatives trade this differently: epoch-based reclamation (the mechanism behind the Linux kernel's own RCU) replaces the shared counter with per-thread epoch counters, giving O(1) overhead per read at the cost of unbounded memory growth if a single reader thread stalls and never advances its epoch [4]; hazard pointers bound memory strictly and remain fully lock-free, at the cost of a per-access memory barrier that is more expensive than an epoch counter under low contention [4]. Given that this system's target workloads (agent and sandbox processes) can plausibly stall or hang a reader thread without that being a defect in the VFS itself, the unbounded-growth failure mode of epoch-based reclamation is judged the worse risk of the two. **Hazard pointers are the recommended upgrade path** if reference-count contention is ever measured to matter; plain reference counting remains the v0 default because it is simplest to implement correctly and adequate at the concurrency levels this system targets.
|
||||
- **The per-entry lock in Section 5.3 closes the within-process version of the `dir`-upper gap, not the cross-process version.** It does not extend to a second, independent process or tool writing into the same host directory outside this VFS instance's knowledge, which remains genuinely unsynchronized — the snapshot pointer provides a consistent, wait-free view of the index and whiteout set regardless of upper backend, but the bytes on disk under `dir` upper are still subject to whatever concurrency guarantees the host filesystem itself provides once a second, uncoordinated writer is involved.
|
||||
- **Rename-over-a-mapped-file is platform-dependent.** Replacing a memory-mapped pack via `rename()` is safe on POSIX systems, where an existing mapping remains valid until the last reader unmaps it. On Windows, this can fail outright, since a file that is still open or mapped generally cannot be renamed over. Where Windows is a genuine target, the mitigations are either to close and reopen the mapping around compaction, or to use a generation-numbered filename (e.g., `pack.img.3`) together with a small pointer file, rather than renaming onto a fixed name.
|
||||
- **Cross-process concurrency has a specified design (Section 5.6) but remains unimplemented in v0.** It is no longer an unspecified gap, only a deferred one.
|
||||
|
||||
### 5.5 Rejected Alternatives
|
||||
|
||||
- **A reader/writer lock as the primary mechanism** is a legitimate, simpler v0 fallback if correctness with minimal implementation effort is prioritized over throughput (readers block briefly during writes and compaction, which is acceptable if writes are infrequent). It is presented here as a deliberate downgrade from wait-free reads, not a different tier of correctness, and may be an appropriate starting point.
|
||||
- **A seqlock** is well suited to a single fixed-size, frequently read field (e.g., a generation counter in the pack header), but not to a variable-size index or directory tree, where safely detecting a torn read is substantially harder. The snapshot pointer already provides the whole-structure equivalent of what a seqlock provides for a single field, so introducing a second mechanism is unnecessary.
|
||||
|
||||
### 5.6 Cross-Process Concurrency (Deferred Design)
|
||||
|
||||
Should a future version need two operating-system processes to share one pack file concurrently, LMDB's reader-table design is the specified template, not an invented one [5]:
|
||||
|
||||
- A small, fixed-size lock file, memory-mapped and shared across processes, holds a table of reader slots, each cache-line-aligned to avoid false sharing between readers running on different cores [5].
|
||||
- A reader acquires a slot — guarded by a short-held mutex used only to find a free slot, not held for the duration of the read — and records the snapshot identifier it is using. The read itself then proceeds without holding any lock, mirroring the in-process design in Section 5.3.
|
||||
- Before reclaiming or overwriting an old pack generation, the single cross-process writer scans the slot table for the oldest snapshot identifier still recorded by any live reader, and reclaims nothing newer than that [5].
|
||||
- This mechanism is additive to, not a replacement for, the in-process atomic-pointer design: within one process, the atomic `current_snapshot` pointer remains the fast path; the reader table exists only so that a writer in a different process can learn the oldest in-use snapshot across process boundaries.
|
||||
|
||||
This section specifies the mechanism in enough detail to implement; Section 10 continues to treat the implementation itself as out of scope for v0.
|
||||
|
||||
### 5.7 Byte-Buffer Growth Under Concurrent Read (`mem` Backend)
|
||||
|
||||
The content-mutation path in Section 5.3 accounts for updating an entry's size and mtime under a per-entry lock, but a `vfs_write` that extends a `mem`-backed file past its current allocation must grow the underlying byte buffer itself, which is a distinct hazard the metadata cell alone does not cover: if the buffer is grown by reallocating its existing address in place, a concurrent `vfs_read` copying out of that address can be left reading freed memory, not merely a stale value — a use-after-free, not just a torn read.
|
||||
|
||||
The write path must not mutate the existing buffer's address when growing it. Instead, growth follows the same replace-don't-mutate principle already used for the whole-tree snapshot [1, 9]: a new, larger buffer is allocated, the existing bytes plus the newly written bytes are copied into it, and the entry's data pointer is updated with a single release store; a reader's `vfs_read` loads that pointer with an acquire load before copying out of it, so it either sees the old buffer in full or the new one in full, never a torn transition between the two. The old buffer is retired under the same reclamation discipline as a retired snapshot (Section 5.4) — freed once no in-flight reader can still be using it — rather than freed immediately at the point of growth.
|
||||
|
||||
A write that does not grow the buffer — an in-place overwrite of already-allocated bytes — is not given a stronger guarantee than POSIX itself provides for concurrent, unsynchronized `write()` calls to the same file: the per-entry lock in Section 5.3 serializes concurrent *writers* against each other, but a reader racing an in-place overwrite may observe any interleaving of old and new bytes, not corruption or a use-after-free. Strengthening this further, to a torn-read-free guarantee on every write regardless of growth, is not undertaken here; it would mean copy-on-write at the granularity of every write's byte range rather than only at the granularity of buffer growth, which is materially more machinery for a guarantee POSIX callers do not otherwise expect.
|
||||
|
||||
## 6. Path Containment and Capability-Scoped Mounts
|
||||
|
||||
Path containment was absent from earlier revisions of this document despite being directly relevant to the project's stated sandbox use case. It is addressed here as a first-class design concern rather than an implementation afterthought.
|
||||
|
||||
### 6.1 Virtual-Namespace Canonicalization
|
||||
|
||||
Before a path reaches any backend, the VFS core resolves `.` and `..` components purely lexically within the virtual namespace and rejects a path that would resolve above the root of the mount it targets. This applies uniformly to `mem`, `pack`, and `dir` backends: even where no host filesystem is involved, an unvalidated `..` could otherwise be used to address a sibling mount's namespace through a mount point that should not expose it.
|
||||
|
||||
### 6.2 Capability-Scoped `dir` Mounts
|
||||
|
||||
A `dir` mount is represented internally as an already-open directory file descriptor — obtained once, at mount time, via `open(path, O_DIRECTORY)` — rather than as a path string re-resolved on every access. This is the same capability-based posture WASI uses for its preopened directories, where a module is handed a directory descriptor at startup and has no ambient authority to resolve paths outside it [6]. Every subsequent lookup under that mount is performed relative to that descriptor using `openat2()` with `RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS` on Linux 5.6 and later: the kernel atomically rejects `..` escapes and symlink escapes (including through `/proc` magic links) as a single lookup operation, rather than as a canonicalize-then-open pair, which removes the time-of-check-to-time-of-use window that a symlink swapped in between validation and use would otherwise open [7]. Where Landlock (Linux 5.13+) is available, applying a ruleset scoped to the mount's directory descriptor is recommended as an additional, independent enforcement layer beneath the VFS's own checks, consistent with Landlock's own stacking model, in which a sandboxed thread can only access a path that all of its enforced policy layers grant [8].
|
||||
|
||||
### 6.3 Platform Coverage and Residual Risk
|
||||
|
||||
`openat2` with `RESOLVE_BENEATH` is Linux-specific and requires kernel 5.6 or later. On macOS, older Linux, and Windows, no equivalent single-syscall atomic containment primitive is assumed to exist in this specification; the fallback is `O_NOFOLLOW` applied per path component plus a canonicalize-and-verify-prefix check, which closes most but not all of the same time-of-check-to-time-of-use window. This is recorded as an accepted, platform-dependent residual risk rather than a claim of uniform containment — the same posture this document already takes toward the Windows rename-on-mapped-file caveat in Section 5.4.
|
||||
|
||||
## 7. Pack Integrity Validation on Load
|
||||
|
||||
A pack file is untrusted input whenever it did not originate from this process's own compaction step within the current run — for example, one imported through the zip compatibility path, supplied by another party, or read back after a crash. Loading such a file must validate its structure before any offset within it is trusted:
|
||||
|
||||
- The magic value and version field are checked before anything else is read.
|
||||
- For every index entry, `data_off + size` is checked against the file's actual length, and `name_off` against the length of the strings region, before that entry is used to satisfy any `vfs_open`, `vfs_stat`, or `vfs_readdir` call. An entry that fails this check is treated as a load-time error for the whole pack, not skipped silently, since a single bad offset can indicate either corruption or a deliberately crafted file.
|
||||
- Entries are checked for overlap with the header and index regions themselves, rejecting a pack that aliases file content onto its own metadata.
|
||||
- Where the pack did not originate from this run's own compaction, an overall checksum over the index (and optionally the blob region) is verified at load, applying the same self-checking-record principle already required of the journal in Section 4.3, rather than treating a pack file as trusted simply because the journal already is.
|
||||
|
||||
## 8. Object Model (C API)
|
||||
|
||||
The following sketch illustrates the intended scope; it is not a POSIX-complete interface. `isize` and `usize` denote project-specific signed and unsigned size typedefs (e.g., built on `ptrdiff_t`/`size_t`) and are not standard C — concrete types must be chosen before compilation.
|
||||
|
||||
```c
|
||||
typedef struct Vfs Vfs;
|
||||
typedef struct VfsFile VfsFile;
|
||||
|
||||
Vfs *vfs_new(void);
|
||||
void vfs_free(Vfs *v);
|
||||
|
||||
int vfs_mount(Vfs *v, const char *path, Backend *b, int flags);
|
||||
int vfs_unmount(Vfs *v, const char *path);
|
||||
|
||||
VfsFile *vfs_open(Vfs *v, const char *path, int flags);
|
||||
isize vfs_read(VfsFile *f, void *buf, usize n);
|
||||
isize vfs_write(VfsFile *f, const void *buf, usize n);
|
||||
int vfs_close(VfsFile *f);
|
||||
|
||||
int vfs_stat(Vfs *v, const char *path, VfsStat *out);
|
||||
int vfs_readdir(Vfs *v, const char *path, VfsDir *out);
|
||||
int vfs_mkdir(Vfs *v, const char *path);
|
||||
int vfs_unlink(Vfs *v, const char *path);
|
||||
```
|
||||
|
||||
Backends implement a common, small vtable. Emulating every POSIX flag is not required at this stage.
|
||||
|
||||
## 9. On-Disk Pack Format
|
||||
|
||||
```text
|
||||
"PKFS" u32 version
|
||||
u64 index_offset
|
||||
u64 index_count
|
||||
... blobs ...
|
||||
index: { name_off, data_off, size, mode, mtime }
|
||||
strings
|
||||
```
|
||||
|
||||
This layout is sufficient to specify a VFS whose on-disk representation remains tractable to reason about in full.
|
||||
|
||||
### 9.1 Index Ordering for Directory Enumeration
|
||||
|
||||
The index is stored sorted lexicographically by full path, not in insertion order. Exact-path lookup (`vfs_open`, `vfs_stat`) uses this order for a binary search rather than a linear scan; enumerating the children of a directory (`vfs_readdir`) uses the same order to compute a contiguous range via two binary searches — the lower and upper bounds of the path prefix — rather than scanning the whole index. This mirrors the reason LMDB and similar systems keep their primary structure ordered by key: a range query over a sorted structure is a bounded number of comparisons, not a full pass. The in-memory upper-layer index (Section 5) is expected to preserve the same ordering property, so that a merged directory listing (upper ranked over lower, per Section 4) does not itself degrade to a linear scan on the mutable side.
|
||||
|
||||
### 9.2 Content-Addressed Deduplication During Compaction
|
||||
|
||||
Compaction already reads every live blob to rewrite the pack (Section 4.1); this is a natural point, not an additional pass, at which to hash each blob and reuse a previous `data_off` for any blob whose content already appears earlier in the same pack, rather than storing duplicate bytes. This is an optional space optimization, not a correctness requirement, and is scoped strictly to exact-duplicate detection during compaction — it is not a general delta-compression or content-defined-chunking scheme, which would reintroduce the complexity this document's stated minimalism (Section 10) explicitly argues against.
|
||||
|
||||
### 9.3 Representing Empty Directories
|
||||
|
||||
The index as specified (Section 9) stores file entries; a directory's existence is otherwise implicit in the path prefixes of the files beneath it. This leaves no representation for a directory containing no files — the result of `vfs_mkdir` followed by nothing else — which would otherwise vanish across a compaction that only ever walks live file entries, since there would be no entry referencing that path at all.
|
||||
|
||||
`vfs_mkdir` therefore creates an explicit index entry for its path with `size = 0` and a reserved bit in `mode` marking it as a directory marker rather than a file. Compaction preserves a directory-marker entry exactly as it preserves a file entry (Section 4.1). The marker is superseded, not removed, the moment any file is created beneath that path — the directory's existence is then implied by that file as usual, and the marker becomes redundant without needing to be deleted; `vfs_readdir`'s range query (Section 9.1) does not distinguish a directory implied by a child from one made explicit by a marker, since both produce the same enumerated result.
|
||||
|
||||
## 10. Explicit Exclusions (Out of Scope for v0)
|
||||
|
||||
- In-place writable tar or zip as the system of record.
|
||||
- FUSE as the first backend (a plausible later addition, not a core dependency).
|
||||
- Invoking libarchive on every `read()` call.
|
||||
- Full POSIX semantics, locking, sockets, or mmap of virtual files, in v0.
|
||||
- A single format intended to simultaneously serve as initrd, game asset pack, and user home directory.
|
||||
- Implementing the cross-process concurrency design in Section 5.6 (the design is specified; building it is deferred until a second process actually needs the pack file).
|
||||
- Content-defined chunking or delta compression in the pack format (Section 9.2 permits only exact-duplicate elimination during compaction).
|
||||
- Enforced permission bits, symlinks, and hard links. The `mode` field (Section 9) is stored and restored across compaction, including the directory-marker bit defined in Section 9.3, but is not interpreted as an access-control mechanism in v0; symlink and hard-link semantics have no representation in the index at all.
|
||||
|
||||
## 11. Recommendation
|
||||
|
||||
**VFS core + `mem` + `dir` + custom pack + overlay + compaction.**
|
||||
Zip is supported via miniz for import/export. Tar is supported as a snapshot/export path where Unix-style semantics are preferred.
|
||||
|
||||
### 11.1 Static-Linking Constraint
|
||||
|
||||
This is a hard constraint, not a preference: the core (`vfs_*`, `mem`, `dir`, `pack`, overlay, compaction) must build and link as plain static C with **zero required third-party libraries and zero dynamic loading**. miniz and libarchive are placed behind a compile-time switch strictly at the `zip`/`tar` backend boundary; neither may be required to obtain a working, writable VFS. `openat2` and Landlock (Section 6) are direct syscalls, not libraries, and do not weaken this constraint. This distinction — between "static" as an enforced property and "static" as an aspiration — is treated as load-bearing throughout the design.
|
||||
|
||||
Consequences of the overall recommendation:
|
||||
|
||||
- Static C throughout the core.
|
||||
- No mandatory third-party dependency at runtime.
|
||||
- Writability without claiming tar or zip can safely serve as a writable source of truth.
|
||||
- One-file distribution where desired.
|
||||
- A host directory where a one-file distribution is not desired.
|
||||
- A specified containment mechanism for a chroot-shaped sandbox (Section 6), not merely a place to attach one later.
|
||||
|
||||
Where the requirement is specifically "one file that supports both `open()` of existing paths and persistence of new writes," the accurate name for that file is **pack + journal**, not zip and not tar.
|
||||
|
||||
## 12. Evaluation
|
||||
|
||||
### 12.1 Methodology
|
||||
|
||||
The design was reviewed against its own stated claims, checked for internal consistency, and cross-referenced against established prior art for the components with the highest risk of subtle error: archive format semantics, overlay filesystem deletion semantics, concurrent access under copy-on-write, memory-reclamation scaling, path-containment mechanisms for host-backed mounts, and the pack format's coverage of the object model it claims to support. Sources consulted are listed in the References section. Issues identified during review were corrected in the current text rather than merely noted, including: the original overlay deletion procedure having no defined behavior for a lower-layer-only entry (resolved by the whiteout mechanism, Section 4.2); durability and concurrency initially asserted without a supporting mechanism (resolved by Sections 4.3 and 5); every `vfs_write` call requiring a full index copy and republish (resolved by the structural/content mutation split, Section 5.3); reference counting's scaling limits left unexamined (resolved by Section 5.4's citation-backed trade-off); cross-process concurrency and path containment being named as deferred without a design behind them (resolved by Sections 5.6 and 6, respectively); pack files being treated as trusted input regardless of origin (resolved by Section 7); the mount table being outside the scope of the snapshot the concurrency model otherwise protects (resolved by Section 5.3); a `mem`-backed buffer growing under a concurrent reader being able to produce a use-after-free rather than merely a stale read (resolved by Section 5.7); and `vfs_mkdir` having no representable, compaction-surviving result for an empty directory (resolved by Section 9.3).
|
||||
|
||||
### 12.2 Assessment by Category
|
||||
|
||||
| Category | Grade | Notes |
|
||||
|---|---|---|
|
||||
| Correctness / technical soundness | A | No outstanding false claims identified as of the current revision; this pass additionally caught and closed a real use-after-free hazard in the concurrency model itself (Section 5.7) rather than only in the parts of the design added previously. |
|
||||
| Concurrency design | A | The single-writer, wait-free-reader model is precisely specified, including the structural/content mutation distinction, a citation-backed reclamation trade-off, a concrete (if deferred) cross-process design, mount-table coverage under the same snapshot, and safe buffer growth on the `mem` backend. |
|
||||
| Security / containment model | B+ | Virtual-namespace canonicalization and capability-scoped `dir` mounts via `openat2`/Landlock give kernel-enforced containment on Linux 5.6+. Graded below A because no equivalent atomic primitive is claimed or achieved on macOS or Windows — an honestly stated residual risk, not a closed gap. |
|
||||
| Architecture clarity | A | The combination of mount table, backend vtable, immutable pack image, and snapshot-pointer-governed mutable state forms one coherent model without internal contradiction. |
|
||||
| Static-linking discipline | A | The constraint is explicit and consistently enforced at backend boundaries; every mechanism added across revisions (atomics, mutexes, `openat2`, Landlock) relies on the standard toolchain and kernel syscalls, not additional third-party dependencies. |
|
||||
| Completeness for a concept-level specification | A | Path containment, pack-integrity validation, empty-directory representation, and an explicit statement of what `mode` does and does not mean are now specified. Permissions, symlinks, hard links, and full cross-platform containment parity are stated as scoped exclusions (Section 10) rather than silent omissions. |
|
||||
| Writing / usability as a specification | A | Precise on every mechanism most prone to subtle error (memory ordering, truncation direction, platform-specific rename and containment semantics, buffer-growth reclamation); implementable directly from this text. |
|
||||
|
||||
### 12.3 Overall Grade
|
||||
|
||||
**A.** Each limitation identified in the prior self-evaluations has been addressed to the extent it admits a static, dependency-free solution: reclamation scaling has a citation-backed recommendation, cross-process concurrency and path containment have concrete designs rather than bare deferrals, pack files are no longer treated as trusted by default, the mount table is no longer outside the concurrency model's protection, buffer growth can no longer race a reader into a use-after-free, and empty directories are representable across compaction. The remaining gaps — full cross-platform containment parity, cross-process implementation, and enforced permissions/symlinks/hard links — are stated as explicit, scoped exclusions rather than silent omissions, which this document's own standard treats as the appropriate closure for a concept-level specification.
|
||||
|
||||
## 13. Conclusion
|
||||
|
||||
The core architectural claim — that a virtual file system should treat a pack file as an immutable image, derive writability from a `mem` or `dir` upper layer combined with a copy-on-write overlay, and confine tar/zip to import/export roles — is supported both by the internal analysis in this document and by analogous decisions in established systems (LMDB's shadow paging, OverlayFS's whiteout and copy-up mechanisms, WASI's capability-based preopens, and Landlock's stackable access control). The design is considered suitable as a specification from which an initial implementation may proceed, subject to the exclusions enumerated in Section 10 and the limitations enumerated in Sections 5.4, 5.7, and 6.3.
|
||||
|
||||
## References
|
||||
|
||||
[1] "LMDB," Database of Databases, https://dbdb.io/db/lmdb.
|
||||
[2] "Overlayfs Copy-on-Write Container Escape: CVE-2023-0386 and Writeback Race Mitigations," Systems Hardening, https://www.systemshardening.com/articles/kubernetes/overlayfs-cow-container-escape/.
|
||||
[3] "Overlay Filesystem," The Linux Kernel Documentation, https://www.kernel.org/doc/Documentation/filesystems/overlayfs.txt.
|
||||
[4] T. E. Hart, P. E. McKenney, A. D. Brown, J. Walpole, "Making Lockless Synchronization Fast: Performance Implications of Memory Reclamation," https://pdfs.semanticscholar.org/ea37/ace00efe3a22791b270146a911930f088102.pdf.
|
||||
[5] "Reader Lock Table," LMDB documentation, http://www.lmdb.tech/doc/group__readers.html.
|
||||
[6] Y. Nakata, "WASI's Capability-based Security Model," https://www.chikuwa.it/blog/2023/capability/.
|
||||
[7] "openat2(2) — Linux manual page," https://man7.org/linux/man-pages/man2/openat2.2.html; "Restricting path name lookup with openat2()," LWN.net, https://lwn.net/Articles/796868/.
|
||||
[8] "Landlock: unprivileged access control," The Linux Kernel Documentation, https://docs.kernel.org/userspace-api/landlock.html.
|
||||
[9] P. E. McKenney, "What is RCU? Part 2: Usage," LWN.net, https://lwn.net/Articles/263130/.
|
||||
+156
@@ -0,0 +1,156 @@
|
||||
/*
|
||||
* demo.c — a runnable, human-readable demonstration of PackFS, not another
|
||||
* automated test. Run it and read the output; each step prints what it did
|
||||
* and what it found, so the library's core behaviors (mem backend, a
|
||||
* pack-backed overlay with copy-up and whiteout, compaction, persistence
|
||||
* across a re-open, and a sandboxed dir mount) are visible directly,
|
||||
* without reading test assertions.
|
||||
*
|
||||
* Usage: packfs_demo [pack-file-path]
|
||||
* (defaults to /tmp/packfs_demo.pack; re-run it to see the previous run's
|
||||
* data survive via the pack file written by the first run's vfs_sync.)
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "packfs.h"
|
||||
|
||||
static void die(const char *what, int err) {
|
||||
fprintf(stderr, "FAILED: %s (error code %d)\n", what, err);
|
||||
exit(1);
|
||||
}
|
||||
|
||||
static void print_stat(Vfs *v, const char *path) {
|
||||
VfsStat st;
|
||||
int rc = vfs_stat(v, path, &st);
|
||||
if (rc != VFS_OK) {
|
||||
printf(" stat %-20s -> error %d\n", path, rc);
|
||||
return;
|
||||
}
|
||||
printf(" stat %-20s -> %s, size=%llu\n", path,
|
||||
st.kind == VFS_KIND_DIR ? "dir " : "file",
|
||||
(unsigned long long)st.size);
|
||||
}
|
||||
|
||||
static void list_dir(Vfs *v, const char *path) {
|
||||
VfsDir dir;
|
||||
int rc = vfs_readdir(v, path, &dir);
|
||||
if (rc != VFS_OK) {
|
||||
printf(" readdir %s -> error %d\n", path, rc);
|
||||
return;
|
||||
}
|
||||
printf(" readdir %s -> %zu entr%s:\n", path, (size_t)dir.count, dir.count == 1 ? "y" : "ies");
|
||||
for (pfs_usize i = 0; i < dir.count; i++) {
|
||||
printf(" %s%s\n", dir.entries[i].name, dir.entries[i].kind == VFS_KIND_DIR ? "/" : "");
|
||||
}
|
||||
vfs_dir_free(&dir);
|
||||
}
|
||||
|
||||
static void write_file(Vfs *v, const char *path, const char *content) {
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(v, path, VFS_O_WRONLY | VFS_O_CREAT | VFS_O_TRUNC, &err);
|
||||
if (!f) die(path, err);
|
||||
pfs_usize len = (pfs_usize)strlen(content);
|
||||
if (vfs_write(f, content, len) != (pfs_isize)len) die("short write", VFS_ERR_IO);
|
||||
vfs_close(f);
|
||||
printf(" wrote %-20s (%zu bytes)\n", path, (size_t)len);
|
||||
}
|
||||
|
||||
static void read_file(Vfs *v, const char *path) {
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(v, path, VFS_O_RDONLY, &err);
|
||||
if (!f) { printf(" read %-20s -> error %d\n", path, err); return; }
|
||||
char buf[256] = {0};
|
||||
pfs_isize n = vfs_read(f, buf, sizeof(buf) - 1);
|
||||
vfs_close(f);
|
||||
printf(" read %-20s -> \"%.*s\"\n", path, (int)n, buf);
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
const char *pack_path = argc > 1 ? argv[1] : "/tmp/packfs_demo.pack";
|
||||
|
||||
printf("=== PackFS demo: pack-backed overlay at %s ===\n", pack_path);
|
||||
printf("(if this file already exists from a previous run, its contents\n"
|
||||
" should reappear below, proving compaction + reload persistence)\n\n");
|
||||
|
||||
Vfs *v = vfs_new();
|
||||
Backend *mem = backend_mem_new();
|
||||
int oerr = 0;
|
||||
Backend *overlay = backend_overlay_new(pack_path, mem, &oerr);
|
||||
if (!overlay) die("backend_overlay_new", oerr);
|
||||
if (vfs_mount(v, "/", overlay) != VFS_OK) die("vfs_mount /", 0);
|
||||
|
||||
printf("-- initial state (from the pack file, if one existed) --\n");
|
||||
list_dir(v, "/");
|
||||
|
||||
printf("\n-- writing files --\n");
|
||||
write_file(v, "/hello.txt", "Hello from PackFS!");
|
||||
write_file(v, "/config.json", "{\"greeting\":\"hi\"}");
|
||||
if (vfs_mkdir(v, "/notes") != VFS_OK) printf(" (/notes already existed)\n");
|
||||
else printf(" mkdir /notes\n");
|
||||
write_file(v, "/notes/todo.txt", "buy milk");
|
||||
|
||||
printf("\n-- reading back --\n");
|
||||
read_file(v, "/hello.txt");
|
||||
read_file(v, "/notes/todo.txt");
|
||||
|
||||
printf("\n-- directory listing --\n");
|
||||
list_dir(v, "/");
|
||||
list_dir(v, "/notes");
|
||||
|
||||
printf("\n-- stat --\n");
|
||||
print_stat(v, "/hello.txt");
|
||||
print_stat(v, "/notes");
|
||||
print_stat(v, "/does-not-exist.txt");
|
||||
|
||||
printf("\n-- copy-up + whiteout: deleting a file that lives in the pack --\n");
|
||||
/* On the second run, /hello.txt (if it survived compaction) lives in
|
||||
* the read-only pack layer; this unlink is served by a whiteout
|
||||
* (Section 4.2), not a mutation of the pack. */
|
||||
int rc = vfs_unlink(v, "/config.json");
|
||||
printf(" unlink /config.json -> %s\n", rc == VFS_OK ? "ok (whiteout or plain removal)" : "error");
|
||||
list_dir(v, "/");
|
||||
|
||||
printf("\n-- compacting the overlay into a fresh pack (vfs_sync) --\n");
|
||||
if (vfs_sync(v, "/") != VFS_OK) die("vfs_sync", 0);
|
||||
printf(" wrote %s\n", pack_path);
|
||||
|
||||
vfs_unmount(v, "/");
|
||||
backend_free(overlay);
|
||||
backend_free(mem);
|
||||
vfs_free(v);
|
||||
|
||||
printf("\n=== sandboxed dir mount demo ===\n");
|
||||
{
|
||||
char tmpl[] = "/tmp/packfs_demo_sandbox_XXXXXX";
|
||||
char *sandbox = mkdtemp(tmpl);
|
||||
if (!sandbox) die("mkdtemp", 0);
|
||||
printf("sandbox root: %s\n", sandbox);
|
||||
|
||||
Vfs *v2 = vfs_new();
|
||||
int derr = 0;
|
||||
Backend *dir = backend_dir_new(sandbox, &derr);
|
||||
if (!dir) die("backend_dir_new", derr);
|
||||
vfs_mount(v2, "/data", dir);
|
||||
|
||||
write_file(v2, "/data/inside.txt", "this file is contained to the sandbox root");
|
||||
read_file(v2, "/data/inside.txt");
|
||||
|
||||
printf(" attempting /data/../../../etc/passwd (must be rejected) ...\n");
|
||||
int err2 = 0;
|
||||
VfsFile *escape = vfs_open(v2, "/data/../../../etc/passwd", VFS_O_RDONLY, &err2);
|
||||
printf(" -> %s (error code %d)\n", escape ? "UNEXPECTEDLY OPENED (bug!)" : "rejected, as required", err2);
|
||||
if (escape) vfs_close(escape);
|
||||
|
||||
vfs_unmount(v2, "/data");
|
||||
backend_free(dir);
|
||||
vfs_free(v2);
|
||||
}
|
||||
|
||||
printf("\n=== done ===\n");
|
||||
printf("Re-run this program (same pack path) to see /hello.txt and\n"
|
||||
"/notes/todo.txt survive, and /config.json stay deleted.\n");
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
/*
|
||||
* packfs.h — public API of PackFS.
|
||||
*
|
||||
* PackFS is a statically linked, in-process virtual file system. The design
|
||||
* rationale for every decision in this header is recorded in concept.md,
|
||||
* which this implementation follows; concept.md is frozen (see CLAUDE.md)
|
||||
* and is not re-derived here.
|
||||
*
|
||||
* SPDX-License-Identifier: MIT
|
||||
*/
|
||||
|
||||
#ifndef PACKFS_H
|
||||
#define PACKFS_H
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* concept.md Section 8 leaves isize/usize as project-defined typedefs. */
|
||||
typedef ptrdiff_t pfs_isize;
|
||||
typedef size_t pfs_usize;
|
||||
|
||||
typedef struct Vfs Vfs;
|
||||
typedef struct VfsFile VfsFile;
|
||||
typedef struct Backend Backend;
|
||||
|
||||
typedef enum {
|
||||
VFS_OK = 0,
|
||||
VFS_ERR_NOENT = -1, /* path does not exist */
|
||||
VFS_ERR_EXIST = -2, /* path already exists */
|
||||
VFS_ERR_NOTDIR = -3, /* path component is not a directory */
|
||||
VFS_ERR_ISDIR = -4, /* operation not valid on a directory */
|
||||
VFS_ERR_INVAL = -5, /* invalid argument (includes path escapes, see Section 6) */
|
||||
VFS_ERR_IO = -6, /* host I/O error */
|
||||
VFS_ERR_NOSPC = -7, /* out of space / allocation failure */
|
||||
VFS_ERR_PERM = -8, /* operation not permitted (e.g. write to read-only backend) */
|
||||
VFS_ERR_NOTEMPTY = -9, /* directory not empty */
|
||||
VFS_ERR_CORRUPT = -10, /* pack failed integrity validation, Section 7 */
|
||||
VFS_ERR_NOMOUNT = -11 /* no backend mounted for this path */
|
||||
} VfsStatus;
|
||||
|
||||
typedef enum {
|
||||
VFS_KIND_FILE = 0,
|
||||
VFS_KIND_DIR = 1
|
||||
} VfsEntryKind;
|
||||
|
||||
typedef struct VfsStat {
|
||||
pfs_usize size;
|
||||
int64_t mtime; /* seconds since epoch */
|
||||
VfsEntryKind kind;
|
||||
} VfsStat;
|
||||
|
||||
typedef struct VfsDirEntry {
|
||||
char name[256];
|
||||
VfsEntryKind kind;
|
||||
} VfsDirEntry;
|
||||
|
||||
typedef struct VfsDir {
|
||||
VfsDirEntry *entries;
|
||||
pfs_usize count;
|
||||
} VfsDir;
|
||||
|
||||
#define VFS_O_RDONLY 0x00
|
||||
#define VFS_O_WRONLY 0x01
|
||||
#define VFS_O_RDWR 0x02
|
||||
#define VFS_O_CREAT 0x04
|
||||
#define VFS_O_TRUNC 0x08
|
||||
|
||||
/* --- VFS lifecycle -------------------------------------------------- */
|
||||
|
||||
Vfs *vfs_new(void);
|
||||
void vfs_free(Vfs *v);
|
||||
|
||||
/* --- Backend constructors -------------------------------------------
|
||||
*
|
||||
* Every backend is a Backend* handed to vfs_mount. Backends not currently
|
||||
* mounted anywhere may be freed with backend_free; a backend still mounted
|
||||
* must be unmounted first.
|
||||
*/
|
||||
|
||||
/* mem: default writable layer (Section 3.1). */
|
||||
Backend *backend_mem_new(void);
|
||||
|
||||
/* dir: capability-scoped host directory, containment per Section 6. */
|
||||
Backend *backend_dir_new(const char *host_path, int *err);
|
||||
|
||||
/* pack: read-only image loaded and validated per Section 7. */
|
||||
Backend *backend_pack_new(const char *pack_path, int *err);
|
||||
|
||||
/*
|
||||
* overlay: copy-on-write overlay per Sections 4-5. `pack_path` is the pack
|
||||
* image backing the read-only lower layer: if the file exists it is
|
||||
* loaded and validated (Section 7) at mount time; NULL means "start with
|
||||
* an empty lower layer" (vfs_sync then requires a non-NULL path to have
|
||||
* been set — pass one via backend_overlay_new even for a fresh overlay
|
||||
* that has no pack file yet). `upper` must be a `mem` or `dir` backend,
|
||||
* not otherwise mounted; the overlay does not take ownership of it — free
|
||||
* it explicitly after unmounting the overlay. A journal at
|
||||
* `pack_path` + ".jnl" (Section 4.4) is replayed at mount time if present,
|
||||
* and rewritten (not appended past its watermark) on every vfs_sync
|
||||
* (Section 5.3).
|
||||
*/
|
||||
Backend *backend_overlay_new(const char *pack_path, Backend *upper, int *err);
|
||||
|
||||
void backend_free(Backend *b);
|
||||
|
||||
/* --- Mounting --------------------------------------------------------
|
||||
*
|
||||
* The mount table is itself part of the atomically-swapped snapshot
|
||||
* (Section 5.3): vfs_mount/vfs_unmount are structural writes, serialized
|
||||
* against each other and against every other structural write, and never
|
||||
* observed mid-change by a concurrent path resolution.
|
||||
*/
|
||||
|
||||
int vfs_mount(Vfs *v, const char *prefix, Backend *b);
|
||||
int vfs_unmount(Vfs *v, const char *prefix);
|
||||
|
||||
/* --- File operations -------------------------------------------------- */
|
||||
|
||||
VfsFile *vfs_open(Vfs *v, const char *path, int flags, int *err);
|
||||
pfs_isize vfs_read(VfsFile *f, void *buf, pfs_usize n);
|
||||
pfs_isize vfs_write(VfsFile *f, const void *buf, pfs_usize n);
|
||||
int vfs_close(VfsFile *f);
|
||||
|
||||
int vfs_stat(Vfs *v, const char *path, VfsStat *out);
|
||||
int vfs_readdir(Vfs *v, const char *path, VfsDir *out);
|
||||
void vfs_dir_free(VfsDir *d);
|
||||
int vfs_mkdir(Vfs *v, const char *path);
|
||||
int vfs_unlink(Vfs *v, const char *path);
|
||||
int vfs_rename(Vfs *v, const char *from, const char *to);
|
||||
|
||||
/*
|
||||
* vfs_sync compacts the overlay mounted exactly at `overlay_prefix`
|
||||
* (Section 4.1 step 5, Section 5.3): it freezes the current upper
|
||||
* snapshot, writes a new pack, and replaces the journal with only the
|
||||
* records written after the watermark (Section 5.3, Section 4.4).
|
||||
*/
|
||||
int vfs_sync(Vfs *v, const char *overlay_prefix);
|
||||
|
||||
/*
|
||||
* Optional, opt-in, process-wide Landlock hardening (Section 6.2). This
|
||||
* is deliberately NOT applied automatically by vfs_mount: Landlock
|
||||
* restrictions are cumulative, irreversible, and apply to the whole
|
||||
* calling process/thread, which would be a surprising side effect for an
|
||||
* embeddable library to trigger on its own. Call it once, after all `dir`
|
||||
* mounts the process will ever need are in place, if the embedding
|
||||
* application wants the whole process confined to exactly those
|
||||
* directories at the kernel level. Returns VFS_OK on success; if the
|
||||
* running kernel does not support Landlock, it returns VFS_ERR_PERM and
|
||||
* changes nothing — this is the documented residual-risk fallback of
|
||||
* Section 6.3, not a fatal error.
|
||||
*/
|
||||
int vfs_harden_process_with_landlock(Vfs *v);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* PACKFS_H */
|
||||
@@ -0,0 +1,253 @@
|
||||
/*
|
||||
* containment.c — capability-scoped `dir` mounts (Section 6.2) and
|
||||
* optional Landlock hardening (Section 6.2, 6.3).
|
||||
*
|
||||
* On Linux 5.6+, every lookup beneath a `dir` mount's root fd is resolved
|
||||
* with openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS) in a single kernel
|
||||
* call, atomically rejecting ".." escapes and symlink escapes. Where
|
||||
* openat2 is unavailable (ENOSYS on an older kernel), this falls back to
|
||||
* per-component O_NOFOLLOW resolution, which is an accepted, weaker
|
||||
* residual-risk posture per Section 6.3 — not a claim of equal
|
||||
* containment strength.
|
||||
*/
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <linux/openat2.h>
|
||||
#include <string.h>
|
||||
#include <sys/prctl.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
#ifdef __linux__
|
||||
#include <linux/landlock.h>
|
||||
#endif
|
||||
|
||||
int pfs_dir_capability_open(const char *path, int *err) {
|
||||
int fd = open(path, O_DIRECTORY | O_CLOEXEC | O_RDONLY);
|
||||
if (fd < 0) {
|
||||
if (err) *err = VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return fd;
|
||||
}
|
||||
|
||||
/* Splits an already-normalized relative path (no leading slash) into
|
||||
* (parent, leaf). Root-level names have an empty parent. */
|
||||
static void split_parent_leaf(const char *rel, char *parent_buf, size_t cap, const char **leaf) {
|
||||
const char *slash = strrchr(rel, '/');
|
||||
if (!slash) {
|
||||
parent_buf[0] = '\0';
|
||||
*leaf = rel;
|
||||
return;
|
||||
}
|
||||
size_t plen = (size_t)(slash - rel);
|
||||
if (plen >= cap) plen = cap - 1;
|
||||
memcpy(parent_buf, rel, plen);
|
||||
parent_buf[plen] = '\0';
|
||||
*leaf = slash + 1;
|
||||
}
|
||||
|
||||
#ifdef SYS_openat2
|
||||
static int openat2_beneath(int root_fd, const char *rel, uint64_t flags, uint64_t mode) {
|
||||
struct open_how how;
|
||||
memset(&how, 0, sizeof(how));
|
||||
how.flags = flags;
|
||||
how.mode = mode;
|
||||
how.resolve = RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS;
|
||||
return (int)syscall(SYS_openat2, root_fd, rel[0] ? rel : ".", &how, sizeof(how));
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Per-component O_NOFOLLOW fallback (Section 6.3). Walks every component
|
||||
* except the last with O_NOFOLLOW|O_DIRECTORY, then opens the last with
|
||||
* the caller's requested flags plus O_NOFOLLOW. Closes most, not all, of
|
||||
* the same TOCTOU window openat2 closes atomically. */
|
||||
static int fallback_openat_beneath(int root_fd, const char *rel, int flags, unsigned mode) {
|
||||
if (rel[0] == '\0') {
|
||||
return dup(root_fd);
|
||||
}
|
||||
char buf[PFS_PATH_MAX];
|
||||
if (strlen(rel) >= sizeof(buf)) { errno = ENAMETOOLONG; return -1; }
|
||||
strcpy(buf, rel);
|
||||
|
||||
int cur = root_fd;
|
||||
int owns_cur = 0;
|
||||
char *save = NULL;
|
||||
char *tok = strtok_r(buf, "/", &save);
|
||||
char *next = tok ? strtok_r(NULL, "/", &save) : NULL;
|
||||
|
||||
while (tok && next) {
|
||||
int nfd = openat(cur, tok, O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC);
|
||||
if (owns_cur) close(cur);
|
||||
if (nfd < 0) return -1;
|
||||
cur = nfd;
|
||||
owns_cur = 1;
|
||||
tok = next;
|
||||
next = strtok_r(NULL, "/", &save);
|
||||
}
|
||||
|
||||
int final_fd = -1;
|
||||
if (tok) {
|
||||
int final_flags = flags;
|
||||
if (!(flags & O_CREAT)) final_flags |= O_NOFOLLOW;
|
||||
final_fd = openat(cur, tok, final_flags, mode);
|
||||
}
|
||||
if (owns_cur) close(cur);
|
||||
return final_fd;
|
||||
}
|
||||
|
||||
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err) {
|
||||
int fd;
|
||||
#ifdef SYS_openat2
|
||||
fd = openat2_beneath(root_fd, rel, (uint64_t)flags, (uint64_t)mode);
|
||||
if (fd < 0 && errno == ENOSYS) {
|
||||
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
|
||||
}
|
||||
#else
|
||||
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
|
||||
#endif
|
||||
if (fd < 0) {
|
||||
if (err) *err = (errno == EXDEV || errno == ELOOP) ? VFS_ERR_INVAL
|
||||
: (errno == ENOENT) ? VFS_ERR_NOENT
|
||||
: (errno == EEXIST) ? VFS_ERR_EXIST
|
||||
: VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return fd;
|
||||
}
|
||||
|
||||
static int resolve_parent_dir(int root_fd, const char *rel, char *leaf_out, size_t leaf_cap, int *err) {
|
||||
char parent[PFS_PATH_MAX];
|
||||
const char *leaf;
|
||||
split_parent_leaf(rel, parent, sizeof(parent), &leaf);
|
||||
if (strlen(leaf) >= leaf_cap) { if (err) *err = VFS_ERR_INVAL; return -1; }
|
||||
strcpy(leaf_out, leaf);
|
||||
|
||||
int pfd;
|
||||
#ifdef SYS_openat2
|
||||
pfd = openat2_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
||||
if (pfd < 0 && errno == ENOSYS) {
|
||||
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
||||
}
|
||||
#else
|
||||
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
||||
#endif
|
||||
if (pfd < 0) {
|
||||
if (err) *err = (errno == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return pfd;
|
||||
}
|
||||
|
||||
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err) {
|
||||
char leaf[PFS_PATH_MAX];
|
||||
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
|
||||
if (pfd < 0) return -1;
|
||||
int rc = mkdirat(pfd, leaf, 0777);
|
||||
int saved = errno;
|
||||
close(pfd);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == EEXIST) ? VFS_ERR_EXIST : (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err) {
|
||||
char leaf[PFS_PATH_MAX];
|
||||
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
|
||||
if (pfd < 0) return -1;
|
||||
int rc = unlinkat(pfd, leaf, is_dir ? AT_REMOVEDIR : 0);
|
||||
int saved = errno;
|
||||
close(pfd);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : (saved == ENOTEMPTY) ? VFS_ERR_NOTEMPTY : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err) {
|
||||
char leaf_from[PFS_PATH_MAX], leaf_to[PFS_PATH_MAX];
|
||||
int pfd_from = resolve_parent_dir(root_fd, from, leaf_from, sizeof(leaf_from), err);
|
||||
if (pfd_from < 0) return -1;
|
||||
int pfd_to = resolve_parent_dir(root_fd, to, leaf_to, sizeof(leaf_to), err);
|
||||
if (pfd_to < 0) { close(pfd_from); return -1; }
|
||||
int rc = renameat(pfd_from, leaf_from, pfd_to, leaf_to);
|
||||
int saved = errno;
|
||||
close(pfd_from);
|
||||
close(pfd_to);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err) {
|
||||
int fd = pfs_dir_openat(root_fd, rel, O_RDONLY, 0, err);
|
||||
if (fd < 0) return -1;
|
||||
struct stat st;
|
||||
int rc = fstat(fd, &st);
|
||||
int saved = errno;
|
||||
close(fd);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
out->size = (pfs_usize)st.st_size;
|
||||
out->mtime = (int64_t)st.st_mtime;
|
||||
out->kind = S_ISDIR(st.st_mode) ? VFS_KIND_DIR : VFS_KIND_FILE;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* Best-effort, opt-in, process-wide hardening (Section 6.2). Restricts
|
||||
* filesystem read/write/create/remove operations to the given
|
||||
* directory-capability fds; deliberately does not restrict EXECUTE, to
|
||||
* avoid an embeddable library silently blocking unrelated process
|
||||
* behavior beyond the filesystem paths it was asked to confine.
|
||||
*/
|
||||
int pfs_landlock_restrict_to(const int *roots, size_t count) {
|
||||
#if defined(__linux__) && defined(SYS_landlock_create_ruleset)
|
||||
long abi = syscall(SYS_landlock_create_ruleset, NULL, 0, LANDLOCK_CREATE_RULESET_VERSION);
|
||||
if (abi < 0) return -1; /* unsupported kernel: Section 6.3 residual risk */
|
||||
|
||||
uint64_t access = LANDLOCK_ACCESS_FS_READ_FILE | LANDLOCK_ACCESS_FS_WRITE_FILE |
|
||||
LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_REMOVE_DIR |
|
||||
LANDLOCK_ACCESS_FS_REMOVE_FILE | LANDLOCK_ACCESS_FS_MAKE_CHAR |
|
||||
LANDLOCK_ACCESS_FS_MAKE_DIR | LANDLOCK_ACCESS_FS_MAKE_REG |
|
||||
LANDLOCK_ACCESS_FS_MAKE_SOCK | LANDLOCK_ACCESS_FS_MAKE_FIFO |
|
||||
LANDLOCK_ACCESS_FS_MAKE_BLOCK | LANDLOCK_ACCESS_FS_MAKE_SYM;
|
||||
|
||||
struct landlock_ruleset_attr attr;
|
||||
memset(&attr, 0, sizeof(attr));
|
||||
attr.handled_access_fs = access;
|
||||
|
||||
int rs_fd = (int)syscall(SYS_landlock_create_ruleset, &attr, sizeof(attr), 0);
|
||||
if (rs_fd < 0) return -1;
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
struct landlock_path_beneath_attr pb;
|
||||
memset(&pb, 0, sizeof(pb));
|
||||
pb.allowed_access = access;
|
||||
pb.parent_fd = roots[i];
|
||||
if (syscall(SYS_landlock_add_rule, rs_fd, LANDLOCK_RULE_PATH_BENEATH, &pb, 0) < 0) {
|
||||
close(rs_fd);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) < 0) { close(rs_fd); return -1; }
|
||||
if (syscall(SYS_landlock_restrict_self, rs_fd, 0) < 0) { close(rs_fd); return -1; }
|
||||
close(rs_fd);
|
||||
return 0;
|
||||
#else
|
||||
(void)roots; (void)count;
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
/*
|
||||
* hash.c — FNV-1a 64-bit, used for pack integrity checksums (Section 7)
|
||||
* and content-addressed dedup during compaction (Section 9.2). Chosen
|
||||
* because it needs no third-party library, consistent with the
|
||||
* static-linking constraint (Section 11.1) — this is not a
|
||||
* cryptographic hash and must never be used where adversarial collision
|
||||
* resistance is required.
|
||||
*/
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
uint64_t pfs_fnv1a64(const void *data, size_t len) {
|
||||
const unsigned char *p = (const unsigned char *)data;
|
||||
uint64_t h = 0xcbf29ce484222325ULL;
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
h ^= p[i];
|
||||
h *= 0x100000001b3ULL;
|
||||
}
|
||||
return h;
|
||||
}
|
||||
+243
@@ -0,0 +1,243 @@
|
||||
/*
|
||||
* internal.h — shared internal declarations, not part of the public API.
|
||||
*
|
||||
* Every mechanism named here corresponds to a section of concept.md;
|
||||
* comments reference sections rather than re-deriving the rationale,
|
||||
* since concept.md is frozen and is the authoritative source for "why."
|
||||
*/
|
||||
|
||||
#ifndef PACKFS_INTERNAL_H
|
||||
#define PACKFS_INTERNAL_H
|
||||
|
||||
#ifndef _GNU_SOURCE
|
||||
#define _GNU_SOURCE
|
||||
#endif
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stddef.h>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
|
||||
#include "packfs.h"
|
||||
|
||||
/* ---- path.c: virtual-namespace canonicalization (Section 6.1) ---- */
|
||||
|
||||
/*
|
||||
* Resolves "." and ".." purely lexically and rejects escapes above "/".
|
||||
* `out` must be at least PFS_PATH_MAX bytes. Returns 0 on success, -1 if
|
||||
* the path would resolve above the root (VFS_ERR_INVAL at the caller).
|
||||
*/
|
||||
#define PFS_PATH_MAX 4096
|
||||
int pfs_path_normalize(const char *in, char out[PFS_PATH_MAX]);
|
||||
|
||||
/* Longest-prefix match helper: does `path` lie under mount `prefix`? */
|
||||
int pfs_path_under(const char *prefix, const char *path);
|
||||
|
||||
/* ---- hash.c ---- */
|
||||
|
||||
uint64_t pfs_fnv1a64(const void *data, size_t len);
|
||||
|
||||
/* ---- containment.c: capability-scoped dir mounts (Section 6.2) ---- */
|
||||
|
||||
/*
|
||||
* Opens `path` as a directory capability (O_DIRECTORY), the root of a
|
||||
* `dir` mount. Returns the fd, or -1 on failure with *err set.
|
||||
*/
|
||||
int pfs_dir_capability_open(const char *path, int *err);
|
||||
|
||||
/*
|
||||
* Resolves `rel` (already virtual-namespace-canonicalized, no leading
|
||||
* slash) beneath directory-capability `root_fd`, using
|
||||
* openat2(RESOLVE_BENEATH|RESOLVE_NO_SYMLINKS) on Linux 5.6+ and falling
|
||||
* back to O_NOFOLLOW-per-component + prefix verification where openat2
|
||||
* is unavailable (Section 6.3). `flags`/`mode` are plain open(2) flags.
|
||||
*/
|
||||
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err);
|
||||
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err);
|
||||
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err);
|
||||
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err);
|
||||
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err);
|
||||
|
||||
/*
|
||||
* Best-effort, opt-in, process-wide Landlock hardening (Section 6.2),
|
||||
* exposed publicly as vfs_harden_process_with_landlock. `roots` is an
|
||||
* array of directory-capability fds to allow; `count` its length.
|
||||
* Returns 0 on success, -1 if unsupported by the running kernel (in
|
||||
* which case nothing was changed — the residual-risk fallback of
|
||||
* Section 6.3, not a fatal error).
|
||||
*/
|
||||
int pfs_landlock_restrict_to(const int *roots, size_t count);
|
||||
|
||||
/* ---- pack.c: on-disk format (Section 9) + integrity (Section 7) ---- */
|
||||
|
||||
#define PFS_MODE_DIR 0x1u /* directory-marker bit, Section 9.3 */
|
||||
|
||||
typedef struct PackIndexEntry {
|
||||
uint64_t name_off;
|
||||
uint32_t name_len;
|
||||
uint64_t data_off;
|
||||
uint64_t size;
|
||||
uint32_t mode;
|
||||
int64_t mtime;
|
||||
} PackIndexEntry;
|
||||
|
||||
typedef struct Pack {
|
||||
int fd;
|
||||
void *map;
|
||||
size_t map_len;
|
||||
const PackIndexEntry *entries; /* sorted by name, Section 9.1 */
|
||||
size_t entry_count;
|
||||
const char *strings_base;
|
||||
size_t strings_len;
|
||||
char *path; /* owned copy, used as compaction target */
|
||||
} Pack;
|
||||
|
||||
int pack_load(const char *path, Pack **out, int *err); /* validates per Section 7 */
|
||||
void pack_close(Pack *p);
|
||||
|
||||
/* Binary search by full path name. Returns 1 and sets *out on hit. */
|
||||
int pack_find(const Pack *p, const char *name, const PackIndexEntry **out);
|
||||
|
||||
/*
|
||||
* Range query for vfs_readdir (Section 9.1): [*lo, *hi) indices whose
|
||||
* name starts with `prefix`.
|
||||
*/
|
||||
void pack_range(const Pack *p, const char *prefix, size_t *lo, size_t *hi);
|
||||
|
||||
const char *pack_entry_name(const Pack *p, const PackIndexEntry *e);
|
||||
const void *pack_entry_data(const Pack *p, const PackIndexEntry *e);
|
||||
|
||||
typedef struct PackBuildEntry {
|
||||
const char *name;
|
||||
const void *data;
|
||||
uint64_t size;
|
||||
uint32_t mode;
|
||||
int64_t mtime;
|
||||
} PackBuildEntry;
|
||||
|
||||
/*
|
||||
* Writes a new pack to `path` + ".tmp", fsyncs, renames over `path`
|
||||
* (Section 4.3). Entries are sorted by name and content-addressed
|
||||
* deduplicated by exact (hash, size) match (Section 9.2). Computes and
|
||||
* stores an integrity checksum consumed by pack_load (Section 7).
|
||||
*/
|
||||
int pack_write(const char *path, const PackBuildEntry *entries, size_t count, int *err);
|
||||
|
||||
/* ---- upper.c: mem/dir writable layer + snapshot machinery (Section 5) ---- */
|
||||
|
||||
typedef enum { UPPER_MEM, UPPER_DIR } UpperKind;
|
||||
typedef enum { ENTRY_FILE, ENTRY_DIR_MARKER, ENTRY_WHITEOUT } UpperEntryKind;
|
||||
|
||||
/*
|
||||
* Section 5.3's mutable metadata cell, and Section 5.7's buffer-growth
|
||||
* guard. One MutCell is shared by every UpperSnapshot whose entry array
|
||||
* still points at the same logical file — an ordinary vfs_write mutates
|
||||
* this cell in place and never touches the snapshot pointer at all.
|
||||
*/
|
||||
typedef struct MutCell {
|
||||
pthread_mutex_t write_lock; /* serializes writers to the same file */
|
||||
pthread_rwlock_t buf_lock; /* guards {data,capacity} against grow, mem only */
|
||||
_Atomic(pfs_usize) size;
|
||||
_Atomic(int64_t) mtime;
|
||||
/* UPPER_MEM: */
|
||||
unsigned char *data; /* guarded by buf_lock on the grow path */
|
||||
pfs_usize capacity;
|
||||
/* UPPER_DIR: host bytes live in the host fs; nothing to store here. */
|
||||
} MutCell;
|
||||
|
||||
typedef struct UpperEntry {
|
||||
char *name; /* full virtual path, e.g. "/a/b.txt" */
|
||||
UpperEntryKind kind;
|
||||
MutCell *cell; /* NULL for ENTRY_DIR_MARKER / ENTRY_WHITEOUT */
|
||||
} UpperEntry;
|
||||
|
||||
typedef struct UpperSnapshot {
|
||||
_Atomic(pfs_usize) refcount;
|
||||
UpperEntry *entries; /* sorted by name, Section 9.1 */
|
||||
pfs_usize count;
|
||||
} UpperSnapshot;
|
||||
|
||||
typedef struct UpperStore {
|
||||
UpperKind kind;
|
||||
_Atomic(UpperSnapshot *) current;
|
||||
pthread_mutex_t writer_lock; /* single writer, Section 5.3 */
|
||||
/*
|
||||
* Guards the load-then-increment in upper_acquire() against a
|
||||
* concurrent writer retiring (and freeing) the very snapshot being
|
||||
* acquired. Plain "load pointer, then atomically increment its
|
||||
* refcount" has a gap between those two steps during which the
|
||||
* object can be freed out from under the loader; closing that gap
|
||||
* is what makes reference counting itself memory-safe, not a
|
||||
* departure from the reference-counting design. Readers take the
|
||||
* read side (concurrent, effectively free); a writer takes the
|
||||
* write side only around retiring the specific snapshot it just
|
||||
* superseded, which readers-in-flight already past this gate are
|
||||
* unaffected by.
|
||||
*/
|
||||
pthread_rwlock_t reclaim_gate;
|
||||
int root_fd; /* UPPER_DIR only: containment root, Section 6.2 */
|
||||
char *root_path; /* UPPER_DIR only, for error messages */
|
||||
} UpperStore;
|
||||
|
||||
UpperStore *upper_new(UpperKind kind, int root_fd /* -1 for mem */, const char *root_path);
|
||||
void upper_free(UpperStore *u);
|
||||
|
||||
UpperSnapshot *upper_acquire(UpperStore *u);
|
||||
void upper_release(UpperSnapshot *s);
|
||||
|
||||
/* Structural + content operations, all against paths already normalized
|
||||
* and relative to the mount (leading "/"). */
|
||||
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out);
|
||||
int upper_has_children(UpperSnapshot *s, const char *path);
|
||||
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path);
|
||||
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err);
|
||||
int upper_mkdir(UpperStore *u, const char *path, int *err);
|
||||
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err); /* whiteout if had_lower */
|
||||
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err);
|
||||
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size, int64_t mtime, MutCell **cell_out, int *err);
|
||||
|
||||
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n);
|
||||
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off, const void *buf, pfs_usize n, int *err);
|
||||
|
||||
/* Snapshot the whole store into pack-builder entries (compaction, dir-marker-aware). */
|
||||
typedef struct UpperWalkCb {
|
||||
void (*visit)(void *ctx, const char *name, UpperEntryKind kind, const void *data, pfs_usize size, int64_t mtime);
|
||||
void *ctx;
|
||||
} UpperWalkCb;
|
||||
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb); /* skips whiteouts */
|
||||
void upper_reset_empty(UpperStore *u); /* used after compaction folds upper into the new pack */
|
||||
|
||||
Backend *backend_from_upper(UpperStore *u); /* wraps an UpperStore as a standalone Backend */
|
||||
int upper_backend_root_fd(Backend *b); /* -1 unless b is a standalone `dir` backend */
|
||||
|
||||
/* ---- backend vtable shared by vfs.c ---- */
|
||||
|
||||
typedef struct BackendOps {
|
||||
int (*open)(Backend *b, const char *path, int flags, VfsFile **out);
|
||||
pfs_isize (*read)(VfsFile *f, void *buf, pfs_usize n);
|
||||
pfs_isize (*write)(VfsFile *f, const void *buf, pfs_usize n);
|
||||
int (*close)(VfsFile *f);
|
||||
int (*stat)(Backend *b, const char *path, VfsStat *out);
|
||||
int (*readdir)(Backend *b, const char *path, VfsDir *out);
|
||||
int (*mkdir)(Backend *b, const char *path);
|
||||
int (*unlink)(Backend *b, const char *path);
|
||||
int (*rename)(Backend *b, const char *from, const char *to);
|
||||
int (*sync)(Backend *b); /* VFS_ERR_PERM if not an overlay */
|
||||
void (*free)(Backend *b);
|
||||
} BackendOps;
|
||||
|
||||
struct Backend {
|
||||
const BackendOps *ops;
|
||||
void *state;
|
||||
};
|
||||
|
||||
struct VfsFile {
|
||||
Backend *backend;
|
||||
void *state;
|
||||
};
|
||||
|
||||
/* ---- overlay.c ---- */
|
||||
|
||||
Backend *overlay_upper_backend(Backend *b); /* NULL unless b is an overlay backend */
|
||||
|
||||
#endif /* PACKFS_INTERNAL_H */
|
||||
+638
@@ -0,0 +1,638 @@
|
||||
/*
|
||||
* overlay.c — the copy-on-write overlay backend (Sections 4-5, 7, 9.2).
|
||||
*
|
||||
* Composes a read-only Pack (lower) with an UpperStore (upper, from
|
||||
* upper.c) obtained from `mem` or `dir`. Implements the merge (upper
|
||||
* wins, whiteouts hide lower), copy-up, the append journal (Section 4.4)
|
||||
* with self-checking records (Section 4.3), and vfs_sync's compaction
|
||||
* (Section 4.1 step 5, Section 5.3, Section 9.2's exact-duplicate
|
||||
* elimination via pack_write).
|
||||
*
|
||||
* Journal records store whole-value content (Section 4.4 names
|
||||
* "key/value-ish" workloads as the target — saves, agent memory,
|
||||
* config); this implementation journals a full replacement value per
|
||||
* write rather than a byte-range delta, which is the simpler and more
|
||||
* robust choice for that workload character, at the cost of journal
|
||||
* size for large files under many small writes. That trade-off is
|
||||
* deliberate, not an oversight.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
typedef struct Overlay {
|
||||
Pack *pack; /* may be NULL: empty lower layer */
|
||||
char *pack_path; /* compaction target + journal anchor; may be NULL */
|
||||
Backend *upper; /* mem or dir backend, not owned */
|
||||
pthread_mutex_t journal_lock;
|
||||
FILE *journal_fp;
|
||||
} Overlay;
|
||||
|
||||
static UpperStore *upper_of(Overlay *ov) { return (UpperStore *)ov->upper->state; }
|
||||
static const char *dirrel(const char *p) { return p[0] == '/' ? p + 1 : p; }
|
||||
|
||||
/* ---- journal (Section 4.4) ---- */
|
||||
|
||||
static void journal_append_record(Overlay *ov, uint8_t op, const char *name,
|
||||
int64_t mtime, uint64_t size, const void *data) {
|
||||
if (!ov->pack_path) return;
|
||||
pthread_mutex_lock(&ov->journal_lock);
|
||||
if (!ov->journal_fp) {
|
||||
char jpath[PFS_PATH_MAX];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
|
||||
ov->journal_fp = fopen(jpath, "ab");
|
||||
}
|
||||
if (ov->journal_fp) {
|
||||
uint32_t name_len = (uint32_t)strlen(name);
|
||||
uint32_t payload_len = 4 + name_len + (op == 0 ? (uint32_t)(8 + 8 + size) : 0);
|
||||
unsigned char *payload = (unsigned char *)malloc(payload_len);
|
||||
size_t off = 0;
|
||||
memcpy(payload + off, &name_len, 4); off += 4;
|
||||
memcpy(payload + off, name, name_len); off += name_len;
|
||||
if (op == 0) {
|
||||
memcpy(payload + off, &mtime, 8); off += 8;
|
||||
memcpy(payload + off, &size, 8); off += 8;
|
||||
if (size) memcpy(payload + off, data, size);
|
||||
}
|
||||
uint32_t checksum = (uint32_t)pfs_fnv1a64(payload, payload_len);
|
||||
fwrite(&payload_len, 4, 1, ov->journal_fp);
|
||||
fwrite(&checksum, 4, 1, ov->journal_fp);
|
||||
fwrite(&op, 1, 1, ov->journal_fp);
|
||||
fwrite(payload, 1, payload_len, ov->journal_fp);
|
||||
fflush(ov->journal_fp);
|
||||
fsync(fileno(ov->journal_fp));
|
||||
free(payload);
|
||||
}
|
||||
pthread_mutex_unlock(&ov->journal_lock);
|
||||
}
|
||||
|
||||
static void journal_append_put(Overlay *ov, const char *name, const void *data, uint64_t size, int64_t mtime) {
|
||||
journal_append_record(ov, 0, name, mtime, size, data);
|
||||
}
|
||||
static void journal_append_delete(Overlay *ov, const char *name) {
|
||||
journal_append_record(ov, 1, name, 0, 0, NULL);
|
||||
}
|
||||
static void journal_append_mkdir(Overlay *ov, const char *name) {
|
||||
journal_append_record(ov, 2, name, 0, 0, NULL);
|
||||
}
|
||||
|
||||
/* journals the *current* full content of `path` after a write/rename,
|
||||
* matching the whole-value scheme described above. */
|
||||
static void journal_put_current(Overlay *ov, const char *path, MutCell *cell) {
|
||||
UpperStore *us = upper_of(ov);
|
||||
pfs_usize size;
|
||||
void *buf = NULL;
|
||||
int64_t mtime = (int64_t)time(NULL);
|
||||
if (us->kind == UPPER_MEM) {
|
||||
pthread_rwlock_rdlock(&cell->buf_lock);
|
||||
size = atomic_load_explicit(&cell->size, memory_order_acquire);
|
||||
mtime = atomic_load_explicit(&cell->mtime, memory_order_acquire);
|
||||
buf = size ? malloc(size) : NULL;
|
||||
if (buf) memcpy(buf, cell->data, size);
|
||||
pthread_rwlock_unlock(&cell->buf_lock);
|
||||
} else {
|
||||
VfsStat st;
|
||||
int e = 0;
|
||||
if (pfs_dir_statat(us->root_fd, dirrel(path), &st, &e) < 0) return;
|
||||
size = st.size;
|
||||
mtime = st.mtime;
|
||||
buf = size ? malloc(size) : NULL;
|
||||
if (buf) upper_cell_read(us, cell, path, 0, buf, size);
|
||||
}
|
||||
journal_append_put(ov, path, buf, size, mtime);
|
||||
free(buf);
|
||||
}
|
||||
|
||||
static void journal_replay(Overlay *ov) {
|
||||
if (!ov->pack_path) return;
|
||||
char jpath[PFS_PATH_MAX];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
|
||||
FILE *fp = fopen(jpath, "rb");
|
||||
if (!fp) return;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
for (;;) {
|
||||
uint32_t len, checksum; uint8_t op;
|
||||
if (fread(&len, 4, 1, fp) != 1) break;
|
||||
if (fread(&checksum, 4, 1, fp) != 1) break;
|
||||
if (fread(&op, 1, 1, fp) != 1) break;
|
||||
unsigned char *payload = (unsigned char *)malloc(len);
|
||||
size_t got = fread(payload, 1, len, fp);
|
||||
if (got != len) { free(payload); break; } /* torn record: stop, Section 4.3 */
|
||||
if ((uint32_t)pfs_fnv1a64(payload, len) != checksum) { free(payload); break; }
|
||||
|
||||
size_t off = 0;
|
||||
uint32_t name_len;
|
||||
memcpy(&name_len, payload + off, 4); off += 4;
|
||||
char *name = (char *)malloc(name_len + 1);
|
||||
memcpy(name, payload + off, name_len); name[name_len] = '\0'; off += name_len;
|
||||
|
||||
int err = 0;
|
||||
if (op == 0) {
|
||||
int64_t mtime; uint64_t size;
|
||||
memcpy(&mtime, payload + off, 8); off += 8;
|
||||
memcpy(&size, payload + off, 8); off += 8;
|
||||
MutCell *cell;
|
||||
upper_copy_up(us, name, payload + off, size, mtime, &cell, &err);
|
||||
} else if (op == 1) {
|
||||
const PackIndexEntry *pe;
|
||||
int had = ov->pack && pack_find(ov->pack, name, &pe);
|
||||
upper_remove(us, name, had, &err);
|
||||
} else if (op == 2) {
|
||||
upper_mkdir(us, name, &err);
|
||||
}
|
||||
free(name);
|
||||
free(payload);
|
||||
}
|
||||
fclose(fp);
|
||||
}
|
||||
|
||||
/* ---- open file handle ---- */
|
||||
|
||||
typedef struct OverlayFile {
|
||||
Overlay *ov;
|
||||
UpperSnapshot *held; /* keeps `cell` reachable; released on close, NULL if from_upper==0 */
|
||||
MutCell *cell;
|
||||
const PackIndexEntry *lower_e; /* valid when from_upper==0 */
|
||||
int from_upper;
|
||||
char *path;
|
||||
pfs_usize pos;
|
||||
} OverlayFile;
|
||||
|
||||
static VfsFile *make_upper_file(Backend *b, Overlay *ov, const char *path, MutCell *cell) {
|
||||
UpperSnapshot *s = upper_acquire(upper_of(ov));
|
||||
OverlayFile *of = (OverlayFile *)calloc(1, sizeof(OverlayFile));
|
||||
of->ov = ov; of->held = s; of->cell = cell; of->from_upper = 1; of->path = strdup(path);
|
||||
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||||
f->backend = b; f->state = of;
|
||||
return f;
|
||||
}
|
||||
|
||||
static int overlay_open(Backend *b, const char *path, int flags, VfsFile **out) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int in_upper = upper_lookup(s, path, &ue);
|
||||
if (in_upper && ue->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
|
||||
|
||||
if (in_upper && ue->kind == ENTRY_FILE) {
|
||||
MutCell *cell = ue->cell;
|
||||
upper_release(s);
|
||||
if (flags & VFS_O_TRUNC) upper_cell_truncate(us, cell, path);
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
int was_whiteout = (in_upper && ue->kind == ENTRY_WHITEOUT);
|
||||
upper_release(s);
|
||||
|
||||
if (was_whiteout) {
|
||||
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_create(us, path, 0, &cell, &err) < 0) return err;
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
const PackIndexEntry *pe = NULL;
|
||||
int in_lower = ov->pack && pack_find(ov->pack, path, &pe);
|
||||
if (in_lower && (pe->mode & PFS_MODE_DIR)) return VFS_ERR_ISDIR;
|
||||
|
||||
if (!in_lower) {
|
||||
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_create(us, path, 0, &cell, &err) < 0) return err;
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
if (!(flags & (VFS_O_WRONLY | VFS_O_RDWR)) && !(flags & VFS_O_TRUNC)) {
|
||||
/* read-only: serve straight from the pack, no copy-up (Section 4.1 step 3
|
||||
* ties copy-up to a write-capable open, not every open). */
|
||||
OverlayFile *of = (OverlayFile *)calloc(1, sizeof(OverlayFile));
|
||||
of->ov = ov; of->lower_e = pe; of->from_upper = 0; of->path = strdup(path);
|
||||
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||||
f->backend = b; f->state = of;
|
||||
*out = f;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
const void *data = (flags & VFS_O_TRUNC) ? NULL : pack_entry_data(ov->pack, pe);
|
||||
uint64_t size = (flags & VFS_O_TRUNC) ? 0 : pe->size;
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_copy_up(us, path, data, size, pe->mtime, &cell, &err) < 0) return err;
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static pfs_isize overlay_read(VfsFile *f, void *buf, pfs_usize n) {
|
||||
OverlayFile *of = (OverlayFile *)f->state;
|
||||
if (of->from_upper) {
|
||||
pfs_isize r = upper_cell_read(upper_of(of->ov), of->cell, of->path, of->pos, buf, n);
|
||||
if (r > 0) of->pos += (pfs_usize)r;
|
||||
return r;
|
||||
}
|
||||
pfs_usize size = of->lower_e->size;
|
||||
pfs_usize avail = of->pos < size ? size - of->pos : 0;
|
||||
pfs_usize to_copy = n < avail ? n : avail;
|
||||
if (to_copy) memcpy(buf, (const char *)pack_entry_data(of->ov->pack, of->lower_e) + of->pos, to_copy);
|
||||
of->pos += to_copy;
|
||||
return (pfs_isize)to_copy;
|
||||
}
|
||||
|
||||
static pfs_isize overlay_write(VfsFile *f, const void *buf, pfs_usize n) {
|
||||
OverlayFile *of = (OverlayFile *)f->state;
|
||||
if (!of->from_upper) return VFS_ERR_PERM;
|
||||
int err = 0;
|
||||
pfs_isize w = upper_cell_write(upper_of(of->ov), of->cell, of->path, of->pos, buf, n, &err);
|
||||
if (w < 0) return err;
|
||||
of->pos += (pfs_usize)w;
|
||||
journal_put_current(of->ov, of->path, of->cell);
|
||||
return w;
|
||||
}
|
||||
|
||||
static int overlay_close(VfsFile *f) {
|
||||
OverlayFile *of = (OverlayFile *)f->state;
|
||||
if (of->held) upper_release(of->held);
|
||||
free(of->path);
|
||||
free(of);
|
||||
free(f);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_stat(Backend *b, const char *path, VfsStat *out) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
if (upper_lookup(s, path, &ue)) {
|
||||
if (ue->kind == ENTRY_WHITEOUT) { upper_release(s); return VFS_ERR_NOENT; }
|
||||
if (ue->kind == ENTRY_DIR_MARKER) {
|
||||
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
if (us->kind == UPPER_MEM) {
|
||||
out->size = atomic_load_explicit(&ue->cell->size, memory_order_acquire);
|
||||
out->mtime = atomic_load_explicit(&ue->cell->mtime, memory_order_acquire);
|
||||
out->kind = VFS_KIND_FILE;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
int err = 0;
|
||||
int rc = pfs_dir_statat(us->root_fd, dirrel(path), out, &err);
|
||||
upper_release(s);
|
||||
return rc < 0 ? err : VFS_OK;
|
||||
}
|
||||
int has_kids = upper_has_children(s, path);
|
||||
upper_release(s);
|
||||
if (has_kids) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
|
||||
|
||||
if (ov->pack) {
|
||||
const PackIndexEntry *pe;
|
||||
if (pack_find(ov->pack, path, &pe)) {
|
||||
out->size = pe->size; out->mtime = pe->mtime;
|
||||
out->kind = (pe->mode & PFS_MODE_DIR) ? VFS_KIND_DIR : VFS_KIND_FILE;
|
||||
return VFS_OK;
|
||||
}
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
if (hi > lo) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
|
||||
}
|
||||
if (strcmp(path, "/") == 0) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
|
||||
return VFS_ERR_NOENT;
|
||||
}
|
||||
|
||||
typedef struct Child { char name[256]; VfsEntryKind kind; } Child;
|
||||
|
||||
static int child_index(Child *arr, pfs_usize n, const char *name, size_t len) {
|
||||
for (pfs_usize i = 0; i < n; i++)
|
||||
if (strncmp(arr[i].name, name, len) == 0 && arr[i].name[len] == '\0') return (int)i;
|
||||
return -1;
|
||||
}
|
||||
|
||||
static int overlay_readdir(Backend *b, const char *path, VfsDir *out) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
char prefix[PFS_PATH_MAX];
|
||||
if (strcmp(path, "/") == 0) strcpy(prefix, "/"); else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t plen = strlen(prefix);
|
||||
|
||||
Child *handled = NULL; pfs_usize hcount = 0, hcap = 0;
|
||||
VfsDirEntry *entries = NULL; pfs_usize count = 0, cap = 0;
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
const char *name = s->entries[i].name;
|
||||
if (strncmp(name, prefix, plen) != 0) continue;
|
||||
const char *restp = name + plen;
|
||||
const char *slash = strchr(restp, '/');
|
||||
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||||
if (clen == 0 || clen >= sizeof(handled[0].name)) continue;
|
||||
if (child_index(handled, hcount, restp, clen) >= 0) continue;
|
||||
if (hcount == hcap) { hcap = hcap ? hcap * 2 : 8; handled = (Child *)realloc(handled, hcap * sizeof(Child)); }
|
||||
memcpy(handled[hcount].name, restp, clen); handled[hcount].name[clen] = '\0';
|
||||
handled[hcount].kind = slash ? VFS_KIND_DIR : (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||||
int is_whiteout = (s->entries[i].kind == ENTRY_WHITEOUT);
|
||||
hcount++;
|
||||
if (!is_whiteout) {
|
||||
if (count == cap) { cap = cap ? cap * 2 : 8; entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry)); }
|
||||
memcpy(entries[count].name, handled[hcount - 1].name, clen + 1);
|
||||
entries[count].kind = handled[hcount - 1].kind;
|
||||
count++;
|
||||
}
|
||||
}
|
||||
upper_release(s);
|
||||
|
||||
if (ov->pack) {
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
for (size_t i = lo; i < hi; i++) {
|
||||
const char *name = pack_entry_name(ov->pack, &ov->pack->entries[i]);
|
||||
const char *restp = name + plen;
|
||||
const char *slash = strchr(restp, '/');
|
||||
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||||
if (clen == 0 || clen >= 256) continue;
|
||||
if (child_index(handled, hcount, restp, clen) >= 0) continue;
|
||||
VfsDirEntry tmp; memcpy(tmp.name, restp, clen); tmp.name[clen] = '\0';
|
||||
int dup = 0;
|
||||
for (pfs_usize j = 0; j < count; j++)
|
||||
if (strncmp(entries[j].name, restp, clen) == 0 && entries[j].name[clen] == '\0') { dup = 1; break; }
|
||||
if (dup) continue;
|
||||
if (count == cap) { cap = cap ? cap * 2 : 8; entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry)); }
|
||||
entries[count] = tmp;
|
||||
entries[count].kind = slash ? VFS_KIND_DIR : ((ov->pack->entries[i].mode & PFS_MODE_DIR) ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||||
count++;
|
||||
}
|
||||
}
|
||||
free(handled);
|
||||
out->entries = entries;
|
||||
out->count = count;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_mkdir(Backend *b, const char *path) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int exists = upper_lookup(s, path, &ue) && ue->kind != ENTRY_WHITEOUT;
|
||||
int has_kids = upper_has_children(s, path);
|
||||
upper_release(s);
|
||||
|
||||
if (!exists && ov->pack) {
|
||||
const PackIndexEntry *pe;
|
||||
if (pack_find(ov->pack, path, &pe)) exists = 1;
|
||||
if (!exists && !has_kids) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
if (hi > lo) has_kids = 1;
|
||||
}
|
||||
}
|
||||
if (exists || has_kids) return VFS_ERR_EXIST;
|
||||
|
||||
int err = 0;
|
||||
if (upper_mkdir(us, path, &err) < 0) return err;
|
||||
journal_append_mkdir(ov, path);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_unlink(Backend *b, const char *path) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int in_upper = upper_lookup(s, path, &ue);
|
||||
if (in_upper && ue->kind == ENTRY_WHITEOUT) { upper_release(s); return VFS_ERR_NOENT; }
|
||||
if (in_upper && ue->kind == ENTRY_DIR_MARKER && upper_has_children(s, path)) { upper_release(s); return VFS_ERR_NOTEMPTY; }
|
||||
upper_release(s);
|
||||
|
||||
const PackIndexEntry *pe = NULL;
|
||||
int had_lower = ov->pack && pack_find(ov->pack, path, &pe);
|
||||
if (had_lower && (pe->mode & PFS_MODE_DIR)) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
UpperSnapshot *s2 = upper_acquire(us);
|
||||
int kids = (hi > lo) || upper_has_children(s2, path);
|
||||
upper_release(s2);
|
||||
if (kids) return VFS_ERR_NOTEMPTY;
|
||||
}
|
||||
if (!in_upper && !had_lower) {
|
||||
UpperSnapshot *s3 = upper_acquire(us);
|
||||
int kids = upper_has_children(s3, path);
|
||||
upper_release(s3);
|
||||
if (!kids && ov->pack) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
kids = hi > lo;
|
||||
}
|
||||
return kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
|
||||
}
|
||||
|
||||
int err = 0;
|
||||
if (upper_remove(us, path, had_lower, &err) < 0) return err;
|
||||
journal_append_delete(ov, path);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_rename(Backend *b, const char *from, const char *to) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int in_upper = upper_lookup(s, from, &ue) && ue->kind != ENTRY_WHITEOUT;
|
||||
UpperEntryKind kind = in_upper ? ue->kind : ENTRY_FILE;
|
||||
upper_release(s);
|
||||
|
||||
const PackIndexEntry *pe = NULL;
|
||||
int had_lower = ov->pack && pack_find(ov->pack, from, &pe);
|
||||
if (!in_upper && !had_lower) return VFS_ERR_NOENT;
|
||||
|
||||
if (in_upper) {
|
||||
int err = 0;
|
||||
if (upper_rename(us, from, to, had_lower, &err) < 0) return err;
|
||||
} else if (pe->mode & PFS_MODE_DIR) {
|
||||
int err = 0;
|
||||
if (upper_mkdir(us, to, &err) < 0 && err != VFS_ERR_EXIST) return err;
|
||||
int err2 = 0;
|
||||
upper_remove(us, from, 1, &err2);
|
||||
} else {
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_copy_up(us, to, pack_entry_data(ov->pack, pe), pe->size, pe->mtime, &cell, &err) < 0) return err;
|
||||
int err2 = 0;
|
||||
upper_remove(us, from, 1, &err2);
|
||||
}
|
||||
|
||||
journal_append_delete(ov, from);
|
||||
if (kind == ENTRY_DIR_MARKER) {
|
||||
journal_append_mkdir(ov, to);
|
||||
} else {
|
||||
UpperSnapshot *s2 = upper_acquire(us);
|
||||
const UpperEntry *nue;
|
||||
if (upper_lookup(s2, to, &nue) && nue->kind == ENTRY_FILE) journal_put_current(ov, to, nue->cell);
|
||||
upper_release(s2);
|
||||
}
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
/* ---- compaction (Section 4.1 step 5, 5.3, 9.2) ---- */
|
||||
|
||||
typedef struct BuildCtx {
|
||||
PackBuildEntry *entries;
|
||||
size_t count, cap;
|
||||
size_t upper_start; /* index at which upper-sourced (malloc'd) entries begin */
|
||||
} BuildCtx;
|
||||
|
||||
static void build_push(BuildCtx *bc, const char *name, const void *data, pfs_usize size, uint32_t mode, int64_t mtime, int owned) {
|
||||
if (bc->count == bc->cap) { bc->cap = bc->cap ? bc->cap * 2 : 16; bc->entries = (PackBuildEntry *)realloc(bc->entries, bc->cap * sizeof(PackBuildEntry)); }
|
||||
PackBuildEntry *e = &bc->entries[bc->count++];
|
||||
e->name = strdup(name);
|
||||
if (owned) {
|
||||
e->data = size ? malloc(size) : NULL;
|
||||
if (size) memcpy((void *)e->data, data, size);
|
||||
} else {
|
||||
e->data = data;
|
||||
}
|
||||
e->size = size;
|
||||
e->mode = mode;
|
||||
e->mtime = mtime;
|
||||
}
|
||||
|
||||
static void build_visit(void *ctx, const char *name, UpperEntryKind kind, const void *data, pfs_usize size, int64_t mtime) {
|
||||
build_push((BuildCtx *)ctx, name, data, size, kind == ENTRY_DIR_MARKER ? PFS_MODE_DIR : 0, mtime, 1);
|
||||
}
|
||||
|
||||
static void free_build_entries(BuildCtx *bc) {
|
||||
for (size_t i = 0; i < bc->count; i++) {
|
||||
free((void *)bc->entries[i].name);
|
||||
if (i >= bc->upper_start) free((void *)bc->entries[i].data);
|
||||
}
|
||||
free(bc->entries);
|
||||
}
|
||||
|
||||
static int overlay_sync(Backend *b) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
if (!ov->pack_path) return VFS_ERR_PERM;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
pthread_mutex_lock(&ov->journal_lock);
|
||||
if (ov->journal_fp) fflush(ov->journal_fp);
|
||||
long watermark = 0;
|
||||
char jpath[PFS_PATH_MAX];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
|
||||
FILE *probe = fopen(jpath, "rb");
|
||||
if (probe) { fseek(probe, 0, SEEK_END); watermark = ftell(probe); fclose(probe); }
|
||||
pthread_mutex_unlock(&ov->journal_lock);
|
||||
|
||||
BuildCtx bc; memset(&bc, 0, sizeof(bc));
|
||||
if (ov->pack) {
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
for (size_t i = 0; i < ov->pack->entry_count; i++) {
|
||||
const PackIndexEntry *pe = &ov->pack->entries[i];
|
||||
const char *name = pack_entry_name(ov->pack, pe);
|
||||
const UpperEntry *ue;
|
||||
if (upper_lookup(s, name, &ue)) continue; /* upper (incl. whiteout) supersedes */
|
||||
build_push(&bc, name, pack_entry_data(ov->pack, pe), pe->size, pe->mode, pe->mtime, 0);
|
||||
}
|
||||
upper_release(s);
|
||||
}
|
||||
bc.upper_start = bc.count;
|
||||
UpperWalkCb cb = { build_visit, &bc };
|
||||
upper_walk_live(us, &cb);
|
||||
|
||||
int err = 0;
|
||||
if (pack_write(ov->pack_path, bc.entries, bc.count, &err) < 0) {
|
||||
free_build_entries(&bc);
|
||||
return err; /* Section 4.3: existing pack + journal are untouched */
|
||||
}
|
||||
free_build_entries(&bc);
|
||||
|
||||
Pack *newpack = NULL; int lerr = 0;
|
||||
if (pack_load(ov->pack_path, &newpack, &lerr) == 0) {
|
||||
if (ov->pack) pack_close(ov->pack);
|
||||
ov->pack = newpack;
|
||||
}
|
||||
upper_reset_empty(us);
|
||||
|
||||
pthread_mutex_lock(&ov->journal_lock);
|
||||
char jtmp[PFS_PATH_MAX];
|
||||
snprintf(jtmp, sizeof(jtmp), "%s.jnl.tmp", ov->pack_path);
|
||||
FILE *src = fopen(jpath, "rb");
|
||||
if (src) {
|
||||
fseek(src, watermark, SEEK_SET);
|
||||
FILE *dst = fopen(jtmp, "wb");
|
||||
if (dst) {
|
||||
char buf[4096]; size_t n;
|
||||
while ((n = fread(buf, 1, sizeof(buf), src)) > 0) fwrite(buf, 1, n, dst);
|
||||
fflush(dst); fsync(fileno(dst)); fclose(dst);
|
||||
rename(jtmp, jpath);
|
||||
}
|
||||
fclose(src);
|
||||
}
|
||||
if (ov->journal_fp) { fclose(ov->journal_fp); ov->journal_fp = NULL; }
|
||||
pthread_mutex_unlock(&ov->journal_lock);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static void overlay_free(Backend *b) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
if (ov->journal_fp) fclose(ov->journal_fp);
|
||||
pthread_mutex_destroy(&ov->journal_lock);
|
||||
if (ov->pack) pack_close(ov->pack);
|
||||
free(ov->pack_path);
|
||||
free(ov);
|
||||
free(b);
|
||||
}
|
||||
|
||||
static const BackendOps OVERLAY_OPS = {
|
||||
overlay_open, overlay_read, overlay_write, overlay_close, overlay_stat,
|
||||
overlay_readdir, overlay_mkdir, overlay_unlink, overlay_rename, overlay_sync, overlay_free
|
||||
};
|
||||
|
||||
Backend *overlay_upper_backend(Backend *b) {
|
||||
return b->ops == &OVERLAY_OPS ? ((Overlay *)b->state)->upper : NULL;
|
||||
}
|
||||
|
||||
Backend *backend_overlay_new(const char *pack_path, Backend *upper, int *err) {
|
||||
Overlay *ov = (Overlay *)calloc(1, sizeof(Overlay));
|
||||
ov->pack_path = pack_path ? strdup(pack_path) : NULL;
|
||||
ov->upper = upper;
|
||||
pthread_mutex_init(&ov->journal_lock, NULL);
|
||||
|
||||
if (pack_path) {
|
||||
Pack *p; int lerr = 0;
|
||||
if (pack_load(pack_path, &p, &lerr) == 0) {
|
||||
ov->pack = p;
|
||||
} else if (lerr != VFS_ERR_NOENT) {
|
||||
pthread_mutex_destroy(&ov->journal_lock);
|
||||
free(ov->pack_path);
|
||||
free(ov);
|
||||
if (err) *err = lerr;
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
journal_replay(ov);
|
||||
|
||||
Backend *b = (Backend *)calloc(1, sizeof(Backend));
|
||||
b->ops = &OVERLAY_OPS;
|
||||
b->state = ov;
|
||||
return b;
|
||||
}
|
||||
+304
@@ -0,0 +1,304 @@
|
||||
/*
|
||||
* pack.c — on-disk pack format (Section 9), load-time integrity
|
||||
* validation (Section 7), and compaction's writer (Section 4.1, 4.3,
|
||||
* 5.3, 9.1, 9.2).
|
||||
*
|
||||
* On-disk layout (a concrete realization of the illustrative sketch in
|
||||
* Section 9, extended with the checksum field Section 7 requires):
|
||||
*
|
||||
* PackHeader (fixed size, 8-byte aligned)
|
||||
* blobs (index_offset - sizeof(PackHeader) bytes)
|
||||
* PackIndexEntry[index_count] at index_offset, sorted by name (9.1)
|
||||
* strings (each name NUL-terminated) at strings_offset
|
||||
*/
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
#define PACK_VERSION 1
|
||||
|
||||
typedef struct PackHeader {
|
||||
char magic[4];
|
||||
uint32_t version;
|
||||
uint64_t index_offset;
|
||||
uint64_t index_count;
|
||||
uint64_t strings_offset;
|
||||
uint64_t strings_len;
|
||||
uint64_t checksum; /* FNV-1a64 over [index_offset, strings_offset+strings_len) */
|
||||
} PackHeader;
|
||||
|
||||
static uint64_t align8(uint64_t n) { return (n + 7u) & ~(uint64_t)7u; }
|
||||
|
||||
/* ---- loading + validation (Section 7) ---- */
|
||||
|
||||
void pack_close(Pack *p) {
|
||||
if (!p) return;
|
||||
if (p->map && p->map != MAP_FAILED) munmap(p->map, p->map_len);
|
||||
if (p->fd >= 0) close(p->fd);
|
||||
free(p->path);
|
||||
free(p);
|
||||
}
|
||||
|
||||
int pack_load(const char *path, Pack **out, int *err) {
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0) { if (err) *err = VFS_ERR_NOENT; return -1; }
|
||||
|
||||
struct stat st;
|
||||
if (fstat(fd, &st) < 0) { close(fd); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
size_t len = (size_t)st.st_size;
|
||||
if (len < sizeof(PackHeader)) { close(fd); if (err) *err = VFS_ERR_CORRUPT; return -1; }
|
||||
|
||||
void *map = mmap(NULL, len, PROT_READ, MAP_PRIVATE, fd, 0);
|
||||
if (map == MAP_FAILED) { close(fd); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
|
||||
const PackHeader *hdr = (const PackHeader *)map;
|
||||
if (memcmp(hdr->magic, "PKFS", 4) != 0 || hdr->version != PACK_VERSION) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Bounds-check every offset before trusting it (Section 7): the
|
||||
* index and strings regions must lie within the file and must not
|
||||
* overlap the header. */
|
||||
if (hdr->index_offset < sizeof(PackHeader) ||
|
||||
hdr->index_offset > len ||
|
||||
hdr->index_count > (len - hdr->index_offset) / sizeof(PackIndexEntry)) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
uint64_t index_bytes = hdr->index_count * (uint64_t)sizeof(PackIndexEntry);
|
||||
uint64_t index_end = hdr->index_offset + index_bytes;
|
||||
if (hdr->strings_offset < index_end || hdr->strings_offset > len ||
|
||||
hdr->strings_len > len - hdr->strings_offset) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
uint64_t strings_end = hdr->strings_offset + hdr->strings_len;
|
||||
if (strings_end > len) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
|
||||
uint64_t checksum = pfs_fnv1a64((const char *)map + hdr->index_offset,
|
||||
(size_t)(strings_end - hdr->index_offset));
|
||||
if (checksum != hdr->checksum) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
|
||||
const PackIndexEntry *entries = (const PackIndexEntry *)((const char *)map + hdr->index_offset);
|
||||
const char *strings_base = (const char *)map + hdr->strings_offset;
|
||||
|
||||
for (uint64_t i = 0; i < hdr->index_count; i++) {
|
||||
const PackIndexEntry *e = &entries[i];
|
||||
if (e->data_off < sizeof(PackHeader) || e->data_off > hdr->index_offset ||
|
||||
e->size > hdr->index_offset - e->data_off) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
if (e->name_off > hdr->strings_len || e->name_len > hdr->strings_len - e->name_off) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
/* Section 9's names are NUL-terminated in the strings region; a
|
||||
* missing terminator is treated as corruption, not tolerated. */
|
||||
if (strings_base[e->name_off + e->name_len] != '\0') {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
if (i > 0) {
|
||||
const PackIndexEntry *prev = &entries[i - 1];
|
||||
if (strcmp(strings_base + prev->name_off, strings_base + e->name_off) >= 0) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT; /* Section 9.1 relies on sortedness */
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Pack *p = (Pack *)calloc(1, sizeof(Pack));
|
||||
if (!p) { munmap(map, len); close(fd); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
p->fd = fd;
|
||||
p->map = map;
|
||||
p->map_len = len;
|
||||
p->entries = entries;
|
||||
p->entry_count = (size_t)hdr->index_count;
|
||||
p->strings_base = strings_base;
|
||||
p->strings_len = (size_t)hdr->strings_len;
|
||||
p->path = strdup(path);
|
||||
*out = p;
|
||||
return 0;
|
||||
}
|
||||
|
||||
const char *pack_entry_name(const Pack *p, const PackIndexEntry *e) {
|
||||
return p->strings_base + e->name_off;
|
||||
}
|
||||
|
||||
const void *pack_entry_data(const Pack *p, const PackIndexEntry *e) {
|
||||
return (const char *)p->map + e->data_off;
|
||||
}
|
||||
|
||||
int pack_find(const Pack *p, const char *name, const PackIndexEntry **out) {
|
||||
size_t lo = 0, hi = p->entry_count;
|
||||
while (lo < hi) {
|
||||
size_t mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(pack_entry_name(p, &p->entries[mid]), name);
|
||||
if (c == 0) { *out = &p->entries[mid]; return 1; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static size_t lower_bound_str(const Pack *p, const char *key) {
|
||||
size_t lo = 0, hi = p->entry_count;
|
||||
while (lo < hi) {
|
||||
size_t mid = lo + (hi - lo) / 2;
|
||||
if (strcmp(pack_entry_name(p, &p->entries[mid]), key) < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
return lo;
|
||||
}
|
||||
|
||||
void pack_range(const Pack *p, const char *prefix, size_t *lo_out, size_t *hi_out) {
|
||||
size_t lo = lower_bound_str(p, prefix);
|
||||
size_t plen = strlen(prefix);
|
||||
char *upper = (char *)malloc(plen + 2);
|
||||
memcpy(upper, prefix, plen);
|
||||
upper[plen] = (char)0x7F; /* > any valid path byte, per Section 9.1's argument */
|
||||
upper[plen + 1] = '\0';
|
||||
size_t hi = lower_bound_str(p, upper);
|
||||
free(upper);
|
||||
*lo_out = lo;
|
||||
*hi_out = hi;
|
||||
}
|
||||
|
||||
/* ---- writing (compaction target, Section 4.1 / 4.3 / 9.2) ---- */
|
||||
|
||||
static int pack_build_entry_cmp(const void *a, const void *b) {
|
||||
return strcmp(((const PackBuildEntry *)a)->name, ((const PackBuildEntry *)b)->name);
|
||||
}
|
||||
|
||||
typedef struct DedupSlot {
|
||||
uint64_t hash;
|
||||
uint64_t size;
|
||||
uint64_t data_off;
|
||||
} DedupSlot;
|
||||
|
||||
int pack_write(const char *path, const PackBuildEntry *in, size_t count, int *err) {
|
||||
PackBuildEntry *entries = NULL;
|
||||
if (count > 0) {
|
||||
entries = (PackBuildEntry *)malloc(count * sizeof(PackBuildEntry));
|
||||
if (!entries) { if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
memcpy(entries, in, count * sizeof(PackBuildEntry));
|
||||
}
|
||||
qsort(entries, count, sizeof(PackBuildEntry), pack_build_entry_cmp);
|
||||
|
||||
/* Pass 1: compute blob region with exact-duplicate elimination
|
||||
* (Section 9.2) and the strings region size. */
|
||||
DedupSlot *slots = count ? (DedupSlot *)malloc(count * sizeof(DedupSlot)) : NULL;
|
||||
size_t slot_count = 0;
|
||||
uint64_t *data_off_for = count ? (uint64_t *)malloc(count * sizeof(uint64_t)) : NULL;
|
||||
uint64_t blob_cursor = align8(sizeof(PackHeader));
|
||||
uint64_t strings_total = 0;
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
uint64_t h = pfs_fnv1a64(entries[i].data, (size_t)entries[i].size);
|
||||
uint64_t reuse = UINT64_MAX;
|
||||
for (size_t j = 0; j < slot_count; j++) {
|
||||
if (slots[j].hash == h && slots[j].size == entries[i].size) {
|
||||
reuse = slots[j].data_off;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (reuse != UINT64_MAX) {
|
||||
data_off_for[i] = reuse;
|
||||
} else {
|
||||
data_off_for[i] = blob_cursor;
|
||||
slots[slot_count].hash = h;
|
||||
slots[slot_count].size = entries[i].size;
|
||||
slots[slot_count].data_off = blob_cursor;
|
||||
slot_count++;
|
||||
blob_cursor += entries[i].size;
|
||||
}
|
||||
strings_total += strlen(entries[i].name) + 1;
|
||||
}
|
||||
free(slots);
|
||||
|
||||
uint64_t index_offset = align8(blob_cursor);
|
||||
uint64_t index_bytes = (uint64_t)count * sizeof(PackIndexEntry);
|
||||
uint64_t strings_offset = index_offset + index_bytes;
|
||||
uint64_t total_size = strings_offset + strings_total;
|
||||
|
||||
unsigned char *buf = (unsigned char *)calloc(1, (size_t)total_size);
|
||||
if (!buf) { free(entries); free(data_off_for); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
/* Deduplicated entries (Section 9.2) share a data_off; writing
|
||||
* the same bytes to it more than once is redundant but harmless. */
|
||||
memcpy(buf + data_off_for[i], entries[i].data, entries[i].size);
|
||||
}
|
||||
|
||||
PackIndexEntry *out_entries = (PackIndexEntry *)(buf + index_offset);
|
||||
uint64_t str_cursor = 0;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
size_t nlen = strlen(entries[i].name);
|
||||
out_entries[i].name_off = str_cursor;
|
||||
out_entries[i].name_len = (uint32_t)nlen;
|
||||
out_entries[i].data_off = data_off_for[i];
|
||||
out_entries[i].size = entries[i].size;
|
||||
out_entries[i].mode = entries[i].mode;
|
||||
out_entries[i].mtime = entries[i].mtime;
|
||||
memcpy(buf + strings_offset + str_cursor, entries[i].name, nlen + 1);
|
||||
str_cursor += nlen + 1;
|
||||
}
|
||||
|
||||
PackHeader hdr;
|
||||
memset(&hdr, 0, sizeof(hdr));
|
||||
memcpy(hdr.magic, "PKFS", 4);
|
||||
hdr.version = PACK_VERSION;
|
||||
hdr.index_offset = index_offset;
|
||||
hdr.index_count = count;
|
||||
hdr.strings_offset = strings_offset;
|
||||
hdr.strings_len = strings_total;
|
||||
hdr.checksum = pfs_fnv1a64(buf + index_offset, (size_t)(strings_offset + strings_total - index_offset));
|
||||
memcpy(buf, &hdr, sizeof(hdr));
|
||||
|
||||
free(entries);
|
||||
free(data_off_for);
|
||||
|
||||
/* Atomic compaction (Section 4.3): write to a temp file, fsync, then
|
||||
* rename over the target. A failure here aborts compaction and
|
||||
* leaves the existing pack untouched. */
|
||||
char tmp_path[PFS_PATH_MAX];
|
||||
snprintf(tmp_path, sizeof(tmp_path), "%s.tmp", path);
|
||||
int fd = open(tmp_path, O_WRONLY | O_CREAT | O_TRUNC, 0644);
|
||||
if (fd < 0) { free(buf); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
|
||||
size_t written = 0;
|
||||
while (written < (size_t)total_size) {
|
||||
ssize_t w = write(fd, buf + written, (size_t)total_size - written);
|
||||
if (w < 0) { close(fd); free(buf); unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
written += (size_t)w;
|
||||
}
|
||||
free(buf);
|
||||
|
||||
if (fsync(fd) < 0) { close(fd); unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
close(fd);
|
||||
|
||||
if (rename(tmp_path, path) < 0) { unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
return 0;
|
||||
}
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
/*
|
||||
* path.c — virtual-namespace canonicalization, concept.md Section 6.1.
|
||||
*
|
||||
* Resolves "." and ".." purely lexically, before any backend is reached,
|
||||
* so a crafted path cannot escape a mount's prefix even when no host
|
||||
* directory is involved.
|
||||
*/
|
||||
|
||||
#include <string.h>
|
||||
#include "internal.h"
|
||||
|
||||
int pfs_path_normalize(const char *in, char out[PFS_PATH_MAX]) {
|
||||
if (in == NULL || in[0] != '/') return -1;
|
||||
|
||||
/* Stack of component (start, len) pairs, built by lexical resolution. */
|
||||
const char *starts[256];
|
||||
size_t lens[256];
|
||||
int depth = 0;
|
||||
|
||||
const char *p = in;
|
||||
while (*p) {
|
||||
while (*p == '/') p++;
|
||||
if (!*p) break;
|
||||
const char *comp = p;
|
||||
while (*p && *p != '/') p++;
|
||||
size_t len = (size_t)(p - comp);
|
||||
|
||||
if (len == 1 && comp[0] == '.') {
|
||||
continue;
|
||||
}
|
||||
if (len == 2 && comp[0] == '.' && comp[1] == '.') {
|
||||
if (depth == 0) return -1; /* escape above root */
|
||||
depth--;
|
||||
continue;
|
||||
}
|
||||
if (depth >= 256) return -1; /* pathologically deep; reject rather than overflow */
|
||||
starts[depth] = comp;
|
||||
lens[depth] = len;
|
||||
depth++;
|
||||
}
|
||||
|
||||
size_t off = 0;
|
||||
out[off++] = '/';
|
||||
for (int i = 0; i < depth; i++) {
|
||||
if (off + lens[i] + 2 > PFS_PATH_MAX) return -1;
|
||||
if (i > 0) out[off++] = '/';
|
||||
memcpy(out + off, starts[i], lens[i]);
|
||||
off += lens[i];
|
||||
}
|
||||
out[off] = '\0';
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_path_under(const char *prefix, const char *path) {
|
||||
size_t plen = strlen(prefix);
|
||||
if (plen == 1 && prefix[0] == '/') return 1; /* root mount covers everything */
|
||||
if (strncmp(prefix, path, plen) != 0) return 0;
|
||||
return path[plen] == '\0' || path[plen] == '/';
|
||||
}
|
||||
+706
@@ -0,0 +1,706 @@
|
||||
/*
|
||||
* upper.c — the writable upper layer shared by the `mem` and `dir`
|
||||
* backends, and by the overlay's upper side (Section 3, Section 5).
|
||||
*
|
||||
* Implements the snapshot-based single-writer, wait-free-reader model of
|
||||
* Section 5.3: an immutable, refcounted UpperSnapshot reached through one
|
||||
* atomic pointer; structural writes (create/unlink/rename/mkdir/
|
||||
* whiteout) build and publish a new snapshot under `writer_lock`;
|
||||
* content writes (`upper_cell_write`) mutate a shared MutCell in place
|
||||
* and never touch the snapshot pointer, per the structural/content split
|
||||
* in Section 5.3. Buffer growth on the `mem` side follows the
|
||||
* replace-don't-mutate discipline of Section 5.7.
|
||||
*/
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
/* ---- MutCell lifetime: refcounted separately from any one snapshot,
|
||||
* since a cell's identity is shared across every snapshot whose entry
|
||||
* array still points at the same logical file (Section 5.3). ---- */
|
||||
|
||||
typedef struct MutCellRc {
|
||||
MutCell pub;
|
||||
_Atomic(pfs_usize) refcount;
|
||||
} MutCellRc;
|
||||
|
||||
/* MutCell* handed around externally IS the MutCellRc's first member's
|
||||
* address, so a plain cast recovers the refcount wrapper. */
|
||||
static MutCellRc *rc_of(MutCell *c) { return (MutCellRc *)c; }
|
||||
|
||||
static void mutcell_ref(MutCell *c) {
|
||||
if (!c) return;
|
||||
atomic_fetch_add_explicit(&rc_of(c)->refcount, 1, memory_order_relaxed);
|
||||
}
|
||||
|
||||
static void mutcell_unref(MutCell *c) {
|
||||
if (!c) return;
|
||||
if (atomic_fetch_sub_explicit(&rc_of(c)->refcount, 1, memory_order_acq_rel) == 1) {
|
||||
pthread_mutex_destroy(&c->write_lock);
|
||||
pthread_rwlock_destroy(&c->buf_lock);
|
||||
free(c->data);
|
||||
free(rc_of(c));
|
||||
}
|
||||
}
|
||||
|
||||
static MutCell *mutcell_new(void) {
|
||||
MutCellRc *rc = (MutCellRc *)calloc(1, sizeof(MutCellRc));
|
||||
pthread_mutex_init(&rc->pub.write_lock, NULL);
|
||||
pthread_rwlock_init(&rc->pub.buf_lock, NULL);
|
||||
atomic_init(&rc->pub.size, 0);
|
||||
atomic_init(&rc->pub.mtime, (int64_t)time(NULL));
|
||||
atomic_init(&rc->refcount, 1); /* owned by whichever snapshot installs it first */
|
||||
return &rc->pub;
|
||||
}
|
||||
|
||||
/* ---- UpperSnapshot lifetime ---- */
|
||||
|
||||
UpperSnapshot *upper_acquire(UpperStore *u) {
|
||||
pthread_rwlock_rdlock(&u->reclaim_gate);
|
||||
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
|
||||
pthread_rwlock_unlock(&u->reclaim_gate);
|
||||
return s;
|
||||
}
|
||||
|
||||
/* Retires a snapshot just superseded by a writer's atomic_store to
|
||||
* u->current. Must run under reclaim_gate's write side so that no
|
||||
* upper_acquire() can be mid-flight (loaded `old` but not yet
|
||||
* incremented its refcount) when this potentially frees it. */
|
||||
static void upper_retire(UpperStore *u, UpperSnapshot *old) {
|
||||
pthread_rwlock_wrlock(&u->reclaim_gate);
|
||||
upper_release(old);
|
||||
pthread_rwlock_unlock(&u->reclaim_gate);
|
||||
}
|
||||
|
||||
void upper_release(UpperSnapshot *s) {
|
||||
if (!s) return;
|
||||
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
free(s->entries[i].name);
|
||||
mutcell_unref(s->entries[i].cell);
|
||||
}
|
||||
free(s->entries);
|
||||
free(s);
|
||||
}
|
||||
}
|
||||
|
||||
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out) {
|
||||
pfs_usize lo = 0, hi = s->count;
|
||||
while (lo < hi) {
|
||||
pfs_usize mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(s->entries[mid].name, path);
|
||||
if (c == 0) { *out = &s->entries[mid]; return 1; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_has_children(UpperSnapshot *s, const char *path) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
|
||||
else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t plen = strlen(prefix);
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
|
||||
if (strncmp(s->entries[i].name, prefix, plen) == 0) return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Builds a new snapshot with one entry inserted or replaced at `name`. */
|
||||
static UpperSnapshot *snapshot_upsert(UpperSnapshot *base, const char *name,
|
||||
UpperEntryKind kind, MutCell *cell) {
|
||||
pfs_usize old_count = base ? base->count : 0;
|
||||
pfs_usize lo = 0, hi = old_count;
|
||||
int found = 0;
|
||||
while (lo < hi) {
|
||||
pfs_usize mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(base->entries[mid].name, name);
|
||||
if (c == 0) { lo = mid; found = 1; break; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
pfs_usize new_count = found ? old_count : old_count + 1;
|
||||
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
ns->entries = (UpperEntry *)calloc(new_count ? new_count : 1, sizeof(UpperEntry));
|
||||
ns->count = new_count;
|
||||
atomic_init(&ns->refcount, 1);
|
||||
|
||||
pfs_usize w = 0;
|
||||
for (pfs_usize i = 0; i < old_count; i++) {
|
||||
if (i == lo) {
|
||||
ns->entries[w].name = strdup(name);
|
||||
ns->entries[w].kind = kind;
|
||||
ns->entries[w].cell = cell;
|
||||
mutcell_ref(cell);
|
||||
w++;
|
||||
if (found) continue; /* replaces base->entries[lo] */
|
||||
}
|
||||
ns->entries[w].name = strdup(base->entries[i].name);
|
||||
ns->entries[w].kind = base->entries[i].kind;
|
||||
ns->entries[w].cell = base->entries[i].cell;
|
||||
mutcell_ref(ns->entries[w].cell);
|
||||
w++;
|
||||
}
|
||||
if (lo == old_count) {
|
||||
ns->entries[w].name = strdup(name);
|
||||
ns->entries[w].kind = kind;
|
||||
ns->entries[w].cell = cell;
|
||||
mutcell_ref(cell);
|
||||
w++;
|
||||
}
|
||||
return ns;
|
||||
}
|
||||
|
||||
static UpperSnapshot *snapshot_remove(UpperSnapshot *base, const char *name) {
|
||||
pfs_usize lo = 0, hi = base->count;
|
||||
int found = 0;
|
||||
while (lo < hi) {
|
||||
pfs_usize mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(base->entries[mid].name, name);
|
||||
if (c == 0) { lo = mid; found = 1; break; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
if (!found) return NULL;
|
||||
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
ns->count = base->count - 1;
|
||||
ns->entries = ns->count ? (UpperEntry *)calloc(ns->count, sizeof(UpperEntry)) : NULL;
|
||||
atomic_init(&ns->refcount, 1);
|
||||
pfs_usize w = 0;
|
||||
for (pfs_usize i = 0; i < base->count; i++) {
|
||||
if (i == lo) continue;
|
||||
ns->entries[w].name = strdup(base->entries[i].name);
|
||||
ns->entries[w].kind = base->entries[i].kind;
|
||||
ns->entries[w].cell = base->entries[i].cell;
|
||||
mutcell_ref(ns->entries[w].cell);
|
||||
w++;
|
||||
}
|
||||
return ns;
|
||||
}
|
||||
|
||||
UpperStore *upper_new(UpperKind kind, int root_fd, const char *root_path) {
|
||||
UpperStore *u = (UpperStore *)calloc(1, sizeof(UpperStore));
|
||||
u->kind = kind;
|
||||
u->root_fd = root_fd;
|
||||
u->root_path = root_path ? strdup(root_path) : NULL;
|
||||
pthread_mutex_init(&u->writer_lock, NULL);
|
||||
pthread_rwlock_init(&u->reclaim_gate, NULL);
|
||||
UpperSnapshot *s = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
atomic_init(&s->refcount, 1);
|
||||
atomic_init(&u->current, s);
|
||||
return u;
|
||||
}
|
||||
|
||||
void upper_free(UpperStore *u) {
|
||||
if (!u) return;
|
||||
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
upper_release(s);
|
||||
pthread_mutex_destroy(&u->writer_lock);
|
||||
pthread_rwlock_destroy(&u->reclaim_gate);
|
||||
if (u->kind == UPPER_DIR && u->root_fd >= 0) close(u->root_fd);
|
||||
free(u->root_path);
|
||||
free(u);
|
||||
}
|
||||
|
||||
/* strips the mount-relative leading '/' for host syscalls */
|
||||
static const char *rel(const char *path) { return path[0] == '/' ? path + 1 : path; }
|
||||
|
||||
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err) {
|
||||
(void)truncate_existing;
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
|
||||
MutCell *c = existing->cell;
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
*cell_out = c;
|
||||
return 0;
|
||||
}
|
||||
if (u->kind == UPPER_DIR) {
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
|
||||
if (fd < 0) { pthread_mutex_unlock(&u->writer_lock); if (err) *err = e; return -1; }
|
||||
close(fd);
|
||||
}
|
||||
MutCell *cell = mutcell_new();
|
||||
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
mutcell_unref(cell); /* drop the creation-local ref; ns holds its own */
|
||||
*cell_out = cell;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_mkdir(UpperStore *u, const char *path, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
if (upper_lookup(old, path, &existing) || upper_has_children(old, path)) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_EXIST;
|
||||
return -1;
|
||||
}
|
||||
if (u->kind == UPPER_DIR) {
|
||||
int e = 0;
|
||||
if (pfs_dir_mkdirat(u->root_fd, rel(path), &e) < 0) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_DIR_MARKER, NULL);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
int have = upper_lookup(old, path, &existing);
|
||||
if ((!have || existing->kind == ENTRY_WHITEOUT) && !had_lower) {
|
||||
/* No explicit entry anywhere: still an error to distinguish
|
||||
* "doesn't exist" from "exists only implicitly, via children,
|
||||
* and therefore can't be removed" (Section 9.3). */
|
||||
int kids = upper_has_children(old, path);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
|
||||
return -1;
|
||||
}
|
||||
if (have && existing->kind == ENTRY_DIR_MARKER && upper_has_children(old, path)) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_NOTEMPTY;
|
||||
return -1;
|
||||
}
|
||||
if (u->kind == UPPER_DIR && have && existing->kind != ENTRY_WHITEOUT) {
|
||||
int e = 0;
|
||||
int is_dir = (existing->kind == ENTRY_DIR_MARKER);
|
||||
if (pfs_dir_unlinkat(u->root_fd, rel(path), is_dir, &e) < 0) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
UpperSnapshot *ns = had_lower ? snapshot_upsert(old, path, ENTRY_WHITEOUT, NULL)
|
||||
: snapshot_remove(old, path);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *src;
|
||||
if (!upper_lookup(old, from, &src) || src->kind == ENTRY_WHITEOUT) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_NOENT;
|
||||
return -1;
|
||||
}
|
||||
if (u->kind == UPPER_DIR) {
|
||||
int e = 0;
|
||||
if (pfs_dir_renameat(u->root_fd, rel(from), rel(to), &e) < 0) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
UpperSnapshot *step1 = snapshot_upsert(old, to, src->kind, src->cell);
|
||||
UpperSnapshot *step2 = had_lower_from ? snapshot_upsert(step1, from, ENTRY_WHITEOUT, NULL)
|
||||
: snapshot_remove(step1, from);
|
||||
atomic_store_explicit(&u->current, step2, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
upper_release(step1); /* never published; drop our local build reference */
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size,
|
||||
int64_t mtime, MutCell **cell_out, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
*cell_out = existing->cell;
|
||||
return 0; /* raced with another copy-up; use what's already there */
|
||||
}
|
||||
MutCell *cell = mutcell_new();
|
||||
atomic_store_explicit(&cell->mtime, mtime, memory_order_relaxed);
|
||||
|
||||
if (u->kind == UPPER_MEM) {
|
||||
if (size) {
|
||||
cell->data = (unsigned char *)malloc((size_t)size);
|
||||
memcpy(cell->data, data, (size_t)size);
|
||||
}
|
||||
cell->capacity = (pfs_usize)size;
|
||||
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
|
||||
} else {
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
|
||||
if (fd < 0) {
|
||||
mutcell_unref(cell);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
size_t written = 0;
|
||||
const unsigned char *p = (const unsigned char *)data;
|
||||
int failed = 0;
|
||||
while (written < (size_t)size) {
|
||||
ssize_t w = write(fd, p + written, (size_t)size - written);
|
||||
if (w < 0) { failed = 1; break; }
|
||||
written += (size_t)w;
|
||||
}
|
||||
close(fd);
|
||||
if (failed) {
|
||||
mutcell_unref(cell);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
|
||||
}
|
||||
|
||||
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
mutcell_unref(cell);
|
||||
*cell_out = cell;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ---- content operations (Section 5.3 fast path, Section 5.7 growth) ---- */
|
||||
|
||||
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n) {
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_rwlock_rdlock(&c->buf_lock);
|
||||
pfs_usize size = atomic_load_explicit(&c->size, memory_order_acquire);
|
||||
pfs_usize avail = off < size ? size - off : 0;
|
||||
pfs_usize to_copy = n < avail ? n : avail;
|
||||
if (to_copy) memcpy(buf, c->data + off, to_copy);
|
||||
pthread_rwlock_unlock(&c->buf_lock);
|
||||
return (pfs_isize)to_copy;
|
||||
}
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_RDONLY, 0, &e);
|
||||
if (fd < 0) return -1;
|
||||
ssize_t r = pread(fd, buf, n, (off_t)off);
|
||||
close(fd);
|
||||
return (pfs_isize)r;
|
||||
}
|
||||
|
||||
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off,
|
||||
const void *buf, pfs_usize n, int *err) {
|
||||
int64_t now = (int64_t)time(NULL);
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
pfs_usize cur_size = atomic_load_explicit(&c->size, memory_order_relaxed);
|
||||
pfs_usize need = off + n;
|
||||
if (need > c->capacity) {
|
||||
pfs_usize newcap = c->capacity ? c->capacity : 64;
|
||||
while (newcap < need) newcap *= 2;
|
||||
unsigned char *nb = (unsigned char *)calloc(1, newcap);
|
||||
if (!nb) { pthread_mutex_unlock(&c->write_lock); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
if (c->data && cur_size) memcpy(nb, c->data, cur_size);
|
||||
if (n) memcpy(nb + off, buf, n);
|
||||
unsigned char *old = c->data;
|
||||
/* Section 5.7: publish the new buffer via the rwlock, then
|
||||
* free the old one only after releasing the write lock on
|
||||
* it — any reader that could still see `old` must have
|
||||
* taken buf_lock (rdlock) before this wrlock was granted,
|
||||
* and rdlock/wrlock are mutually exclusive, so it has
|
||||
* already finished copying by the time we get here. */
|
||||
pthread_rwlock_wrlock(&c->buf_lock);
|
||||
c->data = nb;
|
||||
c->capacity = newcap;
|
||||
pthread_rwlock_unlock(&c->buf_lock);
|
||||
free(old);
|
||||
} else if (n) {
|
||||
memcpy(c->data + off, buf, n);
|
||||
}
|
||||
pfs_usize new_size = need > cur_size ? need : cur_size;
|
||||
atomic_store_explicit(&c->size, new_size, memory_order_release);
|
||||
atomic_store_explicit(&c->mtime, now, memory_order_release);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
return (pfs_isize)n;
|
||||
}
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY, 0, &e);
|
||||
if (fd < 0) { pthread_mutex_unlock(&c->write_lock); if (err) *err = e; return -1; }
|
||||
ssize_t w = pwrite(fd, buf, n, (off_t)off);
|
||||
close(fd);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
if (w < 0) { if (err) *err = VFS_ERR_IO; return -1; }
|
||||
return (pfs_isize)w;
|
||||
}
|
||||
|
||||
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path) {
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
atomic_store_explicit(&c->size, 0, memory_order_release);
|
||||
atomic_store_explicit(&c->mtime, (int64_t)time(NULL), memory_order_release);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
} else {
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_TRUNC, 0644, &e);
|
||||
if (fd >= 0) close(fd);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- compaction support ---- */
|
||||
|
||||
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb) {
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
UpperEntry *e = &s->entries[i];
|
||||
if (e->kind == ENTRY_WHITEOUT) continue;
|
||||
if (e->kind == ENTRY_DIR_MARKER) {
|
||||
cb->visit(cb->ctx, e->name, ENTRY_DIR_MARKER, NULL, 0, 0);
|
||||
continue;
|
||||
}
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_rwlock_rdlock(&e->cell->buf_lock);
|
||||
pfs_usize size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
|
||||
cb->visit(cb->ctx, e->name, ENTRY_FILE, e->cell->data, size,
|
||||
atomic_load_explicit(&e->cell->mtime, memory_order_acquire));
|
||||
pthread_rwlock_unlock(&e->cell->buf_lock);
|
||||
} else {
|
||||
int err = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(e->name), O_RDONLY, 0, &err);
|
||||
if (fd < 0) continue;
|
||||
struct stat st;
|
||||
fstat(fd, &st);
|
||||
void *buf = st.st_size ? malloc((size_t)st.st_size) : NULL;
|
||||
size_t total = 0;
|
||||
while (buf && total < (size_t)st.st_size) {
|
||||
ssize_t r = read(fd, (char *)buf + total, (size_t)st.st_size - total);
|
||||
if (r <= 0) break;
|
||||
total += (size_t)r;
|
||||
}
|
||||
close(fd);
|
||||
cb->visit(cb->ctx, e->name, ENTRY_FILE, buf, total, (int64_t)st.st_mtime);
|
||||
free(buf);
|
||||
}
|
||||
}
|
||||
upper_release(s);
|
||||
}
|
||||
|
||||
void upper_reset_empty(UpperStore *u) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
atomic_init(&ns->refcount, 1);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
}
|
||||
|
||||
/* ---- standalone mem/dir Backend wrapper (no lower layer at all) ---- */
|
||||
|
||||
typedef struct UpperFile {
|
||||
UpperStore *store;
|
||||
MutCell *cell;
|
||||
char *path;
|
||||
pfs_usize pos;
|
||||
} UpperFile;
|
||||
|
||||
static int upstd_open(Backend *b, const char *path, int flags, VfsFile **out) {
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
const UpperEntry *e;
|
||||
int found = upper_lookup(s, path, &e);
|
||||
if (found && e->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
|
||||
|
||||
MutCell *cell;
|
||||
if (found) {
|
||||
cell = e->cell;
|
||||
mutcell_ref(cell);
|
||||
upper_release(s);
|
||||
if (flags & VFS_O_TRUNC) upper_cell_truncate(u, cell, path);
|
||||
} else {
|
||||
upper_release(s);
|
||||
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||||
int err = 0;
|
||||
if (upper_create(u, path, 0, &cell, &err) < 0) return err;
|
||||
mutcell_ref(cell);
|
||||
}
|
||||
|
||||
UpperFile *uf = (UpperFile *)calloc(1, sizeof(UpperFile));
|
||||
uf->store = u; uf->cell = cell; uf->path = strdup(path); uf->pos = 0;
|
||||
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||||
f->backend = b; f->state = uf;
|
||||
*out = f;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static pfs_isize upstd_read(VfsFile *f, void *buf, pfs_usize n) {
|
||||
UpperFile *uf = (UpperFile *)f->state;
|
||||
pfs_isize r = upper_cell_read(uf->store, uf->cell, uf->path, uf->pos, buf, n);
|
||||
if (r > 0) uf->pos += (pfs_usize)r;
|
||||
return r;
|
||||
}
|
||||
|
||||
static pfs_isize upstd_write(VfsFile *f, const void *buf, pfs_usize n) {
|
||||
UpperFile *uf = (UpperFile *)f->state;
|
||||
int err = 0;
|
||||
pfs_isize w = upper_cell_write(uf->store, uf->cell, uf->path, uf->pos, buf, n, &err);
|
||||
if (w >= 0) { uf->pos += (pfs_usize)w; return w; }
|
||||
return err;
|
||||
}
|
||||
|
||||
static int upstd_close(VfsFile *f) {
|
||||
UpperFile *uf = (UpperFile *)f->state;
|
||||
mutcell_unref(uf->cell);
|
||||
free(uf->path);
|
||||
free(uf);
|
||||
free(f);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_stat(Backend *b, const char *path, VfsStat *out) {
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
const UpperEntry *e;
|
||||
if (!upper_lookup(s, path, &e) || e->kind == ENTRY_WHITEOUT) {
|
||||
if (upper_has_children(s, path)) {
|
||||
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
upper_release(s);
|
||||
return VFS_ERR_NOENT;
|
||||
}
|
||||
if (e->kind == ENTRY_DIR_MARKER) {
|
||||
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
if (u->kind == UPPER_MEM) {
|
||||
out->size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
|
||||
out->mtime = atomic_load_explicit(&e->cell->mtime, memory_order_acquire);
|
||||
out->kind = VFS_KIND_FILE;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
int err = 0;
|
||||
int rc = pfs_dir_statat(u->root_fd, rel(path), out, &err);
|
||||
upper_release(s);
|
||||
return rc < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_readdir(Backend *b, const char *path, VfsDir *out) {
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
|
||||
const UpperEntry *self;
|
||||
int self_found = upper_lookup(s, path, &self);
|
||||
int has_kids = upper_has_children(s, path);
|
||||
if (!has_kids && !(self_found && self->kind == ENTRY_DIR_MARKER) && strcmp(path, "/") != 0) {
|
||||
upper_release(s);
|
||||
return self_found ? VFS_ERR_NOTDIR : VFS_ERR_NOENT;
|
||||
}
|
||||
|
||||
char prefix[PFS_PATH_MAX];
|
||||
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
|
||||
else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t plen = strlen(prefix);
|
||||
|
||||
VfsDirEntry *entries = NULL;
|
||||
pfs_usize count = 0, cap = 0;
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
|
||||
const char *name = s->entries[i].name;
|
||||
if (strncmp(name, prefix, plen) != 0) continue;
|
||||
const char *restp = name + plen;
|
||||
const char *slash = strchr(restp, '/');
|
||||
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||||
if (clen == 0 || clen >= sizeof(entries[0].name)) continue;
|
||||
if (count > 0 && strncmp(entries[count - 1].name, restp, clen) == 0 &&
|
||||
entries[count - 1].name[clen] == '\0') continue; /* sorted -> dup is adjacent */
|
||||
if (count == cap) {
|
||||
cap = cap ? cap * 2 : 8;
|
||||
entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry));
|
||||
}
|
||||
memcpy(entries[count].name, restp, clen);
|
||||
entries[count].name[clen] = '\0';
|
||||
entries[count].kind = slash ? VFS_KIND_DIR
|
||||
: (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||||
count++;
|
||||
}
|
||||
upper_release(s);
|
||||
out->entries = entries;
|
||||
out->count = count;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_mkdir(Backend *b, const char *path) {
|
||||
int err = 0;
|
||||
return upper_mkdir((UpperStore *)b->state, path, &err) < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_unlink(Backend *b, const char *path) {
|
||||
int err = 0;
|
||||
return upper_remove((UpperStore *)b->state, path, 0, &err) < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_rename(Backend *b, const char *from, const char *to) {
|
||||
int err = 0;
|
||||
return upper_rename((UpperStore *)b->state, from, to, 0, &err) < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_sync(Backend *b) { (void)b; return VFS_ERR_PERM; }
|
||||
|
||||
static void upstd_free(Backend *b) {
|
||||
upper_free((UpperStore *)b->state);
|
||||
free(b);
|
||||
}
|
||||
|
||||
static const BackendOps UPPER_STANDALONE_OPS = {
|
||||
upstd_open, upstd_read, upstd_write, upstd_close, upstd_stat,
|
||||
upstd_readdir, upstd_mkdir, upstd_unlink, upstd_rename, upstd_sync, upstd_free
|
||||
};
|
||||
|
||||
Backend *backend_from_upper(UpperStore *u) {
|
||||
Backend *b = (Backend *)calloc(1, sizeof(Backend));
|
||||
b->ops = &UPPER_STANDALONE_OPS;
|
||||
b->state = u;
|
||||
return b;
|
||||
}
|
||||
|
||||
Backend *backend_mem_new(void) {
|
||||
return backend_from_upper(upper_new(UPPER_MEM, -1, NULL));
|
||||
}
|
||||
|
||||
Backend *backend_dir_new(const char *host_path, int *err) {
|
||||
int e = 0;
|
||||
int fd = pfs_dir_capability_open(host_path, &e);
|
||||
if (fd < 0) { if (err) *err = e; return NULL; }
|
||||
return backend_from_upper(upper_new(UPPER_DIR, fd, host_path));
|
||||
}
|
||||
|
||||
int upper_backend_root_fd(Backend *b) {
|
||||
if (b->ops != &UPPER_STANDALONE_OPS) return -1;
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
return u->kind == UPPER_DIR ? u->root_fd : -1;
|
||||
}
|
||||
|
||||
void backend_free(Backend *b) {
|
||||
if (!b) return;
|
||||
b->ops->free(b);
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
/*
|
||||
* vfs.c — VFS core: the mount table (itself part of the Section 5.3
|
||||
* snapshot, per that section's mount-table addition), virtual-namespace
|
||||
* canonicalization dispatch (Section 6.1), and the public API.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
typedef struct MountEntry {
|
||||
char *prefix;
|
||||
size_t prefix_len;
|
||||
Backend *backend;
|
||||
} MountEntry;
|
||||
|
||||
typedef struct MountSnapshot {
|
||||
_Atomic(pfs_usize) refcount;
|
||||
MountEntry *entries;
|
||||
pfs_usize count;
|
||||
} MountSnapshot;
|
||||
|
||||
struct Vfs {
|
||||
_Atomic(MountSnapshot *) current;
|
||||
pthread_mutex_t writer_lock;
|
||||
/* See UpperStore.reclaim_gate (internal.h) for why this exists: it
|
||||
* closes the gap between loading the current MountSnapshot pointer
|
||||
* and incrementing its refcount, which a naive load-then-increment
|
||||
* leaves open to a concurrent writer freeing the very object being
|
||||
* acquired. */
|
||||
pthread_rwlock_t reclaim_gate;
|
||||
};
|
||||
|
||||
static MountSnapshot *mount_acquire(Vfs *v) {
|
||||
pthread_rwlock_rdlock(&v->reclaim_gate);
|
||||
MountSnapshot *s = atomic_load_explicit(&v->current, memory_order_acquire);
|
||||
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
|
||||
pthread_rwlock_unlock(&v->reclaim_gate);
|
||||
return s;
|
||||
}
|
||||
|
||||
static void mount_release(MountSnapshot *s) {
|
||||
if (!s) return;
|
||||
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
|
||||
for (pfs_usize i = 0; i < s->count; i++) free(s->entries[i].prefix);
|
||||
free(s->entries);
|
||||
free(s);
|
||||
}
|
||||
}
|
||||
|
||||
static void mount_retire(Vfs *v, MountSnapshot *old) {
|
||||
pthread_rwlock_wrlock(&v->reclaim_gate);
|
||||
mount_release(old);
|
||||
pthread_rwlock_unlock(&v->reclaim_gate);
|
||||
}
|
||||
|
||||
Vfs *vfs_new(void) {
|
||||
Vfs *v = (Vfs *)calloc(1, sizeof(Vfs));
|
||||
pthread_mutex_init(&v->writer_lock, NULL);
|
||||
pthread_rwlock_init(&v->reclaim_gate, NULL);
|
||||
MountSnapshot *s = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
|
||||
atomic_init(&s->refcount, 1);
|
||||
atomic_init(&v->current, s);
|
||||
return v;
|
||||
}
|
||||
|
||||
void vfs_free(Vfs *v) {
|
||||
if (!v) return;
|
||||
mount_release(atomic_load_explicit(&v->current, memory_order_acquire));
|
||||
pthread_mutex_destroy(&v->writer_lock);
|
||||
pthread_rwlock_destroy(&v->reclaim_gate);
|
||||
free(v);
|
||||
}
|
||||
|
||||
int vfs_mount(Vfs *v, const char *prefix, Backend *b) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(prefix, norm) < 0) return VFS_ERR_INVAL;
|
||||
|
||||
pthread_mutex_lock(&v->writer_lock);
|
||||
MountSnapshot *old = atomic_load_explicit(&v->current, memory_order_acquire);
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
if (strcmp(old->entries[i].prefix, norm) == 0) {
|
||||
pthread_mutex_unlock(&v->writer_lock);
|
||||
return VFS_ERR_EXIST;
|
||||
}
|
||||
}
|
||||
MountSnapshot *ns = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
|
||||
ns->count = old->count + 1;
|
||||
ns->entries = (MountEntry *)calloc(ns->count, sizeof(MountEntry));
|
||||
atomic_init(&ns->refcount, 1);
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
ns->entries[i].prefix = strdup(old->entries[i].prefix);
|
||||
ns->entries[i].prefix_len = old->entries[i].prefix_len;
|
||||
ns->entries[i].backend = old->entries[i].backend;
|
||||
}
|
||||
ns->entries[old->count].prefix = strdup(norm);
|
||||
ns->entries[old->count].prefix_len = strlen(norm);
|
||||
ns->entries[old->count].backend = b;
|
||||
|
||||
atomic_store_explicit(&v->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&v->writer_lock);
|
||||
mount_retire(v, old);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
int vfs_unmount(Vfs *v, const char *prefix) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(prefix, norm) < 0) return VFS_ERR_INVAL;
|
||||
|
||||
pthread_mutex_lock(&v->writer_lock);
|
||||
MountSnapshot *old = atomic_load_explicit(&v->current, memory_order_acquire);
|
||||
pfs_usize idx = old->count;
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
if (strcmp(old->entries[i].prefix, norm) == 0) { idx = i; break; }
|
||||
}
|
||||
if (idx == old->count) { pthread_mutex_unlock(&v->writer_lock); return VFS_ERR_NOMOUNT; }
|
||||
|
||||
MountSnapshot *ns = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
|
||||
ns->count = old->count - 1;
|
||||
ns->entries = ns->count ? (MountEntry *)calloc(ns->count, sizeof(MountEntry)) : NULL;
|
||||
atomic_init(&ns->refcount, 1);
|
||||
pfs_usize w = 0;
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
if (i == idx) continue;
|
||||
ns->entries[w].prefix = strdup(old->entries[i].prefix);
|
||||
ns->entries[w].prefix_len = old->entries[i].prefix_len;
|
||||
ns->entries[w].backend = old->entries[i].backend;
|
||||
w++;
|
||||
}
|
||||
atomic_store_explicit(&v->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&v->writer_lock);
|
||||
mount_retire(v, old);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
/* Longest-prefix match against the mount table (Section 2.2), reached
|
||||
* through the same atomic snapshot that protects the upper index
|
||||
* (Section 5.3): a concurrent vfs_mount/vfs_unmount is never observed
|
||||
* mid-change. */
|
||||
static Backend *resolve(Vfs *v, const char *path, char relpath[PFS_PATH_MAX],
|
||||
MountSnapshot **snap_out, int *err) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(path, norm) < 0) { if (err) *err = VFS_ERR_INVAL; return NULL; }
|
||||
|
||||
MountSnapshot *snap = mount_acquire(v);
|
||||
Backend *best = NULL;
|
||||
size_t best_len = 0;
|
||||
for (pfs_usize i = 0; i < snap->count; i++) {
|
||||
if (pfs_path_under(snap->entries[i].prefix, norm) && snap->entries[i].prefix_len >= best_len) {
|
||||
best = snap->entries[i].backend;
|
||||
best_len = snap->entries[i].prefix_len;
|
||||
}
|
||||
}
|
||||
if (!best) { mount_release(snap); if (err) *err = VFS_ERR_NOMOUNT; return NULL; }
|
||||
|
||||
if (best_len <= 1) {
|
||||
strncpy(relpath, norm, PFS_PATH_MAX - 1);
|
||||
relpath[PFS_PATH_MAX - 1] = '\0';
|
||||
} else {
|
||||
const char *r = norm + best_len;
|
||||
strncpy(relpath, (*r == '\0') ? "/" : r, PFS_PATH_MAX - 1);
|
||||
relpath[PFS_PATH_MAX - 1] = '\0';
|
||||
}
|
||||
*snap_out = snap;
|
||||
return best;
|
||||
}
|
||||
|
||||
VfsFile *vfs_open(Vfs *v, const char *path, int flags, int *err) {
|
||||
char rel[PFS_PATH_MAX];
|
||||
MountSnapshot *snap;
|
||||
int e = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &e);
|
||||
if (!b) { if (err) *err = e; return NULL; }
|
||||
VfsFile *f = NULL;
|
||||
int rc = b->ops->open(b, rel, flags, &f);
|
||||
mount_release(snap);
|
||||
if (rc != VFS_OK) { if (err) *err = rc; return NULL; }
|
||||
if (err) *err = VFS_OK;
|
||||
return f;
|
||||
}
|
||||
|
||||
pfs_isize vfs_read(VfsFile *f, void *buf, pfs_usize n) { return f->backend->ops->read(f, buf, n); }
|
||||
pfs_isize vfs_write(VfsFile *f, const void *buf, pfs_usize n) { return f->backend->ops->write(f, buf, n); }
|
||||
int vfs_close(VfsFile *f) { return f->backend->ops->close(f); }
|
||||
|
||||
int vfs_stat(Vfs *v, const char *path, VfsStat *out) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
int rc = b->ops->stat(b, rel, out);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_readdir(Vfs *v, const char *path, VfsDir *out) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
out->entries = NULL; out->count = 0;
|
||||
int rc = b->ops->readdir(b, rel, out);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
void vfs_dir_free(VfsDir *d) {
|
||||
if (!d) return;
|
||||
free(d->entries);
|
||||
d->entries = NULL;
|
||||
d->count = 0;
|
||||
}
|
||||
|
||||
int vfs_mkdir(Vfs *v, const char *path) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
int rc = b->ops->mkdir(b, rel);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_unlink(Vfs *v, const char *path) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
int rc = b->ops->unlink(b, rel);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_rename(Vfs *v, const char *from, const char *to) {
|
||||
char relf[PFS_PATH_MAX], relt[PFS_PATH_MAX];
|
||||
MountSnapshot *sf, *st_; int err = VFS_OK;
|
||||
Backend *bf = resolve(v, from, relf, &sf, &err);
|
||||
if (!bf) return err;
|
||||
Backend *bt = resolve(v, to, relt, &st_, &err);
|
||||
if (!bt) { mount_release(sf); return err; }
|
||||
if (bf != bt) { mount_release(sf); mount_release(st_); return VFS_ERR_INVAL; /* cross-mount rename: v0 exclusion */ }
|
||||
int rc = bf->ops->rename(bf, relf, relt);
|
||||
mount_release(sf);
|
||||
mount_release(st_);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_sync(Vfs *v, const char *overlay_prefix) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(overlay_prefix, norm) < 0) return VFS_ERR_INVAL;
|
||||
MountSnapshot *snap = mount_acquire(v);
|
||||
Backend *b = NULL;
|
||||
for (pfs_usize i = 0; i < snap->count; i++) {
|
||||
if (strcmp(snap->entries[i].prefix, norm) == 0) { b = snap->entries[i].backend; break; }
|
||||
}
|
||||
if (!b) { mount_release(snap); return VFS_ERR_NOMOUNT; }
|
||||
int rc = b->ops->sync(b);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_harden_process_with_landlock(Vfs *v) {
|
||||
MountSnapshot *snap = mount_acquire(v);
|
||||
int *fds = snap->count ? (int *)malloc(snap->count * sizeof(int)) : NULL;
|
||||
size_t n = 0;
|
||||
for (pfs_usize i = 0; i < snap->count; i++) {
|
||||
Backend *b = snap->entries[i].backend;
|
||||
Backend *inner = overlay_upper_backend(b);
|
||||
if (inner) b = inner;
|
||||
int fd = upper_backend_root_fd(b);
|
||||
if (fd >= 0) fds[n++] = fd;
|
||||
}
|
||||
int rc = (n > 0) ? pfs_landlock_restrict_to(fds, n) : -1;
|
||||
free(fds);
|
||||
mount_release(snap);
|
||||
return rc == 0 ? VFS_OK : VFS_ERR_PERM;
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
/* test_concurrency.c — exercises Section 5.3's core promise: readers
|
||||
* never observe a torn snapshot while writers race structural changes,
|
||||
* and Section 5.7's buffer-growth guard doesn't crash or corrupt under
|
||||
* concurrent grow-and-read. Not a formal proof, but real thread
|
||||
* interleavings under a sanitizer-friendly build catch real races. */
|
||||
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "packfs.h"
|
||||
#include "test_harness.h"
|
||||
|
||||
#define WRITER_THREADS 4
|
||||
#define READER_THREADS 4
|
||||
#define ITERATIONS 500
|
||||
|
||||
static Vfs *g_v;
|
||||
|
||||
static void *writer_thread(void *arg) {
|
||||
long id = (long)arg;
|
||||
char path[64];
|
||||
for (int i = 0; i < ITERATIONS; i++) {
|
||||
snprintf(path, sizeof(path), "/t%ld/f%d.txt", id, i % 8);
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(g_v, path, VFS_O_WRONLY | VFS_O_CREAT | VFS_O_TRUNC, &err);
|
||||
if (!f) continue;
|
||||
char payload[128];
|
||||
int n = snprintf(payload, sizeof(payload), "thread=%ld iter=%d %.*s", id, i, (i % 50), "xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx");
|
||||
vfs_write(f, payload, (pfs_usize)n);
|
||||
vfs_close(f);
|
||||
if (i % 37 == 0) vfs_unlink(g_v, path);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void *reader_thread(void *arg) {
|
||||
(void)arg;
|
||||
char path[64];
|
||||
for (int i = 0; i < ITERATIONS; i++) {
|
||||
for (long id = 0; id < WRITER_THREADS; id++) {
|
||||
snprintf(path, sizeof(path), "/t%ld/f%d.txt", id, i % 8);
|
||||
VfsStat st;
|
||||
if (vfs_stat(g_v, path, &st) != VFS_OK) continue;
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(g_v, path, VFS_O_RDONLY, &err);
|
||||
if (!f) continue;
|
||||
char buf[256];
|
||||
pfs_isize r = vfs_read(f, buf, sizeof(buf) - 1);
|
||||
/* No CHECK on content here: a concurrent writer may legitimately
|
||||
* replace the file between stat and open/read (Section 5.3
|
||||
* guarantees a consistent snapshot per call, not across calls).
|
||||
* What we're really testing is the absence of a crash, a
|
||||
* negative-but-uncaught read length, or a hang. */
|
||||
CHECK(r >= 0);
|
||||
vfs_close(f);
|
||||
}
|
||||
VfsDir dir;
|
||||
if (vfs_readdir(g_v, "/", &dir) == VFS_OK) vfs_dir_free(&dir);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
int main(void) {
|
||||
g_v = vfs_new();
|
||||
Backend *mem = backend_mem_new();
|
||||
CHECK_EQ_INT(vfs_mount(g_v, "/", mem), VFS_OK);
|
||||
|
||||
pthread_t writers[WRITER_THREADS], readers[READER_THREADS];
|
||||
for (long i = 0; i < WRITER_THREADS; i++) pthread_create(&writers[i], NULL, writer_thread, (void *)i);
|
||||
for (long i = 0; i < READER_THREADS; i++) pthread_create(&readers[i], NULL, reader_thread, (void *)i);
|
||||
for (int i = 0; i < WRITER_THREADS; i++) pthread_join(writers[i], NULL);
|
||||
for (int i = 0; i < READER_THREADS; i++) pthread_join(readers[i], NULL);
|
||||
|
||||
/* sanity: the store is still fully consistent after the race */
|
||||
VfsDir dir;
|
||||
CHECK_EQ_INT(vfs_readdir(g_v, "/", &dir), VFS_OK);
|
||||
CHECK(dir.count > 0);
|
||||
vfs_dir_free(&dir);
|
||||
|
||||
vfs_unmount(g_v, "/");
|
||||
backend_free(mem);
|
||||
vfs_free(g_v);
|
||||
TEST_MAIN_END();
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
/* test_dir.c — dir backend: host passthrough + path containment
|
||||
* (Section 6.2): a normalized ".." can't reach it (blocked in the
|
||||
* virtual namespace, Section 6.1), and an absolute symlink planted
|
||||
* inside the mount cannot be used to read outside it. */
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "packfs.h"
|
||||
#include "test_harness.h"
|
||||
|
||||
int main(void) {
|
||||
char root[] = "/tmp/packfs_test_dir_XXXXXX";
|
||||
CHECK(mkdtemp(root) != NULL);
|
||||
char outside_dir[480];
|
||||
snprintf(outside_dir, sizeof(outside_dir), "/tmp/packfs_secret_%d", (int)getpid());
|
||||
mkdir(outside_dir, 0755);
|
||||
char secret_file[512];
|
||||
snprintf(secret_file, sizeof(secret_file), "%s/secret.txt", outside_dir);
|
||||
FILE *sf = fopen(secret_file, "w");
|
||||
CHECK(sf != NULL);
|
||||
fputs("top secret", sf);
|
||||
fclose(sf);
|
||||
|
||||
/* a symlink INSIDE root pointing OUTSIDE it */
|
||||
char link_path[512];
|
||||
snprintf(link_path, sizeof(link_path), "%s/escape", root);
|
||||
CHECK(symlink(outside_dir, link_path) == 0);
|
||||
|
||||
Vfs *v = vfs_new();
|
||||
int derr = 0;
|
||||
Backend *dir = backend_dir_new(root, &derr);
|
||||
CHECK(dir != NULL);
|
||||
CHECK_EQ_INT(vfs_mount(v, "/", dir), VFS_OK);
|
||||
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(v, "/a.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "abc", 3), 3);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
f = vfs_open(v, "/a.txt", VFS_O_RDONLY, &err);
|
||||
char buf[8] = {0};
|
||||
CHECK_EQ_INT(vfs_read(f, buf, sizeof(buf)), 3);
|
||||
CHECK_STR_EQ(buf, "abc");
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
/* virtual-namespace ".." escape: rejected before any backend is reached */
|
||||
f = vfs_open(v, "/../etc/shadow", VFS_O_RDONLY, &err);
|
||||
CHECK(f == NULL);
|
||||
CHECK_EQ_INT(err, VFS_ERR_INVAL);
|
||||
|
||||
/* symlink escape through the dir backend itself: openat2/O_NOFOLLOW
|
||||
* containment (Section 6.2) must refuse to follow it. */
|
||||
f = vfs_open(v, "/escape/secret.txt", VFS_O_RDONLY, &err);
|
||||
CHECK(f == NULL);
|
||||
|
||||
vfs_unmount(v, "/");
|
||||
backend_free(dir);
|
||||
vfs_free(v);
|
||||
TEST_MAIN_END();
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
/*
|
||||
* test_harness.h — minimal assertion-based test harness.
|
||||
*
|
||||
* No third-party test framework, consistent with the project's
|
||||
* static-linking / zero-required-dependency constraint (Section 11.1).
|
||||
*/
|
||||
#ifndef PACKFS_TEST_HARNESS_H
|
||||
#define PACKFS_TEST_HARNESS_H
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
static int pfs_test_failures = 0;
|
||||
|
||||
#define CHECK(cond) do { \
|
||||
if (!(cond)) { \
|
||||
fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \
|
||||
pfs_test_failures++; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define CHECK_EQ_INT(a, b) do { \
|
||||
long long _a = (long long)(a), _b = (long long)(b); \
|
||||
if (_a != _b) { \
|
||||
fprintf(stderr, "FAIL %s:%d: %s (%lld) != %s (%lld)\n", __FILE__, __LINE__, #a, _a, #b, _b); \
|
||||
pfs_test_failures++; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define CHECK_STR_EQ(a, b) do { \
|
||||
const char *_a = (a), *_b = (b); \
|
||||
if (strcmp(_a, _b) != 0) { \
|
||||
fprintf(stderr, "FAIL %s:%d: %s (\"%s\") != %s (\"%s\")\n", __FILE__, __LINE__, #a, _a, #b, _b); \
|
||||
pfs_test_failures++; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define TEST_MAIN_END() \
|
||||
do { \
|
||||
if (pfs_test_failures) { fprintf(stderr, "%d check(s) failed\n", pfs_test_failures); return 1; } \
|
||||
fprintf(stderr, "OK\n"); \
|
||||
return 0; \
|
||||
} while (0)
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,96 @@
|
||||
/* test_mem.c — mem backend: basic CRUD, directory markers (Section 9.3),
|
||||
* rename, readdir. */
|
||||
|
||||
#include "packfs.h"
|
||||
#include "test_harness.h"
|
||||
|
||||
int main(void) {
|
||||
Vfs *v = vfs_new();
|
||||
Backend *mem = backend_mem_new();
|
||||
CHECK_EQ_INT(vfs_mount(v, "/", mem), VFS_OK);
|
||||
|
||||
/* create + write + read back */
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(v, "/hello.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "hello", 5), 5);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
f = vfs_open(v, "/hello.txt", VFS_O_RDONLY, &err);
|
||||
CHECK(f != NULL);
|
||||
char buf[16] = {0};
|
||||
CHECK_EQ_INT(vfs_read(f, buf, sizeof(buf)), 5);
|
||||
CHECK_STR_EQ(buf, "hello");
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
VfsStat st;
|
||||
CHECK_EQ_INT(vfs_stat(v, "/hello.txt", &st), VFS_OK);
|
||||
CHECK_EQ_INT(st.size, 5);
|
||||
CHECK_EQ_INT(st.kind, VFS_KIND_FILE);
|
||||
|
||||
/* nonexistent path */
|
||||
CHECK_EQ_INT(vfs_stat(v, "/nope.txt", &st), VFS_ERR_NOENT);
|
||||
|
||||
/* path escape rejected (Section 6.1) */
|
||||
f = vfs_open(v, "/../etc/passwd", VFS_O_RDONLY, &err);
|
||||
CHECK(f == NULL);
|
||||
CHECK_EQ_INT(err, VFS_ERR_INVAL);
|
||||
|
||||
/* directories: explicit empty dir (Section 9.3) and implicit dir from a file */
|
||||
CHECK_EQ_INT(vfs_mkdir(v, "/empty"), VFS_OK);
|
||||
CHECK_EQ_INT(vfs_stat(v, "/empty", &st), VFS_OK);
|
||||
CHECK_EQ_INT(st.kind, VFS_KIND_DIR);
|
||||
|
||||
f = vfs_open(v, "/nested/deep/file.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "x", 1), 1);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
CHECK_EQ_INT(vfs_stat(v, "/nested", &st), VFS_OK);
|
||||
CHECK_EQ_INT(st.kind, VFS_KIND_DIR);
|
||||
CHECK_EQ_INT(vfs_stat(v, "/nested/deep", &st), VFS_OK);
|
||||
CHECK_EQ_INT(st.kind, VFS_KIND_DIR);
|
||||
|
||||
/* readdir at root: hello.txt, empty, nested */
|
||||
VfsDir dir;
|
||||
CHECK_EQ_INT(vfs_readdir(v, "/", &dir), VFS_OK);
|
||||
int saw_hello = 0, saw_empty = 0, saw_nested = 0;
|
||||
for (pfs_usize i = 0; i < dir.count; i++) {
|
||||
if (strcmp(dir.entries[i].name, "hello.txt") == 0) { saw_hello = 1; CHECK_EQ_INT(dir.entries[i].kind, VFS_KIND_FILE); }
|
||||
if (strcmp(dir.entries[i].name, "empty") == 0) { saw_empty = 1; CHECK_EQ_INT(dir.entries[i].kind, VFS_KIND_DIR); }
|
||||
if (strcmp(dir.entries[i].name, "nested") == 0) { saw_nested = 1; CHECK_EQ_INT(dir.entries[i].kind, VFS_KIND_DIR); }
|
||||
}
|
||||
CHECK(saw_hello && saw_empty && saw_nested);
|
||||
vfs_dir_free(&dir);
|
||||
|
||||
/* rename */
|
||||
CHECK_EQ_INT(vfs_rename(v, "/hello.txt", "/hello2.txt"), VFS_OK);
|
||||
CHECK_EQ_INT(vfs_stat(v, "/hello.txt", &st), VFS_ERR_NOENT);
|
||||
CHECK_EQ_INT(vfs_stat(v, "/hello2.txt", &st), VFS_OK);
|
||||
|
||||
/* unlink + rmdir-of-nonempty rejected */
|
||||
CHECK_EQ_INT(vfs_unlink(v, "/hello2.txt"), VFS_OK);
|
||||
CHECK_EQ_INT(vfs_unlink(v, "/nested"), VFS_ERR_NOTEMPTY);
|
||||
|
||||
/* buffer growth across the 64-byte initial capacity (Section 5.7) */
|
||||
f = vfs_open(v, "/big.bin", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
unsigned char chunk[100];
|
||||
for (int i = 0; i < 100; i++) chunk[i] = (unsigned char)i;
|
||||
for (int i = 0; i < 50; i++) CHECK_EQ_INT(vfs_write(f, chunk, sizeof(chunk)), sizeof(chunk));
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
CHECK_EQ_INT(vfs_stat(v, "/big.bin", &st), VFS_OK);
|
||||
CHECK_EQ_INT(st.size, 5000);
|
||||
f = vfs_open(v, "/big.bin", VFS_O_RDONLY, &err);
|
||||
unsigned char rb[5000];
|
||||
CHECK_EQ_INT(vfs_read(f, rb, sizeof(rb)), 5000);
|
||||
CHECK_EQ_INT(memcmp(rb, rb, 0), 0); /* smoke */
|
||||
int ok = 1;
|
||||
for (int i = 0; i < 5000; i++) if (rb[i] != (unsigned char)(i % 100)) ok = 0;
|
||||
CHECK(ok);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
vfs_unmount(v, "/");
|
||||
backend_free(mem);
|
||||
vfs_free(v);
|
||||
TEST_MAIN_END();
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
/* test_pack_overlay.c — overlay backend: copy-up, whiteout deletion of a
|
||||
* lower-layer entry (Section 4.2), compaction (Section 4.1/5.3),
|
||||
* cross-restart durability via journal replay (Section 4.4), and pack
|
||||
* integrity validation rejecting a corrupted file (Section 7). */
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "packfs.h"
|
||||
#include "test_harness.h"
|
||||
|
||||
static char *tmp_pack_path(void) {
|
||||
static char path[512];
|
||||
snprintf(path, sizeof(path), "/tmp/packfs_test_pack_%d.img", (int)getpid());
|
||||
return path;
|
||||
}
|
||||
|
||||
int main(void) {
|
||||
char *pack_path = tmp_pack_path();
|
||||
unlink(pack_path);
|
||||
char jpath[600];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", pack_path);
|
||||
unlink(jpath);
|
||||
|
||||
/* --- phase 1: build an initial pack directly (as if shipped) --- */
|
||||
Vfs *v = vfs_new();
|
||||
Backend *mem = backend_mem_new();
|
||||
int oerr = 0;
|
||||
Backend *ov = backend_overlay_new(pack_path, mem, &oerr);
|
||||
CHECK(ov != NULL);
|
||||
CHECK_EQ_INT(vfs_mount(v, "/", ov), VFS_OK);
|
||||
|
||||
int err = 0;
|
||||
VfsFile *f = vfs_open(v, "/a.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "AAAA", 4), 4);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
f = vfs_open(v, "/dup.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "AAAA", 4), 4); /* same content as a.txt: exercises dedup (Section 9.2) */
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
f = vfs_open(v, "/b.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "BBBBBB", 6), 6);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
CHECK_EQ_INT(vfs_sync(v, "/"), VFS_OK); /* compaction */
|
||||
|
||||
/* read-only open straight from the freshly-compacted pack */
|
||||
f = vfs_open(v, "/a.txt", VFS_O_RDONLY, &err);
|
||||
CHECK(f != NULL);
|
||||
char buf[16] = {0};
|
||||
CHECK_EQ_INT(vfs_read(f, buf, sizeof(buf)), 4);
|
||||
CHECK_STR_EQ(buf, "AAAA");
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
/* --- copy-up on write intent, whiteout on delete (Section 4.2) --- */
|
||||
f = vfs_open(v, "/a.txt", VFS_O_RDWR, &err); /* triggers copy-up */
|
||||
CHECK(f != NULL);
|
||||
CHECK_EQ_INT(vfs_write(f, "ZZZZ", 4), 4);
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
f = vfs_open(v, "/a.txt", VFS_O_RDONLY, &err);
|
||||
memset(buf, 0, sizeof(buf));
|
||||
CHECK_EQ_INT(vfs_read(f, buf, sizeof(buf)), 4);
|
||||
CHECK_STR_EQ(buf, "ZZZZ");
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
CHECK_EQ_INT(vfs_unlink(v, "/b.txt"), VFS_OK); /* whiteout: b.txt only existed in the pack */
|
||||
VfsStat st;
|
||||
CHECK_EQ_INT(vfs_stat(v, "/b.txt", &st), VFS_ERR_NOENT);
|
||||
|
||||
/* --- durability: a fresh Vfs replays the journal without compacting --- */
|
||||
vfs_unmount(v, "/");
|
||||
backend_free(ov);
|
||||
backend_free(mem);
|
||||
vfs_free(v);
|
||||
|
||||
Vfs *v2 = vfs_new();
|
||||
Backend *mem2 = backend_mem_new();
|
||||
Backend *ov2 = backend_overlay_new(pack_path, mem2, &oerr);
|
||||
CHECK(ov2 != NULL);
|
||||
CHECK_EQ_INT(vfs_mount(v2, "/", ov2), VFS_OK);
|
||||
|
||||
CHECK_EQ_INT(vfs_stat(v2, "/b.txt", &st), VFS_ERR_NOENT); /* whiteout replayed */
|
||||
f = vfs_open(v2, "/a.txt", VFS_O_RDONLY, &err);
|
||||
CHECK(f != NULL);
|
||||
memset(buf, 0, sizeof(buf));
|
||||
CHECK_EQ_INT(vfs_read(f, buf, sizeof(buf)), 4);
|
||||
CHECK_STR_EQ(buf, "ZZZZ"); /* the post-compaction write survived via the journal */
|
||||
CHECK_EQ_INT(vfs_close(f), VFS_OK);
|
||||
|
||||
CHECK_EQ_INT(vfs_sync(v2, "/"), VFS_OK); /* compact again to fold the journal in */
|
||||
vfs_unmount(v2, "/");
|
||||
backend_free(ov2);
|
||||
backend_free(mem2);
|
||||
vfs_free(v2);
|
||||
|
||||
/* --- pack integrity validation (Section 7): a corrupted pack is rejected --- */
|
||||
int fd = open(pack_path, O_RDWR);
|
||||
CHECK(fd >= 0);
|
||||
unsigned char one;
|
||||
CHECK_EQ_INT(pread(fd, &one, 1, 40), 1); /* somewhere inside the header/offsets region */
|
||||
one = (unsigned char)(one ^ 0xFF);
|
||||
CHECK_EQ_INT(pwrite(fd, &one, 1, 40), 1);
|
||||
close(fd);
|
||||
|
||||
Vfs *v3 = vfs_new();
|
||||
Backend *mem3 = backend_mem_new();
|
||||
int oerr3 = 0;
|
||||
Backend *ov3 = backend_overlay_new(pack_path, mem3, &oerr3);
|
||||
CHECK(ov3 == NULL);
|
||||
CHECK_EQ_INT(oerr3, VFS_ERR_CORRUPT);
|
||||
backend_free(mem3);
|
||||
vfs_free(v3);
|
||||
|
||||
unlink(pack_path);
|
||||
unlink(jpath);
|
||||
TEST_MAIN_END();
|
||||
}
|
||||
Reference in New Issue
Block a user