Add PackFS v0: statically linked in-process VFS implementing concept.md

Implements the core design: a mount table published as an atomically-
swapped snapshot; mem/dir/pack/overlay backends; copy-on-write overlay
with copy-up and whiteout deletion; a checksummed append journal;
compaction with exact-duplicate elimination; single-writer/wait-free-
reader concurrency with a structural/content write split; openat2/
Landlock path containment for dir mounts; and load-time pack integrity
validation. Zero required third-party dependencies.

Sanitizer testing (ASan/UBSan) caught and led to fixing a genuine
heap-use-after-free in the snapshot-reclamation path: the textbook
"load pointer, then increment its refcount" pattern left a gap a
concurrent writer could free through. Closed with a small reclaim_gate
rwlock, documented in internal.h and CLAUDE.md since it's a pattern
every refcounted structure in the codebase now follows.

zip/tar import/export backends, recommended in concept.md Section 11,
will not be built — a permanent project decision recorded in CLAUDE.md
since concept.md itself is frozen and cannot be edited to reflect it.

Includes a runnable demo (examples/demo.c, `make demo`) exercising the
library end to end and proving cross-run persistence through the pack
file, plus open-source scaffolding: MIT license, README, CONTRIBUTING,
and a CI workflow running the test suite under ASan/UBSan/TSan.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UqJpkdJ6Njnt1pw3CbghzB
This commit is contained in:
2026-09-14 05:44:08 +00:00
co-authored by Claude Sonnet 5
commit 72e3c900f2
23 changed files with 4010 additions and 0 deletions
+253
View File
@@ -0,0 +1,253 @@
/*
* containment.c — capability-scoped `dir` mounts (Section 6.2) and
* optional Landlock hardening (Section 6.2, 6.3).
*
* On Linux 5.6+, every lookup beneath a `dir` mount's root fd is resolved
* with openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS) in a single kernel
* call, atomically rejecting ".." escapes and symlink escapes. Where
* openat2 is unavailable (ENOSYS on an older kernel), this falls back to
* per-component O_NOFOLLOW resolution, which is an accepted, weaker
* residual-risk posture per Section 6.3 — not a claim of equal
* containment strength.
*/
#include <errno.h>
#include <fcntl.h>
#include <linux/openat2.h>
#include <string.h>
#include <sys/prctl.h>
#include <sys/stat.h>
#include <sys/syscall.h>
#include <unistd.h>
#include "internal.h"
#ifdef __linux__
#include <linux/landlock.h>
#endif
int pfs_dir_capability_open(const char *path, int *err) {
int fd = open(path, O_DIRECTORY | O_CLOEXEC | O_RDONLY);
if (fd < 0) {
if (err) *err = VFS_ERR_IO;
return -1;
}
return fd;
}
/* Splits an already-normalized relative path (no leading slash) into
* (parent, leaf). Root-level names have an empty parent. */
static void split_parent_leaf(const char *rel, char *parent_buf, size_t cap, const char **leaf) {
const char *slash = strrchr(rel, '/');
if (!slash) {
parent_buf[0] = '\0';
*leaf = rel;
return;
}
size_t plen = (size_t)(slash - rel);
if (plen >= cap) plen = cap - 1;
memcpy(parent_buf, rel, plen);
parent_buf[plen] = '\0';
*leaf = slash + 1;
}
#ifdef SYS_openat2
static int openat2_beneath(int root_fd, const char *rel, uint64_t flags, uint64_t mode) {
struct open_how how;
memset(&how, 0, sizeof(how));
how.flags = flags;
how.mode = mode;
how.resolve = RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS;
return (int)syscall(SYS_openat2, root_fd, rel[0] ? rel : ".", &how, sizeof(how));
}
#endif
/* Per-component O_NOFOLLOW fallback (Section 6.3). Walks every component
* except the last with O_NOFOLLOW|O_DIRECTORY, then opens the last with
* the caller's requested flags plus O_NOFOLLOW. Closes most, not all, of
* the same TOCTOU window openat2 closes atomically. */
static int fallback_openat_beneath(int root_fd, const char *rel, int flags, unsigned mode) {
if (rel[0] == '\0') {
return dup(root_fd);
}
char buf[PFS_PATH_MAX];
if (strlen(rel) >= sizeof(buf)) { errno = ENAMETOOLONG; return -1; }
strcpy(buf, rel);
int cur = root_fd;
int owns_cur = 0;
char *save = NULL;
char *tok = strtok_r(buf, "/", &save);
char *next = tok ? strtok_r(NULL, "/", &save) : NULL;
while (tok && next) {
int nfd = openat(cur, tok, O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC);
if (owns_cur) close(cur);
if (nfd < 0) return -1;
cur = nfd;
owns_cur = 1;
tok = next;
next = strtok_r(NULL, "/", &save);
}
int final_fd = -1;
if (tok) {
int final_flags = flags;
if (!(flags & O_CREAT)) final_flags |= O_NOFOLLOW;
final_fd = openat(cur, tok, final_flags, mode);
}
if (owns_cur) close(cur);
return final_fd;
}
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err) {
int fd;
#ifdef SYS_openat2
fd = openat2_beneath(root_fd, rel, (uint64_t)flags, (uint64_t)mode);
if (fd < 0 && errno == ENOSYS) {
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
}
#else
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
#endif
if (fd < 0) {
if (err) *err = (errno == EXDEV || errno == ELOOP) ? VFS_ERR_INVAL
: (errno == ENOENT) ? VFS_ERR_NOENT
: (errno == EEXIST) ? VFS_ERR_EXIST
: VFS_ERR_IO;
return -1;
}
return fd;
}
static int resolve_parent_dir(int root_fd, const char *rel, char *leaf_out, size_t leaf_cap, int *err) {
char parent[PFS_PATH_MAX];
const char *leaf;
split_parent_leaf(rel, parent, sizeof(parent), &leaf);
if (strlen(leaf) >= leaf_cap) { if (err) *err = VFS_ERR_INVAL; return -1; }
strcpy(leaf_out, leaf);
int pfd;
#ifdef SYS_openat2
pfd = openat2_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
if (pfd < 0 && errno == ENOSYS) {
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
}
#else
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
#endif
if (pfd < 0) {
if (err) *err = (errno == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
return pfd;
}
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err) {
char leaf[PFS_PATH_MAX];
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
if (pfd < 0) return -1;
int rc = mkdirat(pfd, leaf, 0777);
int saved = errno;
close(pfd);
if (rc < 0) {
if (err) *err = (saved == EEXIST) ? VFS_ERR_EXIST : (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
return 0;
}
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err) {
char leaf[PFS_PATH_MAX];
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
if (pfd < 0) return -1;
int rc = unlinkat(pfd, leaf, is_dir ? AT_REMOVEDIR : 0);
int saved = errno;
close(pfd);
if (rc < 0) {
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : (saved == ENOTEMPTY) ? VFS_ERR_NOTEMPTY : VFS_ERR_IO;
return -1;
}
return 0;
}
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err) {
char leaf_from[PFS_PATH_MAX], leaf_to[PFS_PATH_MAX];
int pfd_from = resolve_parent_dir(root_fd, from, leaf_from, sizeof(leaf_from), err);
if (pfd_from < 0) return -1;
int pfd_to = resolve_parent_dir(root_fd, to, leaf_to, sizeof(leaf_to), err);
if (pfd_to < 0) { close(pfd_from); return -1; }
int rc = renameat(pfd_from, leaf_from, pfd_to, leaf_to);
int saved = errno;
close(pfd_from);
close(pfd_to);
if (rc < 0) {
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
return 0;
}
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err) {
int fd = pfs_dir_openat(root_fd, rel, O_RDONLY, 0, err);
if (fd < 0) return -1;
struct stat st;
int rc = fstat(fd, &st);
int saved = errno;
close(fd);
if (rc < 0) {
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
out->size = (pfs_usize)st.st_size;
out->mtime = (int64_t)st.st_mtime;
out->kind = S_ISDIR(st.st_mode) ? VFS_KIND_DIR : VFS_KIND_FILE;
return 0;
}
/*
* Best-effort, opt-in, process-wide hardening (Section 6.2). Restricts
* filesystem read/write/create/remove operations to the given
* directory-capability fds; deliberately does not restrict EXECUTE, to
* avoid an embeddable library silently blocking unrelated process
* behavior beyond the filesystem paths it was asked to confine.
*/
int pfs_landlock_restrict_to(const int *roots, size_t count) {
#if defined(__linux__) && defined(SYS_landlock_create_ruleset)
long abi = syscall(SYS_landlock_create_ruleset, NULL, 0, LANDLOCK_CREATE_RULESET_VERSION);
if (abi < 0) return -1; /* unsupported kernel: Section 6.3 residual risk */
uint64_t access = LANDLOCK_ACCESS_FS_READ_FILE | LANDLOCK_ACCESS_FS_WRITE_FILE |
LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_REMOVE_DIR |
LANDLOCK_ACCESS_FS_REMOVE_FILE | LANDLOCK_ACCESS_FS_MAKE_CHAR |
LANDLOCK_ACCESS_FS_MAKE_DIR | LANDLOCK_ACCESS_FS_MAKE_REG |
LANDLOCK_ACCESS_FS_MAKE_SOCK | LANDLOCK_ACCESS_FS_MAKE_FIFO |
LANDLOCK_ACCESS_FS_MAKE_BLOCK | LANDLOCK_ACCESS_FS_MAKE_SYM;
struct landlock_ruleset_attr attr;
memset(&attr, 0, sizeof(attr));
attr.handled_access_fs = access;
int rs_fd = (int)syscall(SYS_landlock_create_ruleset, &attr, sizeof(attr), 0);
if (rs_fd < 0) return -1;
for (size_t i = 0; i < count; i++) {
struct landlock_path_beneath_attr pb;
memset(&pb, 0, sizeof(pb));
pb.allowed_access = access;
pb.parent_fd = roots[i];
if (syscall(SYS_landlock_add_rule, rs_fd, LANDLOCK_RULE_PATH_BENEATH, &pb, 0) < 0) {
close(rs_fd);
return -1;
}
}
if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) < 0) { close(rs_fd); return -1; }
if (syscall(SYS_landlock_restrict_self, rs_fd, 0) < 0) { close(rs_fd); return -1; }
close(rs_fd);
return 0;
#else
(void)roots; (void)count;
return -1;
#endif
}
+20
View File
@@ -0,0 +1,20 @@
/*
* hash.c — FNV-1a 64-bit, used for pack integrity checksums (Section 7)
* and content-addressed dedup during compaction (Section 9.2). Chosen
* because it needs no third-party library, consistent with the
* static-linking constraint (Section 11.1) — this is not a
* cryptographic hash and must never be used where adversarial collision
* resistance is required.
*/
#include "internal.h"
uint64_t pfs_fnv1a64(const void *data, size_t len) {
const unsigned char *p = (const unsigned char *)data;
uint64_t h = 0xcbf29ce484222325ULL;
for (size_t i = 0; i < len; i++) {
h ^= p[i];
h *= 0x100000001b3ULL;
}
return h;
}
+243
View File
@@ -0,0 +1,243 @@
/*
* internal.h — shared internal declarations, not part of the public API.
*
* Every mechanism named here corresponds to a section of concept.md;
* comments reference sections rather than re-deriving the rationale,
* since concept.md is frozen and is the authoritative source for "why."
*/
#ifndef PACKFS_INTERNAL_H
#define PACKFS_INTERNAL_H
#ifndef _GNU_SOURCE
#define _GNU_SOURCE
#endif
#include <stdatomic.h>
#include <stdint.h>
#include <stddef.h>
#include <pthread.h>
#include <stdio.h>
#include "packfs.h"
/* ---- path.c: virtual-namespace canonicalization (Section 6.1) ---- */
/*
* Resolves "." and ".." purely lexically and rejects escapes above "/".
* `out` must be at least PFS_PATH_MAX bytes. Returns 0 on success, -1 if
* the path would resolve above the root (VFS_ERR_INVAL at the caller).
*/
#define PFS_PATH_MAX 4096
int pfs_path_normalize(const char *in, char out[PFS_PATH_MAX]);
/* Longest-prefix match helper: does `path` lie under mount `prefix`? */
int pfs_path_under(const char *prefix, const char *path);
/* ---- hash.c ---- */
uint64_t pfs_fnv1a64(const void *data, size_t len);
/* ---- containment.c: capability-scoped dir mounts (Section 6.2) ---- */
/*
* Opens `path` as a directory capability (O_DIRECTORY), the root of a
* `dir` mount. Returns the fd, or -1 on failure with *err set.
*/
int pfs_dir_capability_open(const char *path, int *err);
/*
* Resolves `rel` (already virtual-namespace-canonicalized, no leading
* slash) beneath directory-capability `root_fd`, using
* openat2(RESOLVE_BENEATH|RESOLVE_NO_SYMLINKS) on Linux 5.6+ and falling
* back to O_NOFOLLOW-per-component + prefix verification where openat2
* is unavailable (Section 6.3). `flags`/`mode` are plain open(2) flags.
*/
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err);
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err);
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err);
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err);
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err);
/*
* Best-effort, opt-in, process-wide Landlock hardening (Section 6.2),
* exposed publicly as vfs_harden_process_with_landlock. `roots` is an
* array of directory-capability fds to allow; `count` its length.
* Returns 0 on success, -1 if unsupported by the running kernel (in
* which case nothing was changed — the residual-risk fallback of
* Section 6.3, not a fatal error).
*/
int pfs_landlock_restrict_to(const int *roots, size_t count);
/* ---- pack.c: on-disk format (Section 9) + integrity (Section 7) ---- */
#define PFS_MODE_DIR 0x1u /* directory-marker bit, Section 9.3 */
typedef struct PackIndexEntry {
uint64_t name_off;
uint32_t name_len;
uint64_t data_off;
uint64_t size;
uint32_t mode;
int64_t mtime;
} PackIndexEntry;
typedef struct Pack {
int fd;
void *map;
size_t map_len;
const PackIndexEntry *entries; /* sorted by name, Section 9.1 */
size_t entry_count;
const char *strings_base;
size_t strings_len;
char *path; /* owned copy, used as compaction target */
} Pack;
int pack_load(const char *path, Pack **out, int *err); /* validates per Section 7 */
void pack_close(Pack *p);
/* Binary search by full path name. Returns 1 and sets *out on hit. */
int pack_find(const Pack *p, const char *name, const PackIndexEntry **out);
/*
* Range query for vfs_readdir (Section 9.1): [*lo, *hi) indices whose
* name starts with `prefix`.
*/
void pack_range(const Pack *p, const char *prefix, size_t *lo, size_t *hi);
const char *pack_entry_name(const Pack *p, const PackIndexEntry *e);
const void *pack_entry_data(const Pack *p, const PackIndexEntry *e);
typedef struct PackBuildEntry {
const char *name;
const void *data;
uint64_t size;
uint32_t mode;
int64_t mtime;
} PackBuildEntry;
/*
* Writes a new pack to `path` + ".tmp", fsyncs, renames over `path`
* (Section 4.3). Entries are sorted by name and content-addressed
* deduplicated by exact (hash, size) match (Section 9.2). Computes and
* stores an integrity checksum consumed by pack_load (Section 7).
*/
int pack_write(const char *path, const PackBuildEntry *entries, size_t count, int *err);
/* ---- upper.c: mem/dir writable layer + snapshot machinery (Section 5) ---- */
typedef enum { UPPER_MEM, UPPER_DIR } UpperKind;
typedef enum { ENTRY_FILE, ENTRY_DIR_MARKER, ENTRY_WHITEOUT } UpperEntryKind;
/*
* Section 5.3's mutable metadata cell, and Section 5.7's buffer-growth
* guard. One MutCell is shared by every UpperSnapshot whose entry array
* still points at the same logical file — an ordinary vfs_write mutates
* this cell in place and never touches the snapshot pointer at all.
*/
typedef struct MutCell {
pthread_mutex_t write_lock; /* serializes writers to the same file */
pthread_rwlock_t buf_lock; /* guards {data,capacity} against grow, mem only */
_Atomic(pfs_usize) size;
_Atomic(int64_t) mtime;
/* UPPER_MEM: */
unsigned char *data; /* guarded by buf_lock on the grow path */
pfs_usize capacity;
/* UPPER_DIR: host bytes live in the host fs; nothing to store here. */
} MutCell;
typedef struct UpperEntry {
char *name; /* full virtual path, e.g. "/a/b.txt" */
UpperEntryKind kind;
MutCell *cell; /* NULL for ENTRY_DIR_MARKER / ENTRY_WHITEOUT */
} UpperEntry;
typedef struct UpperSnapshot {
_Atomic(pfs_usize) refcount;
UpperEntry *entries; /* sorted by name, Section 9.1 */
pfs_usize count;
} UpperSnapshot;
typedef struct UpperStore {
UpperKind kind;
_Atomic(UpperSnapshot *) current;
pthread_mutex_t writer_lock; /* single writer, Section 5.3 */
/*
* Guards the load-then-increment in upper_acquire() against a
* concurrent writer retiring (and freeing) the very snapshot being
* acquired. Plain "load pointer, then atomically increment its
* refcount" has a gap between those two steps during which the
* object can be freed out from under the loader; closing that gap
* is what makes reference counting itself memory-safe, not a
* departure from the reference-counting design. Readers take the
* read side (concurrent, effectively free); a writer takes the
* write side only around retiring the specific snapshot it just
* superseded, which readers-in-flight already past this gate are
* unaffected by.
*/
pthread_rwlock_t reclaim_gate;
int root_fd; /* UPPER_DIR only: containment root, Section 6.2 */
char *root_path; /* UPPER_DIR only, for error messages */
} UpperStore;
UpperStore *upper_new(UpperKind kind, int root_fd /* -1 for mem */, const char *root_path);
void upper_free(UpperStore *u);
UpperSnapshot *upper_acquire(UpperStore *u);
void upper_release(UpperSnapshot *s);
/* Structural + content operations, all against paths already normalized
* and relative to the mount (leading "/"). */
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out);
int upper_has_children(UpperSnapshot *s, const char *path);
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path);
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err);
int upper_mkdir(UpperStore *u, const char *path, int *err);
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err); /* whiteout if had_lower */
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err);
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size, int64_t mtime, MutCell **cell_out, int *err);
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n);
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off, const void *buf, pfs_usize n, int *err);
/* Snapshot the whole store into pack-builder entries (compaction, dir-marker-aware). */
typedef struct UpperWalkCb {
void (*visit)(void *ctx, const char *name, UpperEntryKind kind, const void *data, pfs_usize size, int64_t mtime);
void *ctx;
} UpperWalkCb;
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb); /* skips whiteouts */
void upper_reset_empty(UpperStore *u); /* used after compaction folds upper into the new pack */
Backend *backend_from_upper(UpperStore *u); /* wraps an UpperStore as a standalone Backend */
int upper_backend_root_fd(Backend *b); /* -1 unless b is a standalone `dir` backend */
/* ---- backend vtable shared by vfs.c ---- */
typedef struct BackendOps {
int (*open)(Backend *b, const char *path, int flags, VfsFile **out);
pfs_isize (*read)(VfsFile *f, void *buf, pfs_usize n);
pfs_isize (*write)(VfsFile *f, const void *buf, pfs_usize n);
int (*close)(VfsFile *f);
int (*stat)(Backend *b, const char *path, VfsStat *out);
int (*readdir)(Backend *b, const char *path, VfsDir *out);
int (*mkdir)(Backend *b, const char *path);
int (*unlink)(Backend *b, const char *path);
int (*rename)(Backend *b, const char *from, const char *to);
int (*sync)(Backend *b); /* VFS_ERR_PERM if not an overlay */
void (*free)(Backend *b);
} BackendOps;
struct Backend {
const BackendOps *ops;
void *state;
};
struct VfsFile {
Backend *backend;
void *state;
};
/* ---- overlay.c ---- */
Backend *overlay_upper_backend(Backend *b); /* NULL unless b is an overlay backend */
#endif /* PACKFS_INTERNAL_H */
+638
View File
@@ -0,0 +1,638 @@
/*
* overlay.c — the copy-on-write overlay backend (Sections 4-5, 7, 9.2).
*
* Composes a read-only Pack (lower) with an UpperStore (upper, from
* upper.c) obtained from `mem` or `dir`. Implements the merge (upper
* wins, whiteouts hide lower), copy-up, the append journal (Section 4.4)
* with self-checking records (Section 4.3), and vfs_sync's compaction
* (Section 4.1 step 5, Section 5.3, Section 9.2's exact-duplicate
* elimination via pack_write).
*
* Journal records store whole-value content (Section 4.4 names
* "key/value-ish" workloads as the target — saves, agent memory,
* config); this implementation journals a full replacement value per
* write rather than a byte-range delta, which is the simpler and more
* robust choice for that workload character, at the cost of journal
* size for large files under many small writes. That trade-off is
* deliberate, not an oversight.
*/
#include <stdlib.h>
#include <string.h>
#include <time.h>
#include <unistd.h>
#include "internal.h"
typedef struct Overlay {
Pack *pack; /* may be NULL: empty lower layer */
char *pack_path; /* compaction target + journal anchor; may be NULL */
Backend *upper; /* mem or dir backend, not owned */
pthread_mutex_t journal_lock;
FILE *journal_fp;
} Overlay;
static UpperStore *upper_of(Overlay *ov) { return (UpperStore *)ov->upper->state; }
static const char *dirrel(const char *p) { return p[0] == '/' ? p + 1 : p; }
/* ---- journal (Section 4.4) ---- */
static void journal_append_record(Overlay *ov, uint8_t op, const char *name,
int64_t mtime, uint64_t size, const void *data) {
if (!ov->pack_path) return;
pthread_mutex_lock(&ov->journal_lock);
if (!ov->journal_fp) {
char jpath[PFS_PATH_MAX];
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
ov->journal_fp = fopen(jpath, "ab");
}
if (ov->journal_fp) {
uint32_t name_len = (uint32_t)strlen(name);
uint32_t payload_len = 4 + name_len + (op == 0 ? (uint32_t)(8 + 8 + size) : 0);
unsigned char *payload = (unsigned char *)malloc(payload_len);
size_t off = 0;
memcpy(payload + off, &name_len, 4); off += 4;
memcpy(payload + off, name, name_len); off += name_len;
if (op == 0) {
memcpy(payload + off, &mtime, 8); off += 8;
memcpy(payload + off, &size, 8); off += 8;
if (size) memcpy(payload + off, data, size);
}
uint32_t checksum = (uint32_t)pfs_fnv1a64(payload, payload_len);
fwrite(&payload_len, 4, 1, ov->journal_fp);
fwrite(&checksum, 4, 1, ov->journal_fp);
fwrite(&op, 1, 1, ov->journal_fp);
fwrite(payload, 1, payload_len, ov->journal_fp);
fflush(ov->journal_fp);
fsync(fileno(ov->journal_fp));
free(payload);
}
pthread_mutex_unlock(&ov->journal_lock);
}
static void journal_append_put(Overlay *ov, const char *name, const void *data, uint64_t size, int64_t mtime) {
journal_append_record(ov, 0, name, mtime, size, data);
}
static void journal_append_delete(Overlay *ov, const char *name) {
journal_append_record(ov, 1, name, 0, 0, NULL);
}
static void journal_append_mkdir(Overlay *ov, const char *name) {
journal_append_record(ov, 2, name, 0, 0, NULL);
}
/* journals the *current* full content of `path` after a write/rename,
* matching the whole-value scheme described above. */
static void journal_put_current(Overlay *ov, const char *path, MutCell *cell) {
UpperStore *us = upper_of(ov);
pfs_usize size;
void *buf = NULL;
int64_t mtime = (int64_t)time(NULL);
if (us->kind == UPPER_MEM) {
pthread_rwlock_rdlock(&cell->buf_lock);
size = atomic_load_explicit(&cell->size, memory_order_acquire);
mtime = atomic_load_explicit(&cell->mtime, memory_order_acquire);
buf = size ? malloc(size) : NULL;
if (buf) memcpy(buf, cell->data, size);
pthread_rwlock_unlock(&cell->buf_lock);
} else {
VfsStat st;
int e = 0;
if (pfs_dir_statat(us->root_fd, dirrel(path), &st, &e) < 0) return;
size = st.size;
mtime = st.mtime;
buf = size ? malloc(size) : NULL;
if (buf) upper_cell_read(us, cell, path, 0, buf, size);
}
journal_append_put(ov, path, buf, size, mtime);
free(buf);
}
static void journal_replay(Overlay *ov) {
if (!ov->pack_path) return;
char jpath[PFS_PATH_MAX];
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
FILE *fp = fopen(jpath, "rb");
if (!fp) return;
UpperStore *us = upper_of(ov);
for (;;) {
uint32_t len, checksum; uint8_t op;
if (fread(&len, 4, 1, fp) != 1) break;
if (fread(&checksum, 4, 1, fp) != 1) break;
if (fread(&op, 1, 1, fp) != 1) break;
unsigned char *payload = (unsigned char *)malloc(len);
size_t got = fread(payload, 1, len, fp);
if (got != len) { free(payload); break; } /* torn record: stop, Section 4.3 */
if ((uint32_t)pfs_fnv1a64(payload, len) != checksum) { free(payload); break; }
size_t off = 0;
uint32_t name_len;
memcpy(&name_len, payload + off, 4); off += 4;
char *name = (char *)malloc(name_len + 1);
memcpy(name, payload + off, name_len); name[name_len] = '\0'; off += name_len;
int err = 0;
if (op == 0) {
int64_t mtime; uint64_t size;
memcpy(&mtime, payload + off, 8); off += 8;
memcpy(&size, payload + off, 8); off += 8;
MutCell *cell;
upper_copy_up(us, name, payload + off, size, mtime, &cell, &err);
} else if (op == 1) {
const PackIndexEntry *pe;
int had = ov->pack && pack_find(ov->pack, name, &pe);
upper_remove(us, name, had, &err);
} else if (op == 2) {
upper_mkdir(us, name, &err);
}
free(name);
free(payload);
}
fclose(fp);
}
/* ---- open file handle ---- */
typedef struct OverlayFile {
Overlay *ov;
UpperSnapshot *held; /* keeps `cell` reachable; released on close, NULL if from_upper==0 */
MutCell *cell;
const PackIndexEntry *lower_e; /* valid when from_upper==0 */
int from_upper;
char *path;
pfs_usize pos;
} OverlayFile;
static VfsFile *make_upper_file(Backend *b, Overlay *ov, const char *path, MutCell *cell) {
UpperSnapshot *s = upper_acquire(upper_of(ov));
OverlayFile *of = (OverlayFile *)calloc(1, sizeof(OverlayFile));
of->ov = ov; of->held = s; of->cell = cell; of->from_upper = 1; of->path = strdup(path);
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
f->backend = b; f->state = of;
return f;
}
static int overlay_open(Backend *b, const char *path, int flags, VfsFile **out) {
Overlay *ov = (Overlay *)b->state;
UpperStore *us = upper_of(ov);
UpperSnapshot *s = upper_acquire(us);
const UpperEntry *ue;
int in_upper = upper_lookup(s, path, &ue);
if (in_upper && ue->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
if (in_upper && ue->kind == ENTRY_FILE) {
MutCell *cell = ue->cell;
upper_release(s);
if (flags & VFS_O_TRUNC) upper_cell_truncate(us, cell, path);
*out = make_upper_file(b, ov, path, cell);
return VFS_OK;
}
int was_whiteout = (in_upper && ue->kind == ENTRY_WHITEOUT);
upper_release(s);
if (was_whiteout) {
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
MutCell *cell; int err = 0;
if (upper_create(us, path, 0, &cell, &err) < 0) return err;
*out = make_upper_file(b, ov, path, cell);
return VFS_OK;
}
const PackIndexEntry *pe = NULL;
int in_lower = ov->pack && pack_find(ov->pack, path, &pe);
if (in_lower && (pe->mode & PFS_MODE_DIR)) return VFS_ERR_ISDIR;
if (!in_lower) {
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
MutCell *cell; int err = 0;
if (upper_create(us, path, 0, &cell, &err) < 0) return err;
*out = make_upper_file(b, ov, path, cell);
return VFS_OK;
}
if (!(flags & (VFS_O_WRONLY | VFS_O_RDWR)) && !(flags & VFS_O_TRUNC)) {
/* read-only: serve straight from the pack, no copy-up (Section 4.1 step 3
* ties copy-up to a write-capable open, not every open). */
OverlayFile *of = (OverlayFile *)calloc(1, sizeof(OverlayFile));
of->ov = ov; of->lower_e = pe; of->from_upper = 0; of->path = strdup(path);
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
f->backend = b; f->state = of;
*out = f;
return VFS_OK;
}
const void *data = (flags & VFS_O_TRUNC) ? NULL : pack_entry_data(ov->pack, pe);
uint64_t size = (flags & VFS_O_TRUNC) ? 0 : pe->size;
MutCell *cell; int err = 0;
if (upper_copy_up(us, path, data, size, pe->mtime, &cell, &err) < 0) return err;
*out = make_upper_file(b, ov, path, cell);
return VFS_OK;
}
static pfs_isize overlay_read(VfsFile *f, void *buf, pfs_usize n) {
OverlayFile *of = (OverlayFile *)f->state;
if (of->from_upper) {
pfs_isize r = upper_cell_read(upper_of(of->ov), of->cell, of->path, of->pos, buf, n);
if (r > 0) of->pos += (pfs_usize)r;
return r;
}
pfs_usize size = of->lower_e->size;
pfs_usize avail = of->pos < size ? size - of->pos : 0;
pfs_usize to_copy = n < avail ? n : avail;
if (to_copy) memcpy(buf, (const char *)pack_entry_data(of->ov->pack, of->lower_e) + of->pos, to_copy);
of->pos += to_copy;
return (pfs_isize)to_copy;
}
static pfs_isize overlay_write(VfsFile *f, const void *buf, pfs_usize n) {
OverlayFile *of = (OverlayFile *)f->state;
if (!of->from_upper) return VFS_ERR_PERM;
int err = 0;
pfs_isize w = upper_cell_write(upper_of(of->ov), of->cell, of->path, of->pos, buf, n, &err);
if (w < 0) return err;
of->pos += (pfs_usize)w;
journal_put_current(of->ov, of->path, of->cell);
return w;
}
static int overlay_close(VfsFile *f) {
OverlayFile *of = (OverlayFile *)f->state;
if (of->held) upper_release(of->held);
free(of->path);
free(of);
free(f);
return VFS_OK;
}
static int overlay_stat(Backend *b, const char *path, VfsStat *out) {
Overlay *ov = (Overlay *)b->state;
UpperStore *us = upper_of(ov);
UpperSnapshot *s = upper_acquire(us);
const UpperEntry *ue;
if (upper_lookup(s, path, &ue)) {
if (ue->kind == ENTRY_WHITEOUT) { upper_release(s); return VFS_ERR_NOENT; }
if (ue->kind == ENTRY_DIR_MARKER) {
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
upper_release(s);
return VFS_OK;
}
if (us->kind == UPPER_MEM) {
out->size = atomic_load_explicit(&ue->cell->size, memory_order_acquire);
out->mtime = atomic_load_explicit(&ue->cell->mtime, memory_order_acquire);
out->kind = VFS_KIND_FILE;
upper_release(s);
return VFS_OK;
}
int err = 0;
int rc = pfs_dir_statat(us->root_fd, dirrel(path), out, &err);
upper_release(s);
return rc < 0 ? err : VFS_OK;
}
int has_kids = upper_has_children(s, path);
upper_release(s);
if (has_kids) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
if (ov->pack) {
const PackIndexEntry *pe;
if (pack_find(ov->pack, path, &pe)) {
out->size = pe->size; out->mtime = pe->mtime;
out->kind = (pe->mode & PFS_MODE_DIR) ? VFS_KIND_DIR : VFS_KIND_FILE;
return VFS_OK;
}
char prefix[PFS_PATH_MAX];
snprintf(prefix, sizeof(prefix), "%s/", path);
size_t lo, hi;
pack_range(ov->pack, prefix, &lo, &hi);
if (hi > lo) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
}
if (strcmp(path, "/") == 0) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
return VFS_ERR_NOENT;
}
typedef struct Child { char name[256]; VfsEntryKind kind; } Child;
static int child_index(Child *arr, pfs_usize n, const char *name, size_t len) {
for (pfs_usize i = 0; i < n; i++)
if (strncmp(arr[i].name, name, len) == 0 && arr[i].name[len] == '\0') return (int)i;
return -1;
}
static int overlay_readdir(Backend *b, const char *path, VfsDir *out) {
Overlay *ov = (Overlay *)b->state;
UpperStore *us = upper_of(ov);
char prefix[PFS_PATH_MAX];
if (strcmp(path, "/") == 0) strcpy(prefix, "/"); else snprintf(prefix, sizeof(prefix), "%s/", path);
size_t plen = strlen(prefix);
Child *handled = NULL; pfs_usize hcount = 0, hcap = 0;
VfsDirEntry *entries = NULL; pfs_usize count = 0, cap = 0;
UpperSnapshot *s = upper_acquire(us);
for (pfs_usize i = 0; i < s->count; i++) {
const char *name = s->entries[i].name;
if (strncmp(name, prefix, plen) != 0) continue;
const char *restp = name + plen;
const char *slash = strchr(restp, '/');
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
if (clen == 0 || clen >= sizeof(handled[0].name)) continue;
if (child_index(handled, hcount, restp, clen) >= 0) continue;
if (hcount == hcap) { hcap = hcap ? hcap * 2 : 8; handled = (Child *)realloc(handled, hcap * sizeof(Child)); }
memcpy(handled[hcount].name, restp, clen); handled[hcount].name[clen] = '\0';
handled[hcount].kind = slash ? VFS_KIND_DIR : (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
int is_whiteout = (s->entries[i].kind == ENTRY_WHITEOUT);
hcount++;
if (!is_whiteout) {
if (count == cap) { cap = cap ? cap * 2 : 8; entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry)); }
memcpy(entries[count].name, handled[hcount - 1].name, clen + 1);
entries[count].kind = handled[hcount - 1].kind;
count++;
}
}
upper_release(s);
if (ov->pack) {
size_t lo, hi;
pack_range(ov->pack, prefix, &lo, &hi);
for (size_t i = lo; i < hi; i++) {
const char *name = pack_entry_name(ov->pack, &ov->pack->entries[i]);
const char *restp = name + plen;
const char *slash = strchr(restp, '/');
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
if (clen == 0 || clen >= 256) continue;
if (child_index(handled, hcount, restp, clen) >= 0) continue;
VfsDirEntry tmp; memcpy(tmp.name, restp, clen); tmp.name[clen] = '\0';
int dup = 0;
for (pfs_usize j = 0; j < count; j++)
if (strncmp(entries[j].name, restp, clen) == 0 && entries[j].name[clen] == '\0') { dup = 1; break; }
if (dup) continue;
if (count == cap) { cap = cap ? cap * 2 : 8; entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry)); }
entries[count] = tmp;
entries[count].kind = slash ? VFS_KIND_DIR : ((ov->pack->entries[i].mode & PFS_MODE_DIR) ? VFS_KIND_DIR : VFS_KIND_FILE);
count++;
}
}
free(handled);
out->entries = entries;
out->count = count;
return VFS_OK;
}
static int overlay_mkdir(Backend *b, const char *path) {
Overlay *ov = (Overlay *)b->state;
UpperStore *us = upper_of(ov);
UpperSnapshot *s = upper_acquire(us);
const UpperEntry *ue;
int exists = upper_lookup(s, path, &ue) && ue->kind != ENTRY_WHITEOUT;
int has_kids = upper_has_children(s, path);
upper_release(s);
if (!exists && ov->pack) {
const PackIndexEntry *pe;
if (pack_find(ov->pack, path, &pe)) exists = 1;
if (!exists && !has_kids) {
char prefix[PFS_PATH_MAX];
snprintf(prefix, sizeof(prefix), "%s/", path);
size_t lo, hi;
pack_range(ov->pack, prefix, &lo, &hi);
if (hi > lo) has_kids = 1;
}
}
if (exists || has_kids) return VFS_ERR_EXIST;
int err = 0;
if (upper_mkdir(us, path, &err) < 0) return err;
journal_append_mkdir(ov, path);
return VFS_OK;
}
static int overlay_unlink(Backend *b, const char *path) {
Overlay *ov = (Overlay *)b->state;
UpperStore *us = upper_of(ov);
UpperSnapshot *s = upper_acquire(us);
const UpperEntry *ue;
int in_upper = upper_lookup(s, path, &ue);
if (in_upper && ue->kind == ENTRY_WHITEOUT) { upper_release(s); return VFS_ERR_NOENT; }
if (in_upper && ue->kind == ENTRY_DIR_MARKER && upper_has_children(s, path)) { upper_release(s); return VFS_ERR_NOTEMPTY; }
upper_release(s);
const PackIndexEntry *pe = NULL;
int had_lower = ov->pack && pack_find(ov->pack, path, &pe);
if (had_lower && (pe->mode & PFS_MODE_DIR)) {
char prefix[PFS_PATH_MAX];
snprintf(prefix, sizeof(prefix), "%s/", path);
size_t lo, hi;
pack_range(ov->pack, prefix, &lo, &hi);
UpperSnapshot *s2 = upper_acquire(us);
int kids = (hi > lo) || upper_has_children(s2, path);
upper_release(s2);
if (kids) return VFS_ERR_NOTEMPTY;
}
if (!in_upper && !had_lower) {
UpperSnapshot *s3 = upper_acquire(us);
int kids = upper_has_children(s3, path);
upper_release(s3);
if (!kids && ov->pack) {
char prefix[PFS_PATH_MAX];
snprintf(prefix, sizeof(prefix), "%s/", path);
size_t lo, hi;
pack_range(ov->pack, prefix, &lo, &hi);
kids = hi > lo;
}
return kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
}
int err = 0;
if (upper_remove(us, path, had_lower, &err) < 0) return err;
journal_append_delete(ov, path);
return VFS_OK;
}
static int overlay_rename(Backend *b, const char *from, const char *to) {
Overlay *ov = (Overlay *)b->state;
UpperStore *us = upper_of(ov);
UpperSnapshot *s = upper_acquire(us);
const UpperEntry *ue;
int in_upper = upper_lookup(s, from, &ue) && ue->kind != ENTRY_WHITEOUT;
UpperEntryKind kind = in_upper ? ue->kind : ENTRY_FILE;
upper_release(s);
const PackIndexEntry *pe = NULL;
int had_lower = ov->pack && pack_find(ov->pack, from, &pe);
if (!in_upper && !had_lower) return VFS_ERR_NOENT;
if (in_upper) {
int err = 0;
if (upper_rename(us, from, to, had_lower, &err) < 0) return err;
} else if (pe->mode & PFS_MODE_DIR) {
int err = 0;
if (upper_mkdir(us, to, &err) < 0 && err != VFS_ERR_EXIST) return err;
int err2 = 0;
upper_remove(us, from, 1, &err2);
} else {
MutCell *cell; int err = 0;
if (upper_copy_up(us, to, pack_entry_data(ov->pack, pe), pe->size, pe->mtime, &cell, &err) < 0) return err;
int err2 = 0;
upper_remove(us, from, 1, &err2);
}
journal_append_delete(ov, from);
if (kind == ENTRY_DIR_MARKER) {
journal_append_mkdir(ov, to);
} else {
UpperSnapshot *s2 = upper_acquire(us);
const UpperEntry *nue;
if (upper_lookup(s2, to, &nue) && nue->kind == ENTRY_FILE) journal_put_current(ov, to, nue->cell);
upper_release(s2);
}
return VFS_OK;
}
/* ---- compaction (Section 4.1 step 5, 5.3, 9.2) ---- */
typedef struct BuildCtx {
PackBuildEntry *entries;
size_t count, cap;
size_t upper_start; /* index at which upper-sourced (malloc'd) entries begin */
} BuildCtx;
static void build_push(BuildCtx *bc, const char *name, const void *data, pfs_usize size, uint32_t mode, int64_t mtime, int owned) {
if (bc->count == bc->cap) { bc->cap = bc->cap ? bc->cap * 2 : 16; bc->entries = (PackBuildEntry *)realloc(bc->entries, bc->cap * sizeof(PackBuildEntry)); }
PackBuildEntry *e = &bc->entries[bc->count++];
e->name = strdup(name);
if (owned) {
e->data = size ? malloc(size) : NULL;
if (size) memcpy((void *)e->data, data, size);
} else {
e->data = data;
}
e->size = size;
e->mode = mode;
e->mtime = mtime;
}
static void build_visit(void *ctx, const char *name, UpperEntryKind kind, const void *data, pfs_usize size, int64_t mtime) {
build_push((BuildCtx *)ctx, name, data, size, kind == ENTRY_DIR_MARKER ? PFS_MODE_DIR : 0, mtime, 1);
}
static void free_build_entries(BuildCtx *bc) {
for (size_t i = 0; i < bc->count; i++) {
free((void *)bc->entries[i].name);
if (i >= bc->upper_start) free((void *)bc->entries[i].data);
}
free(bc->entries);
}
static int overlay_sync(Backend *b) {
Overlay *ov = (Overlay *)b->state;
if (!ov->pack_path) return VFS_ERR_PERM;
UpperStore *us = upper_of(ov);
pthread_mutex_lock(&ov->journal_lock);
if (ov->journal_fp) fflush(ov->journal_fp);
long watermark = 0;
char jpath[PFS_PATH_MAX];
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
FILE *probe = fopen(jpath, "rb");
if (probe) { fseek(probe, 0, SEEK_END); watermark = ftell(probe); fclose(probe); }
pthread_mutex_unlock(&ov->journal_lock);
BuildCtx bc; memset(&bc, 0, sizeof(bc));
if (ov->pack) {
UpperSnapshot *s = upper_acquire(us);
for (size_t i = 0; i < ov->pack->entry_count; i++) {
const PackIndexEntry *pe = &ov->pack->entries[i];
const char *name = pack_entry_name(ov->pack, pe);
const UpperEntry *ue;
if (upper_lookup(s, name, &ue)) continue; /* upper (incl. whiteout) supersedes */
build_push(&bc, name, pack_entry_data(ov->pack, pe), pe->size, pe->mode, pe->mtime, 0);
}
upper_release(s);
}
bc.upper_start = bc.count;
UpperWalkCb cb = { build_visit, &bc };
upper_walk_live(us, &cb);
int err = 0;
if (pack_write(ov->pack_path, bc.entries, bc.count, &err) < 0) {
free_build_entries(&bc);
return err; /* Section 4.3: existing pack + journal are untouched */
}
free_build_entries(&bc);
Pack *newpack = NULL; int lerr = 0;
if (pack_load(ov->pack_path, &newpack, &lerr) == 0) {
if (ov->pack) pack_close(ov->pack);
ov->pack = newpack;
}
upper_reset_empty(us);
pthread_mutex_lock(&ov->journal_lock);
char jtmp[PFS_PATH_MAX];
snprintf(jtmp, sizeof(jtmp), "%s.jnl.tmp", ov->pack_path);
FILE *src = fopen(jpath, "rb");
if (src) {
fseek(src, watermark, SEEK_SET);
FILE *dst = fopen(jtmp, "wb");
if (dst) {
char buf[4096]; size_t n;
while ((n = fread(buf, 1, sizeof(buf), src)) > 0) fwrite(buf, 1, n, dst);
fflush(dst); fsync(fileno(dst)); fclose(dst);
rename(jtmp, jpath);
}
fclose(src);
}
if (ov->journal_fp) { fclose(ov->journal_fp); ov->journal_fp = NULL; }
pthread_mutex_unlock(&ov->journal_lock);
return VFS_OK;
}
static void overlay_free(Backend *b) {
Overlay *ov = (Overlay *)b->state;
if (ov->journal_fp) fclose(ov->journal_fp);
pthread_mutex_destroy(&ov->journal_lock);
if (ov->pack) pack_close(ov->pack);
free(ov->pack_path);
free(ov);
free(b);
}
static const BackendOps OVERLAY_OPS = {
overlay_open, overlay_read, overlay_write, overlay_close, overlay_stat,
overlay_readdir, overlay_mkdir, overlay_unlink, overlay_rename, overlay_sync, overlay_free
};
Backend *overlay_upper_backend(Backend *b) {
return b->ops == &OVERLAY_OPS ? ((Overlay *)b->state)->upper : NULL;
}
Backend *backend_overlay_new(const char *pack_path, Backend *upper, int *err) {
Overlay *ov = (Overlay *)calloc(1, sizeof(Overlay));
ov->pack_path = pack_path ? strdup(pack_path) : NULL;
ov->upper = upper;
pthread_mutex_init(&ov->journal_lock, NULL);
if (pack_path) {
Pack *p; int lerr = 0;
if (pack_load(pack_path, &p, &lerr) == 0) {
ov->pack = p;
} else if (lerr != VFS_ERR_NOENT) {
pthread_mutex_destroy(&ov->journal_lock);
free(ov->pack_path);
free(ov);
if (err) *err = lerr;
return NULL;
}
}
journal_replay(ov);
Backend *b = (Backend *)calloc(1, sizeof(Backend));
b->ops = &OVERLAY_OPS;
b->state = ov;
return b;
}
+304
View File
@@ -0,0 +1,304 @@
/*
* pack.c — on-disk pack format (Section 9), load-time integrity
* validation (Section 7), and compaction's writer (Section 4.1, 4.3,
* 5.3, 9.1, 9.2).
*
* On-disk layout (a concrete realization of the illustrative sketch in
* Section 9, extended with the checksum field Section 7 requires):
*
* PackHeader (fixed size, 8-byte aligned)
* blobs (index_offset - sizeof(PackHeader) bytes)
* PackIndexEntry[index_count] at index_offset, sorted by name (9.1)
* strings (each name NUL-terminated) at strings_offset
*/
#include <errno.h>
#include <fcntl.h>
#include <stdlib.h>
#include <string.h>
#include <sys/mman.h>
#include <sys/stat.h>
#include <unistd.h>
#include "internal.h"
#define PACK_VERSION 1
typedef struct PackHeader {
char magic[4];
uint32_t version;
uint64_t index_offset;
uint64_t index_count;
uint64_t strings_offset;
uint64_t strings_len;
uint64_t checksum; /* FNV-1a64 over [index_offset, strings_offset+strings_len) */
} PackHeader;
static uint64_t align8(uint64_t n) { return (n + 7u) & ~(uint64_t)7u; }
/* ---- loading + validation (Section 7) ---- */
void pack_close(Pack *p) {
if (!p) return;
if (p->map && p->map != MAP_FAILED) munmap(p->map, p->map_len);
if (p->fd >= 0) close(p->fd);
free(p->path);
free(p);
}
int pack_load(const char *path, Pack **out, int *err) {
int fd = open(path, O_RDONLY | O_CLOEXEC);
if (fd < 0) { if (err) *err = VFS_ERR_NOENT; return -1; }
struct stat st;
if (fstat(fd, &st) < 0) { close(fd); if (err) *err = VFS_ERR_IO; return -1; }
size_t len = (size_t)st.st_size;
if (len < sizeof(PackHeader)) { close(fd); if (err) *err = VFS_ERR_CORRUPT; return -1; }
void *map = mmap(NULL, len, PROT_READ, MAP_PRIVATE, fd, 0);
if (map == MAP_FAILED) { close(fd); if (err) *err = VFS_ERR_IO; return -1; }
const PackHeader *hdr = (const PackHeader *)map;
if (memcmp(hdr->magic, "PKFS", 4) != 0 || hdr->version != PACK_VERSION) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
/* Bounds-check every offset before trusting it (Section 7): the
* index and strings regions must lie within the file and must not
* overlap the header. */
if (hdr->index_offset < sizeof(PackHeader) ||
hdr->index_offset > len ||
hdr->index_count > (len - hdr->index_offset) / sizeof(PackIndexEntry)) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
uint64_t index_bytes = hdr->index_count * (uint64_t)sizeof(PackIndexEntry);
uint64_t index_end = hdr->index_offset + index_bytes;
if (hdr->strings_offset < index_end || hdr->strings_offset > len ||
hdr->strings_len > len - hdr->strings_offset) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
uint64_t strings_end = hdr->strings_offset + hdr->strings_len;
if (strings_end > len) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
uint64_t checksum = pfs_fnv1a64((const char *)map + hdr->index_offset,
(size_t)(strings_end - hdr->index_offset));
if (checksum != hdr->checksum) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
const PackIndexEntry *entries = (const PackIndexEntry *)((const char *)map + hdr->index_offset);
const char *strings_base = (const char *)map + hdr->strings_offset;
for (uint64_t i = 0; i < hdr->index_count; i++) {
const PackIndexEntry *e = &entries[i];
if (e->data_off < sizeof(PackHeader) || e->data_off > hdr->index_offset ||
e->size > hdr->index_offset - e->data_off) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
if (e->name_off > hdr->strings_len || e->name_len > hdr->strings_len - e->name_off) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
/* Section 9's names are NUL-terminated in the strings region; a
* missing terminator is treated as corruption, not tolerated. */
if (strings_base[e->name_off + e->name_len] != '\0') {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT;
return -1;
}
if (i > 0) {
const PackIndexEntry *prev = &entries[i - 1];
if (strcmp(strings_base + prev->name_off, strings_base + e->name_off) >= 0) {
munmap(map, len); close(fd);
if (err) *err = VFS_ERR_CORRUPT; /* Section 9.1 relies on sortedness */
return -1;
}
}
}
Pack *p = (Pack *)calloc(1, sizeof(Pack));
if (!p) { munmap(map, len); close(fd); if (err) *err = VFS_ERR_NOSPC; return -1; }
p->fd = fd;
p->map = map;
p->map_len = len;
p->entries = entries;
p->entry_count = (size_t)hdr->index_count;
p->strings_base = strings_base;
p->strings_len = (size_t)hdr->strings_len;
p->path = strdup(path);
*out = p;
return 0;
}
const char *pack_entry_name(const Pack *p, const PackIndexEntry *e) {
return p->strings_base + e->name_off;
}
const void *pack_entry_data(const Pack *p, const PackIndexEntry *e) {
return (const char *)p->map + e->data_off;
}
int pack_find(const Pack *p, const char *name, const PackIndexEntry **out) {
size_t lo = 0, hi = p->entry_count;
while (lo < hi) {
size_t mid = lo + (hi - lo) / 2;
int c = strcmp(pack_entry_name(p, &p->entries[mid]), name);
if (c == 0) { *out = &p->entries[mid]; return 1; }
if (c < 0) lo = mid + 1; else hi = mid;
}
return 0;
}
static size_t lower_bound_str(const Pack *p, const char *key) {
size_t lo = 0, hi = p->entry_count;
while (lo < hi) {
size_t mid = lo + (hi - lo) / 2;
if (strcmp(pack_entry_name(p, &p->entries[mid]), key) < 0) lo = mid + 1; else hi = mid;
}
return lo;
}
void pack_range(const Pack *p, const char *prefix, size_t *lo_out, size_t *hi_out) {
size_t lo = lower_bound_str(p, prefix);
size_t plen = strlen(prefix);
char *upper = (char *)malloc(plen + 2);
memcpy(upper, prefix, plen);
upper[plen] = (char)0x7F; /* > any valid path byte, per Section 9.1's argument */
upper[plen + 1] = '\0';
size_t hi = lower_bound_str(p, upper);
free(upper);
*lo_out = lo;
*hi_out = hi;
}
/* ---- writing (compaction target, Section 4.1 / 4.3 / 9.2) ---- */
static int pack_build_entry_cmp(const void *a, const void *b) {
return strcmp(((const PackBuildEntry *)a)->name, ((const PackBuildEntry *)b)->name);
}
typedef struct DedupSlot {
uint64_t hash;
uint64_t size;
uint64_t data_off;
} DedupSlot;
int pack_write(const char *path, const PackBuildEntry *in, size_t count, int *err) {
PackBuildEntry *entries = NULL;
if (count > 0) {
entries = (PackBuildEntry *)malloc(count * sizeof(PackBuildEntry));
if (!entries) { if (err) *err = VFS_ERR_NOSPC; return -1; }
memcpy(entries, in, count * sizeof(PackBuildEntry));
}
qsort(entries, count, sizeof(PackBuildEntry), pack_build_entry_cmp);
/* Pass 1: compute blob region with exact-duplicate elimination
* (Section 9.2) and the strings region size. */
DedupSlot *slots = count ? (DedupSlot *)malloc(count * sizeof(DedupSlot)) : NULL;
size_t slot_count = 0;
uint64_t *data_off_for = count ? (uint64_t *)malloc(count * sizeof(uint64_t)) : NULL;
uint64_t blob_cursor = align8(sizeof(PackHeader));
uint64_t strings_total = 0;
for (size_t i = 0; i < count; i++) {
uint64_t h = pfs_fnv1a64(entries[i].data, (size_t)entries[i].size);
uint64_t reuse = UINT64_MAX;
for (size_t j = 0; j < slot_count; j++) {
if (slots[j].hash == h && slots[j].size == entries[i].size) {
reuse = slots[j].data_off;
break;
}
}
if (reuse != UINT64_MAX) {
data_off_for[i] = reuse;
} else {
data_off_for[i] = blob_cursor;
slots[slot_count].hash = h;
slots[slot_count].size = entries[i].size;
slots[slot_count].data_off = blob_cursor;
slot_count++;
blob_cursor += entries[i].size;
}
strings_total += strlen(entries[i].name) + 1;
}
free(slots);
uint64_t index_offset = align8(blob_cursor);
uint64_t index_bytes = (uint64_t)count * sizeof(PackIndexEntry);
uint64_t strings_offset = index_offset + index_bytes;
uint64_t total_size = strings_offset + strings_total;
unsigned char *buf = (unsigned char *)calloc(1, (size_t)total_size);
if (!buf) { free(entries); free(data_off_for); if (err) *err = VFS_ERR_NOSPC; return -1; }
for (size_t i = 0; i < count; i++) {
/* Deduplicated entries (Section 9.2) share a data_off; writing
* the same bytes to it more than once is redundant but harmless. */
memcpy(buf + data_off_for[i], entries[i].data, entries[i].size);
}
PackIndexEntry *out_entries = (PackIndexEntry *)(buf + index_offset);
uint64_t str_cursor = 0;
for (size_t i = 0; i < count; i++) {
size_t nlen = strlen(entries[i].name);
out_entries[i].name_off = str_cursor;
out_entries[i].name_len = (uint32_t)nlen;
out_entries[i].data_off = data_off_for[i];
out_entries[i].size = entries[i].size;
out_entries[i].mode = entries[i].mode;
out_entries[i].mtime = entries[i].mtime;
memcpy(buf + strings_offset + str_cursor, entries[i].name, nlen + 1);
str_cursor += nlen + 1;
}
PackHeader hdr;
memset(&hdr, 0, sizeof(hdr));
memcpy(hdr.magic, "PKFS", 4);
hdr.version = PACK_VERSION;
hdr.index_offset = index_offset;
hdr.index_count = count;
hdr.strings_offset = strings_offset;
hdr.strings_len = strings_total;
hdr.checksum = pfs_fnv1a64(buf + index_offset, (size_t)(strings_offset + strings_total - index_offset));
memcpy(buf, &hdr, sizeof(hdr));
free(entries);
free(data_off_for);
/* Atomic compaction (Section 4.3): write to a temp file, fsync, then
* rename over the target. A failure here aborts compaction and
* leaves the existing pack untouched. */
char tmp_path[PFS_PATH_MAX];
snprintf(tmp_path, sizeof(tmp_path), "%s.tmp", path);
int fd = open(tmp_path, O_WRONLY | O_CREAT | O_TRUNC, 0644);
if (fd < 0) { free(buf); if (err) *err = VFS_ERR_IO; return -1; }
size_t written = 0;
while (written < (size_t)total_size) {
ssize_t w = write(fd, buf + written, (size_t)total_size - written);
if (w < 0) { close(fd); free(buf); unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
written += (size_t)w;
}
free(buf);
if (fsync(fd) < 0) { close(fd); unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
close(fd);
if (rename(tmp_path, path) < 0) { unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
return 0;
}
+59
View File
@@ -0,0 +1,59 @@
/*
* path.c — virtual-namespace canonicalization, concept.md Section 6.1.
*
* Resolves "." and ".." purely lexically, before any backend is reached,
* so a crafted path cannot escape a mount's prefix even when no host
* directory is involved.
*/
#include <string.h>
#include "internal.h"
int pfs_path_normalize(const char *in, char out[PFS_PATH_MAX]) {
if (in == NULL || in[0] != '/') return -1;
/* Stack of component (start, len) pairs, built by lexical resolution. */
const char *starts[256];
size_t lens[256];
int depth = 0;
const char *p = in;
while (*p) {
while (*p == '/') p++;
if (!*p) break;
const char *comp = p;
while (*p && *p != '/') p++;
size_t len = (size_t)(p - comp);
if (len == 1 && comp[0] == '.') {
continue;
}
if (len == 2 && comp[0] == '.' && comp[1] == '.') {
if (depth == 0) return -1; /* escape above root */
depth--;
continue;
}
if (depth >= 256) return -1; /* pathologically deep; reject rather than overflow */
starts[depth] = comp;
lens[depth] = len;
depth++;
}
size_t off = 0;
out[off++] = '/';
for (int i = 0; i < depth; i++) {
if (off + lens[i] + 2 > PFS_PATH_MAX) return -1;
if (i > 0) out[off++] = '/';
memcpy(out + off, starts[i], lens[i]);
off += lens[i];
}
out[off] = '\0';
return 0;
}
int pfs_path_under(const char *prefix, const char *path) {
size_t plen = strlen(prefix);
if (plen == 1 && prefix[0] == '/') return 1; /* root mount covers everything */
if (strncmp(prefix, path, plen) != 0) return 0;
return path[plen] == '\0' || path[plen] == '/';
}
+706
View File
@@ -0,0 +1,706 @@
/*
* upper.c — the writable upper layer shared by the `mem` and `dir`
* backends, and by the overlay's upper side (Section 3, Section 5).
*
* Implements the snapshot-based single-writer, wait-free-reader model of
* Section 5.3: an immutable, refcounted UpperSnapshot reached through one
* atomic pointer; structural writes (create/unlink/rename/mkdir/
* whiteout) build and publish a new snapshot under `writer_lock`;
* content writes (`upper_cell_write`) mutate a shared MutCell in place
* and never touch the snapshot pointer, per the structural/content split
* in Section 5.3. Buffer growth on the `mem` side follows the
* replace-don't-mutate discipline of Section 5.7.
*/
#include <errno.h>
#include <fcntl.h>
#include <stdlib.h>
#include <string.h>
#include <sys/stat.h>
#include <time.h>
#include <unistd.h>
#include "internal.h"
/* ---- MutCell lifetime: refcounted separately from any one snapshot,
* since a cell's identity is shared across every snapshot whose entry
* array still points at the same logical file (Section 5.3). ---- */
typedef struct MutCellRc {
MutCell pub;
_Atomic(pfs_usize) refcount;
} MutCellRc;
/* MutCell* handed around externally IS the MutCellRc's first member's
* address, so a plain cast recovers the refcount wrapper. */
static MutCellRc *rc_of(MutCell *c) { return (MutCellRc *)c; }
static void mutcell_ref(MutCell *c) {
if (!c) return;
atomic_fetch_add_explicit(&rc_of(c)->refcount, 1, memory_order_relaxed);
}
static void mutcell_unref(MutCell *c) {
if (!c) return;
if (atomic_fetch_sub_explicit(&rc_of(c)->refcount, 1, memory_order_acq_rel) == 1) {
pthread_mutex_destroy(&c->write_lock);
pthread_rwlock_destroy(&c->buf_lock);
free(c->data);
free(rc_of(c));
}
}
static MutCell *mutcell_new(void) {
MutCellRc *rc = (MutCellRc *)calloc(1, sizeof(MutCellRc));
pthread_mutex_init(&rc->pub.write_lock, NULL);
pthread_rwlock_init(&rc->pub.buf_lock, NULL);
atomic_init(&rc->pub.size, 0);
atomic_init(&rc->pub.mtime, (int64_t)time(NULL));
atomic_init(&rc->refcount, 1); /* owned by whichever snapshot installs it first */
return &rc->pub;
}
/* ---- UpperSnapshot lifetime ---- */
UpperSnapshot *upper_acquire(UpperStore *u) {
pthread_rwlock_rdlock(&u->reclaim_gate);
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
pthread_rwlock_unlock(&u->reclaim_gate);
return s;
}
/* Retires a snapshot just superseded by a writer's atomic_store to
* u->current. Must run under reclaim_gate's write side so that no
* upper_acquire() can be mid-flight (loaded `old` but not yet
* incremented its refcount) when this potentially frees it. */
static void upper_retire(UpperStore *u, UpperSnapshot *old) {
pthread_rwlock_wrlock(&u->reclaim_gate);
upper_release(old);
pthread_rwlock_unlock(&u->reclaim_gate);
}
void upper_release(UpperSnapshot *s) {
if (!s) return;
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
for (pfs_usize i = 0; i < s->count; i++) {
free(s->entries[i].name);
mutcell_unref(s->entries[i].cell);
}
free(s->entries);
free(s);
}
}
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out) {
pfs_usize lo = 0, hi = s->count;
while (lo < hi) {
pfs_usize mid = lo + (hi - lo) / 2;
int c = strcmp(s->entries[mid].name, path);
if (c == 0) { *out = &s->entries[mid]; return 1; }
if (c < 0) lo = mid + 1; else hi = mid;
}
return 0;
}
int upper_has_children(UpperSnapshot *s, const char *path) {
char prefix[PFS_PATH_MAX];
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
else snprintf(prefix, sizeof(prefix), "%s/", path);
size_t plen = strlen(prefix);
for (pfs_usize i = 0; i < s->count; i++) {
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
if (strncmp(s->entries[i].name, prefix, plen) == 0) return 1;
}
return 0;
}
/* Builds a new snapshot with one entry inserted or replaced at `name`. */
static UpperSnapshot *snapshot_upsert(UpperSnapshot *base, const char *name,
UpperEntryKind kind, MutCell *cell) {
pfs_usize old_count = base ? base->count : 0;
pfs_usize lo = 0, hi = old_count;
int found = 0;
while (lo < hi) {
pfs_usize mid = lo + (hi - lo) / 2;
int c = strcmp(base->entries[mid].name, name);
if (c == 0) { lo = mid; found = 1; break; }
if (c < 0) lo = mid + 1; else hi = mid;
}
pfs_usize new_count = found ? old_count : old_count + 1;
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
ns->entries = (UpperEntry *)calloc(new_count ? new_count : 1, sizeof(UpperEntry));
ns->count = new_count;
atomic_init(&ns->refcount, 1);
pfs_usize w = 0;
for (pfs_usize i = 0; i < old_count; i++) {
if (i == lo) {
ns->entries[w].name = strdup(name);
ns->entries[w].kind = kind;
ns->entries[w].cell = cell;
mutcell_ref(cell);
w++;
if (found) continue; /* replaces base->entries[lo] */
}
ns->entries[w].name = strdup(base->entries[i].name);
ns->entries[w].kind = base->entries[i].kind;
ns->entries[w].cell = base->entries[i].cell;
mutcell_ref(ns->entries[w].cell);
w++;
}
if (lo == old_count) {
ns->entries[w].name = strdup(name);
ns->entries[w].kind = kind;
ns->entries[w].cell = cell;
mutcell_ref(cell);
w++;
}
return ns;
}
static UpperSnapshot *snapshot_remove(UpperSnapshot *base, const char *name) {
pfs_usize lo = 0, hi = base->count;
int found = 0;
while (lo < hi) {
pfs_usize mid = lo + (hi - lo) / 2;
int c = strcmp(base->entries[mid].name, name);
if (c == 0) { lo = mid; found = 1; break; }
if (c < 0) lo = mid + 1; else hi = mid;
}
if (!found) return NULL;
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
ns->count = base->count - 1;
ns->entries = ns->count ? (UpperEntry *)calloc(ns->count, sizeof(UpperEntry)) : NULL;
atomic_init(&ns->refcount, 1);
pfs_usize w = 0;
for (pfs_usize i = 0; i < base->count; i++) {
if (i == lo) continue;
ns->entries[w].name = strdup(base->entries[i].name);
ns->entries[w].kind = base->entries[i].kind;
ns->entries[w].cell = base->entries[i].cell;
mutcell_ref(ns->entries[w].cell);
w++;
}
return ns;
}
UpperStore *upper_new(UpperKind kind, int root_fd, const char *root_path) {
UpperStore *u = (UpperStore *)calloc(1, sizeof(UpperStore));
u->kind = kind;
u->root_fd = root_fd;
u->root_path = root_path ? strdup(root_path) : NULL;
pthread_mutex_init(&u->writer_lock, NULL);
pthread_rwlock_init(&u->reclaim_gate, NULL);
UpperSnapshot *s = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
atomic_init(&s->refcount, 1);
atomic_init(&u->current, s);
return u;
}
void upper_free(UpperStore *u) {
if (!u) return;
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
upper_release(s);
pthread_mutex_destroy(&u->writer_lock);
pthread_rwlock_destroy(&u->reclaim_gate);
if (u->kind == UPPER_DIR && u->root_fd >= 0) close(u->root_fd);
free(u->root_path);
free(u);
}
/* strips the mount-relative leading '/' for host syscalls */
static const char *rel(const char *path) { return path[0] == '/' ? path + 1 : path; }
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err) {
(void)truncate_existing;
pthread_mutex_lock(&u->writer_lock);
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
const UpperEntry *existing;
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
MutCell *c = existing->cell;
pthread_mutex_unlock(&u->writer_lock);
*cell_out = c;
return 0;
}
if (u->kind == UPPER_DIR) {
int e = 0;
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
if (fd < 0) { pthread_mutex_unlock(&u->writer_lock); if (err) *err = e; return -1; }
close(fd);
}
MutCell *cell = mutcell_new();
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
atomic_store_explicit(&u->current, ns, memory_order_release);
pthread_mutex_unlock(&u->writer_lock);
upper_retire(u, old);
mutcell_unref(cell); /* drop the creation-local ref; ns holds its own */
*cell_out = cell;
return 0;
}
int upper_mkdir(UpperStore *u, const char *path, int *err) {
pthread_mutex_lock(&u->writer_lock);
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
const UpperEntry *existing;
if (upper_lookup(old, path, &existing) || upper_has_children(old, path)) {
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = VFS_ERR_EXIST;
return -1;
}
if (u->kind == UPPER_DIR) {
int e = 0;
if (pfs_dir_mkdirat(u->root_fd, rel(path), &e) < 0) {
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = e;
return -1;
}
}
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_DIR_MARKER, NULL);
atomic_store_explicit(&u->current, ns, memory_order_release);
pthread_mutex_unlock(&u->writer_lock);
upper_retire(u, old);
return 0;
}
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err) {
pthread_mutex_lock(&u->writer_lock);
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
const UpperEntry *existing;
int have = upper_lookup(old, path, &existing);
if ((!have || existing->kind == ENTRY_WHITEOUT) && !had_lower) {
/* No explicit entry anywhere: still an error to distinguish
* "doesn't exist" from "exists only implicitly, via children,
* and therefore can't be removed" (Section 9.3). */
int kids = upper_has_children(old, path);
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
return -1;
}
if (have && existing->kind == ENTRY_DIR_MARKER && upper_has_children(old, path)) {
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = VFS_ERR_NOTEMPTY;
return -1;
}
if (u->kind == UPPER_DIR && have && existing->kind != ENTRY_WHITEOUT) {
int e = 0;
int is_dir = (existing->kind == ENTRY_DIR_MARKER);
if (pfs_dir_unlinkat(u->root_fd, rel(path), is_dir, &e) < 0) {
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = e;
return -1;
}
}
UpperSnapshot *ns = had_lower ? snapshot_upsert(old, path, ENTRY_WHITEOUT, NULL)
: snapshot_remove(old, path);
atomic_store_explicit(&u->current, ns, memory_order_release);
pthread_mutex_unlock(&u->writer_lock);
upper_retire(u, old);
return 0;
}
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err) {
pthread_mutex_lock(&u->writer_lock);
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
const UpperEntry *src;
if (!upper_lookup(old, from, &src) || src->kind == ENTRY_WHITEOUT) {
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = VFS_ERR_NOENT;
return -1;
}
if (u->kind == UPPER_DIR) {
int e = 0;
if (pfs_dir_renameat(u->root_fd, rel(from), rel(to), &e) < 0) {
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = e;
return -1;
}
}
UpperSnapshot *step1 = snapshot_upsert(old, to, src->kind, src->cell);
UpperSnapshot *step2 = had_lower_from ? snapshot_upsert(step1, from, ENTRY_WHITEOUT, NULL)
: snapshot_remove(step1, from);
atomic_store_explicit(&u->current, step2, memory_order_release);
pthread_mutex_unlock(&u->writer_lock);
upper_retire(u, old);
upper_release(step1); /* never published; drop our local build reference */
return 0;
}
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size,
int64_t mtime, MutCell **cell_out, int *err) {
pthread_mutex_lock(&u->writer_lock);
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
const UpperEntry *existing;
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
pthread_mutex_unlock(&u->writer_lock);
*cell_out = existing->cell;
return 0; /* raced with another copy-up; use what's already there */
}
MutCell *cell = mutcell_new();
atomic_store_explicit(&cell->mtime, mtime, memory_order_relaxed);
if (u->kind == UPPER_MEM) {
if (size) {
cell->data = (unsigned char *)malloc((size_t)size);
memcpy(cell->data, data, (size_t)size);
}
cell->capacity = (pfs_usize)size;
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
} else {
int e = 0;
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
if (fd < 0) {
mutcell_unref(cell);
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = e;
return -1;
}
size_t written = 0;
const unsigned char *p = (const unsigned char *)data;
int failed = 0;
while (written < (size_t)size) {
ssize_t w = write(fd, p + written, (size_t)size - written);
if (w < 0) { failed = 1; break; }
written += (size_t)w;
}
close(fd);
if (failed) {
mutcell_unref(cell);
pthread_mutex_unlock(&u->writer_lock);
if (err) *err = VFS_ERR_IO;
return -1;
}
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
}
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
atomic_store_explicit(&u->current, ns, memory_order_release);
pthread_mutex_unlock(&u->writer_lock);
upper_retire(u, old);
mutcell_unref(cell);
*cell_out = cell;
return 0;
}
/* ---- content operations (Section 5.3 fast path, Section 5.7 growth) ---- */
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n) {
if (u->kind == UPPER_MEM) {
pthread_rwlock_rdlock(&c->buf_lock);
pfs_usize size = atomic_load_explicit(&c->size, memory_order_acquire);
pfs_usize avail = off < size ? size - off : 0;
pfs_usize to_copy = n < avail ? n : avail;
if (to_copy) memcpy(buf, c->data + off, to_copy);
pthread_rwlock_unlock(&c->buf_lock);
return (pfs_isize)to_copy;
}
int e = 0;
int fd = pfs_dir_openat(u->root_fd, rel(path), O_RDONLY, 0, &e);
if (fd < 0) return -1;
ssize_t r = pread(fd, buf, n, (off_t)off);
close(fd);
return (pfs_isize)r;
}
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off,
const void *buf, pfs_usize n, int *err) {
int64_t now = (int64_t)time(NULL);
if (u->kind == UPPER_MEM) {
pthread_mutex_lock(&c->write_lock);
pfs_usize cur_size = atomic_load_explicit(&c->size, memory_order_relaxed);
pfs_usize need = off + n;
if (need > c->capacity) {
pfs_usize newcap = c->capacity ? c->capacity : 64;
while (newcap < need) newcap *= 2;
unsigned char *nb = (unsigned char *)calloc(1, newcap);
if (!nb) { pthread_mutex_unlock(&c->write_lock); if (err) *err = VFS_ERR_NOSPC; return -1; }
if (c->data && cur_size) memcpy(nb, c->data, cur_size);
if (n) memcpy(nb + off, buf, n);
unsigned char *old = c->data;
/* Section 5.7: publish the new buffer via the rwlock, then
* free the old one only after releasing the write lock on
* it — any reader that could still see `old` must have
* taken buf_lock (rdlock) before this wrlock was granted,
* and rdlock/wrlock are mutually exclusive, so it has
* already finished copying by the time we get here. */
pthread_rwlock_wrlock(&c->buf_lock);
c->data = nb;
c->capacity = newcap;
pthread_rwlock_unlock(&c->buf_lock);
free(old);
} else if (n) {
memcpy(c->data + off, buf, n);
}
pfs_usize new_size = need > cur_size ? need : cur_size;
atomic_store_explicit(&c->size, new_size, memory_order_release);
atomic_store_explicit(&c->mtime, now, memory_order_release);
pthread_mutex_unlock(&c->write_lock);
return (pfs_isize)n;
}
pthread_mutex_lock(&c->write_lock);
int e = 0;
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY, 0, &e);
if (fd < 0) { pthread_mutex_unlock(&c->write_lock); if (err) *err = e; return -1; }
ssize_t w = pwrite(fd, buf, n, (off_t)off);
close(fd);
pthread_mutex_unlock(&c->write_lock);
if (w < 0) { if (err) *err = VFS_ERR_IO; return -1; }
return (pfs_isize)w;
}
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path) {
if (u->kind == UPPER_MEM) {
pthread_mutex_lock(&c->write_lock);
atomic_store_explicit(&c->size, 0, memory_order_release);
atomic_store_explicit(&c->mtime, (int64_t)time(NULL), memory_order_release);
pthread_mutex_unlock(&c->write_lock);
} else {
pthread_mutex_lock(&c->write_lock);
int e = 0;
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_TRUNC, 0644, &e);
if (fd >= 0) close(fd);
pthread_mutex_unlock(&c->write_lock);
}
}
/* ---- compaction support ---- */
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb) {
UpperSnapshot *s = upper_acquire(u);
for (pfs_usize i = 0; i < s->count; i++) {
UpperEntry *e = &s->entries[i];
if (e->kind == ENTRY_WHITEOUT) continue;
if (e->kind == ENTRY_DIR_MARKER) {
cb->visit(cb->ctx, e->name, ENTRY_DIR_MARKER, NULL, 0, 0);
continue;
}
if (u->kind == UPPER_MEM) {
pthread_rwlock_rdlock(&e->cell->buf_lock);
pfs_usize size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
cb->visit(cb->ctx, e->name, ENTRY_FILE, e->cell->data, size,
atomic_load_explicit(&e->cell->mtime, memory_order_acquire));
pthread_rwlock_unlock(&e->cell->buf_lock);
} else {
int err = 0;
int fd = pfs_dir_openat(u->root_fd, rel(e->name), O_RDONLY, 0, &err);
if (fd < 0) continue;
struct stat st;
fstat(fd, &st);
void *buf = st.st_size ? malloc((size_t)st.st_size) : NULL;
size_t total = 0;
while (buf && total < (size_t)st.st_size) {
ssize_t r = read(fd, (char *)buf + total, (size_t)st.st_size - total);
if (r <= 0) break;
total += (size_t)r;
}
close(fd);
cb->visit(cb->ctx, e->name, ENTRY_FILE, buf, total, (int64_t)st.st_mtime);
free(buf);
}
}
upper_release(s);
}
void upper_reset_empty(UpperStore *u) {
pthread_mutex_lock(&u->writer_lock);
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
atomic_init(&ns->refcount, 1);
atomic_store_explicit(&u->current, ns, memory_order_release);
pthread_mutex_unlock(&u->writer_lock);
upper_retire(u, old);
}
/* ---- standalone mem/dir Backend wrapper (no lower layer at all) ---- */
typedef struct UpperFile {
UpperStore *store;
MutCell *cell;
char *path;
pfs_usize pos;
} UpperFile;
static int upstd_open(Backend *b, const char *path, int flags, VfsFile **out) {
UpperStore *u = (UpperStore *)b->state;
UpperSnapshot *s = upper_acquire(u);
const UpperEntry *e;
int found = upper_lookup(s, path, &e);
if (found && e->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
MutCell *cell;
if (found) {
cell = e->cell;
mutcell_ref(cell);
upper_release(s);
if (flags & VFS_O_TRUNC) upper_cell_truncate(u, cell, path);
} else {
upper_release(s);
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
int err = 0;
if (upper_create(u, path, 0, &cell, &err) < 0) return err;
mutcell_ref(cell);
}
UpperFile *uf = (UpperFile *)calloc(1, sizeof(UpperFile));
uf->store = u; uf->cell = cell; uf->path = strdup(path); uf->pos = 0;
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
f->backend = b; f->state = uf;
*out = f;
return VFS_OK;
}
static pfs_isize upstd_read(VfsFile *f, void *buf, pfs_usize n) {
UpperFile *uf = (UpperFile *)f->state;
pfs_isize r = upper_cell_read(uf->store, uf->cell, uf->path, uf->pos, buf, n);
if (r > 0) uf->pos += (pfs_usize)r;
return r;
}
static pfs_isize upstd_write(VfsFile *f, const void *buf, pfs_usize n) {
UpperFile *uf = (UpperFile *)f->state;
int err = 0;
pfs_isize w = upper_cell_write(uf->store, uf->cell, uf->path, uf->pos, buf, n, &err);
if (w >= 0) { uf->pos += (pfs_usize)w; return w; }
return err;
}
static int upstd_close(VfsFile *f) {
UpperFile *uf = (UpperFile *)f->state;
mutcell_unref(uf->cell);
free(uf->path);
free(uf);
free(f);
return VFS_OK;
}
static int upstd_stat(Backend *b, const char *path, VfsStat *out) {
UpperStore *u = (UpperStore *)b->state;
UpperSnapshot *s = upper_acquire(u);
const UpperEntry *e;
if (!upper_lookup(s, path, &e) || e->kind == ENTRY_WHITEOUT) {
if (upper_has_children(s, path)) {
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
upper_release(s);
return VFS_OK;
}
upper_release(s);
return VFS_ERR_NOENT;
}
if (e->kind == ENTRY_DIR_MARKER) {
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
upper_release(s);
return VFS_OK;
}
if (u->kind == UPPER_MEM) {
out->size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
out->mtime = atomic_load_explicit(&e->cell->mtime, memory_order_acquire);
out->kind = VFS_KIND_FILE;
upper_release(s);
return VFS_OK;
}
int err = 0;
int rc = pfs_dir_statat(u->root_fd, rel(path), out, &err);
upper_release(s);
return rc < 0 ? err : VFS_OK;
}
static int upstd_readdir(Backend *b, const char *path, VfsDir *out) {
UpperStore *u = (UpperStore *)b->state;
UpperSnapshot *s = upper_acquire(u);
const UpperEntry *self;
int self_found = upper_lookup(s, path, &self);
int has_kids = upper_has_children(s, path);
if (!has_kids && !(self_found && self->kind == ENTRY_DIR_MARKER) && strcmp(path, "/") != 0) {
upper_release(s);
return self_found ? VFS_ERR_NOTDIR : VFS_ERR_NOENT;
}
char prefix[PFS_PATH_MAX];
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
else snprintf(prefix, sizeof(prefix), "%s/", path);
size_t plen = strlen(prefix);
VfsDirEntry *entries = NULL;
pfs_usize count = 0, cap = 0;
for (pfs_usize i = 0; i < s->count; i++) {
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
const char *name = s->entries[i].name;
if (strncmp(name, prefix, plen) != 0) continue;
const char *restp = name + plen;
const char *slash = strchr(restp, '/');
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
if (clen == 0 || clen >= sizeof(entries[0].name)) continue;
if (count > 0 && strncmp(entries[count - 1].name, restp, clen) == 0 &&
entries[count - 1].name[clen] == '\0') continue; /* sorted -> dup is adjacent */
if (count == cap) {
cap = cap ? cap * 2 : 8;
entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry));
}
memcpy(entries[count].name, restp, clen);
entries[count].name[clen] = '\0';
entries[count].kind = slash ? VFS_KIND_DIR
: (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
count++;
}
upper_release(s);
out->entries = entries;
out->count = count;
return VFS_OK;
}
static int upstd_mkdir(Backend *b, const char *path) {
int err = 0;
return upper_mkdir((UpperStore *)b->state, path, &err) < 0 ? err : VFS_OK;
}
static int upstd_unlink(Backend *b, const char *path) {
int err = 0;
return upper_remove((UpperStore *)b->state, path, 0, &err) < 0 ? err : VFS_OK;
}
static int upstd_rename(Backend *b, const char *from, const char *to) {
int err = 0;
return upper_rename((UpperStore *)b->state, from, to, 0, &err) < 0 ? err : VFS_OK;
}
static int upstd_sync(Backend *b) { (void)b; return VFS_ERR_PERM; }
static void upstd_free(Backend *b) {
upper_free((UpperStore *)b->state);
free(b);
}
static const BackendOps UPPER_STANDALONE_OPS = {
upstd_open, upstd_read, upstd_write, upstd_close, upstd_stat,
upstd_readdir, upstd_mkdir, upstd_unlink, upstd_rename, upstd_sync, upstd_free
};
Backend *backend_from_upper(UpperStore *u) {
Backend *b = (Backend *)calloc(1, sizeof(Backend));
b->ops = &UPPER_STANDALONE_OPS;
b->state = u;
return b;
}
Backend *backend_mem_new(void) {
return backend_from_upper(upper_new(UPPER_MEM, -1, NULL));
}
Backend *backend_dir_new(const char *host_path, int *err) {
int e = 0;
int fd = pfs_dir_capability_open(host_path, &e);
if (fd < 0) { if (err) *err = e; return NULL; }
return backend_from_upper(upper_new(UPPER_DIR, fd, host_path));
}
int upper_backend_root_fd(Backend *b) {
if (b->ops != &UPPER_STANDALONE_OPS) return -1;
UpperStore *u = (UpperStore *)b->state;
return u->kind == UPPER_DIR ? u->root_fd : -1;
}
void backend_free(Backend *b) {
if (!b) return;
b->ops->free(b);
}
+275
View File
@@ -0,0 +1,275 @@
/*
* vfs.c — VFS core: the mount table (itself part of the Section 5.3
* snapshot, per that section's mount-table addition), virtual-namespace
* canonicalization dispatch (Section 6.1), and the public API.
*/
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
#include "internal.h"
typedef struct MountEntry {
char *prefix;
size_t prefix_len;
Backend *backend;
} MountEntry;
typedef struct MountSnapshot {
_Atomic(pfs_usize) refcount;
MountEntry *entries;
pfs_usize count;
} MountSnapshot;
struct Vfs {
_Atomic(MountSnapshot *) current;
pthread_mutex_t writer_lock;
/* See UpperStore.reclaim_gate (internal.h) for why this exists: it
* closes the gap between loading the current MountSnapshot pointer
* and incrementing its refcount, which a naive load-then-increment
* leaves open to a concurrent writer freeing the very object being
* acquired. */
pthread_rwlock_t reclaim_gate;
};
static MountSnapshot *mount_acquire(Vfs *v) {
pthread_rwlock_rdlock(&v->reclaim_gate);
MountSnapshot *s = atomic_load_explicit(&v->current, memory_order_acquire);
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
pthread_rwlock_unlock(&v->reclaim_gate);
return s;
}
static void mount_release(MountSnapshot *s) {
if (!s) return;
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
for (pfs_usize i = 0; i < s->count; i++) free(s->entries[i].prefix);
free(s->entries);
free(s);
}
}
static void mount_retire(Vfs *v, MountSnapshot *old) {
pthread_rwlock_wrlock(&v->reclaim_gate);
mount_release(old);
pthread_rwlock_unlock(&v->reclaim_gate);
}
Vfs *vfs_new(void) {
Vfs *v = (Vfs *)calloc(1, sizeof(Vfs));
pthread_mutex_init(&v->writer_lock, NULL);
pthread_rwlock_init(&v->reclaim_gate, NULL);
MountSnapshot *s = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
atomic_init(&s->refcount, 1);
atomic_init(&v->current, s);
return v;
}
void vfs_free(Vfs *v) {
if (!v) return;
mount_release(atomic_load_explicit(&v->current, memory_order_acquire));
pthread_mutex_destroy(&v->writer_lock);
pthread_rwlock_destroy(&v->reclaim_gate);
free(v);
}
int vfs_mount(Vfs *v, const char *prefix, Backend *b) {
char norm[PFS_PATH_MAX];
if (pfs_path_normalize(prefix, norm) < 0) return VFS_ERR_INVAL;
pthread_mutex_lock(&v->writer_lock);
MountSnapshot *old = atomic_load_explicit(&v->current, memory_order_acquire);
for (pfs_usize i = 0; i < old->count; i++) {
if (strcmp(old->entries[i].prefix, norm) == 0) {
pthread_mutex_unlock(&v->writer_lock);
return VFS_ERR_EXIST;
}
}
MountSnapshot *ns = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
ns->count = old->count + 1;
ns->entries = (MountEntry *)calloc(ns->count, sizeof(MountEntry));
atomic_init(&ns->refcount, 1);
for (pfs_usize i = 0; i < old->count; i++) {
ns->entries[i].prefix = strdup(old->entries[i].prefix);
ns->entries[i].prefix_len = old->entries[i].prefix_len;
ns->entries[i].backend = old->entries[i].backend;
}
ns->entries[old->count].prefix = strdup(norm);
ns->entries[old->count].prefix_len = strlen(norm);
ns->entries[old->count].backend = b;
atomic_store_explicit(&v->current, ns, memory_order_release);
pthread_mutex_unlock(&v->writer_lock);
mount_retire(v, old);
return VFS_OK;
}
int vfs_unmount(Vfs *v, const char *prefix) {
char norm[PFS_PATH_MAX];
if (pfs_path_normalize(prefix, norm) < 0) return VFS_ERR_INVAL;
pthread_mutex_lock(&v->writer_lock);
MountSnapshot *old = atomic_load_explicit(&v->current, memory_order_acquire);
pfs_usize idx = old->count;
for (pfs_usize i = 0; i < old->count; i++) {
if (strcmp(old->entries[i].prefix, norm) == 0) { idx = i; break; }
}
if (idx == old->count) { pthread_mutex_unlock(&v->writer_lock); return VFS_ERR_NOMOUNT; }
MountSnapshot *ns = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
ns->count = old->count - 1;
ns->entries = ns->count ? (MountEntry *)calloc(ns->count, sizeof(MountEntry)) : NULL;
atomic_init(&ns->refcount, 1);
pfs_usize w = 0;
for (pfs_usize i = 0; i < old->count; i++) {
if (i == idx) continue;
ns->entries[w].prefix = strdup(old->entries[i].prefix);
ns->entries[w].prefix_len = old->entries[i].prefix_len;
ns->entries[w].backend = old->entries[i].backend;
w++;
}
atomic_store_explicit(&v->current, ns, memory_order_release);
pthread_mutex_unlock(&v->writer_lock);
mount_retire(v, old);
return VFS_OK;
}
/* Longest-prefix match against the mount table (Section 2.2), reached
* through the same atomic snapshot that protects the upper index
* (Section 5.3): a concurrent vfs_mount/vfs_unmount is never observed
* mid-change. */
static Backend *resolve(Vfs *v, const char *path, char relpath[PFS_PATH_MAX],
MountSnapshot **snap_out, int *err) {
char norm[PFS_PATH_MAX];
if (pfs_path_normalize(path, norm) < 0) { if (err) *err = VFS_ERR_INVAL; return NULL; }
MountSnapshot *snap = mount_acquire(v);
Backend *best = NULL;
size_t best_len = 0;
for (pfs_usize i = 0; i < snap->count; i++) {
if (pfs_path_under(snap->entries[i].prefix, norm) && snap->entries[i].prefix_len >= best_len) {
best = snap->entries[i].backend;
best_len = snap->entries[i].prefix_len;
}
}
if (!best) { mount_release(snap); if (err) *err = VFS_ERR_NOMOUNT; return NULL; }
if (best_len <= 1) {
strncpy(relpath, norm, PFS_PATH_MAX - 1);
relpath[PFS_PATH_MAX - 1] = '\0';
} else {
const char *r = norm + best_len;
strncpy(relpath, (*r == '\0') ? "/" : r, PFS_PATH_MAX - 1);
relpath[PFS_PATH_MAX - 1] = '\0';
}
*snap_out = snap;
return best;
}
VfsFile *vfs_open(Vfs *v, const char *path, int flags, int *err) {
char rel[PFS_PATH_MAX];
MountSnapshot *snap;
int e = VFS_OK;
Backend *b = resolve(v, path, rel, &snap, &e);
if (!b) { if (err) *err = e; return NULL; }
VfsFile *f = NULL;
int rc = b->ops->open(b, rel, flags, &f);
mount_release(snap);
if (rc != VFS_OK) { if (err) *err = rc; return NULL; }
if (err) *err = VFS_OK;
return f;
}
pfs_isize vfs_read(VfsFile *f, void *buf, pfs_usize n) { return f->backend->ops->read(f, buf, n); }
pfs_isize vfs_write(VfsFile *f, const void *buf, pfs_usize n) { return f->backend->ops->write(f, buf, n); }
int vfs_close(VfsFile *f) { return f->backend->ops->close(f); }
int vfs_stat(Vfs *v, const char *path, VfsStat *out) {
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
Backend *b = resolve(v, path, rel, &snap, &err);
if (!b) return err;
int rc = b->ops->stat(b, rel, out);
mount_release(snap);
return rc;
}
int vfs_readdir(Vfs *v, const char *path, VfsDir *out) {
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
Backend *b = resolve(v, path, rel, &snap, &err);
if (!b) return err;
out->entries = NULL; out->count = 0;
int rc = b->ops->readdir(b, rel, out);
mount_release(snap);
return rc;
}
void vfs_dir_free(VfsDir *d) {
if (!d) return;
free(d->entries);
d->entries = NULL;
d->count = 0;
}
int vfs_mkdir(Vfs *v, const char *path) {
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
Backend *b = resolve(v, path, rel, &snap, &err);
if (!b) return err;
int rc = b->ops->mkdir(b, rel);
mount_release(snap);
return rc;
}
int vfs_unlink(Vfs *v, const char *path) {
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
Backend *b = resolve(v, path, rel, &snap, &err);
if (!b) return err;
int rc = b->ops->unlink(b, rel);
mount_release(snap);
return rc;
}
int vfs_rename(Vfs *v, const char *from, const char *to) {
char relf[PFS_PATH_MAX], relt[PFS_PATH_MAX];
MountSnapshot *sf, *st_; int err = VFS_OK;
Backend *bf = resolve(v, from, relf, &sf, &err);
if (!bf) return err;
Backend *bt = resolve(v, to, relt, &st_, &err);
if (!bt) { mount_release(sf); return err; }
if (bf != bt) { mount_release(sf); mount_release(st_); return VFS_ERR_INVAL; /* cross-mount rename: v0 exclusion */ }
int rc = bf->ops->rename(bf, relf, relt);
mount_release(sf);
mount_release(st_);
return rc;
}
int vfs_sync(Vfs *v, const char *overlay_prefix) {
char norm[PFS_PATH_MAX];
if (pfs_path_normalize(overlay_prefix, norm) < 0) return VFS_ERR_INVAL;
MountSnapshot *snap = mount_acquire(v);
Backend *b = NULL;
for (pfs_usize i = 0; i < snap->count; i++) {
if (strcmp(snap->entries[i].prefix, norm) == 0) { b = snap->entries[i].backend; break; }
}
if (!b) { mount_release(snap); return VFS_ERR_NOMOUNT; }
int rc = b->ops->sync(b);
mount_release(snap);
return rc;
}
int vfs_harden_process_with_landlock(Vfs *v) {
MountSnapshot *snap = mount_acquire(v);
int *fds = snap->count ? (int *)malloc(snap->count * sizeof(int)) : NULL;
size_t n = 0;
for (pfs_usize i = 0; i < snap->count; i++) {
Backend *b = snap->entries[i].backend;
Backend *inner = overlay_upper_backend(b);
if (inner) b = inner;
int fd = upper_backend_root_fd(b);
if (fd >= 0) fds[n++] = fd;
}
int rc = (n > 0) ? pfs_landlock_restrict_to(fds, n) : -1;
free(fds);
mount_release(snap);
return rc == 0 ? VFS_OK : VFS_ERR_PERM;
}