Add PackFS v0: statically linked in-process VFS implementing concept.md
Implements the core design: a mount table published as an atomically- swapped snapshot; mem/dir/pack/overlay backends; copy-on-write overlay with copy-up and whiteout deletion; a checksummed append journal; compaction with exact-duplicate elimination; single-writer/wait-free- reader concurrency with a structural/content write split; openat2/ Landlock path containment for dir mounts; and load-time pack integrity validation. Zero required third-party dependencies. Sanitizer testing (ASan/UBSan) caught and led to fixing a genuine heap-use-after-free in the snapshot-reclamation path: the textbook "load pointer, then increment its refcount" pattern left a gap a concurrent writer could free through. Closed with a small reclaim_gate rwlock, documented in internal.h and CLAUDE.md since it's a pattern every refcounted structure in the codebase now follows. zip/tar import/export backends, recommended in concept.md Section 11, will not be built — a permanent project decision recorded in CLAUDE.md since concept.md itself is frozen and cannot be edited to reflect it. Includes a runnable demo (examples/demo.c, `make demo`) exercising the library end to end and proving cross-run persistence through the pack file, plus open-source scaffolding: MIT license, README, CONTRIBUTING, and a CI workflow running the test suite under ASan/UBSan/TSan. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UqJpkdJ6Njnt1pw3CbghzB
This commit is contained in:
@@ -0,0 +1,253 @@
|
||||
/*
|
||||
* containment.c — capability-scoped `dir` mounts (Section 6.2) and
|
||||
* optional Landlock hardening (Section 6.2, 6.3).
|
||||
*
|
||||
* On Linux 5.6+, every lookup beneath a `dir` mount's root fd is resolved
|
||||
* with openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS) in a single kernel
|
||||
* call, atomically rejecting ".." escapes and symlink escapes. Where
|
||||
* openat2 is unavailable (ENOSYS on an older kernel), this falls back to
|
||||
* per-component O_NOFOLLOW resolution, which is an accepted, weaker
|
||||
* residual-risk posture per Section 6.3 — not a claim of equal
|
||||
* containment strength.
|
||||
*/
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <linux/openat2.h>
|
||||
#include <string.h>
|
||||
#include <sys/prctl.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
#ifdef __linux__
|
||||
#include <linux/landlock.h>
|
||||
#endif
|
||||
|
||||
int pfs_dir_capability_open(const char *path, int *err) {
|
||||
int fd = open(path, O_DIRECTORY | O_CLOEXEC | O_RDONLY);
|
||||
if (fd < 0) {
|
||||
if (err) *err = VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return fd;
|
||||
}
|
||||
|
||||
/* Splits an already-normalized relative path (no leading slash) into
|
||||
* (parent, leaf). Root-level names have an empty parent. */
|
||||
static void split_parent_leaf(const char *rel, char *parent_buf, size_t cap, const char **leaf) {
|
||||
const char *slash = strrchr(rel, '/');
|
||||
if (!slash) {
|
||||
parent_buf[0] = '\0';
|
||||
*leaf = rel;
|
||||
return;
|
||||
}
|
||||
size_t plen = (size_t)(slash - rel);
|
||||
if (plen >= cap) plen = cap - 1;
|
||||
memcpy(parent_buf, rel, plen);
|
||||
parent_buf[plen] = '\0';
|
||||
*leaf = slash + 1;
|
||||
}
|
||||
|
||||
#ifdef SYS_openat2
|
||||
static int openat2_beneath(int root_fd, const char *rel, uint64_t flags, uint64_t mode) {
|
||||
struct open_how how;
|
||||
memset(&how, 0, sizeof(how));
|
||||
how.flags = flags;
|
||||
how.mode = mode;
|
||||
how.resolve = RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS;
|
||||
return (int)syscall(SYS_openat2, root_fd, rel[0] ? rel : ".", &how, sizeof(how));
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Per-component O_NOFOLLOW fallback (Section 6.3). Walks every component
|
||||
* except the last with O_NOFOLLOW|O_DIRECTORY, then opens the last with
|
||||
* the caller's requested flags plus O_NOFOLLOW. Closes most, not all, of
|
||||
* the same TOCTOU window openat2 closes atomically. */
|
||||
static int fallback_openat_beneath(int root_fd, const char *rel, int flags, unsigned mode) {
|
||||
if (rel[0] == '\0') {
|
||||
return dup(root_fd);
|
||||
}
|
||||
char buf[PFS_PATH_MAX];
|
||||
if (strlen(rel) >= sizeof(buf)) { errno = ENAMETOOLONG; return -1; }
|
||||
strcpy(buf, rel);
|
||||
|
||||
int cur = root_fd;
|
||||
int owns_cur = 0;
|
||||
char *save = NULL;
|
||||
char *tok = strtok_r(buf, "/", &save);
|
||||
char *next = tok ? strtok_r(NULL, "/", &save) : NULL;
|
||||
|
||||
while (tok && next) {
|
||||
int nfd = openat(cur, tok, O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC);
|
||||
if (owns_cur) close(cur);
|
||||
if (nfd < 0) return -1;
|
||||
cur = nfd;
|
||||
owns_cur = 1;
|
||||
tok = next;
|
||||
next = strtok_r(NULL, "/", &save);
|
||||
}
|
||||
|
||||
int final_fd = -1;
|
||||
if (tok) {
|
||||
int final_flags = flags;
|
||||
if (!(flags & O_CREAT)) final_flags |= O_NOFOLLOW;
|
||||
final_fd = openat(cur, tok, final_flags, mode);
|
||||
}
|
||||
if (owns_cur) close(cur);
|
||||
return final_fd;
|
||||
}
|
||||
|
||||
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err) {
|
||||
int fd;
|
||||
#ifdef SYS_openat2
|
||||
fd = openat2_beneath(root_fd, rel, (uint64_t)flags, (uint64_t)mode);
|
||||
if (fd < 0 && errno == ENOSYS) {
|
||||
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
|
||||
}
|
||||
#else
|
||||
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
|
||||
#endif
|
||||
if (fd < 0) {
|
||||
if (err) *err = (errno == EXDEV || errno == ELOOP) ? VFS_ERR_INVAL
|
||||
: (errno == ENOENT) ? VFS_ERR_NOENT
|
||||
: (errno == EEXIST) ? VFS_ERR_EXIST
|
||||
: VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return fd;
|
||||
}
|
||||
|
||||
static int resolve_parent_dir(int root_fd, const char *rel, char *leaf_out, size_t leaf_cap, int *err) {
|
||||
char parent[PFS_PATH_MAX];
|
||||
const char *leaf;
|
||||
split_parent_leaf(rel, parent, sizeof(parent), &leaf);
|
||||
if (strlen(leaf) >= leaf_cap) { if (err) *err = VFS_ERR_INVAL; return -1; }
|
||||
strcpy(leaf_out, leaf);
|
||||
|
||||
int pfd;
|
||||
#ifdef SYS_openat2
|
||||
pfd = openat2_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
||||
if (pfd < 0 && errno == ENOSYS) {
|
||||
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
||||
}
|
||||
#else
|
||||
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
||||
#endif
|
||||
if (pfd < 0) {
|
||||
if (err) *err = (errno == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return pfd;
|
||||
}
|
||||
|
||||
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err) {
|
||||
char leaf[PFS_PATH_MAX];
|
||||
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
|
||||
if (pfd < 0) return -1;
|
||||
int rc = mkdirat(pfd, leaf, 0777);
|
||||
int saved = errno;
|
||||
close(pfd);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == EEXIST) ? VFS_ERR_EXIST : (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err) {
|
||||
char leaf[PFS_PATH_MAX];
|
||||
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
|
||||
if (pfd < 0) return -1;
|
||||
int rc = unlinkat(pfd, leaf, is_dir ? AT_REMOVEDIR : 0);
|
||||
int saved = errno;
|
||||
close(pfd);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : (saved == ENOTEMPTY) ? VFS_ERR_NOTEMPTY : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err) {
|
||||
char leaf_from[PFS_PATH_MAX], leaf_to[PFS_PATH_MAX];
|
||||
int pfd_from = resolve_parent_dir(root_fd, from, leaf_from, sizeof(leaf_from), err);
|
||||
if (pfd_from < 0) return -1;
|
||||
int pfd_to = resolve_parent_dir(root_fd, to, leaf_to, sizeof(leaf_to), err);
|
||||
if (pfd_to < 0) { close(pfd_from); return -1; }
|
||||
int rc = renameat(pfd_from, leaf_from, pfd_to, leaf_to);
|
||||
int saved = errno;
|
||||
close(pfd_from);
|
||||
close(pfd_to);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err) {
|
||||
int fd = pfs_dir_openat(root_fd, rel, O_RDONLY, 0, err);
|
||||
if (fd < 0) return -1;
|
||||
struct stat st;
|
||||
int rc = fstat(fd, &st);
|
||||
int saved = errno;
|
||||
close(fd);
|
||||
if (rc < 0) {
|
||||
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
out->size = (pfs_usize)st.st_size;
|
||||
out->mtime = (int64_t)st.st_mtime;
|
||||
out->kind = S_ISDIR(st.st_mode) ? VFS_KIND_DIR : VFS_KIND_FILE;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* Best-effort, opt-in, process-wide hardening (Section 6.2). Restricts
|
||||
* filesystem read/write/create/remove operations to the given
|
||||
* directory-capability fds; deliberately does not restrict EXECUTE, to
|
||||
* avoid an embeddable library silently blocking unrelated process
|
||||
* behavior beyond the filesystem paths it was asked to confine.
|
||||
*/
|
||||
int pfs_landlock_restrict_to(const int *roots, size_t count) {
|
||||
#if defined(__linux__) && defined(SYS_landlock_create_ruleset)
|
||||
long abi = syscall(SYS_landlock_create_ruleset, NULL, 0, LANDLOCK_CREATE_RULESET_VERSION);
|
||||
if (abi < 0) return -1; /* unsupported kernel: Section 6.3 residual risk */
|
||||
|
||||
uint64_t access = LANDLOCK_ACCESS_FS_READ_FILE | LANDLOCK_ACCESS_FS_WRITE_FILE |
|
||||
LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_REMOVE_DIR |
|
||||
LANDLOCK_ACCESS_FS_REMOVE_FILE | LANDLOCK_ACCESS_FS_MAKE_CHAR |
|
||||
LANDLOCK_ACCESS_FS_MAKE_DIR | LANDLOCK_ACCESS_FS_MAKE_REG |
|
||||
LANDLOCK_ACCESS_FS_MAKE_SOCK | LANDLOCK_ACCESS_FS_MAKE_FIFO |
|
||||
LANDLOCK_ACCESS_FS_MAKE_BLOCK | LANDLOCK_ACCESS_FS_MAKE_SYM;
|
||||
|
||||
struct landlock_ruleset_attr attr;
|
||||
memset(&attr, 0, sizeof(attr));
|
||||
attr.handled_access_fs = access;
|
||||
|
||||
int rs_fd = (int)syscall(SYS_landlock_create_ruleset, &attr, sizeof(attr), 0);
|
||||
if (rs_fd < 0) return -1;
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
struct landlock_path_beneath_attr pb;
|
||||
memset(&pb, 0, sizeof(pb));
|
||||
pb.allowed_access = access;
|
||||
pb.parent_fd = roots[i];
|
||||
if (syscall(SYS_landlock_add_rule, rs_fd, LANDLOCK_RULE_PATH_BENEATH, &pb, 0) < 0) {
|
||||
close(rs_fd);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) < 0) { close(rs_fd); return -1; }
|
||||
if (syscall(SYS_landlock_restrict_self, rs_fd, 0) < 0) { close(rs_fd); return -1; }
|
||||
close(rs_fd);
|
||||
return 0;
|
||||
#else
|
||||
(void)roots; (void)count;
|
||||
return -1;
|
||||
#endif
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
/*
|
||||
* hash.c — FNV-1a 64-bit, used for pack integrity checksums (Section 7)
|
||||
* and content-addressed dedup during compaction (Section 9.2). Chosen
|
||||
* because it needs no third-party library, consistent with the
|
||||
* static-linking constraint (Section 11.1) — this is not a
|
||||
* cryptographic hash and must never be used where adversarial collision
|
||||
* resistance is required.
|
||||
*/
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
uint64_t pfs_fnv1a64(const void *data, size_t len) {
|
||||
const unsigned char *p = (const unsigned char *)data;
|
||||
uint64_t h = 0xcbf29ce484222325ULL;
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
h ^= p[i];
|
||||
h *= 0x100000001b3ULL;
|
||||
}
|
||||
return h;
|
||||
}
|
||||
+243
@@ -0,0 +1,243 @@
|
||||
/*
|
||||
* internal.h — shared internal declarations, not part of the public API.
|
||||
*
|
||||
* Every mechanism named here corresponds to a section of concept.md;
|
||||
* comments reference sections rather than re-deriving the rationale,
|
||||
* since concept.md is frozen and is the authoritative source for "why."
|
||||
*/
|
||||
|
||||
#ifndef PACKFS_INTERNAL_H
|
||||
#define PACKFS_INTERNAL_H
|
||||
|
||||
#ifndef _GNU_SOURCE
|
||||
#define _GNU_SOURCE
|
||||
#endif
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stddef.h>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
|
||||
#include "packfs.h"
|
||||
|
||||
/* ---- path.c: virtual-namespace canonicalization (Section 6.1) ---- */
|
||||
|
||||
/*
|
||||
* Resolves "." and ".." purely lexically and rejects escapes above "/".
|
||||
* `out` must be at least PFS_PATH_MAX bytes. Returns 0 on success, -1 if
|
||||
* the path would resolve above the root (VFS_ERR_INVAL at the caller).
|
||||
*/
|
||||
#define PFS_PATH_MAX 4096
|
||||
int pfs_path_normalize(const char *in, char out[PFS_PATH_MAX]);
|
||||
|
||||
/* Longest-prefix match helper: does `path` lie under mount `prefix`? */
|
||||
int pfs_path_under(const char *prefix, const char *path);
|
||||
|
||||
/* ---- hash.c ---- */
|
||||
|
||||
uint64_t pfs_fnv1a64(const void *data, size_t len);
|
||||
|
||||
/* ---- containment.c: capability-scoped dir mounts (Section 6.2) ---- */
|
||||
|
||||
/*
|
||||
* Opens `path` as a directory capability (O_DIRECTORY), the root of a
|
||||
* `dir` mount. Returns the fd, or -1 on failure with *err set.
|
||||
*/
|
||||
int pfs_dir_capability_open(const char *path, int *err);
|
||||
|
||||
/*
|
||||
* Resolves `rel` (already virtual-namespace-canonicalized, no leading
|
||||
* slash) beneath directory-capability `root_fd`, using
|
||||
* openat2(RESOLVE_BENEATH|RESOLVE_NO_SYMLINKS) on Linux 5.6+ and falling
|
||||
* back to O_NOFOLLOW-per-component + prefix verification where openat2
|
||||
* is unavailable (Section 6.3). `flags`/`mode` are plain open(2) flags.
|
||||
*/
|
||||
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err);
|
||||
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err);
|
||||
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err);
|
||||
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err);
|
||||
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err);
|
||||
|
||||
/*
|
||||
* Best-effort, opt-in, process-wide Landlock hardening (Section 6.2),
|
||||
* exposed publicly as vfs_harden_process_with_landlock. `roots` is an
|
||||
* array of directory-capability fds to allow; `count` its length.
|
||||
* Returns 0 on success, -1 if unsupported by the running kernel (in
|
||||
* which case nothing was changed — the residual-risk fallback of
|
||||
* Section 6.3, not a fatal error).
|
||||
*/
|
||||
int pfs_landlock_restrict_to(const int *roots, size_t count);
|
||||
|
||||
/* ---- pack.c: on-disk format (Section 9) + integrity (Section 7) ---- */
|
||||
|
||||
#define PFS_MODE_DIR 0x1u /* directory-marker bit, Section 9.3 */
|
||||
|
||||
typedef struct PackIndexEntry {
|
||||
uint64_t name_off;
|
||||
uint32_t name_len;
|
||||
uint64_t data_off;
|
||||
uint64_t size;
|
||||
uint32_t mode;
|
||||
int64_t mtime;
|
||||
} PackIndexEntry;
|
||||
|
||||
typedef struct Pack {
|
||||
int fd;
|
||||
void *map;
|
||||
size_t map_len;
|
||||
const PackIndexEntry *entries; /* sorted by name, Section 9.1 */
|
||||
size_t entry_count;
|
||||
const char *strings_base;
|
||||
size_t strings_len;
|
||||
char *path; /* owned copy, used as compaction target */
|
||||
} Pack;
|
||||
|
||||
int pack_load(const char *path, Pack **out, int *err); /* validates per Section 7 */
|
||||
void pack_close(Pack *p);
|
||||
|
||||
/* Binary search by full path name. Returns 1 and sets *out on hit. */
|
||||
int pack_find(const Pack *p, const char *name, const PackIndexEntry **out);
|
||||
|
||||
/*
|
||||
* Range query for vfs_readdir (Section 9.1): [*lo, *hi) indices whose
|
||||
* name starts with `prefix`.
|
||||
*/
|
||||
void pack_range(const Pack *p, const char *prefix, size_t *lo, size_t *hi);
|
||||
|
||||
const char *pack_entry_name(const Pack *p, const PackIndexEntry *e);
|
||||
const void *pack_entry_data(const Pack *p, const PackIndexEntry *e);
|
||||
|
||||
typedef struct PackBuildEntry {
|
||||
const char *name;
|
||||
const void *data;
|
||||
uint64_t size;
|
||||
uint32_t mode;
|
||||
int64_t mtime;
|
||||
} PackBuildEntry;
|
||||
|
||||
/*
|
||||
* Writes a new pack to `path` + ".tmp", fsyncs, renames over `path`
|
||||
* (Section 4.3). Entries are sorted by name and content-addressed
|
||||
* deduplicated by exact (hash, size) match (Section 9.2). Computes and
|
||||
* stores an integrity checksum consumed by pack_load (Section 7).
|
||||
*/
|
||||
int pack_write(const char *path, const PackBuildEntry *entries, size_t count, int *err);
|
||||
|
||||
/* ---- upper.c: mem/dir writable layer + snapshot machinery (Section 5) ---- */
|
||||
|
||||
typedef enum { UPPER_MEM, UPPER_DIR } UpperKind;
|
||||
typedef enum { ENTRY_FILE, ENTRY_DIR_MARKER, ENTRY_WHITEOUT } UpperEntryKind;
|
||||
|
||||
/*
|
||||
* Section 5.3's mutable metadata cell, and Section 5.7's buffer-growth
|
||||
* guard. One MutCell is shared by every UpperSnapshot whose entry array
|
||||
* still points at the same logical file — an ordinary vfs_write mutates
|
||||
* this cell in place and never touches the snapshot pointer at all.
|
||||
*/
|
||||
typedef struct MutCell {
|
||||
pthread_mutex_t write_lock; /* serializes writers to the same file */
|
||||
pthread_rwlock_t buf_lock; /* guards {data,capacity} against grow, mem only */
|
||||
_Atomic(pfs_usize) size;
|
||||
_Atomic(int64_t) mtime;
|
||||
/* UPPER_MEM: */
|
||||
unsigned char *data; /* guarded by buf_lock on the grow path */
|
||||
pfs_usize capacity;
|
||||
/* UPPER_DIR: host bytes live in the host fs; nothing to store here. */
|
||||
} MutCell;
|
||||
|
||||
typedef struct UpperEntry {
|
||||
char *name; /* full virtual path, e.g. "/a/b.txt" */
|
||||
UpperEntryKind kind;
|
||||
MutCell *cell; /* NULL for ENTRY_DIR_MARKER / ENTRY_WHITEOUT */
|
||||
} UpperEntry;
|
||||
|
||||
typedef struct UpperSnapshot {
|
||||
_Atomic(pfs_usize) refcount;
|
||||
UpperEntry *entries; /* sorted by name, Section 9.1 */
|
||||
pfs_usize count;
|
||||
} UpperSnapshot;
|
||||
|
||||
typedef struct UpperStore {
|
||||
UpperKind kind;
|
||||
_Atomic(UpperSnapshot *) current;
|
||||
pthread_mutex_t writer_lock; /* single writer, Section 5.3 */
|
||||
/*
|
||||
* Guards the load-then-increment in upper_acquire() against a
|
||||
* concurrent writer retiring (and freeing) the very snapshot being
|
||||
* acquired. Plain "load pointer, then atomically increment its
|
||||
* refcount" has a gap between those two steps during which the
|
||||
* object can be freed out from under the loader; closing that gap
|
||||
* is what makes reference counting itself memory-safe, not a
|
||||
* departure from the reference-counting design. Readers take the
|
||||
* read side (concurrent, effectively free); a writer takes the
|
||||
* write side only around retiring the specific snapshot it just
|
||||
* superseded, which readers-in-flight already past this gate are
|
||||
* unaffected by.
|
||||
*/
|
||||
pthread_rwlock_t reclaim_gate;
|
||||
int root_fd; /* UPPER_DIR only: containment root, Section 6.2 */
|
||||
char *root_path; /* UPPER_DIR only, for error messages */
|
||||
} UpperStore;
|
||||
|
||||
UpperStore *upper_new(UpperKind kind, int root_fd /* -1 for mem */, const char *root_path);
|
||||
void upper_free(UpperStore *u);
|
||||
|
||||
UpperSnapshot *upper_acquire(UpperStore *u);
|
||||
void upper_release(UpperSnapshot *s);
|
||||
|
||||
/* Structural + content operations, all against paths already normalized
|
||||
* and relative to the mount (leading "/"). */
|
||||
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out);
|
||||
int upper_has_children(UpperSnapshot *s, const char *path);
|
||||
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path);
|
||||
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err);
|
||||
int upper_mkdir(UpperStore *u, const char *path, int *err);
|
||||
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err); /* whiteout if had_lower */
|
||||
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err);
|
||||
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size, int64_t mtime, MutCell **cell_out, int *err);
|
||||
|
||||
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n);
|
||||
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off, const void *buf, pfs_usize n, int *err);
|
||||
|
||||
/* Snapshot the whole store into pack-builder entries (compaction, dir-marker-aware). */
|
||||
typedef struct UpperWalkCb {
|
||||
void (*visit)(void *ctx, const char *name, UpperEntryKind kind, const void *data, pfs_usize size, int64_t mtime);
|
||||
void *ctx;
|
||||
} UpperWalkCb;
|
||||
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb); /* skips whiteouts */
|
||||
void upper_reset_empty(UpperStore *u); /* used after compaction folds upper into the new pack */
|
||||
|
||||
Backend *backend_from_upper(UpperStore *u); /* wraps an UpperStore as a standalone Backend */
|
||||
int upper_backend_root_fd(Backend *b); /* -1 unless b is a standalone `dir` backend */
|
||||
|
||||
/* ---- backend vtable shared by vfs.c ---- */
|
||||
|
||||
typedef struct BackendOps {
|
||||
int (*open)(Backend *b, const char *path, int flags, VfsFile **out);
|
||||
pfs_isize (*read)(VfsFile *f, void *buf, pfs_usize n);
|
||||
pfs_isize (*write)(VfsFile *f, const void *buf, pfs_usize n);
|
||||
int (*close)(VfsFile *f);
|
||||
int (*stat)(Backend *b, const char *path, VfsStat *out);
|
||||
int (*readdir)(Backend *b, const char *path, VfsDir *out);
|
||||
int (*mkdir)(Backend *b, const char *path);
|
||||
int (*unlink)(Backend *b, const char *path);
|
||||
int (*rename)(Backend *b, const char *from, const char *to);
|
||||
int (*sync)(Backend *b); /* VFS_ERR_PERM if not an overlay */
|
||||
void (*free)(Backend *b);
|
||||
} BackendOps;
|
||||
|
||||
struct Backend {
|
||||
const BackendOps *ops;
|
||||
void *state;
|
||||
};
|
||||
|
||||
struct VfsFile {
|
||||
Backend *backend;
|
||||
void *state;
|
||||
};
|
||||
|
||||
/* ---- overlay.c ---- */
|
||||
|
||||
Backend *overlay_upper_backend(Backend *b); /* NULL unless b is an overlay backend */
|
||||
|
||||
#endif /* PACKFS_INTERNAL_H */
|
||||
+638
@@ -0,0 +1,638 @@
|
||||
/*
|
||||
* overlay.c — the copy-on-write overlay backend (Sections 4-5, 7, 9.2).
|
||||
*
|
||||
* Composes a read-only Pack (lower) with an UpperStore (upper, from
|
||||
* upper.c) obtained from `mem` or `dir`. Implements the merge (upper
|
||||
* wins, whiteouts hide lower), copy-up, the append journal (Section 4.4)
|
||||
* with self-checking records (Section 4.3), and vfs_sync's compaction
|
||||
* (Section 4.1 step 5, Section 5.3, Section 9.2's exact-duplicate
|
||||
* elimination via pack_write).
|
||||
*
|
||||
* Journal records store whole-value content (Section 4.4 names
|
||||
* "key/value-ish" workloads as the target — saves, agent memory,
|
||||
* config); this implementation journals a full replacement value per
|
||||
* write rather than a byte-range delta, which is the simpler and more
|
||||
* robust choice for that workload character, at the cost of journal
|
||||
* size for large files under many small writes. That trade-off is
|
||||
* deliberate, not an oversight.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
typedef struct Overlay {
|
||||
Pack *pack; /* may be NULL: empty lower layer */
|
||||
char *pack_path; /* compaction target + journal anchor; may be NULL */
|
||||
Backend *upper; /* mem or dir backend, not owned */
|
||||
pthread_mutex_t journal_lock;
|
||||
FILE *journal_fp;
|
||||
} Overlay;
|
||||
|
||||
static UpperStore *upper_of(Overlay *ov) { return (UpperStore *)ov->upper->state; }
|
||||
static const char *dirrel(const char *p) { return p[0] == '/' ? p + 1 : p; }
|
||||
|
||||
/* ---- journal (Section 4.4) ---- */
|
||||
|
||||
static void journal_append_record(Overlay *ov, uint8_t op, const char *name,
|
||||
int64_t mtime, uint64_t size, const void *data) {
|
||||
if (!ov->pack_path) return;
|
||||
pthread_mutex_lock(&ov->journal_lock);
|
||||
if (!ov->journal_fp) {
|
||||
char jpath[PFS_PATH_MAX];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
|
||||
ov->journal_fp = fopen(jpath, "ab");
|
||||
}
|
||||
if (ov->journal_fp) {
|
||||
uint32_t name_len = (uint32_t)strlen(name);
|
||||
uint32_t payload_len = 4 + name_len + (op == 0 ? (uint32_t)(8 + 8 + size) : 0);
|
||||
unsigned char *payload = (unsigned char *)malloc(payload_len);
|
||||
size_t off = 0;
|
||||
memcpy(payload + off, &name_len, 4); off += 4;
|
||||
memcpy(payload + off, name, name_len); off += name_len;
|
||||
if (op == 0) {
|
||||
memcpy(payload + off, &mtime, 8); off += 8;
|
||||
memcpy(payload + off, &size, 8); off += 8;
|
||||
if (size) memcpy(payload + off, data, size);
|
||||
}
|
||||
uint32_t checksum = (uint32_t)pfs_fnv1a64(payload, payload_len);
|
||||
fwrite(&payload_len, 4, 1, ov->journal_fp);
|
||||
fwrite(&checksum, 4, 1, ov->journal_fp);
|
||||
fwrite(&op, 1, 1, ov->journal_fp);
|
||||
fwrite(payload, 1, payload_len, ov->journal_fp);
|
||||
fflush(ov->journal_fp);
|
||||
fsync(fileno(ov->journal_fp));
|
||||
free(payload);
|
||||
}
|
||||
pthread_mutex_unlock(&ov->journal_lock);
|
||||
}
|
||||
|
||||
static void journal_append_put(Overlay *ov, const char *name, const void *data, uint64_t size, int64_t mtime) {
|
||||
journal_append_record(ov, 0, name, mtime, size, data);
|
||||
}
|
||||
static void journal_append_delete(Overlay *ov, const char *name) {
|
||||
journal_append_record(ov, 1, name, 0, 0, NULL);
|
||||
}
|
||||
static void journal_append_mkdir(Overlay *ov, const char *name) {
|
||||
journal_append_record(ov, 2, name, 0, 0, NULL);
|
||||
}
|
||||
|
||||
/* journals the *current* full content of `path` after a write/rename,
|
||||
* matching the whole-value scheme described above. */
|
||||
static void journal_put_current(Overlay *ov, const char *path, MutCell *cell) {
|
||||
UpperStore *us = upper_of(ov);
|
||||
pfs_usize size;
|
||||
void *buf = NULL;
|
||||
int64_t mtime = (int64_t)time(NULL);
|
||||
if (us->kind == UPPER_MEM) {
|
||||
pthread_rwlock_rdlock(&cell->buf_lock);
|
||||
size = atomic_load_explicit(&cell->size, memory_order_acquire);
|
||||
mtime = atomic_load_explicit(&cell->mtime, memory_order_acquire);
|
||||
buf = size ? malloc(size) : NULL;
|
||||
if (buf) memcpy(buf, cell->data, size);
|
||||
pthread_rwlock_unlock(&cell->buf_lock);
|
||||
} else {
|
||||
VfsStat st;
|
||||
int e = 0;
|
||||
if (pfs_dir_statat(us->root_fd, dirrel(path), &st, &e) < 0) return;
|
||||
size = st.size;
|
||||
mtime = st.mtime;
|
||||
buf = size ? malloc(size) : NULL;
|
||||
if (buf) upper_cell_read(us, cell, path, 0, buf, size);
|
||||
}
|
||||
journal_append_put(ov, path, buf, size, mtime);
|
||||
free(buf);
|
||||
}
|
||||
|
||||
static void journal_replay(Overlay *ov) {
|
||||
if (!ov->pack_path) return;
|
||||
char jpath[PFS_PATH_MAX];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
|
||||
FILE *fp = fopen(jpath, "rb");
|
||||
if (!fp) return;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
for (;;) {
|
||||
uint32_t len, checksum; uint8_t op;
|
||||
if (fread(&len, 4, 1, fp) != 1) break;
|
||||
if (fread(&checksum, 4, 1, fp) != 1) break;
|
||||
if (fread(&op, 1, 1, fp) != 1) break;
|
||||
unsigned char *payload = (unsigned char *)malloc(len);
|
||||
size_t got = fread(payload, 1, len, fp);
|
||||
if (got != len) { free(payload); break; } /* torn record: stop, Section 4.3 */
|
||||
if ((uint32_t)pfs_fnv1a64(payload, len) != checksum) { free(payload); break; }
|
||||
|
||||
size_t off = 0;
|
||||
uint32_t name_len;
|
||||
memcpy(&name_len, payload + off, 4); off += 4;
|
||||
char *name = (char *)malloc(name_len + 1);
|
||||
memcpy(name, payload + off, name_len); name[name_len] = '\0'; off += name_len;
|
||||
|
||||
int err = 0;
|
||||
if (op == 0) {
|
||||
int64_t mtime; uint64_t size;
|
||||
memcpy(&mtime, payload + off, 8); off += 8;
|
||||
memcpy(&size, payload + off, 8); off += 8;
|
||||
MutCell *cell;
|
||||
upper_copy_up(us, name, payload + off, size, mtime, &cell, &err);
|
||||
} else if (op == 1) {
|
||||
const PackIndexEntry *pe;
|
||||
int had = ov->pack && pack_find(ov->pack, name, &pe);
|
||||
upper_remove(us, name, had, &err);
|
||||
} else if (op == 2) {
|
||||
upper_mkdir(us, name, &err);
|
||||
}
|
||||
free(name);
|
||||
free(payload);
|
||||
}
|
||||
fclose(fp);
|
||||
}
|
||||
|
||||
/* ---- open file handle ---- */
|
||||
|
||||
typedef struct OverlayFile {
|
||||
Overlay *ov;
|
||||
UpperSnapshot *held; /* keeps `cell` reachable; released on close, NULL if from_upper==0 */
|
||||
MutCell *cell;
|
||||
const PackIndexEntry *lower_e; /* valid when from_upper==0 */
|
||||
int from_upper;
|
||||
char *path;
|
||||
pfs_usize pos;
|
||||
} OverlayFile;
|
||||
|
||||
static VfsFile *make_upper_file(Backend *b, Overlay *ov, const char *path, MutCell *cell) {
|
||||
UpperSnapshot *s = upper_acquire(upper_of(ov));
|
||||
OverlayFile *of = (OverlayFile *)calloc(1, sizeof(OverlayFile));
|
||||
of->ov = ov; of->held = s; of->cell = cell; of->from_upper = 1; of->path = strdup(path);
|
||||
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||||
f->backend = b; f->state = of;
|
||||
return f;
|
||||
}
|
||||
|
||||
static int overlay_open(Backend *b, const char *path, int flags, VfsFile **out) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int in_upper = upper_lookup(s, path, &ue);
|
||||
if (in_upper && ue->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
|
||||
|
||||
if (in_upper && ue->kind == ENTRY_FILE) {
|
||||
MutCell *cell = ue->cell;
|
||||
upper_release(s);
|
||||
if (flags & VFS_O_TRUNC) upper_cell_truncate(us, cell, path);
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
int was_whiteout = (in_upper && ue->kind == ENTRY_WHITEOUT);
|
||||
upper_release(s);
|
||||
|
||||
if (was_whiteout) {
|
||||
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_create(us, path, 0, &cell, &err) < 0) return err;
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
const PackIndexEntry *pe = NULL;
|
||||
int in_lower = ov->pack && pack_find(ov->pack, path, &pe);
|
||||
if (in_lower && (pe->mode & PFS_MODE_DIR)) return VFS_ERR_ISDIR;
|
||||
|
||||
if (!in_lower) {
|
||||
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_create(us, path, 0, &cell, &err) < 0) return err;
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
if (!(flags & (VFS_O_WRONLY | VFS_O_RDWR)) && !(flags & VFS_O_TRUNC)) {
|
||||
/* read-only: serve straight from the pack, no copy-up (Section 4.1 step 3
|
||||
* ties copy-up to a write-capable open, not every open). */
|
||||
OverlayFile *of = (OverlayFile *)calloc(1, sizeof(OverlayFile));
|
||||
of->ov = ov; of->lower_e = pe; of->from_upper = 0; of->path = strdup(path);
|
||||
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||||
f->backend = b; f->state = of;
|
||||
*out = f;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
const void *data = (flags & VFS_O_TRUNC) ? NULL : pack_entry_data(ov->pack, pe);
|
||||
uint64_t size = (flags & VFS_O_TRUNC) ? 0 : pe->size;
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_copy_up(us, path, data, size, pe->mtime, &cell, &err) < 0) return err;
|
||||
*out = make_upper_file(b, ov, path, cell);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static pfs_isize overlay_read(VfsFile *f, void *buf, pfs_usize n) {
|
||||
OverlayFile *of = (OverlayFile *)f->state;
|
||||
if (of->from_upper) {
|
||||
pfs_isize r = upper_cell_read(upper_of(of->ov), of->cell, of->path, of->pos, buf, n);
|
||||
if (r > 0) of->pos += (pfs_usize)r;
|
||||
return r;
|
||||
}
|
||||
pfs_usize size = of->lower_e->size;
|
||||
pfs_usize avail = of->pos < size ? size - of->pos : 0;
|
||||
pfs_usize to_copy = n < avail ? n : avail;
|
||||
if (to_copy) memcpy(buf, (const char *)pack_entry_data(of->ov->pack, of->lower_e) + of->pos, to_copy);
|
||||
of->pos += to_copy;
|
||||
return (pfs_isize)to_copy;
|
||||
}
|
||||
|
||||
static pfs_isize overlay_write(VfsFile *f, const void *buf, pfs_usize n) {
|
||||
OverlayFile *of = (OverlayFile *)f->state;
|
||||
if (!of->from_upper) return VFS_ERR_PERM;
|
||||
int err = 0;
|
||||
pfs_isize w = upper_cell_write(upper_of(of->ov), of->cell, of->path, of->pos, buf, n, &err);
|
||||
if (w < 0) return err;
|
||||
of->pos += (pfs_usize)w;
|
||||
journal_put_current(of->ov, of->path, of->cell);
|
||||
return w;
|
||||
}
|
||||
|
||||
static int overlay_close(VfsFile *f) {
|
||||
OverlayFile *of = (OverlayFile *)f->state;
|
||||
if (of->held) upper_release(of->held);
|
||||
free(of->path);
|
||||
free(of);
|
||||
free(f);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_stat(Backend *b, const char *path, VfsStat *out) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
if (upper_lookup(s, path, &ue)) {
|
||||
if (ue->kind == ENTRY_WHITEOUT) { upper_release(s); return VFS_ERR_NOENT; }
|
||||
if (ue->kind == ENTRY_DIR_MARKER) {
|
||||
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
if (us->kind == UPPER_MEM) {
|
||||
out->size = atomic_load_explicit(&ue->cell->size, memory_order_acquire);
|
||||
out->mtime = atomic_load_explicit(&ue->cell->mtime, memory_order_acquire);
|
||||
out->kind = VFS_KIND_FILE;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
int err = 0;
|
||||
int rc = pfs_dir_statat(us->root_fd, dirrel(path), out, &err);
|
||||
upper_release(s);
|
||||
return rc < 0 ? err : VFS_OK;
|
||||
}
|
||||
int has_kids = upper_has_children(s, path);
|
||||
upper_release(s);
|
||||
if (has_kids) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
|
||||
|
||||
if (ov->pack) {
|
||||
const PackIndexEntry *pe;
|
||||
if (pack_find(ov->pack, path, &pe)) {
|
||||
out->size = pe->size; out->mtime = pe->mtime;
|
||||
out->kind = (pe->mode & PFS_MODE_DIR) ? VFS_KIND_DIR : VFS_KIND_FILE;
|
||||
return VFS_OK;
|
||||
}
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
if (hi > lo) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
|
||||
}
|
||||
if (strcmp(path, "/") == 0) { out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR; return VFS_OK; }
|
||||
return VFS_ERR_NOENT;
|
||||
}
|
||||
|
||||
typedef struct Child { char name[256]; VfsEntryKind kind; } Child;
|
||||
|
||||
static int child_index(Child *arr, pfs_usize n, const char *name, size_t len) {
|
||||
for (pfs_usize i = 0; i < n; i++)
|
||||
if (strncmp(arr[i].name, name, len) == 0 && arr[i].name[len] == '\0') return (int)i;
|
||||
return -1;
|
||||
}
|
||||
|
||||
static int overlay_readdir(Backend *b, const char *path, VfsDir *out) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
char prefix[PFS_PATH_MAX];
|
||||
if (strcmp(path, "/") == 0) strcpy(prefix, "/"); else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t plen = strlen(prefix);
|
||||
|
||||
Child *handled = NULL; pfs_usize hcount = 0, hcap = 0;
|
||||
VfsDirEntry *entries = NULL; pfs_usize count = 0, cap = 0;
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
const char *name = s->entries[i].name;
|
||||
if (strncmp(name, prefix, plen) != 0) continue;
|
||||
const char *restp = name + plen;
|
||||
const char *slash = strchr(restp, '/');
|
||||
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||||
if (clen == 0 || clen >= sizeof(handled[0].name)) continue;
|
||||
if (child_index(handled, hcount, restp, clen) >= 0) continue;
|
||||
if (hcount == hcap) { hcap = hcap ? hcap * 2 : 8; handled = (Child *)realloc(handled, hcap * sizeof(Child)); }
|
||||
memcpy(handled[hcount].name, restp, clen); handled[hcount].name[clen] = '\0';
|
||||
handled[hcount].kind = slash ? VFS_KIND_DIR : (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||||
int is_whiteout = (s->entries[i].kind == ENTRY_WHITEOUT);
|
||||
hcount++;
|
||||
if (!is_whiteout) {
|
||||
if (count == cap) { cap = cap ? cap * 2 : 8; entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry)); }
|
||||
memcpy(entries[count].name, handled[hcount - 1].name, clen + 1);
|
||||
entries[count].kind = handled[hcount - 1].kind;
|
||||
count++;
|
||||
}
|
||||
}
|
||||
upper_release(s);
|
||||
|
||||
if (ov->pack) {
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
for (size_t i = lo; i < hi; i++) {
|
||||
const char *name = pack_entry_name(ov->pack, &ov->pack->entries[i]);
|
||||
const char *restp = name + plen;
|
||||
const char *slash = strchr(restp, '/');
|
||||
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||||
if (clen == 0 || clen >= 256) continue;
|
||||
if (child_index(handled, hcount, restp, clen) >= 0) continue;
|
||||
VfsDirEntry tmp; memcpy(tmp.name, restp, clen); tmp.name[clen] = '\0';
|
||||
int dup = 0;
|
||||
for (pfs_usize j = 0; j < count; j++)
|
||||
if (strncmp(entries[j].name, restp, clen) == 0 && entries[j].name[clen] == '\0') { dup = 1; break; }
|
||||
if (dup) continue;
|
||||
if (count == cap) { cap = cap ? cap * 2 : 8; entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry)); }
|
||||
entries[count] = tmp;
|
||||
entries[count].kind = slash ? VFS_KIND_DIR : ((ov->pack->entries[i].mode & PFS_MODE_DIR) ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||||
count++;
|
||||
}
|
||||
}
|
||||
free(handled);
|
||||
out->entries = entries;
|
||||
out->count = count;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_mkdir(Backend *b, const char *path) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int exists = upper_lookup(s, path, &ue) && ue->kind != ENTRY_WHITEOUT;
|
||||
int has_kids = upper_has_children(s, path);
|
||||
upper_release(s);
|
||||
|
||||
if (!exists && ov->pack) {
|
||||
const PackIndexEntry *pe;
|
||||
if (pack_find(ov->pack, path, &pe)) exists = 1;
|
||||
if (!exists && !has_kids) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
if (hi > lo) has_kids = 1;
|
||||
}
|
||||
}
|
||||
if (exists || has_kids) return VFS_ERR_EXIST;
|
||||
|
||||
int err = 0;
|
||||
if (upper_mkdir(us, path, &err) < 0) return err;
|
||||
journal_append_mkdir(ov, path);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_unlink(Backend *b, const char *path) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int in_upper = upper_lookup(s, path, &ue);
|
||||
if (in_upper && ue->kind == ENTRY_WHITEOUT) { upper_release(s); return VFS_ERR_NOENT; }
|
||||
if (in_upper && ue->kind == ENTRY_DIR_MARKER && upper_has_children(s, path)) { upper_release(s); return VFS_ERR_NOTEMPTY; }
|
||||
upper_release(s);
|
||||
|
||||
const PackIndexEntry *pe = NULL;
|
||||
int had_lower = ov->pack && pack_find(ov->pack, path, &pe);
|
||||
if (had_lower && (pe->mode & PFS_MODE_DIR)) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
UpperSnapshot *s2 = upper_acquire(us);
|
||||
int kids = (hi > lo) || upper_has_children(s2, path);
|
||||
upper_release(s2);
|
||||
if (kids) return VFS_ERR_NOTEMPTY;
|
||||
}
|
||||
if (!in_upper && !had_lower) {
|
||||
UpperSnapshot *s3 = upper_acquire(us);
|
||||
int kids = upper_has_children(s3, path);
|
||||
upper_release(s3);
|
||||
if (!kids && ov->pack) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t lo, hi;
|
||||
pack_range(ov->pack, prefix, &lo, &hi);
|
||||
kids = hi > lo;
|
||||
}
|
||||
return kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
|
||||
}
|
||||
|
||||
int err = 0;
|
||||
if (upper_remove(us, path, had_lower, &err) < 0) return err;
|
||||
journal_append_delete(ov, path);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int overlay_rename(Backend *b, const char *from, const char *to) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
const UpperEntry *ue;
|
||||
int in_upper = upper_lookup(s, from, &ue) && ue->kind != ENTRY_WHITEOUT;
|
||||
UpperEntryKind kind = in_upper ? ue->kind : ENTRY_FILE;
|
||||
upper_release(s);
|
||||
|
||||
const PackIndexEntry *pe = NULL;
|
||||
int had_lower = ov->pack && pack_find(ov->pack, from, &pe);
|
||||
if (!in_upper && !had_lower) return VFS_ERR_NOENT;
|
||||
|
||||
if (in_upper) {
|
||||
int err = 0;
|
||||
if (upper_rename(us, from, to, had_lower, &err) < 0) return err;
|
||||
} else if (pe->mode & PFS_MODE_DIR) {
|
||||
int err = 0;
|
||||
if (upper_mkdir(us, to, &err) < 0 && err != VFS_ERR_EXIST) return err;
|
||||
int err2 = 0;
|
||||
upper_remove(us, from, 1, &err2);
|
||||
} else {
|
||||
MutCell *cell; int err = 0;
|
||||
if (upper_copy_up(us, to, pack_entry_data(ov->pack, pe), pe->size, pe->mtime, &cell, &err) < 0) return err;
|
||||
int err2 = 0;
|
||||
upper_remove(us, from, 1, &err2);
|
||||
}
|
||||
|
||||
journal_append_delete(ov, from);
|
||||
if (kind == ENTRY_DIR_MARKER) {
|
||||
journal_append_mkdir(ov, to);
|
||||
} else {
|
||||
UpperSnapshot *s2 = upper_acquire(us);
|
||||
const UpperEntry *nue;
|
||||
if (upper_lookup(s2, to, &nue) && nue->kind == ENTRY_FILE) journal_put_current(ov, to, nue->cell);
|
||||
upper_release(s2);
|
||||
}
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
/* ---- compaction (Section 4.1 step 5, 5.3, 9.2) ---- */
|
||||
|
||||
typedef struct BuildCtx {
|
||||
PackBuildEntry *entries;
|
||||
size_t count, cap;
|
||||
size_t upper_start; /* index at which upper-sourced (malloc'd) entries begin */
|
||||
} BuildCtx;
|
||||
|
||||
static void build_push(BuildCtx *bc, const char *name, const void *data, pfs_usize size, uint32_t mode, int64_t mtime, int owned) {
|
||||
if (bc->count == bc->cap) { bc->cap = bc->cap ? bc->cap * 2 : 16; bc->entries = (PackBuildEntry *)realloc(bc->entries, bc->cap * sizeof(PackBuildEntry)); }
|
||||
PackBuildEntry *e = &bc->entries[bc->count++];
|
||||
e->name = strdup(name);
|
||||
if (owned) {
|
||||
e->data = size ? malloc(size) : NULL;
|
||||
if (size) memcpy((void *)e->data, data, size);
|
||||
} else {
|
||||
e->data = data;
|
||||
}
|
||||
e->size = size;
|
||||
e->mode = mode;
|
||||
e->mtime = mtime;
|
||||
}
|
||||
|
||||
static void build_visit(void *ctx, const char *name, UpperEntryKind kind, const void *data, pfs_usize size, int64_t mtime) {
|
||||
build_push((BuildCtx *)ctx, name, data, size, kind == ENTRY_DIR_MARKER ? PFS_MODE_DIR : 0, mtime, 1);
|
||||
}
|
||||
|
||||
static void free_build_entries(BuildCtx *bc) {
|
||||
for (size_t i = 0; i < bc->count; i++) {
|
||||
free((void *)bc->entries[i].name);
|
||||
if (i >= bc->upper_start) free((void *)bc->entries[i].data);
|
||||
}
|
||||
free(bc->entries);
|
||||
}
|
||||
|
||||
static int overlay_sync(Backend *b) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
if (!ov->pack_path) return VFS_ERR_PERM;
|
||||
UpperStore *us = upper_of(ov);
|
||||
|
||||
pthread_mutex_lock(&ov->journal_lock);
|
||||
if (ov->journal_fp) fflush(ov->journal_fp);
|
||||
long watermark = 0;
|
||||
char jpath[PFS_PATH_MAX];
|
||||
snprintf(jpath, sizeof(jpath), "%s.jnl", ov->pack_path);
|
||||
FILE *probe = fopen(jpath, "rb");
|
||||
if (probe) { fseek(probe, 0, SEEK_END); watermark = ftell(probe); fclose(probe); }
|
||||
pthread_mutex_unlock(&ov->journal_lock);
|
||||
|
||||
BuildCtx bc; memset(&bc, 0, sizeof(bc));
|
||||
if (ov->pack) {
|
||||
UpperSnapshot *s = upper_acquire(us);
|
||||
for (size_t i = 0; i < ov->pack->entry_count; i++) {
|
||||
const PackIndexEntry *pe = &ov->pack->entries[i];
|
||||
const char *name = pack_entry_name(ov->pack, pe);
|
||||
const UpperEntry *ue;
|
||||
if (upper_lookup(s, name, &ue)) continue; /* upper (incl. whiteout) supersedes */
|
||||
build_push(&bc, name, pack_entry_data(ov->pack, pe), pe->size, pe->mode, pe->mtime, 0);
|
||||
}
|
||||
upper_release(s);
|
||||
}
|
||||
bc.upper_start = bc.count;
|
||||
UpperWalkCb cb = { build_visit, &bc };
|
||||
upper_walk_live(us, &cb);
|
||||
|
||||
int err = 0;
|
||||
if (pack_write(ov->pack_path, bc.entries, bc.count, &err) < 0) {
|
||||
free_build_entries(&bc);
|
||||
return err; /* Section 4.3: existing pack + journal are untouched */
|
||||
}
|
||||
free_build_entries(&bc);
|
||||
|
||||
Pack *newpack = NULL; int lerr = 0;
|
||||
if (pack_load(ov->pack_path, &newpack, &lerr) == 0) {
|
||||
if (ov->pack) pack_close(ov->pack);
|
||||
ov->pack = newpack;
|
||||
}
|
||||
upper_reset_empty(us);
|
||||
|
||||
pthread_mutex_lock(&ov->journal_lock);
|
||||
char jtmp[PFS_PATH_MAX];
|
||||
snprintf(jtmp, sizeof(jtmp), "%s.jnl.tmp", ov->pack_path);
|
||||
FILE *src = fopen(jpath, "rb");
|
||||
if (src) {
|
||||
fseek(src, watermark, SEEK_SET);
|
||||
FILE *dst = fopen(jtmp, "wb");
|
||||
if (dst) {
|
||||
char buf[4096]; size_t n;
|
||||
while ((n = fread(buf, 1, sizeof(buf), src)) > 0) fwrite(buf, 1, n, dst);
|
||||
fflush(dst); fsync(fileno(dst)); fclose(dst);
|
||||
rename(jtmp, jpath);
|
||||
}
|
||||
fclose(src);
|
||||
}
|
||||
if (ov->journal_fp) { fclose(ov->journal_fp); ov->journal_fp = NULL; }
|
||||
pthread_mutex_unlock(&ov->journal_lock);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static void overlay_free(Backend *b) {
|
||||
Overlay *ov = (Overlay *)b->state;
|
||||
if (ov->journal_fp) fclose(ov->journal_fp);
|
||||
pthread_mutex_destroy(&ov->journal_lock);
|
||||
if (ov->pack) pack_close(ov->pack);
|
||||
free(ov->pack_path);
|
||||
free(ov);
|
||||
free(b);
|
||||
}
|
||||
|
||||
static const BackendOps OVERLAY_OPS = {
|
||||
overlay_open, overlay_read, overlay_write, overlay_close, overlay_stat,
|
||||
overlay_readdir, overlay_mkdir, overlay_unlink, overlay_rename, overlay_sync, overlay_free
|
||||
};
|
||||
|
||||
Backend *overlay_upper_backend(Backend *b) {
|
||||
return b->ops == &OVERLAY_OPS ? ((Overlay *)b->state)->upper : NULL;
|
||||
}
|
||||
|
||||
Backend *backend_overlay_new(const char *pack_path, Backend *upper, int *err) {
|
||||
Overlay *ov = (Overlay *)calloc(1, sizeof(Overlay));
|
||||
ov->pack_path = pack_path ? strdup(pack_path) : NULL;
|
||||
ov->upper = upper;
|
||||
pthread_mutex_init(&ov->journal_lock, NULL);
|
||||
|
||||
if (pack_path) {
|
||||
Pack *p; int lerr = 0;
|
||||
if (pack_load(pack_path, &p, &lerr) == 0) {
|
||||
ov->pack = p;
|
||||
} else if (lerr != VFS_ERR_NOENT) {
|
||||
pthread_mutex_destroy(&ov->journal_lock);
|
||||
free(ov->pack_path);
|
||||
free(ov);
|
||||
if (err) *err = lerr;
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
journal_replay(ov);
|
||||
|
||||
Backend *b = (Backend *)calloc(1, sizeof(Backend));
|
||||
b->ops = &OVERLAY_OPS;
|
||||
b->state = ov;
|
||||
return b;
|
||||
}
|
||||
+304
@@ -0,0 +1,304 @@
|
||||
/*
|
||||
* pack.c — on-disk pack format (Section 9), load-time integrity
|
||||
* validation (Section 7), and compaction's writer (Section 4.1, 4.3,
|
||||
* 5.3, 9.1, 9.2).
|
||||
*
|
||||
* On-disk layout (a concrete realization of the illustrative sketch in
|
||||
* Section 9, extended with the checksum field Section 7 requires):
|
||||
*
|
||||
* PackHeader (fixed size, 8-byte aligned)
|
||||
* blobs (index_offset - sizeof(PackHeader) bytes)
|
||||
* PackIndexEntry[index_count] at index_offset, sorted by name (9.1)
|
||||
* strings (each name NUL-terminated) at strings_offset
|
||||
*/
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
#define PACK_VERSION 1
|
||||
|
||||
typedef struct PackHeader {
|
||||
char magic[4];
|
||||
uint32_t version;
|
||||
uint64_t index_offset;
|
||||
uint64_t index_count;
|
||||
uint64_t strings_offset;
|
||||
uint64_t strings_len;
|
||||
uint64_t checksum; /* FNV-1a64 over [index_offset, strings_offset+strings_len) */
|
||||
} PackHeader;
|
||||
|
||||
static uint64_t align8(uint64_t n) { return (n + 7u) & ~(uint64_t)7u; }
|
||||
|
||||
/* ---- loading + validation (Section 7) ---- */
|
||||
|
||||
void pack_close(Pack *p) {
|
||||
if (!p) return;
|
||||
if (p->map && p->map != MAP_FAILED) munmap(p->map, p->map_len);
|
||||
if (p->fd >= 0) close(p->fd);
|
||||
free(p->path);
|
||||
free(p);
|
||||
}
|
||||
|
||||
int pack_load(const char *path, Pack **out, int *err) {
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0) { if (err) *err = VFS_ERR_NOENT; return -1; }
|
||||
|
||||
struct stat st;
|
||||
if (fstat(fd, &st) < 0) { close(fd); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
size_t len = (size_t)st.st_size;
|
||||
if (len < sizeof(PackHeader)) { close(fd); if (err) *err = VFS_ERR_CORRUPT; return -1; }
|
||||
|
||||
void *map = mmap(NULL, len, PROT_READ, MAP_PRIVATE, fd, 0);
|
||||
if (map == MAP_FAILED) { close(fd); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
|
||||
const PackHeader *hdr = (const PackHeader *)map;
|
||||
if (memcmp(hdr->magic, "PKFS", 4) != 0 || hdr->version != PACK_VERSION) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Bounds-check every offset before trusting it (Section 7): the
|
||||
* index and strings regions must lie within the file and must not
|
||||
* overlap the header. */
|
||||
if (hdr->index_offset < sizeof(PackHeader) ||
|
||||
hdr->index_offset > len ||
|
||||
hdr->index_count > (len - hdr->index_offset) / sizeof(PackIndexEntry)) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
uint64_t index_bytes = hdr->index_count * (uint64_t)sizeof(PackIndexEntry);
|
||||
uint64_t index_end = hdr->index_offset + index_bytes;
|
||||
if (hdr->strings_offset < index_end || hdr->strings_offset > len ||
|
||||
hdr->strings_len > len - hdr->strings_offset) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
uint64_t strings_end = hdr->strings_offset + hdr->strings_len;
|
||||
if (strings_end > len) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
|
||||
uint64_t checksum = pfs_fnv1a64((const char *)map + hdr->index_offset,
|
||||
(size_t)(strings_end - hdr->index_offset));
|
||||
if (checksum != hdr->checksum) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
|
||||
const PackIndexEntry *entries = (const PackIndexEntry *)((const char *)map + hdr->index_offset);
|
||||
const char *strings_base = (const char *)map + hdr->strings_offset;
|
||||
|
||||
for (uint64_t i = 0; i < hdr->index_count; i++) {
|
||||
const PackIndexEntry *e = &entries[i];
|
||||
if (e->data_off < sizeof(PackHeader) || e->data_off > hdr->index_offset ||
|
||||
e->size > hdr->index_offset - e->data_off) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
if (e->name_off > hdr->strings_len || e->name_len > hdr->strings_len - e->name_off) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
/* Section 9's names are NUL-terminated in the strings region; a
|
||||
* missing terminator is treated as corruption, not tolerated. */
|
||||
if (strings_base[e->name_off + e->name_len] != '\0') {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT;
|
||||
return -1;
|
||||
}
|
||||
if (i > 0) {
|
||||
const PackIndexEntry *prev = &entries[i - 1];
|
||||
if (strcmp(strings_base + prev->name_off, strings_base + e->name_off) >= 0) {
|
||||
munmap(map, len); close(fd);
|
||||
if (err) *err = VFS_ERR_CORRUPT; /* Section 9.1 relies on sortedness */
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Pack *p = (Pack *)calloc(1, sizeof(Pack));
|
||||
if (!p) { munmap(map, len); close(fd); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
p->fd = fd;
|
||||
p->map = map;
|
||||
p->map_len = len;
|
||||
p->entries = entries;
|
||||
p->entry_count = (size_t)hdr->index_count;
|
||||
p->strings_base = strings_base;
|
||||
p->strings_len = (size_t)hdr->strings_len;
|
||||
p->path = strdup(path);
|
||||
*out = p;
|
||||
return 0;
|
||||
}
|
||||
|
||||
const char *pack_entry_name(const Pack *p, const PackIndexEntry *e) {
|
||||
return p->strings_base + e->name_off;
|
||||
}
|
||||
|
||||
const void *pack_entry_data(const Pack *p, const PackIndexEntry *e) {
|
||||
return (const char *)p->map + e->data_off;
|
||||
}
|
||||
|
||||
int pack_find(const Pack *p, const char *name, const PackIndexEntry **out) {
|
||||
size_t lo = 0, hi = p->entry_count;
|
||||
while (lo < hi) {
|
||||
size_t mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(pack_entry_name(p, &p->entries[mid]), name);
|
||||
if (c == 0) { *out = &p->entries[mid]; return 1; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static size_t lower_bound_str(const Pack *p, const char *key) {
|
||||
size_t lo = 0, hi = p->entry_count;
|
||||
while (lo < hi) {
|
||||
size_t mid = lo + (hi - lo) / 2;
|
||||
if (strcmp(pack_entry_name(p, &p->entries[mid]), key) < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
return lo;
|
||||
}
|
||||
|
||||
void pack_range(const Pack *p, const char *prefix, size_t *lo_out, size_t *hi_out) {
|
||||
size_t lo = lower_bound_str(p, prefix);
|
||||
size_t plen = strlen(prefix);
|
||||
char *upper = (char *)malloc(plen + 2);
|
||||
memcpy(upper, prefix, plen);
|
||||
upper[plen] = (char)0x7F; /* > any valid path byte, per Section 9.1's argument */
|
||||
upper[plen + 1] = '\0';
|
||||
size_t hi = lower_bound_str(p, upper);
|
||||
free(upper);
|
||||
*lo_out = lo;
|
||||
*hi_out = hi;
|
||||
}
|
||||
|
||||
/* ---- writing (compaction target, Section 4.1 / 4.3 / 9.2) ---- */
|
||||
|
||||
static int pack_build_entry_cmp(const void *a, const void *b) {
|
||||
return strcmp(((const PackBuildEntry *)a)->name, ((const PackBuildEntry *)b)->name);
|
||||
}
|
||||
|
||||
typedef struct DedupSlot {
|
||||
uint64_t hash;
|
||||
uint64_t size;
|
||||
uint64_t data_off;
|
||||
} DedupSlot;
|
||||
|
||||
int pack_write(const char *path, const PackBuildEntry *in, size_t count, int *err) {
|
||||
PackBuildEntry *entries = NULL;
|
||||
if (count > 0) {
|
||||
entries = (PackBuildEntry *)malloc(count * sizeof(PackBuildEntry));
|
||||
if (!entries) { if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
memcpy(entries, in, count * sizeof(PackBuildEntry));
|
||||
}
|
||||
qsort(entries, count, sizeof(PackBuildEntry), pack_build_entry_cmp);
|
||||
|
||||
/* Pass 1: compute blob region with exact-duplicate elimination
|
||||
* (Section 9.2) and the strings region size. */
|
||||
DedupSlot *slots = count ? (DedupSlot *)malloc(count * sizeof(DedupSlot)) : NULL;
|
||||
size_t slot_count = 0;
|
||||
uint64_t *data_off_for = count ? (uint64_t *)malloc(count * sizeof(uint64_t)) : NULL;
|
||||
uint64_t blob_cursor = align8(sizeof(PackHeader));
|
||||
uint64_t strings_total = 0;
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
uint64_t h = pfs_fnv1a64(entries[i].data, (size_t)entries[i].size);
|
||||
uint64_t reuse = UINT64_MAX;
|
||||
for (size_t j = 0; j < slot_count; j++) {
|
||||
if (slots[j].hash == h && slots[j].size == entries[i].size) {
|
||||
reuse = slots[j].data_off;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (reuse != UINT64_MAX) {
|
||||
data_off_for[i] = reuse;
|
||||
} else {
|
||||
data_off_for[i] = blob_cursor;
|
||||
slots[slot_count].hash = h;
|
||||
slots[slot_count].size = entries[i].size;
|
||||
slots[slot_count].data_off = blob_cursor;
|
||||
slot_count++;
|
||||
blob_cursor += entries[i].size;
|
||||
}
|
||||
strings_total += strlen(entries[i].name) + 1;
|
||||
}
|
||||
free(slots);
|
||||
|
||||
uint64_t index_offset = align8(blob_cursor);
|
||||
uint64_t index_bytes = (uint64_t)count * sizeof(PackIndexEntry);
|
||||
uint64_t strings_offset = index_offset + index_bytes;
|
||||
uint64_t total_size = strings_offset + strings_total;
|
||||
|
||||
unsigned char *buf = (unsigned char *)calloc(1, (size_t)total_size);
|
||||
if (!buf) { free(entries); free(data_off_for); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
/* Deduplicated entries (Section 9.2) share a data_off; writing
|
||||
* the same bytes to it more than once is redundant but harmless. */
|
||||
memcpy(buf + data_off_for[i], entries[i].data, entries[i].size);
|
||||
}
|
||||
|
||||
PackIndexEntry *out_entries = (PackIndexEntry *)(buf + index_offset);
|
||||
uint64_t str_cursor = 0;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
size_t nlen = strlen(entries[i].name);
|
||||
out_entries[i].name_off = str_cursor;
|
||||
out_entries[i].name_len = (uint32_t)nlen;
|
||||
out_entries[i].data_off = data_off_for[i];
|
||||
out_entries[i].size = entries[i].size;
|
||||
out_entries[i].mode = entries[i].mode;
|
||||
out_entries[i].mtime = entries[i].mtime;
|
||||
memcpy(buf + strings_offset + str_cursor, entries[i].name, nlen + 1);
|
||||
str_cursor += nlen + 1;
|
||||
}
|
||||
|
||||
PackHeader hdr;
|
||||
memset(&hdr, 0, sizeof(hdr));
|
||||
memcpy(hdr.magic, "PKFS", 4);
|
||||
hdr.version = PACK_VERSION;
|
||||
hdr.index_offset = index_offset;
|
||||
hdr.index_count = count;
|
||||
hdr.strings_offset = strings_offset;
|
||||
hdr.strings_len = strings_total;
|
||||
hdr.checksum = pfs_fnv1a64(buf + index_offset, (size_t)(strings_offset + strings_total - index_offset));
|
||||
memcpy(buf, &hdr, sizeof(hdr));
|
||||
|
||||
free(entries);
|
||||
free(data_off_for);
|
||||
|
||||
/* Atomic compaction (Section 4.3): write to a temp file, fsync, then
|
||||
* rename over the target. A failure here aborts compaction and
|
||||
* leaves the existing pack untouched. */
|
||||
char tmp_path[PFS_PATH_MAX];
|
||||
snprintf(tmp_path, sizeof(tmp_path), "%s.tmp", path);
|
||||
int fd = open(tmp_path, O_WRONLY | O_CREAT | O_TRUNC, 0644);
|
||||
if (fd < 0) { free(buf); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
|
||||
size_t written = 0;
|
||||
while (written < (size_t)total_size) {
|
||||
ssize_t w = write(fd, buf + written, (size_t)total_size - written);
|
||||
if (w < 0) { close(fd); free(buf); unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
written += (size_t)w;
|
||||
}
|
||||
free(buf);
|
||||
|
||||
if (fsync(fd) < 0) { close(fd); unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
close(fd);
|
||||
|
||||
if (rename(tmp_path, path) < 0) { unlink(tmp_path); if (err) *err = VFS_ERR_IO; return -1; }
|
||||
return 0;
|
||||
}
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
/*
|
||||
* path.c — virtual-namespace canonicalization, concept.md Section 6.1.
|
||||
*
|
||||
* Resolves "." and ".." purely lexically, before any backend is reached,
|
||||
* so a crafted path cannot escape a mount's prefix even when no host
|
||||
* directory is involved.
|
||||
*/
|
||||
|
||||
#include <string.h>
|
||||
#include "internal.h"
|
||||
|
||||
int pfs_path_normalize(const char *in, char out[PFS_PATH_MAX]) {
|
||||
if (in == NULL || in[0] != '/') return -1;
|
||||
|
||||
/* Stack of component (start, len) pairs, built by lexical resolution. */
|
||||
const char *starts[256];
|
||||
size_t lens[256];
|
||||
int depth = 0;
|
||||
|
||||
const char *p = in;
|
||||
while (*p) {
|
||||
while (*p == '/') p++;
|
||||
if (!*p) break;
|
||||
const char *comp = p;
|
||||
while (*p && *p != '/') p++;
|
||||
size_t len = (size_t)(p - comp);
|
||||
|
||||
if (len == 1 && comp[0] == '.') {
|
||||
continue;
|
||||
}
|
||||
if (len == 2 && comp[0] == '.' && comp[1] == '.') {
|
||||
if (depth == 0) return -1; /* escape above root */
|
||||
depth--;
|
||||
continue;
|
||||
}
|
||||
if (depth >= 256) return -1; /* pathologically deep; reject rather than overflow */
|
||||
starts[depth] = comp;
|
||||
lens[depth] = len;
|
||||
depth++;
|
||||
}
|
||||
|
||||
size_t off = 0;
|
||||
out[off++] = '/';
|
||||
for (int i = 0; i < depth; i++) {
|
||||
if (off + lens[i] + 2 > PFS_PATH_MAX) return -1;
|
||||
if (i > 0) out[off++] = '/';
|
||||
memcpy(out + off, starts[i], lens[i]);
|
||||
off += lens[i];
|
||||
}
|
||||
out[off] = '\0';
|
||||
return 0;
|
||||
}
|
||||
|
||||
int pfs_path_under(const char *prefix, const char *path) {
|
||||
size_t plen = strlen(prefix);
|
||||
if (plen == 1 && prefix[0] == '/') return 1; /* root mount covers everything */
|
||||
if (strncmp(prefix, path, plen) != 0) return 0;
|
||||
return path[plen] == '\0' || path[plen] == '/';
|
||||
}
|
||||
+706
@@ -0,0 +1,706 @@
|
||||
/*
|
||||
* upper.c — the writable upper layer shared by the `mem` and `dir`
|
||||
* backends, and by the overlay's upper side (Section 3, Section 5).
|
||||
*
|
||||
* Implements the snapshot-based single-writer, wait-free-reader model of
|
||||
* Section 5.3: an immutable, refcounted UpperSnapshot reached through one
|
||||
* atomic pointer; structural writes (create/unlink/rename/mkdir/
|
||||
* whiteout) build and publish a new snapshot under `writer_lock`;
|
||||
* content writes (`upper_cell_write`) mutate a shared MutCell in place
|
||||
* and never touch the snapshot pointer, per the structural/content split
|
||||
* in Section 5.3. Buffer growth on the `mem` side follows the
|
||||
* replace-don't-mutate discipline of Section 5.7.
|
||||
*/
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
/* ---- MutCell lifetime: refcounted separately from any one snapshot,
|
||||
* since a cell's identity is shared across every snapshot whose entry
|
||||
* array still points at the same logical file (Section 5.3). ---- */
|
||||
|
||||
typedef struct MutCellRc {
|
||||
MutCell pub;
|
||||
_Atomic(pfs_usize) refcount;
|
||||
} MutCellRc;
|
||||
|
||||
/* MutCell* handed around externally IS the MutCellRc's first member's
|
||||
* address, so a plain cast recovers the refcount wrapper. */
|
||||
static MutCellRc *rc_of(MutCell *c) { return (MutCellRc *)c; }
|
||||
|
||||
static void mutcell_ref(MutCell *c) {
|
||||
if (!c) return;
|
||||
atomic_fetch_add_explicit(&rc_of(c)->refcount, 1, memory_order_relaxed);
|
||||
}
|
||||
|
||||
static void mutcell_unref(MutCell *c) {
|
||||
if (!c) return;
|
||||
if (atomic_fetch_sub_explicit(&rc_of(c)->refcount, 1, memory_order_acq_rel) == 1) {
|
||||
pthread_mutex_destroy(&c->write_lock);
|
||||
pthread_rwlock_destroy(&c->buf_lock);
|
||||
free(c->data);
|
||||
free(rc_of(c));
|
||||
}
|
||||
}
|
||||
|
||||
static MutCell *mutcell_new(void) {
|
||||
MutCellRc *rc = (MutCellRc *)calloc(1, sizeof(MutCellRc));
|
||||
pthread_mutex_init(&rc->pub.write_lock, NULL);
|
||||
pthread_rwlock_init(&rc->pub.buf_lock, NULL);
|
||||
atomic_init(&rc->pub.size, 0);
|
||||
atomic_init(&rc->pub.mtime, (int64_t)time(NULL));
|
||||
atomic_init(&rc->refcount, 1); /* owned by whichever snapshot installs it first */
|
||||
return &rc->pub;
|
||||
}
|
||||
|
||||
/* ---- UpperSnapshot lifetime ---- */
|
||||
|
||||
UpperSnapshot *upper_acquire(UpperStore *u) {
|
||||
pthread_rwlock_rdlock(&u->reclaim_gate);
|
||||
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
|
||||
pthread_rwlock_unlock(&u->reclaim_gate);
|
||||
return s;
|
||||
}
|
||||
|
||||
/* Retires a snapshot just superseded by a writer's atomic_store to
|
||||
* u->current. Must run under reclaim_gate's write side so that no
|
||||
* upper_acquire() can be mid-flight (loaded `old` but not yet
|
||||
* incremented its refcount) when this potentially frees it. */
|
||||
static void upper_retire(UpperStore *u, UpperSnapshot *old) {
|
||||
pthread_rwlock_wrlock(&u->reclaim_gate);
|
||||
upper_release(old);
|
||||
pthread_rwlock_unlock(&u->reclaim_gate);
|
||||
}
|
||||
|
||||
void upper_release(UpperSnapshot *s) {
|
||||
if (!s) return;
|
||||
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
free(s->entries[i].name);
|
||||
mutcell_unref(s->entries[i].cell);
|
||||
}
|
||||
free(s->entries);
|
||||
free(s);
|
||||
}
|
||||
}
|
||||
|
||||
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out) {
|
||||
pfs_usize lo = 0, hi = s->count;
|
||||
while (lo < hi) {
|
||||
pfs_usize mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(s->entries[mid].name, path);
|
||||
if (c == 0) { *out = &s->entries[mid]; return 1; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_has_children(UpperSnapshot *s, const char *path) {
|
||||
char prefix[PFS_PATH_MAX];
|
||||
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
|
||||
else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t plen = strlen(prefix);
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
|
||||
if (strncmp(s->entries[i].name, prefix, plen) == 0) return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Builds a new snapshot with one entry inserted or replaced at `name`. */
|
||||
static UpperSnapshot *snapshot_upsert(UpperSnapshot *base, const char *name,
|
||||
UpperEntryKind kind, MutCell *cell) {
|
||||
pfs_usize old_count = base ? base->count : 0;
|
||||
pfs_usize lo = 0, hi = old_count;
|
||||
int found = 0;
|
||||
while (lo < hi) {
|
||||
pfs_usize mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(base->entries[mid].name, name);
|
||||
if (c == 0) { lo = mid; found = 1; break; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
pfs_usize new_count = found ? old_count : old_count + 1;
|
||||
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
ns->entries = (UpperEntry *)calloc(new_count ? new_count : 1, sizeof(UpperEntry));
|
||||
ns->count = new_count;
|
||||
atomic_init(&ns->refcount, 1);
|
||||
|
||||
pfs_usize w = 0;
|
||||
for (pfs_usize i = 0; i < old_count; i++) {
|
||||
if (i == lo) {
|
||||
ns->entries[w].name = strdup(name);
|
||||
ns->entries[w].kind = kind;
|
||||
ns->entries[w].cell = cell;
|
||||
mutcell_ref(cell);
|
||||
w++;
|
||||
if (found) continue; /* replaces base->entries[lo] */
|
||||
}
|
||||
ns->entries[w].name = strdup(base->entries[i].name);
|
||||
ns->entries[w].kind = base->entries[i].kind;
|
||||
ns->entries[w].cell = base->entries[i].cell;
|
||||
mutcell_ref(ns->entries[w].cell);
|
||||
w++;
|
||||
}
|
||||
if (lo == old_count) {
|
||||
ns->entries[w].name = strdup(name);
|
||||
ns->entries[w].kind = kind;
|
||||
ns->entries[w].cell = cell;
|
||||
mutcell_ref(cell);
|
||||
w++;
|
||||
}
|
||||
return ns;
|
||||
}
|
||||
|
||||
static UpperSnapshot *snapshot_remove(UpperSnapshot *base, const char *name) {
|
||||
pfs_usize lo = 0, hi = base->count;
|
||||
int found = 0;
|
||||
while (lo < hi) {
|
||||
pfs_usize mid = lo + (hi - lo) / 2;
|
||||
int c = strcmp(base->entries[mid].name, name);
|
||||
if (c == 0) { lo = mid; found = 1; break; }
|
||||
if (c < 0) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
if (!found) return NULL;
|
||||
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
ns->count = base->count - 1;
|
||||
ns->entries = ns->count ? (UpperEntry *)calloc(ns->count, sizeof(UpperEntry)) : NULL;
|
||||
atomic_init(&ns->refcount, 1);
|
||||
pfs_usize w = 0;
|
||||
for (pfs_usize i = 0; i < base->count; i++) {
|
||||
if (i == lo) continue;
|
||||
ns->entries[w].name = strdup(base->entries[i].name);
|
||||
ns->entries[w].kind = base->entries[i].kind;
|
||||
ns->entries[w].cell = base->entries[i].cell;
|
||||
mutcell_ref(ns->entries[w].cell);
|
||||
w++;
|
||||
}
|
||||
return ns;
|
||||
}
|
||||
|
||||
UpperStore *upper_new(UpperKind kind, int root_fd, const char *root_path) {
|
||||
UpperStore *u = (UpperStore *)calloc(1, sizeof(UpperStore));
|
||||
u->kind = kind;
|
||||
u->root_fd = root_fd;
|
||||
u->root_path = root_path ? strdup(root_path) : NULL;
|
||||
pthread_mutex_init(&u->writer_lock, NULL);
|
||||
pthread_rwlock_init(&u->reclaim_gate, NULL);
|
||||
UpperSnapshot *s = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
atomic_init(&s->refcount, 1);
|
||||
atomic_init(&u->current, s);
|
||||
return u;
|
||||
}
|
||||
|
||||
void upper_free(UpperStore *u) {
|
||||
if (!u) return;
|
||||
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
upper_release(s);
|
||||
pthread_mutex_destroy(&u->writer_lock);
|
||||
pthread_rwlock_destroy(&u->reclaim_gate);
|
||||
if (u->kind == UPPER_DIR && u->root_fd >= 0) close(u->root_fd);
|
||||
free(u->root_path);
|
||||
free(u);
|
||||
}
|
||||
|
||||
/* strips the mount-relative leading '/' for host syscalls */
|
||||
static const char *rel(const char *path) { return path[0] == '/' ? path + 1 : path; }
|
||||
|
||||
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err) {
|
||||
(void)truncate_existing;
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
|
||||
MutCell *c = existing->cell;
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
*cell_out = c;
|
||||
return 0;
|
||||
}
|
||||
if (u->kind == UPPER_DIR) {
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
|
||||
if (fd < 0) { pthread_mutex_unlock(&u->writer_lock); if (err) *err = e; return -1; }
|
||||
close(fd);
|
||||
}
|
||||
MutCell *cell = mutcell_new();
|
||||
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
mutcell_unref(cell); /* drop the creation-local ref; ns holds its own */
|
||||
*cell_out = cell;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_mkdir(UpperStore *u, const char *path, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
if (upper_lookup(old, path, &existing) || upper_has_children(old, path)) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_EXIST;
|
||||
return -1;
|
||||
}
|
||||
if (u->kind == UPPER_DIR) {
|
||||
int e = 0;
|
||||
if (pfs_dir_mkdirat(u->root_fd, rel(path), &e) < 0) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_DIR_MARKER, NULL);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
int have = upper_lookup(old, path, &existing);
|
||||
if ((!have || existing->kind == ENTRY_WHITEOUT) && !had_lower) {
|
||||
/* No explicit entry anywhere: still an error to distinguish
|
||||
* "doesn't exist" from "exists only implicitly, via children,
|
||||
* and therefore can't be removed" (Section 9.3). */
|
||||
int kids = upper_has_children(old, path);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
|
||||
return -1;
|
||||
}
|
||||
if (have && existing->kind == ENTRY_DIR_MARKER && upper_has_children(old, path)) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_NOTEMPTY;
|
||||
return -1;
|
||||
}
|
||||
if (u->kind == UPPER_DIR && have && existing->kind != ENTRY_WHITEOUT) {
|
||||
int e = 0;
|
||||
int is_dir = (existing->kind == ENTRY_DIR_MARKER);
|
||||
if (pfs_dir_unlinkat(u->root_fd, rel(path), is_dir, &e) < 0) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
UpperSnapshot *ns = had_lower ? snapshot_upsert(old, path, ENTRY_WHITEOUT, NULL)
|
||||
: snapshot_remove(old, path);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *src;
|
||||
if (!upper_lookup(old, from, &src) || src->kind == ENTRY_WHITEOUT) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_NOENT;
|
||||
return -1;
|
||||
}
|
||||
if (u->kind == UPPER_DIR) {
|
||||
int e = 0;
|
||||
if (pfs_dir_renameat(u->root_fd, rel(from), rel(to), &e) < 0) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
UpperSnapshot *step1 = snapshot_upsert(old, to, src->kind, src->cell);
|
||||
UpperSnapshot *step2 = had_lower_from ? snapshot_upsert(step1, from, ENTRY_WHITEOUT, NULL)
|
||||
: snapshot_remove(step1, from);
|
||||
atomic_store_explicit(&u->current, step2, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
upper_release(step1); /* never published; drop our local build reference */
|
||||
return 0;
|
||||
}
|
||||
|
||||
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size,
|
||||
int64_t mtime, MutCell **cell_out, int *err) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
const UpperEntry *existing;
|
||||
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
*cell_out = existing->cell;
|
||||
return 0; /* raced with another copy-up; use what's already there */
|
||||
}
|
||||
MutCell *cell = mutcell_new();
|
||||
atomic_store_explicit(&cell->mtime, mtime, memory_order_relaxed);
|
||||
|
||||
if (u->kind == UPPER_MEM) {
|
||||
if (size) {
|
||||
cell->data = (unsigned char *)malloc((size_t)size);
|
||||
memcpy(cell->data, data, (size_t)size);
|
||||
}
|
||||
cell->capacity = (pfs_usize)size;
|
||||
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
|
||||
} else {
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
|
||||
if (fd < 0) {
|
||||
mutcell_unref(cell);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = e;
|
||||
return -1;
|
||||
}
|
||||
size_t written = 0;
|
||||
const unsigned char *p = (const unsigned char *)data;
|
||||
int failed = 0;
|
||||
while (written < (size_t)size) {
|
||||
ssize_t w = write(fd, p + written, (size_t)size - written);
|
||||
if (w < 0) { failed = 1; break; }
|
||||
written += (size_t)w;
|
||||
}
|
||||
close(fd);
|
||||
if (failed) {
|
||||
mutcell_unref(cell);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
if (err) *err = VFS_ERR_IO;
|
||||
return -1;
|
||||
}
|
||||
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
|
||||
}
|
||||
|
||||
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
mutcell_unref(cell);
|
||||
*cell_out = cell;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ---- content operations (Section 5.3 fast path, Section 5.7 growth) ---- */
|
||||
|
||||
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n) {
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_rwlock_rdlock(&c->buf_lock);
|
||||
pfs_usize size = atomic_load_explicit(&c->size, memory_order_acquire);
|
||||
pfs_usize avail = off < size ? size - off : 0;
|
||||
pfs_usize to_copy = n < avail ? n : avail;
|
||||
if (to_copy) memcpy(buf, c->data + off, to_copy);
|
||||
pthread_rwlock_unlock(&c->buf_lock);
|
||||
return (pfs_isize)to_copy;
|
||||
}
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_RDONLY, 0, &e);
|
||||
if (fd < 0) return -1;
|
||||
ssize_t r = pread(fd, buf, n, (off_t)off);
|
||||
close(fd);
|
||||
return (pfs_isize)r;
|
||||
}
|
||||
|
||||
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off,
|
||||
const void *buf, pfs_usize n, int *err) {
|
||||
int64_t now = (int64_t)time(NULL);
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
pfs_usize cur_size = atomic_load_explicit(&c->size, memory_order_relaxed);
|
||||
pfs_usize need = off + n;
|
||||
if (need > c->capacity) {
|
||||
pfs_usize newcap = c->capacity ? c->capacity : 64;
|
||||
while (newcap < need) newcap *= 2;
|
||||
unsigned char *nb = (unsigned char *)calloc(1, newcap);
|
||||
if (!nb) { pthread_mutex_unlock(&c->write_lock); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||||
if (c->data && cur_size) memcpy(nb, c->data, cur_size);
|
||||
if (n) memcpy(nb + off, buf, n);
|
||||
unsigned char *old = c->data;
|
||||
/* Section 5.7: publish the new buffer via the rwlock, then
|
||||
* free the old one only after releasing the write lock on
|
||||
* it — any reader that could still see `old` must have
|
||||
* taken buf_lock (rdlock) before this wrlock was granted,
|
||||
* and rdlock/wrlock are mutually exclusive, so it has
|
||||
* already finished copying by the time we get here. */
|
||||
pthread_rwlock_wrlock(&c->buf_lock);
|
||||
c->data = nb;
|
||||
c->capacity = newcap;
|
||||
pthread_rwlock_unlock(&c->buf_lock);
|
||||
free(old);
|
||||
} else if (n) {
|
||||
memcpy(c->data + off, buf, n);
|
||||
}
|
||||
pfs_usize new_size = need > cur_size ? need : cur_size;
|
||||
atomic_store_explicit(&c->size, new_size, memory_order_release);
|
||||
atomic_store_explicit(&c->mtime, now, memory_order_release);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
return (pfs_isize)n;
|
||||
}
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY, 0, &e);
|
||||
if (fd < 0) { pthread_mutex_unlock(&c->write_lock); if (err) *err = e; return -1; }
|
||||
ssize_t w = pwrite(fd, buf, n, (off_t)off);
|
||||
close(fd);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
if (w < 0) { if (err) *err = VFS_ERR_IO; return -1; }
|
||||
return (pfs_isize)w;
|
||||
}
|
||||
|
||||
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path) {
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
atomic_store_explicit(&c->size, 0, memory_order_release);
|
||||
atomic_store_explicit(&c->mtime, (int64_t)time(NULL), memory_order_release);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
} else {
|
||||
pthread_mutex_lock(&c->write_lock);
|
||||
int e = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_TRUNC, 0644, &e);
|
||||
if (fd >= 0) close(fd);
|
||||
pthread_mutex_unlock(&c->write_lock);
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- compaction support ---- */
|
||||
|
||||
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb) {
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
UpperEntry *e = &s->entries[i];
|
||||
if (e->kind == ENTRY_WHITEOUT) continue;
|
||||
if (e->kind == ENTRY_DIR_MARKER) {
|
||||
cb->visit(cb->ctx, e->name, ENTRY_DIR_MARKER, NULL, 0, 0);
|
||||
continue;
|
||||
}
|
||||
if (u->kind == UPPER_MEM) {
|
||||
pthread_rwlock_rdlock(&e->cell->buf_lock);
|
||||
pfs_usize size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
|
||||
cb->visit(cb->ctx, e->name, ENTRY_FILE, e->cell->data, size,
|
||||
atomic_load_explicit(&e->cell->mtime, memory_order_acquire));
|
||||
pthread_rwlock_unlock(&e->cell->buf_lock);
|
||||
} else {
|
||||
int err = 0;
|
||||
int fd = pfs_dir_openat(u->root_fd, rel(e->name), O_RDONLY, 0, &err);
|
||||
if (fd < 0) continue;
|
||||
struct stat st;
|
||||
fstat(fd, &st);
|
||||
void *buf = st.st_size ? malloc((size_t)st.st_size) : NULL;
|
||||
size_t total = 0;
|
||||
while (buf && total < (size_t)st.st_size) {
|
||||
ssize_t r = read(fd, (char *)buf + total, (size_t)st.st_size - total);
|
||||
if (r <= 0) break;
|
||||
total += (size_t)r;
|
||||
}
|
||||
close(fd);
|
||||
cb->visit(cb->ctx, e->name, ENTRY_FILE, buf, total, (int64_t)st.st_mtime);
|
||||
free(buf);
|
||||
}
|
||||
}
|
||||
upper_release(s);
|
||||
}
|
||||
|
||||
void upper_reset_empty(UpperStore *u) {
|
||||
pthread_mutex_lock(&u->writer_lock);
|
||||
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||||
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||||
atomic_init(&ns->refcount, 1);
|
||||
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&u->writer_lock);
|
||||
upper_retire(u, old);
|
||||
}
|
||||
|
||||
/* ---- standalone mem/dir Backend wrapper (no lower layer at all) ---- */
|
||||
|
||||
typedef struct UpperFile {
|
||||
UpperStore *store;
|
||||
MutCell *cell;
|
||||
char *path;
|
||||
pfs_usize pos;
|
||||
} UpperFile;
|
||||
|
||||
static int upstd_open(Backend *b, const char *path, int flags, VfsFile **out) {
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
const UpperEntry *e;
|
||||
int found = upper_lookup(s, path, &e);
|
||||
if (found && e->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
|
||||
|
||||
MutCell *cell;
|
||||
if (found) {
|
||||
cell = e->cell;
|
||||
mutcell_ref(cell);
|
||||
upper_release(s);
|
||||
if (flags & VFS_O_TRUNC) upper_cell_truncate(u, cell, path);
|
||||
} else {
|
||||
upper_release(s);
|
||||
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||||
int err = 0;
|
||||
if (upper_create(u, path, 0, &cell, &err) < 0) return err;
|
||||
mutcell_ref(cell);
|
||||
}
|
||||
|
||||
UpperFile *uf = (UpperFile *)calloc(1, sizeof(UpperFile));
|
||||
uf->store = u; uf->cell = cell; uf->path = strdup(path); uf->pos = 0;
|
||||
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||||
f->backend = b; f->state = uf;
|
||||
*out = f;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static pfs_isize upstd_read(VfsFile *f, void *buf, pfs_usize n) {
|
||||
UpperFile *uf = (UpperFile *)f->state;
|
||||
pfs_isize r = upper_cell_read(uf->store, uf->cell, uf->path, uf->pos, buf, n);
|
||||
if (r > 0) uf->pos += (pfs_usize)r;
|
||||
return r;
|
||||
}
|
||||
|
||||
static pfs_isize upstd_write(VfsFile *f, const void *buf, pfs_usize n) {
|
||||
UpperFile *uf = (UpperFile *)f->state;
|
||||
int err = 0;
|
||||
pfs_isize w = upper_cell_write(uf->store, uf->cell, uf->path, uf->pos, buf, n, &err);
|
||||
if (w >= 0) { uf->pos += (pfs_usize)w; return w; }
|
||||
return err;
|
||||
}
|
||||
|
||||
static int upstd_close(VfsFile *f) {
|
||||
UpperFile *uf = (UpperFile *)f->state;
|
||||
mutcell_unref(uf->cell);
|
||||
free(uf->path);
|
||||
free(uf);
|
||||
free(f);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_stat(Backend *b, const char *path, VfsStat *out) {
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
const UpperEntry *e;
|
||||
if (!upper_lookup(s, path, &e) || e->kind == ENTRY_WHITEOUT) {
|
||||
if (upper_has_children(s, path)) {
|
||||
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
upper_release(s);
|
||||
return VFS_ERR_NOENT;
|
||||
}
|
||||
if (e->kind == ENTRY_DIR_MARKER) {
|
||||
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
if (u->kind == UPPER_MEM) {
|
||||
out->size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
|
||||
out->mtime = atomic_load_explicit(&e->cell->mtime, memory_order_acquire);
|
||||
out->kind = VFS_KIND_FILE;
|
||||
upper_release(s);
|
||||
return VFS_OK;
|
||||
}
|
||||
int err = 0;
|
||||
int rc = pfs_dir_statat(u->root_fd, rel(path), out, &err);
|
||||
upper_release(s);
|
||||
return rc < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_readdir(Backend *b, const char *path, VfsDir *out) {
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
UpperSnapshot *s = upper_acquire(u);
|
||||
|
||||
const UpperEntry *self;
|
||||
int self_found = upper_lookup(s, path, &self);
|
||||
int has_kids = upper_has_children(s, path);
|
||||
if (!has_kids && !(self_found && self->kind == ENTRY_DIR_MARKER) && strcmp(path, "/") != 0) {
|
||||
upper_release(s);
|
||||
return self_found ? VFS_ERR_NOTDIR : VFS_ERR_NOENT;
|
||||
}
|
||||
|
||||
char prefix[PFS_PATH_MAX];
|
||||
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
|
||||
else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||||
size_t plen = strlen(prefix);
|
||||
|
||||
VfsDirEntry *entries = NULL;
|
||||
pfs_usize count = 0, cap = 0;
|
||||
for (pfs_usize i = 0; i < s->count; i++) {
|
||||
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
|
||||
const char *name = s->entries[i].name;
|
||||
if (strncmp(name, prefix, plen) != 0) continue;
|
||||
const char *restp = name + plen;
|
||||
const char *slash = strchr(restp, '/');
|
||||
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||||
if (clen == 0 || clen >= sizeof(entries[0].name)) continue;
|
||||
if (count > 0 && strncmp(entries[count - 1].name, restp, clen) == 0 &&
|
||||
entries[count - 1].name[clen] == '\0') continue; /* sorted -> dup is adjacent */
|
||||
if (count == cap) {
|
||||
cap = cap ? cap * 2 : 8;
|
||||
entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry));
|
||||
}
|
||||
memcpy(entries[count].name, restp, clen);
|
||||
entries[count].name[clen] = '\0';
|
||||
entries[count].kind = slash ? VFS_KIND_DIR
|
||||
: (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||||
count++;
|
||||
}
|
||||
upper_release(s);
|
||||
out->entries = entries;
|
||||
out->count = count;
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_mkdir(Backend *b, const char *path) {
|
||||
int err = 0;
|
||||
return upper_mkdir((UpperStore *)b->state, path, &err) < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_unlink(Backend *b, const char *path) {
|
||||
int err = 0;
|
||||
return upper_remove((UpperStore *)b->state, path, 0, &err) < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_rename(Backend *b, const char *from, const char *to) {
|
||||
int err = 0;
|
||||
return upper_rename((UpperStore *)b->state, from, to, 0, &err) < 0 ? err : VFS_OK;
|
||||
}
|
||||
|
||||
static int upstd_sync(Backend *b) { (void)b; return VFS_ERR_PERM; }
|
||||
|
||||
static void upstd_free(Backend *b) {
|
||||
upper_free((UpperStore *)b->state);
|
||||
free(b);
|
||||
}
|
||||
|
||||
static const BackendOps UPPER_STANDALONE_OPS = {
|
||||
upstd_open, upstd_read, upstd_write, upstd_close, upstd_stat,
|
||||
upstd_readdir, upstd_mkdir, upstd_unlink, upstd_rename, upstd_sync, upstd_free
|
||||
};
|
||||
|
||||
Backend *backend_from_upper(UpperStore *u) {
|
||||
Backend *b = (Backend *)calloc(1, sizeof(Backend));
|
||||
b->ops = &UPPER_STANDALONE_OPS;
|
||||
b->state = u;
|
||||
return b;
|
||||
}
|
||||
|
||||
Backend *backend_mem_new(void) {
|
||||
return backend_from_upper(upper_new(UPPER_MEM, -1, NULL));
|
||||
}
|
||||
|
||||
Backend *backend_dir_new(const char *host_path, int *err) {
|
||||
int e = 0;
|
||||
int fd = pfs_dir_capability_open(host_path, &e);
|
||||
if (fd < 0) { if (err) *err = e; return NULL; }
|
||||
return backend_from_upper(upper_new(UPPER_DIR, fd, host_path));
|
||||
}
|
||||
|
||||
int upper_backend_root_fd(Backend *b) {
|
||||
if (b->ops != &UPPER_STANDALONE_OPS) return -1;
|
||||
UpperStore *u = (UpperStore *)b->state;
|
||||
return u->kind == UPPER_DIR ? u->root_fd : -1;
|
||||
}
|
||||
|
||||
void backend_free(Backend *b) {
|
||||
if (!b) return;
|
||||
b->ops->free(b);
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
/*
|
||||
* vfs.c — VFS core: the mount table (itself part of the Section 5.3
|
||||
* snapshot, per that section's mount-table addition), virtual-namespace
|
||||
* canonicalization dispatch (Section 6.1), and the public API.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
typedef struct MountEntry {
|
||||
char *prefix;
|
||||
size_t prefix_len;
|
||||
Backend *backend;
|
||||
} MountEntry;
|
||||
|
||||
typedef struct MountSnapshot {
|
||||
_Atomic(pfs_usize) refcount;
|
||||
MountEntry *entries;
|
||||
pfs_usize count;
|
||||
} MountSnapshot;
|
||||
|
||||
struct Vfs {
|
||||
_Atomic(MountSnapshot *) current;
|
||||
pthread_mutex_t writer_lock;
|
||||
/* See UpperStore.reclaim_gate (internal.h) for why this exists: it
|
||||
* closes the gap between loading the current MountSnapshot pointer
|
||||
* and incrementing its refcount, which a naive load-then-increment
|
||||
* leaves open to a concurrent writer freeing the very object being
|
||||
* acquired. */
|
||||
pthread_rwlock_t reclaim_gate;
|
||||
};
|
||||
|
||||
static MountSnapshot *mount_acquire(Vfs *v) {
|
||||
pthread_rwlock_rdlock(&v->reclaim_gate);
|
||||
MountSnapshot *s = atomic_load_explicit(&v->current, memory_order_acquire);
|
||||
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
|
||||
pthread_rwlock_unlock(&v->reclaim_gate);
|
||||
return s;
|
||||
}
|
||||
|
||||
static void mount_release(MountSnapshot *s) {
|
||||
if (!s) return;
|
||||
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
|
||||
for (pfs_usize i = 0; i < s->count; i++) free(s->entries[i].prefix);
|
||||
free(s->entries);
|
||||
free(s);
|
||||
}
|
||||
}
|
||||
|
||||
static void mount_retire(Vfs *v, MountSnapshot *old) {
|
||||
pthread_rwlock_wrlock(&v->reclaim_gate);
|
||||
mount_release(old);
|
||||
pthread_rwlock_unlock(&v->reclaim_gate);
|
||||
}
|
||||
|
||||
Vfs *vfs_new(void) {
|
||||
Vfs *v = (Vfs *)calloc(1, sizeof(Vfs));
|
||||
pthread_mutex_init(&v->writer_lock, NULL);
|
||||
pthread_rwlock_init(&v->reclaim_gate, NULL);
|
||||
MountSnapshot *s = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
|
||||
atomic_init(&s->refcount, 1);
|
||||
atomic_init(&v->current, s);
|
||||
return v;
|
||||
}
|
||||
|
||||
void vfs_free(Vfs *v) {
|
||||
if (!v) return;
|
||||
mount_release(atomic_load_explicit(&v->current, memory_order_acquire));
|
||||
pthread_mutex_destroy(&v->writer_lock);
|
||||
pthread_rwlock_destroy(&v->reclaim_gate);
|
||||
free(v);
|
||||
}
|
||||
|
||||
int vfs_mount(Vfs *v, const char *prefix, Backend *b) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(prefix, norm) < 0) return VFS_ERR_INVAL;
|
||||
|
||||
pthread_mutex_lock(&v->writer_lock);
|
||||
MountSnapshot *old = atomic_load_explicit(&v->current, memory_order_acquire);
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
if (strcmp(old->entries[i].prefix, norm) == 0) {
|
||||
pthread_mutex_unlock(&v->writer_lock);
|
||||
return VFS_ERR_EXIST;
|
||||
}
|
||||
}
|
||||
MountSnapshot *ns = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
|
||||
ns->count = old->count + 1;
|
||||
ns->entries = (MountEntry *)calloc(ns->count, sizeof(MountEntry));
|
||||
atomic_init(&ns->refcount, 1);
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
ns->entries[i].prefix = strdup(old->entries[i].prefix);
|
||||
ns->entries[i].prefix_len = old->entries[i].prefix_len;
|
||||
ns->entries[i].backend = old->entries[i].backend;
|
||||
}
|
||||
ns->entries[old->count].prefix = strdup(norm);
|
||||
ns->entries[old->count].prefix_len = strlen(norm);
|
||||
ns->entries[old->count].backend = b;
|
||||
|
||||
atomic_store_explicit(&v->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&v->writer_lock);
|
||||
mount_retire(v, old);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
int vfs_unmount(Vfs *v, const char *prefix) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(prefix, norm) < 0) return VFS_ERR_INVAL;
|
||||
|
||||
pthread_mutex_lock(&v->writer_lock);
|
||||
MountSnapshot *old = atomic_load_explicit(&v->current, memory_order_acquire);
|
||||
pfs_usize idx = old->count;
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
if (strcmp(old->entries[i].prefix, norm) == 0) { idx = i; break; }
|
||||
}
|
||||
if (idx == old->count) { pthread_mutex_unlock(&v->writer_lock); return VFS_ERR_NOMOUNT; }
|
||||
|
||||
MountSnapshot *ns = (MountSnapshot *)calloc(1, sizeof(MountSnapshot));
|
||||
ns->count = old->count - 1;
|
||||
ns->entries = ns->count ? (MountEntry *)calloc(ns->count, sizeof(MountEntry)) : NULL;
|
||||
atomic_init(&ns->refcount, 1);
|
||||
pfs_usize w = 0;
|
||||
for (pfs_usize i = 0; i < old->count; i++) {
|
||||
if (i == idx) continue;
|
||||
ns->entries[w].prefix = strdup(old->entries[i].prefix);
|
||||
ns->entries[w].prefix_len = old->entries[i].prefix_len;
|
||||
ns->entries[w].backend = old->entries[i].backend;
|
||||
w++;
|
||||
}
|
||||
atomic_store_explicit(&v->current, ns, memory_order_release);
|
||||
pthread_mutex_unlock(&v->writer_lock);
|
||||
mount_retire(v, old);
|
||||
return VFS_OK;
|
||||
}
|
||||
|
||||
/* Longest-prefix match against the mount table (Section 2.2), reached
|
||||
* through the same atomic snapshot that protects the upper index
|
||||
* (Section 5.3): a concurrent vfs_mount/vfs_unmount is never observed
|
||||
* mid-change. */
|
||||
static Backend *resolve(Vfs *v, const char *path, char relpath[PFS_PATH_MAX],
|
||||
MountSnapshot **snap_out, int *err) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(path, norm) < 0) { if (err) *err = VFS_ERR_INVAL; return NULL; }
|
||||
|
||||
MountSnapshot *snap = mount_acquire(v);
|
||||
Backend *best = NULL;
|
||||
size_t best_len = 0;
|
||||
for (pfs_usize i = 0; i < snap->count; i++) {
|
||||
if (pfs_path_under(snap->entries[i].prefix, norm) && snap->entries[i].prefix_len >= best_len) {
|
||||
best = snap->entries[i].backend;
|
||||
best_len = snap->entries[i].prefix_len;
|
||||
}
|
||||
}
|
||||
if (!best) { mount_release(snap); if (err) *err = VFS_ERR_NOMOUNT; return NULL; }
|
||||
|
||||
if (best_len <= 1) {
|
||||
strncpy(relpath, norm, PFS_PATH_MAX - 1);
|
||||
relpath[PFS_PATH_MAX - 1] = '\0';
|
||||
} else {
|
||||
const char *r = norm + best_len;
|
||||
strncpy(relpath, (*r == '\0') ? "/" : r, PFS_PATH_MAX - 1);
|
||||
relpath[PFS_PATH_MAX - 1] = '\0';
|
||||
}
|
||||
*snap_out = snap;
|
||||
return best;
|
||||
}
|
||||
|
||||
VfsFile *vfs_open(Vfs *v, const char *path, int flags, int *err) {
|
||||
char rel[PFS_PATH_MAX];
|
||||
MountSnapshot *snap;
|
||||
int e = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &e);
|
||||
if (!b) { if (err) *err = e; return NULL; }
|
||||
VfsFile *f = NULL;
|
||||
int rc = b->ops->open(b, rel, flags, &f);
|
||||
mount_release(snap);
|
||||
if (rc != VFS_OK) { if (err) *err = rc; return NULL; }
|
||||
if (err) *err = VFS_OK;
|
||||
return f;
|
||||
}
|
||||
|
||||
pfs_isize vfs_read(VfsFile *f, void *buf, pfs_usize n) { return f->backend->ops->read(f, buf, n); }
|
||||
pfs_isize vfs_write(VfsFile *f, const void *buf, pfs_usize n) { return f->backend->ops->write(f, buf, n); }
|
||||
int vfs_close(VfsFile *f) { return f->backend->ops->close(f); }
|
||||
|
||||
int vfs_stat(Vfs *v, const char *path, VfsStat *out) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
int rc = b->ops->stat(b, rel, out);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_readdir(Vfs *v, const char *path, VfsDir *out) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
out->entries = NULL; out->count = 0;
|
||||
int rc = b->ops->readdir(b, rel, out);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
void vfs_dir_free(VfsDir *d) {
|
||||
if (!d) return;
|
||||
free(d->entries);
|
||||
d->entries = NULL;
|
||||
d->count = 0;
|
||||
}
|
||||
|
||||
int vfs_mkdir(Vfs *v, const char *path) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
int rc = b->ops->mkdir(b, rel);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_unlink(Vfs *v, const char *path) {
|
||||
char rel[PFS_PATH_MAX]; MountSnapshot *snap; int err = VFS_OK;
|
||||
Backend *b = resolve(v, path, rel, &snap, &err);
|
||||
if (!b) return err;
|
||||
int rc = b->ops->unlink(b, rel);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_rename(Vfs *v, const char *from, const char *to) {
|
||||
char relf[PFS_PATH_MAX], relt[PFS_PATH_MAX];
|
||||
MountSnapshot *sf, *st_; int err = VFS_OK;
|
||||
Backend *bf = resolve(v, from, relf, &sf, &err);
|
||||
if (!bf) return err;
|
||||
Backend *bt = resolve(v, to, relt, &st_, &err);
|
||||
if (!bt) { mount_release(sf); return err; }
|
||||
if (bf != bt) { mount_release(sf); mount_release(st_); return VFS_ERR_INVAL; /* cross-mount rename: v0 exclusion */ }
|
||||
int rc = bf->ops->rename(bf, relf, relt);
|
||||
mount_release(sf);
|
||||
mount_release(st_);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_sync(Vfs *v, const char *overlay_prefix) {
|
||||
char norm[PFS_PATH_MAX];
|
||||
if (pfs_path_normalize(overlay_prefix, norm) < 0) return VFS_ERR_INVAL;
|
||||
MountSnapshot *snap = mount_acquire(v);
|
||||
Backend *b = NULL;
|
||||
for (pfs_usize i = 0; i < snap->count; i++) {
|
||||
if (strcmp(snap->entries[i].prefix, norm) == 0) { b = snap->entries[i].backend; break; }
|
||||
}
|
||||
if (!b) { mount_release(snap); return VFS_ERR_NOMOUNT; }
|
||||
int rc = b->ops->sync(b);
|
||||
mount_release(snap);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int vfs_harden_process_with_landlock(Vfs *v) {
|
||||
MountSnapshot *snap = mount_acquire(v);
|
||||
int *fds = snap->count ? (int *)malloc(snap->count * sizeof(int)) : NULL;
|
||||
size_t n = 0;
|
||||
for (pfs_usize i = 0; i < snap->count; i++) {
|
||||
Backend *b = snap->entries[i].backend;
|
||||
Backend *inner = overlay_upper_backend(b);
|
||||
if (inner) b = inner;
|
||||
int fd = upper_backend_root_fd(b);
|
||||
if (fd >= 0) fds[n++] = fd;
|
||||
}
|
||||
int rc = (n > 0) ? pfs_landlock_restrict_to(fds, n) : -1;
|
||||
free(fds);
|
||||
mount_release(snap);
|
||||
return rc == 0 ? VFS_OK : VFS_ERR_PERM;
|
||||
}
|
||||
Reference in New Issue
Block a user