707 lines
26 KiB
C
707 lines
26 KiB
C
/*
|
|||
|
|
* upper.c — the writable upper layer shared by the `mem` and `dir`
|
||
|
|
* backends, and by the overlay's upper side (Section 3, Section 5).
|
||
|
|
*
|
||
|
|
* Implements the snapshot-based single-writer, wait-free-reader model of
|
||
|
|
* Section 5.3: an immutable, refcounted UpperSnapshot reached through one
|
||
|
|
* atomic pointer; structural writes (create/unlink/rename/mkdir/
|
||
|
|
* whiteout) build and publish a new snapshot under `writer_lock`;
|
||
|
|
* content writes (`upper_cell_write`) mutate a shared MutCell in place
|
||
|
|
* and never touch the snapshot pointer, per the structural/content split
|
||
|
|
* in Section 5.3. Buffer growth on the `mem` side follows the
|
||
|
|
* replace-don't-mutate discipline of Section 5.7.
|
||
|
|
*/
|
||
|
|
|
||
|
|
#include <errno.h>
|
||
|
|
#include <fcntl.h>
|
||
|
|
#include <stdlib.h>
|
||
|
|
#include <string.h>
|
||
|
|
#include <sys/stat.h>
|
||
|
|
#include <time.h>
|
||
|
|
#include <unistd.h>
|
||
|
|
|
||
|
|
#include "internal.h"
|
||
|
|
|
||
|
|
/* ---- MutCell lifetime: refcounted separately from any one snapshot,
|
||
|
|
* since a cell's identity is shared across every snapshot whose entry
|
||
|
|
* array still points at the same logical file (Section 5.3). ---- */
|
||
|
|
|
||
|
|
typedef struct MutCellRc {
|
||
|
|
MutCell pub;
|
||
|
|
_Atomic(pfs_usize) refcount;
|
||
|
|
} MutCellRc;
|
||
|
|
|
||
|
|
/* MutCell* handed around externally IS the MutCellRc's first member's
|
||
|
|
* address, so a plain cast recovers the refcount wrapper. */
|
||
|
|
static MutCellRc *rc_of(MutCell *c) { return (MutCellRc *)c; }
|
||
|
|
|
||
|
|
static void mutcell_ref(MutCell *c) {
|
||
|
|
if (!c) return;
|
||
|
|
atomic_fetch_add_explicit(&rc_of(c)->refcount, 1, memory_order_relaxed);
|
||
|
|
}
|
||
|
|
|
||
|
|
static void mutcell_unref(MutCell *c) {
|
||
|
|
if (!c) return;
|
||
|
|
if (atomic_fetch_sub_explicit(&rc_of(c)->refcount, 1, memory_order_acq_rel) == 1) {
|
||
|
|
pthread_mutex_destroy(&c->write_lock);
|
||
|
|
pthread_rwlock_destroy(&c->buf_lock);
|
||
|
|
free(c->data);
|
||
|
|
free(rc_of(c));
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
static MutCell *mutcell_new(void) {
|
||
|
|
MutCellRc *rc = (MutCellRc *)calloc(1, sizeof(MutCellRc));
|
||
|
|
pthread_mutex_init(&rc->pub.write_lock, NULL);
|
||
|
|
pthread_rwlock_init(&rc->pub.buf_lock, NULL);
|
||
|
|
atomic_init(&rc->pub.size, 0);
|
||
|
|
atomic_init(&rc->pub.mtime, (int64_t)time(NULL));
|
||
|
|
atomic_init(&rc->refcount, 1); /* owned by whichever snapshot installs it first */
|
||
|
|
return &rc->pub;
|
||
|
|
}
|
||
|
|
|
||
|
|
/* ---- UpperSnapshot lifetime ---- */
|
||
|
|
|
||
|
|
UpperSnapshot *upper_acquire(UpperStore *u) {
|
||
|
|
pthread_rwlock_rdlock(&u->reclaim_gate);
|
||
|
|
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
atomic_fetch_add_explicit(&s->refcount, 1, memory_order_relaxed);
|
||
|
|
pthread_rwlock_unlock(&u->reclaim_gate);
|
||
|
|
return s;
|
||
|
|
}
|
||
|
|
|
||
|
|
/* Retires a snapshot just superseded by a writer's atomic_store to
|
||
|
|
* u->current. Must run under reclaim_gate's write side so that no
|
||
|
|
* upper_acquire() can be mid-flight (loaded `old` but not yet
|
||
|
|
* incremented its refcount) when this potentially frees it. */
|
||
|
|
static void upper_retire(UpperStore *u, UpperSnapshot *old) {
|
||
|
|
pthread_rwlock_wrlock(&u->reclaim_gate);
|
||
|
|
upper_release(old);
|
||
|
|
pthread_rwlock_unlock(&u->reclaim_gate);
|
||
|
|
}
|
||
|
|
|
||
|
|
void upper_release(UpperSnapshot *s) {
|
||
|
|
if (!s) return;
|
||
|
|
if (atomic_fetch_sub_explicit(&s->refcount, 1, memory_order_acq_rel) == 1) {
|
||
|
|
for (pfs_usize i = 0; i < s->count; i++) {
|
||
|
|
free(s->entries[i].name);
|
||
|
|
mutcell_unref(s->entries[i].cell);
|
||
|
|
}
|
||
|
|
free(s->entries);
|
||
|
|
free(s);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_lookup(UpperSnapshot *s, const char *path, const UpperEntry **out) {
|
||
|
|
pfs_usize lo = 0, hi = s->count;
|
||
|
|
while (lo < hi) {
|
||
|
|
pfs_usize mid = lo + (hi - lo) / 2;
|
||
|
|
int c = strcmp(s->entries[mid].name, path);
|
||
|
|
if (c == 0) { *out = &s->entries[mid]; return 1; }
|
||
|
|
if (c < 0) lo = mid + 1; else hi = mid;
|
||
|
|
}
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_has_children(UpperSnapshot *s, const char *path) {
|
||
|
|
char prefix[PFS_PATH_MAX];
|
||
|
|
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
|
||
|
|
else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||
|
|
size_t plen = strlen(prefix);
|
||
|
|
for (pfs_usize i = 0; i < s->count; i++) {
|
||
|
|
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
|
||
|
|
if (strncmp(s->entries[i].name, prefix, plen) == 0) return 1;
|
||
|
|
}
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
/* Builds a new snapshot with one entry inserted or replaced at `name`. */
|
||
|
|
static UpperSnapshot *snapshot_upsert(UpperSnapshot *base, const char *name,
|
||
|
|
UpperEntryKind kind, MutCell *cell) {
|
||
|
|
pfs_usize old_count = base ? base->count : 0;
|
||
|
|
pfs_usize lo = 0, hi = old_count;
|
||
|
|
int found = 0;
|
||
|
|
while (lo < hi) {
|
||
|
|
pfs_usize mid = lo + (hi - lo) / 2;
|
||
|
|
int c = strcmp(base->entries[mid].name, name);
|
||
|
|
if (c == 0) { lo = mid; found = 1; break; }
|
||
|
|
if (c < 0) lo = mid + 1; else hi = mid;
|
||
|
|
}
|
||
|
|
pfs_usize new_count = found ? old_count : old_count + 1;
|
||
|
|
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||
|
|
ns->entries = (UpperEntry *)calloc(new_count ? new_count : 1, sizeof(UpperEntry));
|
||
|
|
ns->count = new_count;
|
||
|
|
atomic_init(&ns->refcount, 1);
|
||
|
|
|
||
|
|
pfs_usize w = 0;
|
||
|
|
for (pfs_usize i = 0; i < old_count; i++) {
|
||
|
|
if (i == lo) {
|
||
|
|
ns->entries[w].name = strdup(name);
|
||
|
|
ns->entries[w].kind = kind;
|
||
|
|
ns->entries[w].cell = cell;
|
||
|
|
mutcell_ref(cell);
|
||
|
|
w++;
|
||
|
|
if (found) continue; /* replaces base->entries[lo] */
|
||
|
|
}
|
||
|
|
ns->entries[w].name = strdup(base->entries[i].name);
|
||
|
|
ns->entries[w].kind = base->entries[i].kind;
|
||
|
|
ns->entries[w].cell = base->entries[i].cell;
|
||
|
|
mutcell_ref(ns->entries[w].cell);
|
||
|
|
w++;
|
||
|
|
}
|
||
|
|
if (lo == old_count) {
|
||
|
|
ns->entries[w].name = strdup(name);
|
||
|
|
ns->entries[w].kind = kind;
|
||
|
|
ns->entries[w].cell = cell;
|
||
|
|
mutcell_ref(cell);
|
||
|
|
w++;
|
||
|
|
}
|
||
|
|
return ns;
|
||
|
|
}
|
||
|
|
|
||
|
|
static UpperSnapshot *snapshot_remove(UpperSnapshot *base, const char *name) {
|
||
|
|
pfs_usize lo = 0, hi = base->count;
|
||
|
|
int found = 0;
|
||
|
|
while (lo < hi) {
|
||
|
|
pfs_usize mid = lo + (hi - lo) / 2;
|
||
|
|
int c = strcmp(base->entries[mid].name, name);
|
||
|
|
if (c == 0) { lo = mid; found = 1; break; }
|
||
|
|
if (c < 0) lo = mid + 1; else hi = mid;
|
||
|
|
}
|
||
|
|
if (!found) return NULL;
|
||
|
|
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||
|
|
ns->count = base->count - 1;
|
||
|
|
ns->entries = ns->count ? (UpperEntry *)calloc(ns->count, sizeof(UpperEntry)) : NULL;
|
||
|
|
atomic_init(&ns->refcount, 1);
|
||
|
|
pfs_usize w = 0;
|
||
|
|
for (pfs_usize i = 0; i < base->count; i++) {
|
||
|
|
if (i == lo) continue;
|
||
|
|
ns->entries[w].name = strdup(base->entries[i].name);
|
||
|
|
ns->entries[w].kind = base->entries[i].kind;
|
||
|
|
ns->entries[w].cell = base->entries[i].cell;
|
||
|
|
mutcell_ref(ns->entries[w].cell);
|
||
|
|
w++;
|
||
|
|
}
|
||
|
|
return ns;
|
||
|
|
}
|
||
|
|
|
||
|
|
UpperStore *upper_new(UpperKind kind, int root_fd, const char *root_path) {
|
||
|
|
UpperStore *u = (UpperStore *)calloc(1, sizeof(UpperStore));
|
||
|
|
u->kind = kind;
|
||
|
|
u->root_fd = root_fd;
|
||
|
|
u->root_path = root_path ? strdup(root_path) : NULL;
|
||
|
|
pthread_mutex_init(&u->writer_lock, NULL);
|
||
|
|
pthread_rwlock_init(&u->reclaim_gate, NULL);
|
||
|
|
UpperSnapshot *s = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||
|
|
atomic_init(&s->refcount, 1);
|
||
|
|
atomic_init(&u->current, s);
|
||
|
|
return u;
|
||
|
|
}
|
||
|
|
|
||
|
|
void upper_free(UpperStore *u) {
|
||
|
|
if (!u) return;
|
||
|
|
UpperSnapshot *s = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
upper_release(s);
|
||
|
|
pthread_mutex_destroy(&u->writer_lock);
|
||
|
|
pthread_rwlock_destroy(&u->reclaim_gate);
|
||
|
|
if (u->kind == UPPER_DIR && u->root_fd >= 0) close(u->root_fd);
|
||
|
|
free(u->root_path);
|
||
|
|
free(u);
|
||
|
|
}
|
||
|
|
|
||
|
|
/* strips the mount-relative leading '/' for host syscalls */
|
||
|
|
static const char *rel(const char *path) { return path[0] == '/' ? path + 1 : path; }
|
||
|
|
|
||
|
|
int upper_create(UpperStore *u, const char *path, int truncate_existing, MutCell **cell_out, int *err) {
|
||
|
|
(void)truncate_existing;
|
||
|
|
pthread_mutex_lock(&u->writer_lock);
|
||
|
|
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
const UpperEntry *existing;
|
||
|
|
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
|
||
|
|
MutCell *c = existing->cell;
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
*cell_out = c;
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
if (u->kind == UPPER_DIR) {
|
||
|
|
int e = 0;
|
||
|
|
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
|
||
|
|
if (fd < 0) { pthread_mutex_unlock(&u->writer_lock); if (err) *err = e; return -1; }
|
||
|
|
close(fd);
|
||
|
|
}
|
||
|
|
MutCell *cell = mutcell_new();
|
||
|
|
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
|
||
|
|
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
upper_retire(u, old);
|
||
|
|
mutcell_unref(cell); /* drop the creation-local ref; ns holds its own */
|
||
|
|
*cell_out = cell;
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_mkdir(UpperStore *u, const char *path, int *err) {
|
||
|
|
pthread_mutex_lock(&u->writer_lock);
|
||
|
|
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
const UpperEntry *existing;
|
||
|
|
if (upper_lookup(old, path, &existing) || upper_has_children(old, path)) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = VFS_ERR_EXIST;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
if (u->kind == UPPER_DIR) {
|
||
|
|
int e = 0;
|
||
|
|
if (pfs_dir_mkdirat(u->root_fd, rel(path), &e) < 0) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = e;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_DIR_MARKER, NULL);
|
||
|
|
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
upper_retire(u, old);
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_remove(UpperStore *u, const char *path, int had_lower, int *err) {
|
||
|
|
pthread_mutex_lock(&u->writer_lock);
|
||
|
|
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
const UpperEntry *existing;
|
||
|
|
int have = upper_lookup(old, path, &existing);
|
||
|
|
if ((!have || existing->kind == ENTRY_WHITEOUT) && !had_lower) {
|
||
|
|
/* No explicit entry anywhere: still an error to distinguish
|
||
|
|
* "doesn't exist" from "exists only implicitly, via children,
|
||
|
|
* and therefore can't be removed" (Section 9.3). */
|
||
|
|
int kids = upper_has_children(old, path);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = kids ? VFS_ERR_NOTEMPTY : VFS_ERR_NOENT;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
if (have && existing->kind == ENTRY_DIR_MARKER && upper_has_children(old, path)) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = VFS_ERR_NOTEMPTY;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
if (u->kind == UPPER_DIR && have && existing->kind != ENTRY_WHITEOUT) {
|
||
|
|
int e = 0;
|
||
|
|
int is_dir = (existing->kind == ENTRY_DIR_MARKER);
|
||
|
|
if (pfs_dir_unlinkat(u->root_fd, rel(path), is_dir, &e) < 0) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = e;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
UpperSnapshot *ns = had_lower ? snapshot_upsert(old, path, ENTRY_WHITEOUT, NULL)
|
||
|
|
: snapshot_remove(old, path);
|
||
|
|
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
upper_retire(u, old);
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_rename(UpperStore *u, const char *from, const char *to, int had_lower_from, int *err) {
|
||
|
|
pthread_mutex_lock(&u->writer_lock);
|
||
|
|
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
const UpperEntry *src;
|
||
|
|
if (!upper_lookup(old, from, &src) || src->kind == ENTRY_WHITEOUT) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = VFS_ERR_NOENT;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
if (u->kind == UPPER_DIR) {
|
||
|
|
int e = 0;
|
||
|
|
if (pfs_dir_renameat(u->root_fd, rel(from), rel(to), &e) < 0) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = e;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
UpperSnapshot *step1 = snapshot_upsert(old, to, src->kind, src->cell);
|
||
|
|
UpperSnapshot *step2 = had_lower_from ? snapshot_upsert(step1, from, ENTRY_WHITEOUT, NULL)
|
||
|
|
: snapshot_remove(step1, from);
|
||
|
|
atomic_store_explicit(&u->current, step2, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
upper_retire(u, old);
|
||
|
|
upper_release(step1); /* never published; drop our local build reference */
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_copy_up(UpperStore *u, const char *path, const void *data, uint64_t size,
|
||
|
|
int64_t mtime, MutCell **cell_out, int *err) {
|
||
|
|
pthread_mutex_lock(&u->writer_lock);
|
||
|
|
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
const UpperEntry *existing;
|
||
|
|
if (upper_lookup(old, path, &existing) && existing->kind == ENTRY_FILE) {
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
*cell_out = existing->cell;
|
||
|
|
return 0; /* raced with another copy-up; use what's already there */
|
||
|
|
}
|
||
|
|
MutCell *cell = mutcell_new();
|
||
|
|
atomic_store_explicit(&cell->mtime, mtime, memory_order_relaxed);
|
||
|
|
|
||
|
|
if (u->kind == UPPER_MEM) {
|
||
|
|
if (size) {
|
||
|
|
cell->data = (unsigned char *)malloc((size_t)size);
|
||
|
|
memcpy(cell->data, data, (size_t)size);
|
||
|
|
}
|
||
|
|
cell->capacity = (pfs_usize)size;
|
||
|
|
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
|
||
|
|
} else {
|
||
|
|
int e = 0;
|
||
|
|
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_CREAT | O_TRUNC, 0644, &e);
|
||
|
|
if (fd < 0) {
|
||
|
|
mutcell_unref(cell);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = e;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
size_t written = 0;
|
||
|
|
const unsigned char *p = (const unsigned char *)data;
|
||
|
|
int failed = 0;
|
||
|
|
while (written < (size_t)size) {
|
||
|
|
ssize_t w = write(fd, p + written, (size_t)size - written);
|
||
|
|
if (w < 0) { failed = 1; break; }
|
||
|
|
written += (size_t)w;
|
||
|
|
}
|
||
|
|
close(fd);
|
||
|
|
if (failed) {
|
||
|
|
mutcell_unref(cell);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
if (err) *err = VFS_ERR_IO;
|
||
|
|
return -1;
|
||
|
|
}
|
||
|
|
atomic_store_explicit(&cell->size, (pfs_usize)size, memory_order_relaxed);
|
||
|
|
}
|
||
|
|
|
||
|
|
UpperSnapshot *ns = snapshot_upsert(old, path, ENTRY_FILE, cell);
|
||
|
|
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
upper_retire(u, old);
|
||
|
|
mutcell_unref(cell);
|
||
|
|
*cell_out = cell;
|
||
|
|
return 0;
|
||
|
|
}
|
||
|
|
|
||
|
|
/* ---- content operations (Section 5.3 fast path, Section 5.7 growth) ---- */
|
||
|
|
|
||
|
|
pfs_isize upper_cell_read(UpperStore *u, MutCell *c, const char *path, pfs_usize off, void *buf, pfs_usize n) {
|
||
|
|
if (u->kind == UPPER_MEM) {
|
||
|
|
pthread_rwlock_rdlock(&c->buf_lock);
|
||
|
|
pfs_usize size = atomic_load_explicit(&c->size, memory_order_acquire);
|
||
|
|
pfs_usize avail = off < size ? size - off : 0;
|
||
|
|
pfs_usize to_copy = n < avail ? n : avail;
|
||
|
|
if (to_copy) memcpy(buf, c->data + off, to_copy);
|
||
|
|
pthread_rwlock_unlock(&c->buf_lock);
|
||
|
|
return (pfs_isize)to_copy;
|
||
|
|
}
|
||
|
|
int e = 0;
|
||
|
|
int fd = pfs_dir_openat(u->root_fd, rel(path), O_RDONLY, 0, &e);
|
||
|
|
if (fd < 0) return -1;
|
||
|
|
ssize_t r = pread(fd, buf, n, (off_t)off);
|
||
|
|
close(fd);
|
||
|
|
return (pfs_isize)r;
|
||
|
|
}
|
||
|
|
|
||
|
|
pfs_isize upper_cell_write(UpperStore *u, MutCell *c, const char *path, pfs_usize off,
|
||
|
|
const void *buf, pfs_usize n, int *err) {
|
||
|
|
int64_t now = (int64_t)time(NULL);
|
||
|
|
if (u->kind == UPPER_MEM) {
|
||
|
|
pthread_mutex_lock(&c->write_lock);
|
||
|
|
pfs_usize cur_size = atomic_load_explicit(&c->size, memory_order_relaxed);
|
||
|
|
pfs_usize need = off + n;
|
||
|
|
if (need > c->capacity) {
|
||
|
|
pfs_usize newcap = c->capacity ? c->capacity : 64;
|
||
|
|
while (newcap < need) newcap *= 2;
|
||
|
|
unsigned char *nb = (unsigned char *)calloc(1, newcap);
|
||
|
|
if (!nb) { pthread_mutex_unlock(&c->write_lock); if (err) *err = VFS_ERR_NOSPC; return -1; }
|
||
|
|
if (c->data && cur_size) memcpy(nb, c->data, cur_size);
|
||
|
|
if (n) memcpy(nb + off, buf, n);
|
||
|
|
unsigned char *old = c->data;
|
||
|
|
/* Section 5.7: publish the new buffer via the rwlock, then
|
||
|
|
* free the old one only after releasing the write lock on
|
||
|
|
* it — any reader that could still see `old` must have
|
||
|
|
* taken buf_lock (rdlock) before this wrlock was granted,
|
||
|
|
* and rdlock/wrlock are mutually exclusive, so it has
|
||
|
|
* already finished copying by the time we get here. */
|
||
|
|
pthread_rwlock_wrlock(&c->buf_lock);
|
||
|
|
c->data = nb;
|
||
|
|
c->capacity = newcap;
|
||
|
|
pthread_rwlock_unlock(&c->buf_lock);
|
||
|
|
free(old);
|
||
|
|
} else if (n) {
|
||
|
|
memcpy(c->data + off, buf, n);
|
||
|
|
}
|
||
|
|
pfs_usize new_size = need > cur_size ? need : cur_size;
|
||
|
|
atomic_store_explicit(&c->size, new_size, memory_order_release);
|
||
|
|
atomic_store_explicit(&c->mtime, now, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&c->write_lock);
|
||
|
|
return (pfs_isize)n;
|
||
|
|
}
|
||
|
|
pthread_mutex_lock(&c->write_lock);
|
||
|
|
int e = 0;
|
||
|
|
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY, 0, &e);
|
||
|
|
if (fd < 0) { pthread_mutex_unlock(&c->write_lock); if (err) *err = e; return -1; }
|
||
|
|
ssize_t w = pwrite(fd, buf, n, (off_t)off);
|
||
|
|
close(fd);
|
||
|
|
pthread_mutex_unlock(&c->write_lock);
|
||
|
|
if (w < 0) { if (err) *err = VFS_ERR_IO; return -1; }
|
||
|
|
return (pfs_isize)w;
|
||
|
|
}
|
||
|
|
|
||
|
|
void upper_cell_truncate(UpperStore *u, MutCell *c, const char *path) {
|
||
|
|
if (u->kind == UPPER_MEM) {
|
||
|
|
pthread_mutex_lock(&c->write_lock);
|
||
|
|
atomic_store_explicit(&c->size, 0, memory_order_release);
|
||
|
|
atomic_store_explicit(&c->mtime, (int64_t)time(NULL), memory_order_release);
|
||
|
|
pthread_mutex_unlock(&c->write_lock);
|
||
|
|
} else {
|
||
|
|
pthread_mutex_lock(&c->write_lock);
|
||
|
|
int e = 0;
|
||
|
|
int fd = pfs_dir_openat(u->root_fd, rel(path), O_WRONLY | O_TRUNC, 0644, &e);
|
||
|
|
if (fd >= 0) close(fd);
|
||
|
|
pthread_mutex_unlock(&c->write_lock);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
/* ---- compaction support ---- */
|
||
|
|
|
||
|
|
void upper_walk_live(UpperStore *u, const UpperWalkCb *cb) {
|
||
|
|
UpperSnapshot *s = upper_acquire(u);
|
||
|
|
for (pfs_usize i = 0; i < s->count; i++) {
|
||
|
|
UpperEntry *e = &s->entries[i];
|
||
|
|
if (e->kind == ENTRY_WHITEOUT) continue;
|
||
|
|
if (e->kind == ENTRY_DIR_MARKER) {
|
||
|
|
cb->visit(cb->ctx, e->name, ENTRY_DIR_MARKER, NULL, 0, 0);
|
||
|
|
continue;
|
||
|
|
}
|
||
|
|
if (u->kind == UPPER_MEM) {
|
||
|
|
pthread_rwlock_rdlock(&e->cell->buf_lock);
|
||
|
|
pfs_usize size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
|
||
|
|
cb->visit(cb->ctx, e->name, ENTRY_FILE, e->cell->data, size,
|
||
|
|
atomic_load_explicit(&e->cell->mtime, memory_order_acquire));
|
||
|
|
pthread_rwlock_unlock(&e->cell->buf_lock);
|
||
|
|
} else {
|
||
|
|
int err = 0;
|
||
|
|
int fd = pfs_dir_openat(u->root_fd, rel(e->name), O_RDONLY, 0, &err);
|
||
|
|
if (fd < 0) continue;
|
||
|
|
struct stat st;
|
||
|
|
fstat(fd, &st);
|
||
|
|
void *buf = st.st_size ? malloc((size_t)st.st_size) : NULL;
|
||
|
|
size_t total = 0;
|
||
|
|
while (buf && total < (size_t)st.st_size) {
|
||
|
|
ssize_t r = read(fd, (char *)buf + total, (size_t)st.st_size - total);
|
||
|
|
if (r <= 0) break;
|
||
|
|
total += (size_t)r;
|
||
|
|
}
|
||
|
|
close(fd);
|
||
|
|
cb->visit(cb->ctx, e->name, ENTRY_FILE, buf, total, (int64_t)st.st_mtime);
|
||
|
|
free(buf);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
upper_release(s);
|
||
|
|
}
|
||
|
|
|
||
|
|
void upper_reset_empty(UpperStore *u) {
|
||
|
|
pthread_mutex_lock(&u->writer_lock);
|
||
|
|
UpperSnapshot *old = atomic_load_explicit(&u->current, memory_order_acquire);
|
||
|
|
UpperSnapshot *ns = (UpperSnapshot *)calloc(1, sizeof(UpperSnapshot));
|
||
|
|
atomic_init(&ns->refcount, 1);
|
||
|
|
atomic_store_explicit(&u->current, ns, memory_order_release);
|
||
|
|
pthread_mutex_unlock(&u->writer_lock);
|
||
|
|
upper_retire(u, old);
|
||
|
|
}
|
||
|
|
|
||
|
|
/* ---- standalone mem/dir Backend wrapper (no lower layer at all) ---- */
|
||
|
|
|
||
|
|
typedef struct UpperFile {
|
||
|
|
UpperStore *store;
|
||
|
|
MutCell *cell;
|
||
|
|
char *path;
|
||
|
|
pfs_usize pos;
|
||
|
|
} UpperFile;
|
||
|
|
|
||
|
|
static int upstd_open(Backend *b, const char *path, int flags, VfsFile **out) {
|
||
|
|
UpperStore *u = (UpperStore *)b->state;
|
||
|
|
UpperSnapshot *s = upper_acquire(u);
|
||
|
|
const UpperEntry *e;
|
||
|
|
int found = upper_lookup(s, path, &e);
|
||
|
|
if (found && e->kind == ENTRY_DIR_MARKER) { upper_release(s); return VFS_ERR_ISDIR; }
|
||
|
|
|
||
|
|
MutCell *cell;
|
||
|
|
if (found) {
|
||
|
|
cell = e->cell;
|
||
|
|
mutcell_ref(cell);
|
||
|
|
upper_release(s);
|
||
|
|
if (flags & VFS_O_TRUNC) upper_cell_truncate(u, cell, path);
|
||
|
|
} else {
|
||
|
|
upper_release(s);
|
||
|
|
if (!(flags & VFS_O_CREAT)) return VFS_ERR_NOENT;
|
||
|
|
int err = 0;
|
||
|
|
if (upper_create(u, path, 0, &cell, &err) < 0) return err;
|
||
|
|
mutcell_ref(cell);
|
||
|
|
}
|
||
|
|
|
||
|
|
UpperFile *uf = (UpperFile *)calloc(1, sizeof(UpperFile));
|
||
|
|
uf->store = u; uf->cell = cell; uf->path = strdup(path); uf->pos = 0;
|
||
|
|
VfsFile *f = (VfsFile *)calloc(1, sizeof(VfsFile));
|
||
|
|
f->backend = b; f->state = uf;
|
||
|
|
*out = f;
|
||
|
|
return VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static pfs_isize upstd_read(VfsFile *f, void *buf, pfs_usize n) {
|
||
|
|
UpperFile *uf = (UpperFile *)f->state;
|
||
|
|
pfs_isize r = upper_cell_read(uf->store, uf->cell, uf->path, uf->pos, buf, n);
|
||
|
|
if (r > 0) uf->pos += (pfs_usize)r;
|
||
|
|
return r;
|
||
|
|
}
|
||
|
|
|
||
|
|
static pfs_isize upstd_write(VfsFile *f, const void *buf, pfs_usize n) {
|
||
|
|
UpperFile *uf = (UpperFile *)f->state;
|
||
|
|
int err = 0;
|
||
|
|
pfs_isize w = upper_cell_write(uf->store, uf->cell, uf->path, uf->pos, buf, n, &err);
|
||
|
|
if (w >= 0) { uf->pos += (pfs_usize)w; return w; }
|
||
|
|
return err;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_close(VfsFile *f) {
|
||
|
|
UpperFile *uf = (UpperFile *)f->state;
|
||
|
|
mutcell_unref(uf->cell);
|
||
|
|
free(uf->path);
|
||
|
|
free(uf);
|
||
|
|
free(f);
|
||
|
|
return VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_stat(Backend *b, const char *path, VfsStat *out) {
|
||
|
|
UpperStore *u = (UpperStore *)b->state;
|
||
|
|
UpperSnapshot *s = upper_acquire(u);
|
||
|
|
const UpperEntry *e;
|
||
|
|
if (!upper_lookup(s, path, &e) || e->kind == ENTRY_WHITEOUT) {
|
||
|
|
if (upper_has_children(s, path)) {
|
||
|
|
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||
|
|
upper_release(s);
|
||
|
|
return VFS_OK;
|
||
|
|
}
|
||
|
|
upper_release(s);
|
||
|
|
return VFS_ERR_NOENT;
|
||
|
|
}
|
||
|
|
if (e->kind == ENTRY_DIR_MARKER) {
|
||
|
|
out->size = 0; out->mtime = 0; out->kind = VFS_KIND_DIR;
|
||
|
|
upper_release(s);
|
||
|
|
return VFS_OK;
|
||
|
|
}
|
||
|
|
if (u->kind == UPPER_MEM) {
|
||
|
|
out->size = atomic_load_explicit(&e->cell->size, memory_order_acquire);
|
||
|
|
out->mtime = atomic_load_explicit(&e->cell->mtime, memory_order_acquire);
|
||
|
|
out->kind = VFS_KIND_FILE;
|
||
|
|
upper_release(s);
|
||
|
|
return VFS_OK;
|
||
|
|
}
|
||
|
|
int err = 0;
|
||
|
|
int rc = pfs_dir_statat(u->root_fd, rel(path), out, &err);
|
||
|
|
upper_release(s);
|
||
|
|
return rc < 0 ? err : VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_readdir(Backend *b, const char *path, VfsDir *out) {
|
||
|
|
UpperStore *u = (UpperStore *)b->state;
|
||
|
|
UpperSnapshot *s = upper_acquire(u);
|
||
|
|
|
||
|
|
const UpperEntry *self;
|
||
|
|
int self_found = upper_lookup(s, path, &self);
|
||
|
|
int has_kids = upper_has_children(s, path);
|
||
|
|
if (!has_kids && !(self_found && self->kind == ENTRY_DIR_MARKER) && strcmp(path, "/") != 0) {
|
||
|
|
upper_release(s);
|
||
|
|
return self_found ? VFS_ERR_NOTDIR : VFS_ERR_NOENT;
|
||
|
|
}
|
||
|
|
|
||
|
|
char prefix[PFS_PATH_MAX];
|
||
|
|
if (strcmp(path, "/") == 0) strcpy(prefix, "/");
|
||
|
|
else snprintf(prefix, sizeof(prefix), "%s/", path);
|
||
|
|
size_t plen = strlen(prefix);
|
||
|
|
|
||
|
|
VfsDirEntry *entries = NULL;
|
||
|
|
pfs_usize count = 0, cap = 0;
|
||
|
|
for (pfs_usize i = 0; i < s->count; i++) {
|
||
|
|
if (s->entries[i].kind == ENTRY_WHITEOUT) continue;
|
||
|
|
const char *name = s->entries[i].name;
|
||
|
|
if (strncmp(name, prefix, plen) != 0) continue;
|
||
|
|
const char *restp = name + plen;
|
||
|
|
const char *slash = strchr(restp, '/');
|
||
|
|
size_t clen = slash ? (size_t)(slash - restp) : strlen(restp);
|
||
|
|
if (clen == 0 || clen >= sizeof(entries[0].name)) continue;
|
||
|
|
if (count > 0 && strncmp(entries[count - 1].name, restp, clen) == 0 &&
|
||
|
|
entries[count - 1].name[clen] == '\0') continue; /* sorted -> dup is adjacent */
|
||
|
|
if (count == cap) {
|
||
|
|
cap = cap ? cap * 2 : 8;
|
||
|
|
entries = (VfsDirEntry *)realloc(entries, cap * sizeof(VfsDirEntry));
|
||
|
|
}
|
||
|
|
memcpy(entries[count].name, restp, clen);
|
||
|
|
entries[count].name[clen] = '\0';
|
||
|
|
entries[count].kind = slash ? VFS_KIND_DIR
|
||
|
|
: (s->entries[i].kind == ENTRY_DIR_MARKER ? VFS_KIND_DIR : VFS_KIND_FILE);
|
||
|
|
count++;
|
||
|
|
}
|
||
|
|
upper_release(s);
|
||
|
|
out->entries = entries;
|
||
|
|
out->count = count;
|
||
|
|
return VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_mkdir(Backend *b, const char *path) {
|
||
|
|
int err = 0;
|
||
|
|
return upper_mkdir((UpperStore *)b->state, path, &err) < 0 ? err : VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_unlink(Backend *b, const char *path) {
|
||
|
|
int err = 0;
|
||
|
|
return upper_remove((UpperStore *)b->state, path, 0, &err) < 0 ? err : VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_rename(Backend *b, const char *from, const char *to) {
|
||
|
|
int err = 0;
|
||
|
|
return upper_rename((UpperStore *)b->state, from, to, 0, &err) < 0 ? err : VFS_OK;
|
||
|
|
}
|
||
|
|
|
||
|
|
static int upstd_sync(Backend *b) { (void)b; return VFS_ERR_PERM; }
|
||
|
|
|
||
|
|
static void upstd_free(Backend *b) {
|
||
|
|
upper_free((UpperStore *)b->state);
|
||
|
|
free(b);
|
||
|
|
}
|
||
|
|
|
||
|
|
static const BackendOps UPPER_STANDALONE_OPS = {
|
||
|
|
upstd_open, upstd_read, upstd_write, upstd_close, upstd_stat,
|
||
|
|
upstd_readdir, upstd_mkdir, upstd_unlink, upstd_rename, upstd_sync, upstd_free
|
||
|
|
};
|
||
|
|
|
||
|
|
Backend *backend_from_upper(UpperStore *u) {
|
||
|
|
Backend *b = (Backend *)calloc(1, sizeof(Backend));
|
||
|
|
b->ops = &UPPER_STANDALONE_OPS;
|
||
|
|
b->state = u;
|
||
|
|
return b;
|
||
|
|
}
|
||
|
|
|
||
|
|
Backend *backend_mem_new(void) {
|
||
|
|
return backend_from_upper(upper_new(UPPER_MEM, -1, NULL));
|
||
|
|
}
|
||
|
|
|
||
|
|
Backend *backend_dir_new(const char *host_path, int *err) {
|
||
|
|
int e = 0;
|
||
|
|
int fd = pfs_dir_capability_open(host_path, &e);
|
||
|
|
if (fd < 0) { if (err) *err = e; return NULL; }
|
||
|
|
return backend_from_upper(upper_new(UPPER_DIR, fd, host_path));
|
||
|
|
}
|
||
|
|
|
||
|
|
int upper_backend_root_fd(Backend *b) {
|
||
|
|
if (b->ops != &UPPER_STANDALONE_OPS) return -1;
|
||
|
|
UpperStore *u = (UpperStore *)b->state;
|
||
|
|
return u->kind == UPPER_DIR ? u->root_fd : -1;
|
||
|
|
}
|
||
|
|
|
||
|
|
void backend_free(Backend *b) {
|
||
|
|
if (!b) return;
|
||
|
|
b->ops->free(b);
|
||
|
|
}
|