Project-hygiene pass toward being a properly citable, embeddable, professionally-packaged C library rather than just working code: - PACKFS_VERSION_MAJOR/MINOR/PATCH/STRING in include/packfs.h, the single source of truth for the project's version, plus a runtime pfs_version() (src/vfs.c, next to vfs_new/vfs_free) so a dynamically-linked consumer can check ABI/API compatibility without recompiling. Covered by a new assertion in tests/test_mem.c that the macro and the runtime function never disagree. - packfs.pc.in + a `make install` rule that generates packfs.pc with its Version: field derived from PACKFS_VERSION_STRING via a Makefile-level grep/sed, never hand-maintained separately -- verified end-to-end with a scratch `make install PREFIX=...` + `pkg-config --cflags --libs packfs` + `make uninstall`, not just by reading the rule. - SPDX-License-Identifier: MIT added to every src/*.c and src/internal.h (include/packfs.h already had one); the whole distributed source tree now carries consistent machine-readable license metadata. - SECURITY.md, stating precisely what this project's containment and pack- integrity code actually claims as a security boundary (concept.md Section 6/7) versus what it explicitly does not (unenforced `mode`, no cross-process concurrency) -- not generic boilerplate -- with a real reporting contact rather than a placeholder. - CHANGELOG.md (Keep a Changelog format), summarizing the real history in git log to date; explicitly notes no version is tagged yet. Verified: clean `make all` + `make test` (all 6 binaries, including the new version-check assertion), and a full ASan/UBSan sweep of all 6 binaries with zero real findings (one run hit the already-documented DEADLYSIGNAL sandbox flake on test_pack_write_perf across all 3 retries; re-verified directly afterward with 5/5 additional clean passes and timing well under any plausible timeout, confirming it was the flake, not a regression, before treating this as done). Deliberately not done here, by the user's explicit choice: no CODE_OF_CONDUCT.md, and no git remote/publishing -- this repository still has neither, and both are decisions left to the maintainer. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UqJpkdJ6Njnt1pw3CbghzB
256 lines
8.6 KiB
C
256 lines
8.6 KiB
C
/*
|
|
* containment.c — capability-scoped `dir` mounts (Section 6.2) and
|
|
* optional Landlock hardening (Section 6.2, 6.3).
|
|
*
|
|
* On Linux 5.6+, every lookup beneath a `dir` mount's root fd is resolved
|
|
* with openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS) in a single kernel
|
|
* call, atomically rejecting ".." escapes and symlink escapes. Where
|
|
* openat2 is unavailable (ENOSYS on an older kernel), this falls back to
|
|
* per-component O_NOFOLLOW resolution, which is an accepted, weaker
|
|
* residual-risk posture per Section 6.3 — not a claim of equal
|
|
* containment strength.
|
|
*
|
|
* SPDX-License-Identifier: MIT
|
|
*/
|
|
|
|
#include <errno.h>
|
|
#include <fcntl.h>
|
|
#include <linux/openat2.h>
|
|
#include <string.h>
|
|
#include <sys/prctl.h>
|
|
#include <sys/stat.h>
|
|
#include <sys/syscall.h>
|
|
#include <unistd.h>
|
|
|
|
#include "internal.h"
|
|
|
|
#ifdef __linux__
|
|
#include <linux/landlock.h>
|
|
#endif
|
|
|
|
int pfs_dir_capability_open(const char *path, int *err) {
|
|
int fd = open(path, O_DIRECTORY | O_CLOEXEC | O_RDONLY);
|
|
if (fd < 0) {
|
|
if (err) *err = VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
return fd;
|
|
}
|
|
|
|
/* Splits an already-normalized relative path (no leading slash) into
|
|
* (parent, leaf). Root-level names have an empty parent. */
|
|
static void split_parent_leaf(const char *rel, char *parent_buf, size_t cap, const char **leaf) {
|
|
const char *slash = strrchr(rel, '/');
|
|
if (!slash) {
|
|
parent_buf[0] = '\0';
|
|
*leaf = rel;
|
|
return;
|
|
}
|
|
size_t plen = (size_t)(slash - rel);
|
|
if (plen >= cap) plen = cap - 1;
|
|
memcpy(parent_buf, rel, plen);
|
|
parent_buf[plen] = '\0';
|
|
*leaf = slash + 1;
|
|
}
|
|
|
|
#ifdef SYS_openat2
|
|
static int openat2_beneath(int root_fd, const char *rel, uint64_t flags, uint64_t mode) {
|
|
struct open_how how;
|
|
memset(&how, 0, sizeof(how));
|
|
how.flags = flags;
|
|
how.mode = mode;
|
|
how.resolve = RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS;
|
|
return (int)syscall(SYS_openat2, root_fd, rel[0] ? rel : ".", &how, sizeof(how));
|
|
}
|
|
#endif
|
|
|
|
/* Per-component O_NOFOLLOW fallback (Section 6.3). Walks every component
|
|
* except the last with O_NOFOLLOW|O_DIRECTORY, then opens the last with
|
|
* the caller's requested flags plus O_NOFOLLOW. Closes most, not all, of
|
|
* the same TOCTOU window openat2 closes atomically. */
|
|
static int fallback_openat_beneath(int root_fd, const char *rel, int flags, unsigned mode) {
|
|
if (rel[0] == '\0') {
|
|
return dup(root_fd);
|
|
}
|
|
char buf[PFS_PATH_MAX];
|
|
if (strlen(rel) >= sizeof(buf)) { errno = ENAMETOOLONG; return -1; }
|
|
strcpy(buf, rel);
|
|
|
|
int cur = root_fd;
|
|
int owns_cur = 0;
|
|
char *save = NULL;
|
|
char *tok = strtok_r(buf, "/", &save);
|
|
char *next = tok ? strtok_r(NULL, "/", &save) : NULL;
|
|
|
|
while (tok && next) {
|
|
int nfd = openat(cur, tok, O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC);
|
|
if (owns_cur) close(cur);
|
|
if (nfd < 0) return -1;
|
|
cur = nfd;
|
|
owns_cur = 1;
|
|
tok = next;
|
|
next = strtok_r(NULL, "/", &save);
|
|
}
|
|
|
|
int final_fd = -1;
|
|
if (tok) {
|
|
int final_flags = flags;
|
|
if (!(flags & O_CREAT)) final_flags |= O_NOFOLLOW;
|
|
final_fd = openat(cur, tok, final_flags, mode);
|
|
}
|
|
if (owns_cur) close(cur);
|
|
return final_fd;
|
|
}
|
|
|
|
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err) {
|
|
int fd;
|
|
#ifdef SYS_openat2
|
|
fd = openat2_beneath(root_fd, rel, (uint64_t)flags, (uint64_t)mode);
|
|
if (fd < 0 && errno == ENOSYS) {
|
|
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
|
|
}
|
|
#else
|
|
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
|
|
#endif
|
|
if (fd < 0) {
|
|
if (err) *err = (errno == EXDEV || errno == ELOOP) ? VFS_ERR_INVAL
|
|
: (errno == ENOENT) ? VFS_ERR_NOENT
|
|
: (errno == EEXIST) ? VFS_ERR_EXIST
|
|
: VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
return fd;
|
|
}
|
|
|
|
static int resolve_parent_dir(int root_fd, const char *rel, char *leaf_out, size_t leaf_cap, int *err) {
|
|
char parent[PFS_PATH_MAX];
|
|
const char *leaf;
|
|
split_parent_leaf(rel, parent, sizeof(parent), &leaf);
|
|
if (strlen(leaf) >= leaf_cap) { if (err) *err = VFS_ERR_INVAL; return -1; }
|
|
strcpy(leaf_out, leaf);
|
|
|
|
int pfd;
|
|
#ifdef SYS_openat2
|
|
pfd = openat2_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
|
if (pfd < 0 && errno == ENOSYS) {
|
|
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
|
}
|
|
#else
|
|
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
|
|
#endif
|
|
if (pfd < 0) {
|
|
if (err) *err = (errno == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
return pfd;
|
|
}
|
|
|
|
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err) {
|
|
char leaf[PFS_PATH_MAX];
|
|
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
|
|
if (pfd < 0) return -1;
|
|
int rc = mkdirat(pfd, leaf, 0777);
|
|
int saved = errno;
|
|
close(pfd);
|
|
if (rc < 0) {
|
|
if (err) *err = (saved == EEXIST) ? VFS_ERR_EXIST : (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err) {
|
|
char leaf[PFS_PATH_MAX];
|
|
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
|
|
if (pfd < 0) return -1;
|
|
int rc = unlinkat(pfd, leaf, is_dir ? AT_REMOVEDIR : 0);
|
|
int saved = errno;
|
|
close(pfd);
|
|
if (rc < 0) {
|
|
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : (saved == ENOTEMPTY) ? VFS_ERR_NOTEMPTY : VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err) {
|
|
char leaf_from[PFS_PATH_MAX], leaf_to[PFS_PATH_MAX];
|
|
int pfd_from = resolve_parent_dir(root_fd, from, leaf_from, sizeof(leaf_from), err);
|
|
if (pfd_from < 0) return -1;
|
|
int pfd_to = resolve_parent_dir(root_fd, to, leaf_to, sizeof(leaf_to), err);
|
|
if (pfd_to < 0) { close(pfd_from); return -1; }
|
|
int rc = renameat(pfd_from, leaf_from, pfd_to, leaf_to);
|
|
int saved = errno;
|
|
close(pfd_from);
|
|
close(pfd_to);
|
|
if (rc < 0) {
|
|
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err) {
|
|
int fd = pfs_dir_openat(root_fd, rel, O_RDONLY, 0, err);
|
|
if (fd < 0) return -1;
|
|
struct stat st;
|
|
int rc = fstat(fd, &st);
|
|
int saved = errno;
|
|
close(fd);
|
|
if (rc < 0) {
|
|
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
|
|
return -1;
|
|
}
|
|
out->size = (pfs_usize)st.st_size;
|
|
out->mtime = (int64_t)st.st_mtime;
|
|
out->kind = S_ISDIR(st.st_mode) ? VFS_KIND_DIR : VFS_KIND_FILE;
|
|
return 0;
|
|
}
|
|
|
|
/*
|
|
* Best-effort, opt-in, process-wide hardening (Section 6.2). Restricts
|
|
* filesystem read/write/create/remove operations to the given
|
|
* directory-capability fds; deliberately does not restrict EXECUTE, to
|
|
* avoid an embeddable library silently blocking unrelated process
|
|
* behavior beyond the filesystem paths it was asked to confine.
|
|
*/
|
|
int pfs_landlock_restrict_to(const int *roots, size_t count) {
|
|
#if defined(__linux__) && defined(SYS_landlock_create_ruleset)
|
|
long abi = syscall(SYS_landlock_create_ruleset, NULL, 0, LANDLOCK_CREATE_RULESET_VERSION);
|
|
if (abi < 0) return -1; /* unsupported kernel: Section 6.3 residual risk */
|
|
|
|
uint64_t access = LANDLOCK_ACCESS_FS_READ_FILE | LANDLOCK_ACCESS_FS_WRITE_FILE |
|
|
LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_REMOVE_DIR |
|
|
LANDLOCK_ACCESS_FS_REMOVE_FILE | LANDLOCK_ACCESS_FS_MAKE_CHAR |
|
|
LANDLOCK_ACCESS_FS_MAKE_DIR | LANDLOCK_ACCESS_FS_MAKE_REG |
|
|
LANDLOCK_ACCESS_FS_MAKE_SOCK | LANDLOCK_ACCESS_FS_MAKE_FIFO |
|
|
LANDLOCK_ACCESS_FS_MAKE_BLOCK | LANDLOCK_ACCESS_FS_MAKE_SYM;
|
|
|
|
struct landlock_ruleset_attr attr;
|
|
memset(&attr, 0, sizeof(attr));
|
|
attr.handled_access_fs = access;
|
|
|
|
int rs_fd = (int)syscall(SYS_landlock_create_ruleset, &attr, sizeof(attr), 0);
|
|
if (rs_fd < 0) return -1;
|
|
|
|
for (size_t i = 0; i < count; i++) {
|
|
struct landlock_path_beneath_attr pb;
|
|
memset(&pb, 0, sizeof(pb));
|
|
pb.allowed_access = access;
|
|
pb.parent_fd = roots[i];
|
|
if (syscall(SYS_landlock_add_rule, rs_fd, LANDLOCK_RULE_PATH_BENEATH, &pb, 0) < 0) {
|
|
close(rs_fd);
|
|
return -1;
|
|
}
|
|
}
|
|
|
|
if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) < 0) { close(rs_fd); return -1; }
|
|
if (syscall(SYS_landlock_restrict_self, rs_fd, 0) < 0) { close(rs_fd); return -1; }
|
|
close(rs_fd);
|
|
return 0;
|
|
#else
|
|
(void)roots; (void)count;
|
|
return -1;
|
|
#endif
|
|
}
|