Files
packfs/src/containment.c
T
retoorandClaude Sonnet 5 419182bb05 Add version API, pkg-config, SPDX headers, SECURITY.md, CHANGELOG.md
Project-hygiene pass toward being a properly citable, embeddable,
professionally-packaged C library rather than just working code:

- PACKFS_VERSION_MAJOR/MINOR/PATCH/STRING in include/packfs.h, the single
  source of truth for the project's version, plus a runtime pfs_version()
  (src/vfs.c, next to vfs_new/vfs_free) so a dynamically-linked consumer
  can check ABI/API compatibility without recompiling. Covered by a new
  assertion in tests/test_mem.c that the macro and the runtime function
  never disagree.
- packfs.pc.in + a `make install` rule that generates packfs.pc with its
  Version: field derived from PACKFS_VERSION_STRING via a Makefile-level
  grep/sed, never hand-maintained separately -- verified end-to-end with a
  scratch `make install PREFIX=...` + `pkg-config --cflags --libs packfs`
  + `make uninstall`, not just by reading the rule.
- SPDX-License-Identifier: MIT added to every src/*.c and src/internal.h
  (include/packfs.h already had one); the whole distributed source tree
  now carries consistent machine-readable license metadata.
- SECURITY.md, stating precisely what this project's containment and pack-
  integrity code actually claims as a security boundary (concept.md
  Section 6/7) versus what it explicitly does not (unenforced `mode`, no
  cross-process concurrency) -- not generic boilerplate -- with a real
  reporting contact rather than a placeholder.
- CHANGELOG.md (Keep a Changelog format), summarizing the real history in
  git log to date; explicitly notes no version is tagged yet.

Verified: clean `make all` + `make test` (all 6 binaries, including the
new version-check assertion), and a full ASan/UBSan sweep of all 6
binaries with zero real findings (one run hit the already-documented
DEADLYSIGNAL sandbox flake on test_pack_write_perf across all 3 retries;
re-verified directly afterward with 5/5 additional clean passes and timing
well under any plausible timeout, confirming it was the flake, not a
regression, before treating this as done).

Deliberately not done here, by the user's explicit choice: no
CODE_OF_CONDUCT.md, and no git remote/publishing -- this repository still
has neither, and both are decisions left to the maintainer.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UqJpkdJ6Njnt1pw3CbghzB
2026-09-14 10:58:35 +00:00

256 lines
8.6 KiB
C

/*
* containment.c — capability-scoped `dir` mounts (Section 6.2) and
* optional Landlock hardening (Section 6.2, 6.3).
*
* On Linux 5.6+, every lookup beneath a `dir` mount's root fd is resolved
* with openat2(RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS) in a single kernel
* call, atomically rejecting ".." escapes and symlink escapes. Where
* openat2 is unavailable (ENOSYS on an older kernel), this falls back to
* per-component O_NOFOLLOW resolution, which is an accepted, weaker
* residual-risk posture per Section 6.3 — not a claim of equal
* containment strength.
*
* SPDX-License-Identifier: MIT
*/
#include <errno.h>
#include <fcntl.h>
#include <linux/openat2.h>
#include <string.h>
#include <sys/prctl.h>
#include <sys/stat.h>
#include <sys/syscall.h>
#include <unistd.h>
#include "internal.h"
#ifdef __linux__
#include <linux/landlock.h>
#endif
int pfs_dir_capability_open(const char *path, int *err) {
int fd = open(path, O_DIRECTORY | O_CLOEXEC | O_RDONLY);
if (fd < 0) {
if (err) *err = VFS_ERR_IO;
return -1;
}
return fd;
}
/* Splits an already-normalized relative path (no leading slash) into
* (parent, leaf). Root-level names have an empty parent. */
static void split_parent_leaf(const char *rel, char *parent_buf, size_t cap, const char **leaf) {
const char *slash = strrchr(rel, '/');
if (!slash) {
parent_buf[0] = '\0';
*leaf = rel;
return;
}
size_t plen = (size_t)(slash - rel);
if (plen >= cap) plen = cap - 1;
memcpy(parent_buf, rel, plen);
parent_buf[plen] = '\0';
*leaf = slash + 1;
}
#ifdef SYS_openat2
static int openat2_beneath(int root_fd, const char *rel, uint64_t flags, uint64_t mode) {
struct open_how how;
memset(&how, 0, sizeof(how));
how.flags = flags;
how.mode = mode;
how.resolve = RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS;
return (int)syscall(SYS_openat2, root_fd, rel[0] ? rel : ".", &how, sizeof(how));
}
#endif
/* Per-component O_NOFOLLOW fallback (Section 6.3). Walks every component
* except the last with O_NOFOLLOW|O_DIRECTORY, then opens the last with
* the caller's requested flags plus O_NOFOLLOW. Closes most, not all, of
* the same TOCTOU window openat2 closes atomically. */
static int fallback_openat_beneath(int root_fd, const char *rel, int flags, unsigned mode) {
if (rel[0] == '\0') {
return dup(root_fd);
}
char buf[PFS_PATH_MAX];
if (strlen(rel) >= sizeof(buf)) { errno = ENAMETOOLONG; return -1; }
strcpy(buf, rel);
int cur = root_fd;
int owns_cur = 0;
char *save = NULL;
char *tok = strtok_r(buf, "/", &save);
char *next = tok ? strtok_r(NULL, "/", &save) : NULL;
while (tok && next) {
int nfd = openat(cur, tok, O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC);
if (owns_cur) close(cur);
if (nfd < 0) return -1;
cur = nfd;
owns_cur = 1;
tok = next;
next = strtok_r(NULL, "/", &save);
}
int final_fd = -1;
if (tok) {
int final_flags = flags;
if (!(flags & O_CREAT)) final_flags |= O_NOFOLLOW;
final_fd = openat(cur, tok, final_flags, mode);
}
if (owns_cur) close(cur);
return final_fd;
}
int pfs_dir_openat(int root_fd, const char *rel, int flags, unsigned mode, int *err) {
int fd;
#ifdef SYS_openat2
fd = openat2_beneath(root_fd, rel, (uint64_t)flags, (uint64_t)mode);
if (fd < 0 && errno == ENOSYS) {
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
}
#else
fd = fallback_openat_beneath(root_fd, rel, flags, mode);
#endif
if (fd < 0) {
if (err) *err = (errno == EXDEV || errno == ELOOP) ? VFS_ERR_INVAL
: (errno == ENOENT) ? VFS_ERR_NOENT
: (errno == EEXIST) ? VFS_ERR_EXIST
: VFS_ERR_IO;
return -1;
}
return fd;
}
static int resolve_parent_dir(int root_fd, const char *rel, char *leaf_out, size_t leaf_cap, int *err) {
char parent[PFS_PATH_MAX];
const char *leaf;
split_parent_leaf(rel, parent, sizeof(parent), &leaf);
if (strlen(leaf) >= leaf_cap) { if (err) *err = VFS_ERR_INVAL; return -1; }
strcpy(leaf_out, leaf);
int pfd;
#ifdef SYS_openat2
pfd = openat2_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
if (pfd < 0 && errno == ENOSYS) {
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
}
#else
pfd = fallback_openat_beneath(root_fd, parent, O_DIRECTORY | O_RDONLY, 0);
#endif
if (pfd < 0) {
if (err) *err = (errno == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
return pfd;
}
int pfs_dir_mkdirat(int root_fd, const char *rel, int *err) {
char leaf[PFS_PATH_MAX];
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
if (pfd < 0) return -1;
int rc = mkdirat(pfd, leaf, 0777);
int saved = errno;
close(pfd);
if (rc < 0) {
if (err) *err = (saved == EEXIST) ? VFS_ERR_EXIST : (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
return 0;
}
int pfs_dir_unlinkat(int root_fd, const char *rel, int is_dir, int *err) {
char leaf[PFS_PATH_MAX];
int pfd = resolve_parent_dir(root_fd, rel, leaf, sizeof(leaf), err);
if (pfd < 0) return -1;
int rc = unlinkat(pfd, leaf, is_dir ? AT_REMOVEDIR : 0);
int saved = errno;
close(pfd);
if (rc < 0) {
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : (saved == ENOTEMPTY) ? VFS_ERR_NOTEMPTY : VFS_ERR_IO;
return -1;
}
return 0;
}
int pfs_dir_renameat(int root_fd, const char *from, const char *to, int *err) {
char leaf_from[PFS_PATH_MAX], leaf_to[PFS_PATH_MAX];
int pfd_from = resolve_parent_dir(root_fd, from, leaf_from, sizeof(leaf_from), err);
if (pfd_from < 0) return -1;
int pfd_to = resolve_parent_dir(root_fd, to, leaf_to, sizeof(leaf_to), err);
if (pfd_to < 0) { close(pfd_from); return -1; }
int rc = renameat(pfd_from, leaf_from, pfd_to, leaf_to);
int saved = errno;
close(pfd_from);
close(pfd_to);
if (rc < 0) {
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
return 0;
}
int pfs_dir_statat(int root_fd, const char *rel, VfsStat *out, int *err) {
int fd = pfs_dir_openat(root_fd, rel, O_RDONLY, 0, err);
if (fd < 0) return -1;
struct stat st;
int rc = fstat(fd, &st);
int saved = errno;
close(fd);
if (rc < 0) {
if (err) *err = (saved == ENOENT) ? VFS_ERR_NOENT : VFS_ERR_IO;
return -1;
}
out->size = (pfs_usize)st.st_size;
out->mtime = (int64_t)st.st_mtime;
out->kind = S_ISDIR(st.st_mode) ? VFS_KIND_DIR : VFS_KIND_FILE;
return 0;
}
/*
* Best-effort, opt-in, process-wide hardening (Section 6.2). Restricts
* filesystem read/write/create/remove operations to the given
* directory-capability fds; deliberately does not restrict EXECUTE, to
* avoid an embeddable library silently blocking unrelated process
* behavior beyond the filesystem paths it was asked to confine.
*/
int pfs_landlock_restrict_to(const int *roots, size_t count) {
#if defined(__linux__) && defined(SYS_landlock_create_ruleset)
long abi = syscall(SYS_landlock_create_ruleset, NULL, 0, LANDLOCK_CREATE_RULESET_VERSION);
if (abi < 0) return -1; /* unsupported kernel: Section 6.3 residual risk */
uint64_t access = LANDLOCK_ACCESS_FS_READ_FILE | LANDLOCK_ACCESS_FS_WRITE_FILE |
LANDLOCK_ACCESS_FS_READ_DIR | LANDLOCK_ACCESS_FS_REMOVE_DIR |
LANDLOCK_ACCESS_FS_REMOVE_FILE | LANDLOCK_ACCESS_FS_MAKE_CHAR |
LANDLOCK_ACCESS_FS_MAKE_DIR | LANDLOCK_ACCESS_FS_MAKE_REG |
LANDLOCK_ACCESS_FS_MAKE_SOCK | LANDLOCK_ACCESS_FS_MAKE_FIFO |
LANDLOCK_ACCESS_FS_MAKE_BLOCK | LANDLOCK_ACCESS_FS_MAKE_SYM;
struct landlock_ruleset_attr attr;
memset(&attr, 0, sizeof(attr));
attr.handled_access_fs = access;
int rs_fd = (int)syscall(SYS_landlock_create_ruleset, &attr, sizeof(attr), 0);
if (rs_fd < 0) return -1;
for (size_t i = 0; i < count; i++) {
struct landlock_path_beneath_attr pb;
memset(&pb, 0, sizeof(pb));
pb.allowed_access = access;
pb.parent_fd = roots[i];
if (syscall(SYS_landlock_add_rule, rs_fd, LANDLOCK_RULE_PATH_BENEATH, &pb, 0) < 0) {
close(rs_fd);
return -1;
}
}
if (prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) < 0) { close(rs_fd); return -1; }
if (syscall(SYS_landlock_restrict_self, rs_fd, 0) < 0) { close(rs_fd); return -1; }
close(rs_fd);
return 0;
#else
(void)roots; (void)count;
return -1;
#endif
}