Files
packfs/tests/test_journal_failure.c
T
retoorandClaude Sonnet 5 4aad7d6f64
CI / build-and-test (push) Failing after 38s
Fix real memory leak Gitea CI caught, that local testing had been masking
Gitea CI flagged two things on the last push:

1. A -Wunused-result warning on an intentionally-ignored write() return
   value in test_crash_consistency.c's progress side-channel. Fixed with
   an explicit (void) cast and a comment explaining why ignoring it is
   safe (a short write there only makes the progress count more
   conservative, per that function's own existing documented tolerance).

2. A real LeakSanitizer failure -- 256 bytes across 4 allocations from
   vfs_new/vfs_unmount. Root cause: tests/test_journal_failure.c's two
   "reopen after the failure, verify recovery" blocks called
   vfs_unmount/backend_free/backend_free inside their `if (ov2)` branch
   (the normal, expected path) but vfs_free(v2) only on the `else`
   branch, which is never actually reached in practice. Fixed by moving
   vfs_free(v2) to run unconditionally after the if, in both blocks.

This bug was invisible locally across many runs because local sanitizer
verification had been using ASAN_OPTIONS=detect_leaks=0 -- adopted
originally for a real reason (a SIGKILLed forked child in
test_crash_consistency.c never runs its own exit-time leak check, so its
allocations were never the actual concern) but applied to the whole test
run, which also suppressed detection of this real bug in the *parent*
process's own code. CI doesn't set that option, so it caught what local
runs couldn't. Documented in CONTRIBUTING.md as a real process gap, not
just a code bug: a local verification habit that diverges from what CI
actually runs can let a real finding through until it reaches CI.

Verified: confirmed the leak directly first (reproduced locally by
dropping the detect_leaks=0 override, matching CI exactly, before
touching any code), then confirmed the fix by re-running the same
no-override sweep across all 8 test binaries with zero leaks found, plus
a clean make all + make test.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UqJpkdJ6Njnt1pw3CbghzB
2026-09-14 20:10:25 +00:00

207 lines
8.2 KiB
C

/*
* test_journal_failure.c — a real, forced journal-write I/O failure
* (RLIMIT_FSIZE + SIGXFSZ ignored, so write() fails with EFBIG instead of
* killing the process), verifying two real bugs found and fixed in
* overlay.c while building this test, not by this test catching them
* unprompted -- there was no test before this one that exercised a
* failing journal write at all:
*
* 1. journal_append_* and journal_put_current used to be void, and every
* fwrite/fflush/fsync inside journal_append_record went unchecked --
* a real write failure (disk full, quota, this test's RLIMIT_FSIZE)
* was silently swallowed and vfs_write/vfs_mkdir/vfs_unlink/
* vfs_rename all reported success on an overlay-backed file even
* though the change was never made durable.
*
* 2. Fixing (1) by itself was not enough: a partial write leaves a torn
* record sitting in the middle of the journal file, and
* journal_replay stops at the first record it can't fully read
* (Section 4.3) -- meaning every record appended *after* the failed
* one, including ones that complete successfully later, was silently
* unreachable on every future reopen. Confirmed by direct
* reproduction before this half of the fix: a forced-failed write
* followed by a genuinely successful one was unrecoverable, because
* replay never got past the torn record in between. Fixed by rolling
* the journal file back to its exact pre-record length on any failed
* write, keeping "valid up to EOF" true even when a write fails.
*
* What this test does NOT cover: the parallel bug in journal_put_current
* (a malloc(size) failure used to leave a NULL buffer that got passed
* into journal_append_record's memcpy anyway -- an OOM-triggered
* NULL-pointer dereference, not a "write failed" one). Reliably forcing
* malloc() to fail in a portable, safe way isn't practical here; that
* half of the fix is verified by code inspection (the `if (size && !buf)
* return -1;` guard in journal_put_current), not by a test triggering it.
*/
#include <signal.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/resource.h>
#include <unistd.h>
#include "packfs.h"
#include "test_harness.h"
static void reset_paths(const char *pack_path, const char *jpath) {
unlink(pack_path);
unlink(jpath);
}
static void set_fsize_limit(rlim_t bytes) {
struct rlimit rl;
rl.rlim_cur = bytes;
rl.rlim_max = RLIM_INFINITY;
CHECK_EQ_INT(setrlimit(RLIMIT_FSIZE, &rl), 0);
}
int main(void) {
/* SIGXFSZ's default action terminates the process; ignoring it makes
* the offending write() return -1/EFBIG instead, which is what this
* test needs to observe -- a failed write, not a killed process. */
signal(SIGXFSZ, SIG_IGN);
char pack_path[512];
snprintf(pack_path, sizeof(pack_path), "/tmp/packfs_test_journal_fail_%d.img", (int)getpid());
char jpath[600];
snprintf(jpath, sizeof(jpath), "%s.jnl", pack_path);
/* --- scenario 1: vfs_write's journal append fails; the failure must
* be reported (not swallowed), the in-memory content must still be
* readable in this process, and -- critically -- a fresh reopen must
* NOT see the torn/rolled-back record as if it were valid. --- */
reset_paths(pack_path, jpath);
{
Vfs *v = vfs_new();
Backend *mem = backend_mem_new();
int oerr = 0;
Backend *ov = backend_overlay_new(pack_path, mem, &oerr);
CHECK(ov != NULL);
CHECK_EQ_INT(vfs_mount(v, "/", ov), VFS_OK);
set_fsize_limit(32); /* far smaller than any real journal record */
int err = 0;
VfsFile *f = vfs_open(v, "/big.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
CHECK(f != NULL);
const char *payload = "this content is deliberately longer than 32 bytes";
pfs_isize wr = vfs_write(f, payload, strlen(payload));
CHECK_EQ_INT(wr, VFS_ERR_IO); /* durability failure must be reported */
CHECK_EQ_INT(vfs_close(f), VFS_OK);
/* in-memory: still readable in this process, same as a page-cache
* write a later fsync() fails on -- only durability was refused */
f = vfs_open(v, "/big.txt", VFS_O_RDONLY, &err);
CHECK(f != NULL);
if (f) {
char buf[128] = {0};
pfs_isize rd = vfs_read(f, buf, sizeof(buf));
CHECK_EQ_INT(rd, (pfs_isize)strlen(payload));
CHECK_EQ_INT(memcmp(buf, payload, strlen(payload)), 0);
CHECK_EQ_INT(vfs_close(f), VFS_OK);
}
vfs_unmount(v, "/");
backend_free(ov);
backend_free(mem);
vfs_free(v);
set_fsize_limit(RLIM_INFINITY); /* reopening below only reads, but be safe */
Vfs *v2 = vfs_new();
Backend *mem2 = backend_mem_new();
int oerr2 = 0;
Backend *ov2 = backend_overlay_new(pack_path, mem2, &oerr2);
CHECK(ov2 != NULL); /* must never fail to load, even with a rolled-back journal */
if (ov2) {
CHECK_EQ_INT(vfs_mount(v2, "/", ov2), VFS_OK);
int err2 = 0;
VfsFile *f2 = vfs_open(v2, "/big.txt", VFS_O_RDONLY, &err2);
/* the failed write must not have been replayed as if valid */
CHECK(f2 == NULL);
CHECK_EQ_INT(err2, VFS_ERR_NOENT);
if (f2) vfs_close(f2);
vfs_unmount(v2, "/");
backend_free(ov2);
backend_free(mem2);
}
vfs_free(v2);
}
/* --- scenario 2: vfs_mkdir's journal append fails --- */
reset_paths(pack_path, jpath);
{
Vfs *v = vfs_new();
Backend *mem = backend_mem_new();
int oerr = 0;
Backend *ov = backend_overlay_new(pack_path, mem, &oerr);
CHECK(ov != NULL);
CHECK_EQ_INT(vfs_mount(v, "/", ov), VFS_OK);
set_fsize_limit(4); /* smaller than even a bare mkdir record */
CHECK_EQ_INT(vfs_mkdir(v, "/a-directory-name-long-enough"), VFS_ERR_IO);
vfs_unmount(v, "/");
backend_free(ov);
backend_free(mem);
vfs_free(v);
set_fsize_limit(RLIM_INFINITY);
}
/* --- scenario 3: after a failed write, lifting the limit lets a
* later, unrelated write on the SAME overlay session succeed and
* survive a fresh reopen -- proving the rollback keeps the journal
* usable afterward, not merely "not corrupted." --- */
reset_paths(pack_path, jpath);
{
Vfs *v = vfs_new();
Backend *mem = backend_mem_new();
int oerr = 0;
Backend *ov = backend_overlay_new(pack_path, mem, &oerr);
CHECK(ov != NULL);
CHECK_EQ_INT(vfs_mount(v, "/", ov), VFS_OK);
set_fsize_limit(32);
int err = 0;
VfsFile *f = vfs_open(v, "/big.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
CHECK(f != NULL);
const char *payload = "this content is deliberately longer than 32 bytes";
CHECK_EQ_INT(vfs_write(f, payload, strlen(payload)), VFS_ERR_IO);
CHECK_EQ_INT(vfs_close(f), VFS_OK);
set_fsize_limit(RLIM_INFINITY);
f = vfs_open(v, "/after.txt", VFS_O_WRONLY | VFS_O_CREAT, &err);
CHECK(f != NULL);
CHECK_EQ_INT(vfs_write(f, "durable now", 11), 11);
CHECK_EQ_INT(vfs_close(f), VFS_OK);
vfs_unmount(v, "/");
backend_free(ov);
backend_free(mem);
vfs_free(v);
Vfs *v2 = vfs_new();
Backend *mem2 = backend_mem_new();
int oerr2 = 0;
Backend *ov2 = backend_overlay_new(pack_path, mem2, &oerr2);
CHECK(ov2 != NULL);
if (ov2) {
CHECK_EQ_INT(vfs_mount(v2, "/", ov2), VFS_OK);
int err2 = 0;
VfsFile *f2 = vfs_open(v2, "/after.txt", VFS_O_RDONLY, &err2);
CHECK(f2 != NULL); /* this is the exact case that was broken before the rollback fix */
if (f2) {
char buf[32] = {0};
CHECK_EQ_INT(vfs_read(f2, buf, sizeof(buf)), 11);
CHECK_EQ_INT(memcmp(buf, "durable now", 11), 0);
CHECK_EQ_INT(vfs_close(f2), VFS_OK);
}
vfs_unmount(v2, "/");
backend_free(ov2);
backend_free(mem2);
}
vfs_free(v2);
}
reset_paths(pack_path, jpath);
TEST_MAIN_END();
}