/* * test_pack_write_perf.c — regression guard for a real bug: pack_write's * Section 9.2 exact-duplicate elimination used to be a linear scan of * every previously-seen (hash, size) pair per entry — O(n) per entry, * O(n^2) total across n distinct blobs, found and fixed alongside the * file index's own O(n^2) (see CLAUDE.md's "Known performance * characteristics" and BENCH.md). It is now a hash table (load factor * 1/2, linear probing), which this asserts stays fast at a scale where * the old code was already measurably slow (BENCH.md: 80,000 unique * entries took 1.74s pre-fix, 0.044s post-fix). * * Deliberately bypasses the VFS/overlay/journal path entirely (calls * pack_write directly via internal.h) rather than creating N files * through an overlay: that path journals and fsyncs every write * (Section 4.4), which is a separate, unrelated cost this test has no * reason to pay, and which is pathologically slow on some sandboxed * environments (see CONTRIBUTING.md's ThreadSanitizer note for another * example of the same class of environment quirk) — conflating the two * costs is exactly the mistake that delayed finding this bug in the * first place, so this test deliberately does not repeat it. */ #include #include #include #include #include #include "internal.h" #include "test_harness.h" #define N 10000 static double now_sec(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return (double)ts.tv_sec + (double)ts.tv_nsec / 1e9; } int main(void) { PackBuildEntry *entries = (PackBuildEntry *)malloc(sizeof(PackBuildEntry) * N); char **names = (char **)malloc(sizeof(char *) * N); char **datas = (char **)malloc(sizeof(char *) * N); /* every entry's content is unique: this is the linear scan's actual * worst case (every comparison fails until the empty tail), and * exactly the case the original benchmark's identical-content test * files accidentally never exercised. */ for (int i = 0; i < N; i++) { names[i] = (char *)malloc(32); snprintf(names[i], 32, "/f%06d.dat", i); datas[i] = (char *)malloc(48); int len = snprintf(datas[i], 48, "unique-content-block-%06d", i); entries[i].name = names[i]; entries[i].data = datas[i]; entries[i].size = (uint64_t)len; entries[i].mode = 0; entries[i].mtime = 0; } char path[128]; snprintf(path, sizeof(path), "/tmp/packfs_test_dedup_perf_%d.img", (int)getpid()); unlink(path); int err = 0; double t0 = now_sec(); int rc = pack_write(path, entries, N, &err); double elapsed = now_sec() - t0; CHECK_EQ_INT(rc, 0); /* Generous bound: post-fix this takes well under 0.1s for N=10,000 * on ordinary hardware (BENCH.md: 0.013s for N=20,000). The old * O(n^2) code took long enough at this N to be a clear, unmissable * fail, not a borderline one — this is a regression tripwire, not a * tight performance assertion. */ CHECK(elapsed < 5.0); if (elapsed >= 5.0) { fprintf(stderr, "pack_write(%d unique entries) took %.3fs -- the O(n^2) dedup regression may be back\n", N, elapsed); } /* correctness, not just speed: load it back and confirm every entry * is findable with its own distinct content intact. */ Pack *p = NULL; int lerr = 0; CHECK_EQ_INT(pack_load(path, &p, &lerr), 0); if (p) { for (int i = 0; i < N; i += 997) { /* sample, not all 10,000, to keep this fast */ const PackIndexEntry *e = NULL; CHECK(pack_find(p, names[i], &e)); if (e) { CHECK_EQ_INT(e->size, strlen(datas[i])); CHECK_EQ_INT(memcmp(pack_entry_data(p, e), datas[i], e->size), 0); } } pack_close(p); } unlink(path); for (int i = 0; i < N; i++) { free(names[i]); free(datas[i]); } free(names); free(datas); free(entries); TEST_MAIN_END(); }