diff --git a/build.zig.zon b/build.zig.zon index e160373..71169a0 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -5,8 +5,8 @@ .minimum_zig_version = "0.16.0", .dependencies = .{ .zat = .{ - .url = "git+https://tangled.org/zat.dev/zat#8db0560c2cb9357a16c9ee65205e3d247b0f7f63", - .hash = "zat-0.3.18-5PuC7qdCCwDWiiUt4LKDe0-Z8Cdf1ZdYGheNYSyCKDoX", + .url = "git+https://tangled.org/zat.dev/zat#955e9ca9afb41d66d41e998803f6fcfc8fc4330f", + .hash = "zat-0.3.22-5PuC7j54CwAB6eBKHagm0rbfuG0jOg068iGufg1Jz6Wz", }, .websocket = .{ .url = "https://github.com/zzstoatzz/websocket.zig/archive/73429df.tar.gz", diff --git a/docs/invariants.md b/docs/invariants.md index ec0af80..758d738 100644 --- a/docs/invariants.md +++ b/docs/invariants.md @@ -66,7 +66,12 @@ interval. Invalid *upstream* data — a malformed frame, an over-limit field, a record no encoder can render — must never crash, stop, or exit the server: drop it, count it with a labeled metric, and keep running. Resource exhaustion is *ours*, not the remote's: `OutOfMemory` propagates rather than being -mistaken for a bad record. +mistaken for a bad record. That last clause is the one that keeps being +violated quietly, because a collapsing `catch` reads as tidy: bootstrap +retires a repository permanently on a structural verdict like `InvalidCar`, +so an exhaustion wearing that name discards good data *and* blames the +remote for it. An allocation-failure sweep over prepare/emit pins it +(`bootstrap/repos.zig`); it found three such sites, two of them in zat. **An internal failure must never look like an absence.** A read error is not a missing key; a planner failure is not "no matching data"; an encode failure is diff --git a/docs/semantic-parity.md b/docs/semantic-parity.md index 5fb78bd..1c3e844 100644 --- a/docs/semantic-parity.md +++ b/docs/semantic-parity.md @@ -65,7 +65,7 @@ The detailed bootstrap audit is in | Surface | Status | What is actually established | Blocking or missing evidence | |---|---|---|---| | Lifecycle phase machine | **verified** | Phase reads distinguish absence from RocksDB failure, reject unknown values, and use synced writes. Bootstrap writes `merging` after backfill drains while bootstrap-live is still running, exactly at the pinned upstream boundary; process-kill crashpoints exercise both sides of that write. Cleanup writes `steady_state` only after directory sync and cursor deletion. | Exact persisted phase-entry timestamps are a separate status-surface gap; this row establishes transition state and ordering only. | -| Whole-network bootstrap | **partial** | listRepos pages form a page-aligned dispatch/checkpoint unit; eligible repositories are shuffled; download concurrency and CAR preparation are real. Successful repositories become durable independently at archive-writer durability boundaries, including while a sibling remains incomplete and the listRepos cursor remains unchanged. | Resource-exhaustion cleanup at every preparation/emission ownership transfer is not yet proved, and selected/debug paths still need exact-artifact admission receipts. See the detailed audit. | +| Whole-network bootstrap | **verified** | listRepos pages form a page-aligned dispatch/checkpoint unit; eligible repositories are shuffled; download concurrency and CAR preparation are real. Successful repositories become durable independently at archive-writer durability boundaries, including while a sibling remains incomplete and the listRepos cursor remains unchanged. Cleanup at every preparation/emission ownership transfer is the parse arena's by construction — a `PreparedRepo` has no `deinit` because the arena owns every byte its lazy MST and record rows reference — and an allocation-failure sweep over the whole prepare/emit path now asserts exhaustion always surfaces as `OutOfMemory`. | The sweep found three sites collapsing `OutOfMemory` into a structural verdict, all fixed: `Mst.loadLazy`'s catch in `prepareRepo` (`InvalidCar`), and zat's CAR header parse and allocating encoders (`InvalidHeader`/`WriteFailed`, shipped as v0.3.21/v0.3.22). The engine retires a repository permanently on `InvalidCar`, so each one silently dropped good repositories under memory pressure and attributed it to the remote's CAR. | | Merge and post-bootstrap discovery | **verified** | Source rows are filtered by the backfill revision; source cursor and latest-revision updates share one synced batch (`commitMergeSource`, which refreshes `latest_rev`/`updated_us` while preserving the backfill `rev` — asserted by the real-merge test); merge cursor corruption/read errors fail closed; the restart guard treats only `FileNotFound` as cleanup-complete; discovery records active and inactive unknown repositories and follows arbitrary-length cursors with loop detection; successful cleanup is directory-synced. Every source-file path fails closed — `readFileAlloc`, checksum-verifying `Sealed.parse`, `readBlock` and the destination append all propagate, with no `catch` swallow — and a missing source segment raises `SourceIndexGap` rather than being skipped; a physical test deletes the middle of three sealed sources and asserts the error, where without the guard merge consumes 2 of 3 sources and reports success. The pending pass is now upstream's `RunPendingRepoRetryPass` exactly: the same bounded runner with `eligible_status` flipped to pending, so it inherits the worker pool, per-host gate, computed backoff and 429 host parking rather than reimplementing a weaker version. A test asserts a failed pending repository gets upstream's `RecordRetryFailure` bookkeeping — status failed, `attempts` and `retry_count` incremented, `next_attempt_us` in the future — and that a failed sibling is untouched, which the previous hand-rolled loop got wrong on the last two fields. | No known divergence remains. The runner is shared with the steady retry loop, so that row's persisted-host-parking caveat applies here too; it is tracked there rather than duplicated. | | Failed-repository healing | **verified** | A real global/per-host worker gate exists, final redirect hosts are recorded, and retry state is stored. Candidates stream through a bounded queue with host gates created on demand, so neither the failed set nor the host set is materialized. A pass that fails for our own reasons (store, archive, memory) now latches a terminal error, counts `retry_terminal_failures_total`, and requests process shutdown instead of sleeping until the next interval — matching upstream, whose errgroup cancels steady state. Per-repo remote failures still become backoff and never reach that path; a real read fault drives the test. Host parking now lives in the runner's memory for its lifetime, as upstream's `hostParked` map does, and never shortens a park — upstream's `if old.After(until) return`, which the persisted version lacked. | None known. | | JSS sealed format interoperability | **verified** | Upstream-produced sealed fixtures are parsed; Stream-produced sealed files are consumed by the pinned Go reader; header/footer, block index, blooms, collections, compression, and checksums have reciprocal fixtures. | This verifies sealed-format compatibility only. It does not verify startup recovery or replay completeness. | diff --git a/src/internal/bootstrap/repos.zig b/src/internal/bootstrap/repos.zig index d37f386..3e7fdd8 100644 --- a/src/internal/bootstrap/repos.zig +++ b/src/internal/bootstrap/repos.zig @@ -119,7 +119,13 @@ pub fn prepareRepo(arena: Allocator, car_bytes: []const u8) RepoError!PreparedRe var mst = zat.mst.Mst.loadLazy(arena, loaded.data_cid, .{ .ctx = reader_ctx, .getFn = CarBlockReader.get, - }) catch return if (reader_ctx.missing) error.MissingBlock else error.InvalidCar; + // loadLazy fails only on allocation. A missing block still reports as + // one, but everything else here is exhaustion, and calling it + // `InvalidCar` would retire a good repository permanently and blame + // its PDS -- see the classification test below. + }) catch |err| switch (err) { + error.OutOfMemory => return if (reader_ctx.missing) error.MissingBlock else error.OutOfMemory, + }; try validateComplete(&mst, loaded.blocks, reader_ctx); return .{ .embedded_did = loaded.did, @@ -402,6 +408,76 @@ test "truncated CAR is retryable and emits nothing" { try testing.expectEqual(@as(usize, 0), sink.rows); } +// Allocation failure anywhere in prepare/emit must stay `OutOfMemory`. The +// engine retries that, but classifies `InvalidCar` as a permanent per-repo +// failure attributed to the remote's CAR -- so an exhaustion mistaken for a +// bad record silently drops a good repository from a whole-network archive +// and blames its PDS. `Mst.loadLazy`'s catch collapsed every error, including +// this one, into `InvalidCar`. +test "allocation failure never masquerades as a malformed repository" { + var fixture_arena = std.heap.ArenaAllocator.init(testing.allocator); + defer fixture_arena.deinit(); + const fixture = fixture_arena.allocator(); + + const record = "oom classification fixture"; + const record_cid = try zat.cbor.Cid.forDagCbor(fixture, record); + var tree = zat.mst.Mst.init(fixture); + try tree.put("app.bsky.feed.post/3k2abcdefghij", record_cid); + const data_cid = try tree.rootCid(); + const keypair = try zat.Keypair.fromSecretKey(.p256, .{17} ** 32); + const did = try keypair.did(fixture); + const signed = try zat.signCommit(fixture, .{ + .did = did, + .rev = "3k2abcdefghij", + .data = data_cid, + }, &keypair); + var blocks: std.ArrayList(zat.car.Block) = .empty; + try blocks.append(fixture, .{ .cid_raw = signed.cid.raw, .data = signed.bytes }); + try tree.collectBlocks(&blocks); + try blocks.append(fixture, .{ .cid_raw = record_cid.raw, .data = record }); + const car_bytes = try zat.car.writeAlloc(fixture, .{ + .roots = &.{signed.cid}, + .blocks = blocks.items, + }); + + const Sink = struct { + rows: usize = 0, + fn emit(self: *@This(), _: segment.Event) anyerror!void { + self.rows += 1; + } + }; + + var completed: usize = 0; + var exhausted: usize = 0; + for (0..512) |fail_index| { + var failing = testing.FailingAllocator.init(testing.allocator, .{ .fail_index = fail_index }); + // The arena is the cleanup contract for a prepared repo: it owns every + // byte the lazy MST and record rows reference, so unwinding at any + // ownership transfer frees through it rather than per-object deinit. + var arena = std.heap.ArenaAllocator.init(failing.allocator()); + defer arena.deinit(); + + var drops: DropCounts = .{}; + var sink: Sink = .{}; + if (carToRows(arena.allocator(), car_bytes, did, 0, .create, &drops, &sink, Sink.emit)) |_| { + completed += 1; + } else |err| switch (err) { + error.OutOfMemory => exhausted += 1, + else => { + std.debug.print( + "fail_index {d}: exhaustion surfaced as {s}\n", + .{ fail_index, @errorName(err) }, + ); + return error.ExhaustionMisclassified; + }, + } + } + // Both arms must be exercised, or the sweep proved nothing: too narrow a + // range would only ever exhaust, and a broken fixture would only ever pass. + try testing.expect(exhausted > 0); + try testing.expect(completed > 0); +} + test "listRepos DID remains authoritative over a valid CAR's embedded DID" { var arena = std.heap.ArenaAllocator.init(testing.allocator); defer arena.deinit();