diff --git a/bench/README.md b/bench/README.md index 5a0087b..4574735 100644 --- a/bench/README.md +++ b/bench/README.md @@ -24,6 +24,7 @@ just bench render-record 10 1000 just bench get-record 10 1000 just bench list-records 10 1000 just bench repo-size 5000 +just bench get-repo 5000 4 5 just bench repo-large 100000 just bench space 1000 just bench http @@ -54,6 +55,9 @@ just bench run --scenario write --records 10000 `listRecords(limit=100)` and full repo CAR materialization. The default Just target keeps this moderate; use `repo-large` for intentional 100k+ record runs. +- `get-repo`: focused `com.atproto.sync.getRepo` storage-pressure probe. It + measures full repo CAR construction, `since` diff CAR construction after one + additional commit, and concurrent full exports against one seeded repo. - `space`: experimental permissioned-data storage probes for space discovery, private record writes, private record reads/lists, repo oplog catch-up, and blob readback. @@ -121,6 +125,12 @@ These are synthetic local trend numbers, not cross-PDS comparisons. The old whole-DID range-scan export path is intentionally no longer the production full export behavior because it could include retained stale blocks. +`just bench get-repo` is the narrower operator-facing lane for the same sync +surface. It belongs here first because ZDS owns the storage and route behavior +for serving repo exports. Cross-PDS HTTP comparisons, compression behavior, and +live backfill traffic should live in `atproto-bench` once this local seam is +stable enough to compare against other implementations. + ## methodology - Build ZDS with `-Doptimize=ReleaseFast`. @@ -201,7 +211,7 @@ High-level ZDS surface area represented in `bench/`: |---|---| | repo writes | `write`, `metastore`, `write-profile` | | repo reads | `get-cid`, `get-block`, `get-record`, `list-records` | -| repo export/sync snapshots | `repo`, `repo-size` | +| repo export/sync snapshots | `repo`, `repo-size`, `get-repo` | | sync event storage | `metastore` write path plus smoke tests | | blob storage | `blob`, `space` blob readback | | HTTP routing/serialization/auth overhead | `http` | diff --git a/bench/justfile b/bench/justfile index f60699d..c2dfd93 100644 --- a/bench/justfile +++ b/bench/justfile @@ -61,6 +61,10 @@ list-records callers="10" ops_per_caller="1000": repo-size max_records="5000": {{zig}} build bench -Doptimize=ReleaseFast -- --scenario repo-size --records {{max_records}} +# benchmark sync.getRepo storage pressure: full export, since export, and concurrent full exports +get-repo records="5000" callers="4" ops_per_caller="5": + {{zig}} build bench -Doptimize=ReleaseFast -- --scenario get-repo --records {{records}} --callers {{callers}} --ops-per-caller {{ops_per_caller}} + # deliberately slow large-repo lane for Paul/Jerry-shaped synthetic repositories repo-large records="100000": {{zig}} build bench -Doptimize=ReleaseFast -- --scenario repo-size --records {{records}} diff --git a/bench/main.zig b/bench/main.zig index 0f201ff..8388cc2 100644 --- a/bench/main.zig +++ b/bench/main.zig @@ -15,6 +15,7 @@ const Scenario = enum { get_record, list_records, repo_size, + get_repo, space, write_profile, }; @@ -123,6 +124,7 @@ pub fn main(init: std.process.Init) !void { .get_record => try benchGetRecord(allocator, options), .list_records => try benchListRecords(allocator, options), .repo_size => try benchRepoSizes(allocator, options.records), + .get_repo => try benchGetRepo(allocator, options), .space => try benchSpace(allocator, options), .write_profile => try benchWriteProfile(allocator, options), } @@ -544,6 +546,75 @@ fn benchRepoCarForRepo( }; } +fn benchGetRepo(allocator: std.mem.Allocator, options: Options) !void { + const level: ConcurrencyLevel = .{ + .callers = options.callers orelse 4, + .ops_per_caller = options.ops_per_caller orelse if (options.records >= 100_000) 1 else 5, + }; + + std.debug.print("\n=== zds getRepo benchmarks ===\n", .{}); + std.debug.print("seeding repo with {d} records\n", .{options.records}); + var state = try initBench(allocator); + defer state.deinit(); + try seedIndexedRecords(allocator, state.account, options.records); + + const since_rev = try appendIndexedRecordRev(allocator, state.account, options.records); + defer allocator.free(since_rev); + const latest_rev = try appendIndexedRecordRev(allocator, state.account, options.records + 1); + defer allocator.free(latest_rev); + + (try benchRepoCarForRepo(allocator, state.account, options.records + 2)).print(); + (try benchRepoCarSince(allocator, state.account, since_rev)).print(); + (try benchConcurrentRepoCar(allocator, state.account, level)).print(); +} + +fn benchRepoCarSince( + allocator: std.mem.Allocator, + account: zds.auth.tokens.Account, + since_rev: []const u8, +) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 20; + var total_bytes: usize = 0; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const car = try zds.storage.store.writeRepoCarSince(arena.allocator(), account.did, since_rev); + if (car.len == 0) return error.EmptyRepoCar; + total_bytes += car.len; + } + return .{ + .name = "getRepo since", + .ops = iterations, + .bytes = total_bytes, + .elapsed_ns = nowNs() - start, + }; +} + +fn benchConcurrentRepoCar( + allocator: std.mem.Allocator, + account: zds.auth.tokens.Account, + level: ConcurrencyLevel, +) !ConcurrentResult { + const total_ops = level.callers * level.ops_per_caller; + const latencies = try allocator.alloc(u64, total_ops); + defer allocator.free(latencies); + const threads = try allocator.alloc(std.Thread, level.callers); + defer allocator.free(threads); + + const start = nowNs(); + for (threads, 0..) |*thread, i| { + thread.* = try std.Thread.spawn(.{}, repoCarWorker, .{ + account.did, + latencies[i * level.ops_per_caller ..][0..level.ops_per_caller], + }); + } + for (threads) |thread| thread.join(); + + return concurrentResult("getRepo full", level, total_ops, nowNs() - start, latencies); +} + fn benchSpace(allocator: std.mem.Allocator, options: Options) !void { var state = try initBench(allocator); defer state.deinit(); @@ -842,6 +913,18 @@ fn listWorker(did: []const u8, latencies: []u64) !void { } } +fn repoCarWorker(did: []const u8, latencies: []u64) !void { + var arena = std.heap.ArenaAllocator.init(std.heap.page_allocator); + defer arena.deinit(); + for (latencies) |*latency| { + _ = arena.reset(.retain_capacity); + const start = nowNs(); + const car = try zds.storage.store.writeRepoCar(arena.allocator(), did); + if (car.len == 0) return error.EmptyRepoCar; + latency.* = nowNs() - start; + } +} + const CpuPath = enum { decode, render }; fn benchRecordBlockCpu(allocator: std.mem.Allocator, options: Options, path: CpuPath) !void { @@ -913,6 +996,19 @@ fn seedIndexedRecords(allocator: std.mem.Allocator, account: zds.auth.tokens.Acc } } +fn appendIndexedRecordRev(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, index: usize) ![]const u8 { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const value = try benchRecordValue(arena.allocator(), index); + const result = try zds.storage.store.applyWrites(arena.allocator(), account, &.{.{ .create = .{ + .collection = bench_collection, + .rkey = try std.fmt.allocPrint(arena.allocator(), "rec{d:0>8}", .{index}), + .value = value, + } }}); + if (result.records.len != 1) return error.UnexpectedWriteResult; + return try allocator.dupe(u8, result.commit.rev); +} + fn concurrentResult(name: []const u8, level: ConcurrencyLevel, total_ops: usize, elapsed_ns: u64, latencies: []u64) ConcurrentResult { return .{ .name = name, @@ -1027,6 +1123,7 @@ fn parseScenario(value: []const u8) !Scenario { if (std.mem.eql(u8, value, "get-record")) return .get_record; if (std.mem.eql(u8, value, "list-records")) return .list_records; if (std.mem.eql(u8, value, "repo-size")) return .repo_size; + if (std.mem.eql(u8, value, "get-repo")) return .get_repo; if (std.mem.eql(u8, value, "space")) return .space; if (std.mem.eql(u8, value, "write-profile")) return .write_profile; return error.UnknownScenario; @@ -1034,7 +1131,7 @@ fn parseScenario(value: []const u8) !Scenario { fn usage() void { std.debug.print( - \\usage: zds-bench [--scenario all|write|read|blob|repo|metastore|get-cid|get-block|decode-record|render-record|get-record|list-records|repo-size|space|write-profile] [--records N] [--blobs N] [--blob-size BYTES] [--callers N --ops-per-caller N] + \\usage: zds-bench [--scenario all|write|read|blob|repo|metastore|get-cid|get-block|decode-record|render-record|get-record|list-records|repo-size|get-repo|space|write-profile] [--records N] [--blobs N] [--blob-size BYTES] [--callers N --ops-per-caller N] \\ , .{}); } diff --git a/docs/getrepo-notes.md b/docs/getrepo-notes.md index d82cd8e..ed3b625 100644 --- a/docs/getrepo-notes.md +++ b/docs/getrepo-notes.md @@ -25,7 +25,9 @@ Done in this pass: 3. Changed full export from whole-DID `repo_blocks` scan to current-root reachability. 4. Added lightweight route backpressure for full `getRepo`, configurable with `ZDS_MAX_CONCURRENT_REPO_EXPORTS`. 5. Ran `just bench repo-size`; reachable full export measured about 1.4 ms for 100 records, 16.8 ms for 1,000 records, and 95.7 ms for 5,000 records on the local synthetic benchmark. +6. Added `just bench get-repo` as the focused storage-pressure lane for this route. It measures full export construction, one-commit `since` export construction, and concurrent full export latency against one seeded repo. Suggested next pass: -1. Add a larger deliberate `just bench repo-large 100000` run when we want stress numbers comparable to large public repos. +1. Add a larger deliberate `just bench get-repo 100000 4 1` run when we want stress numbers comparable to large public repos. +2. Add a cross-PDS HTTP/export comparison in `atproto-bench` once the ZDS-local storage seam is stable.