From 60e0206250122fbb78ea69fd9d66cc084490eae0 Mon Sep 17 00:00:00 2001 From: zzstoatzz Date: Mon, 15 Jun 2026 11:22:24 -0500 Subject: [PATCH] Expand PDS benchmark coverage --- bench/README.md | 77 ++++++++++++++-- bench/justfile | 12 +++ bench/main.zig | 232 +++++++++++++++++++++++++++++++++++++++++++++++- 3 files changed, 311 insertions(+), 10 deletions(-) diff --git a/bench/README.md b/bench/README.md index a58b4ff..633552d 100644 --- a/bench/README.md +++ b/bench/README.md @@ -23,6 +23,9 @@ just bench decode-record 10 1000 just bench render-record 10 1000 just bench get-record 10 1000 just bench list-records 10 1000 +just bench repo-size 5000 +just bench repo-large 100000 +just bench space 1000 just bench http just bench official-pds just bench write-profile 10 500 @@ -47,6 +50,13 @@ just bench run --scenario write --records 10000 - `render-record`: DAG-CBOR decode plus JSON rendering, without SQLite. - `get-record`: full `com.atproto.repo.getRecord` storage materialization. - `list-records`: full `com.atproto.repo.listRecords` storage materialization. +- `repo-size`: synthetic small/medium/large repo probes for + `listRecords(limit=100)` and full repo CAR materialization. The default Just + target keeps this moderate; use `repo-large` for intentional 100k+ record + runs. +- `space`: experimental permissioned-data storage probes for space discovery, + private record writes, private record reads/lists, repo oplog catch-up, and + blob readback. - `http`: starts a local ReleaseFast ZDS server and measures a seeded HTTP route matrix with ApacheBench. This is the route-level check for the `httpz` server boundary and should be tracked separately from storage @@ -64,22 +74,39 @@ benchmarks. The goal is not broad coverage yet; it is to catch obviously bad access patterns introduced by the experimental `com.atproto.space.*` storage model. -When adding the first permissioned-data probe, keep it out of `all` until the -protocol shape settles. A useful first pass should seed one self-owned space, -one writer repo, and one small blob, then measure these paths as distinct units -of work: +`just bench space` seeds one self-owned space, one writer repo, and one small +blob, then measures these paths as distinct units of work: -- `createSpace` -- `createRecord` or `applyWrites` into a space writer repo +- `listSpaces` +- `createRecord` into a space writer repo - `getRecord` by `(space, repo, collection, rkey)` - `listRecords`, limit 50, index-only response shape -- ranged `getBlob` +- `getBlob` storage readback - `getRepoOplog` catch-up reads Comparison against Daniel's permissioned-data branch should live in this section once we have a repeatable way to run that PDS locally. Until then, keep ZDS numbers as local trend data and compare only equivalent operations. +## repo sizes + +The normal write/read benchmarks use small synthetic repos so they stay fast +enough to run while developing. `just bench repo-size` adds explicit +small/medium/large tiers for repo-size-sensitive paths: + +- small: 100 records +- medium: 1,000 records, unless the selected maximum is lower +- large: the `max_records` argument + +Use `just bench repo-large 100000` for a deliberate larger run. Large public +accounts are useful reference shapes because they stress repo export, record +indexes, and cursor/list behavior. As of June 15, 2026, the public appview +reports about 38k posts for `pfrazee.com` and about 87k posts for +`jcsalterego.bsky.social`, so a 100k synthetic tier is the right order of +magnitude for large-repo stress. Live network corpora should be captured into +fixtures before they become stable benchmarks. Do not make routine CI or +pre-commit checks depend on live PDS availability. + ## methodology - Build ZDS with `-Doptimize=ReleaseFast`. @@ -152,7 +179,41 @@ The pre-httpz column below is the original narrow route set run against commit | `getSession` | 19.9k req/s, p95 8 ms | 20.8k req/s, p95 6 ms | | `listRecords`, limit 50 | 19.8k req/s, p95 11 ms | 22.4k req/s, p95 7 ms | -## reference matrix +## coverage map + +High-level ZDS surface area represented in `bench/`: + +| PDS surface | representative benchmark | +|---|---| +| repo writes | `write`, `metastore`, `write-profile` | +| repo reads | `get-cid`, `get-block`, `get-record`, `list-records` | +| repo export/sync snapshots | `repo`, `repo-size` | +| sync event storage | `metastore` write path plus smoke tests | +| blob storage | `blob`, `space` blob readback | +| HTTP routing/serialization/auth overhead | `http` | +| OAuth/session/account routes | `http` route matrix | +| permissioned spaces | `space` | +| official PDS read-path comparison | `official-pds` | + +Still intentionally outside the routine benchmark gate: + +- passkey ceremony cryptography and browser UX +- mail delivery providers +- live appview proxy latency +- WebSocket subscriber fanout under network backpressure +- live public-repo corpora until captured as stable fixtures + +## implementation comparison + +The main PDS implementations do not expose identical benchmark seams, so keep +comparisons grouped by equivalent work: + +| implementation | comparable bench surface | notes | +|---|---|---| +| ZDS | `bench/main.zig`, `tools/http_bench.sh` | local SQLite/blobstore, storage microbenchmarks, HTTP route matrix, permissioned-space storage | +| Tranquil | `crates/tranquil-store/benches/metastore*.rs`, `blockstore.rs`, `eventlog.rs` | best match for metastore-shaped record/index/event/blockstore comparisons | +| official Bluesky PDS | `bench/official-pds-records.bench.ts` copied into `packages/pds` | best match for actor-store record CID/get/list/write paths | +| Pegasus | `pegasus/bench/bench_repository.ml`, `mist/bench/bench_mst.ml` | best match for repository block/MST internals; not a full HTTP/PDS route comparison yet | Run Tranquil's metastore benchmark: diff --git a/bench/justfile b/bench/justfile index 565c196..f60699d 100644 --- a/bench/justfile +++ b/bench/justfile @@ -57,6 +57,18 @@ get-record callers="10" ops_per_caller="1000": list-records callers="10" ops_per_caller="1000": {{zig}} build bench -Doptimize=ReleaseFast -- --scenario list-records --callers {{callers}} --ops-per-caller {{ops_per_caller}} +# benchmark listRecords and repo CAR materialization across synthetic repo sizes +repo-size max_records="5000": + {{zig}} build bench -Doptimize=ReleaseFast -- --scenario repo-size --records {{max_records}} + +# deliberately slow large-repo lane for Paul/Jerry-shaped synthetic repositories +repo-large records="100000": + {{zig}} build bench -Doptimize=ReleaseFast -- --scenario repo-size --records {{records}} + +# benchmark experimental permissioned-space storage paths +space records="1000": + {{zig}} build bench -Doptimize=ReleaseFast -- --scenario space --records {{records}} + # run the official Bluesky PDS read probe from a local atproto checkout official-pds: ./run-official-pds.sh diff --git a/bench/main.zig b/bench/main.zig index d8834f5..44d19ac 100644 --- a/bench/main.zig +++ b/bench/main.zig @@ -14,10 +14,15 @@ const Scenario = enum { render_record, get_record, list_records, + repo_size, + space, write_profile, }; const bench_collection = "dev.zds.bench.record"; +const space_collection = "fm.plyr.stg.track"; +const space_type = "fm.plyr.stg.privateMedia"; +const valid_bench_did = "did:plc:zdsbenchzdsbenchzdsbenchzd"; const Options = struct { scenario: Scenario = .all, @@ -117,6 +122,8 @@ pub fn main(init: std.process.Init) !void { .render_record => try benchRenderRecord(allocator, options), .get_record => try benchGetRecord(allocator, options), .list_records => try benchListRecords(allocator, options), + .repo_size => try benchRepoSizes(allocator, options.records), + .space => try benchSpace(allocator, options), .write_profile => try benchWriteProfile(allocator, options), } } @@ -138,7 +145,7 @@ fn initBench(allocator: std.mem.Allocator) !BenchState { "bench.test", "bench@test.com", "password", - "did:plc:zdsbench", + valid_bench_did, true, ); @@ -467,6 +474,216 @@ fn benchListRecords(allocator: std.mem.Allocator, options: Options) !void { result.print(); } +fn benchRepoSizes(allocator: std.mem.Allocator, max_records: usize) !void { + const tiers = [_]struct { + name: []const u8, + records: usize, + }{ + .{ .name = "repo small", .records = 100 }, + .{ .name = "repo medium", .records = @min(1000, max_records) }, + .{ .name = "repo large", .records = max_records }, + }; + + std.debug.print("\n=== zds repo-size benchmarks ===\n", .{}); + var last_records: usize = 0; + for (tiers) |tier| { + if (tier.records == 0 or tier.records == last_records) continue; + last_records = tier.records; + std.debug.print("seeding {s} with {d} records\n", .{ tier.name, tier.records }); + var state = try initBench(allocator); + errdefer state.deinit(); + try seedIndexedRecords(allocator, state.account, tier.records); + (try benchListRecordsForRepo(allocator, state.account, tier.records)).print(); + (try benchRepoCarForRepo(allocator, state.account, tier.records)).print(); + state.deinit(); + } +} + +fn benchListRecordsForRepo( + allocator: std.mem.Allocator, + account: zds.auth.tokens.Account, + records: usize, +) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = if (records >= 100_000) 50 else 200; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const found = try zds.storage.store.listRecords(arena.allocator(), account.did, bench_collection, 100); + if (found.len == 0) return error.MissingRecords; + } + return .{ + .name = "repo tier list", + .ops = iterations, + .elapsed_ns = nowNs() - start, + }; +} + +fn benchRepoCarForRepo( + allocator: std.mem.Allocator, + account: zds.auth.tokens.Account, + records: usize, +) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = if (records >= 100_000) 2 else 5; + var total_bytes: usize = 0; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const car = try zds.storage.store.writeRepoCar(arena.allocator(), account.did); + if (car.len == 0) return error.EmptyRepoCar; + total_bytes += car.len; + } + return .{ + .name = "repo tier CAR", + .ops = iterations, + .bytes = total_bytes, + .elapsed_ns = nowNs() - start, + }; +} + +fn benchSpace(allocator: std.mem.Allocator, options: Options) !void { + var state = try initBench(allocator); + defer state.deinit(); + + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const a = arena.allocator(); + const records = options.records; + const space = try seedSpaceRecords(a, state.account, records); + const cid = try zds.storage.store.putBlob(a, std.Options.debug_io, state.account, "hello permissioned audio", "audio/mpeg"); + + std.debug.print("\n=== zds permissioned-space benchmarks ===\n", .{}); + (try benchSpaceListSpaces(allocator, state.account, space.uri)).print(); + (try benchSpaceWrite(allocator, state.account, space.uri, records)).print(); + (try benchSpaceGetRecord(allocator, state.account, space.uri, records)).print(); + (try benchSpaceListRecords(allocator, state.account, space.uri, records)).print(); + (try benchSpaceOplog(allocator, state.account, space.uri)).print(); + (try benchSpaceBlob(allocator, state.account, cid)).print(); +} + +fn seedSpaceRecords( + allocator: std.mem.Allocator, + account: zds.auth.tokens.Account, + records: usize, +) !zds.storage.store.SpaceConfig { + const space = try zds.storage.store.createSpace(allocator, .{ + .actor_did = account.did, + .owner_did = account.did, + .space_type = space_type, + .skey = "self", + .is_owner = true, + .managing_app = "https://api-stg.plyr.fm", + .is_public = false, + .app_access_mode = "allow", + .app_exceptions_json = "[]", + }); + for (0..records) |i| { + const value = try spaceRecordValue(allocator, i); + const prepared = try zds.storage.store.prepareRecordValue( + allocator, + space_collection, + try std.fmt.allocPrint(allocator, "track{d:0>8}", .{i}), + value, + ); + _ = try zds.storage.store.createSpaceRecord( + allocator, + space.uri, + account.did, + space_collection, + try std.fmt.allocPrint(allocator, "track{d:0>8}", .{i}), + prepared, + ); + } + return space; +} + +fn benchSpaceListSpaces(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, space: []const u8) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 500; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const spaces = try zds.storage.store.listSpaces(arena.allocator(), account.did, account.did, space_type, null, 50); + if (spaces.len == 0 or !std.mem.eql(u8, spaces[0].uri, space)) return error.MissingSpace; + } + return .{ .name = "space listSpaces", .ops = iterations, .elapsed_ns = nowNs() - start }; +} + +fn benchSpaceWrite(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, space: []const u8, offset: usize) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 100; + const start = nowNs(); + for (0..iterations) |i| { + _ = arena.reset(.retain_capacity); + const idx = offset + i; + const rkey = try std.fmt.allocPrint(arena.allocator(), "track{d:0>8}", .{idx}); + const value = try spaceRecordValue(arena.allocator(), idx); + const prepared = try zds.storage.store.prepareRecordValue(arena.allocator(), space_collection, rkey, value); + const result = try zds.storage.store.createSpaceRecord(arena.allocator(), space, account.did, space_collection, rkey, prepared); + if (result.cid.len == 0) return error.UnexpectedWriteResult; + } + return .{ .name = "space createRecord", .ops = iterations, .elapsed_ns = nowNs() - start }; +} + +fn benchSpaceGetRecord(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, space: []const u8, records: usize) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 1000; + const start = nowNs(); + for (0..iterations) |i| { + _ = arena.reset(.retain_capacity); + const idx = if (records == 0) 0 else (i * 13) % records; + const rkey = try std.fmt.allocPrint(arena.allocator(), "track{d:0>8}", .{idx}); + const record = (try zds.storage.store.getSpaceRecord(arena.allocator(), space, account.did, space_collection, rkey)) orelse return error.MissingRecord; + if (record.value_json.len == 0) return error.MissingRecord; + } + return .{ .name = "space getRecord", .ops = iterations, .elapsed_ns = nowNs() - start }; +} + +fn benchSpaceListRecords(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, space: []const u8, records: usize) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 500; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const found = try zds.storage.store.listSpaceRecords(arena.allocator(), space, account.did, space_collection, null, false, 50); + if (records > 0 and found.len == 0) return error.MissingRecords; + } + return .{ .name = "space listRecords", .ops = iterations, .elapsed_ns = nowNs() - start }; +} + +fn benchSpaceOplog(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, space: []const u8) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 500; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const ops = try zds.storage.store.listSpaceRecordOplog(arena.allocator(), space, account.did, null, 100); + if (ops.len == 0) return error.MissingRecords; + } + return .{ .name = "space getRepoOplog", .ops = iterations, .elapsed_ns = nowNs() - start }; +} + +fn benchSpaceBlob(allocator: std.mem.Allocator, account: zds.auth.tokens.Account, cid: []const u8) !BenchResult { + var arena = std.heap.ArenaAllocator.init(allocator); + defer arena.deinit(); + const iterations: usize = 1000; + const start = nowNs(); + for (0..iterations) |_| { + _ = arena.reset(.retain_capacity); + const blob = zds.storage.store.getBlob(arena.allocator(), account.did, cid) orelse return error.MissingBlob; + if (blob.data.len == 0) return error.MissingBlob; + } + return .{ .name = "space getBlob", .ops = iterations, .bytes = iterations * "hello permissioned audio".len, .elapsed_ns = nowNs() - start }; +} + fn benchDecodeRecord(allocator: std.mem.Allocator, options: Options) !void { try benchRecordBlockCpu(allocator, options, .decode); } @@ -748,6 +965,15 @@ fn benchRecordValue(allocator: std.mem.Allocator, index: usize) !std.json.Value return try std.json.parseFromSliceLeaky(std.json.Value, allocator, json, .{}); } +fn spaceRecordValue(allocator: std.mem.Allocator, index: usize) !std.json.Value { + const json = try std.fmt.allocPrint( + allocator, + "{{\"$type\":\"{s}\",\"title\":\"private track {d}\",\"createdAt\":\"2026-06-15T00:{d:0>2}:00.000Z\",\"audioRef\":\"bench-audio-{d}\",\"size\":{d}}}", + .{ space_collection, index, index % 60, index, 1024 + index }, + ); + return try std.json.parseFromSliceLeaky(std.json.Value, allocator, json, .{}); +} + fn parseOptions(init: std.process.Init) !Options { var options: Options = .{}; var args = std.process.Args.Iterator.init(init.minimal.args); @@ -800,13 +1026,15 @@ fn parseScenario(value: []const u8) !Scenario { if (std.mem.eql(u8, value, "render-record")) return .render_record; if (std.mem.eql(u8, value, "get-record")) return .get_record; if (std.mem.eql(u8, value, "list-records")) return .list_records; + if (std.mem.eql(u8, value, "repo-size")) return .repo_size; + if (std.mem.eql(u8, value, "space")) return .space; if (std.mem.eql(u8, value, "write-profile")) return .write_profile; return error.UnknownScenario; } fn usage() void { std.debug.print( - \\usage: zds-bench [--scenario all|write|read|blob|repo|metastore|get-cid|get-block|decode-record|render-record|get-record|list-records|write-profile] [--records N] [--blobs N] [--blob-size BYTES] [--callers N --ops-per-caller N] + \\usage: zds-bench [--scenario all|write|read|blob|repo|metastore|get-cid|get-block|decode-record|render-record|get-record|list-records|repo-size|space|write-profile] [--records N] [--blobs N] [--blob-size BYTES] [--callers N --ops-per-caller N] \\ , .{}); } -- 2.51.2