From fd704eedf50f58ee85fd12ad3d71ddf233c6900d Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Wed, 30 Sep 2026 13:55:50 +0300 Subject: [PATCH] [bench] counterbalance compare-ref across both run orders --- AGENTS.md | 2 +- tests/bench.nu | 279 ++++++++++++++++++++++++++++++++++--------------- 2 files changed, 196 insertions(+), 85 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 5d155ec..68e2f8c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -139,7 +139,7 @@ Hydrant uses multiple `fjall` keyspaces: ### Testing - `nu tests/run_all.nu` - Runs all tests in parallel with automatically assigned free ports. Pass `--skip-creds` to skip tests requiring `.env` credentials or external account fixtures, or `--only=[...]` to run a subset. - `nu tests/feature_matrix.nu` - Checks every supported cargo feature combination compiles error- and warning-free (`--only [...]` for a subset). Run this after touching any `#[cfg(feature = ...)]` gate or adding/removing items in shared modules. The supported matrix lives in the script: `indexer` and `relay` are mutually exclusive modes; `indexer_stream`/`backlinks` require `indexer`; `jetstream` requires `indexer_stream` or `relay`; the bare no-features core must also build. -- `nu tests/bench.nu (fetch | save | compare | compare-ref )` - Criterion benchmarks (`benches/repo.rs`) for CAR parsing with and without CID checks, the full MST walk, the sparse scan, the record-storage check, and a full backfill minus the database, on real repos that `fetch` downloads once into `target/bench-data`. `compare` exits 1 when a benchmark is slower than the baseline by more than `--threshold` percent (default 5) at the low end of its 95% interval, twice in a row; `compare-ref main` benchmarks another ref in a worktree first. Baselines only compare on the machine that saved them. Run after changing any of those paths, from a normal-priority shell on a quiet machine. +- `nu tests/bench.nu (fetch | compare-ref [filter] | save | compare )` - Criterion benchmarks (`benches/repo.rs`) for CAR parsing with and without CID checks, the full MST walk, the sparse scan, the record-storage check, and a full backfill minus the database, on real repos that `fetch` downloads once into `target/bench-data`. `compare-ref main` is the regression gate: it builds `main` in a worktree and this tree, runs each benchmark `main`, this tree, this tree, `main` back to back, and exits 1 when the geometric mean of the two comparisons is slower by more than `--threshold` percent (default 10) at the low end of its 95% interval; the full set takes about 15 minutes. `save`/`compare` against a named baseline is only a rough check within one sitting, on the machine that saved it. Run after changing any of those paths, from a normal-priority shell on a quiet machine. - `nu tests/api_crawler_sources.nu` - Tests `/crawler/sources` CRUD plus dynamic/configured source restart behavior. - `nu tests/api_firehose_sources.nu` - Tests `/firehose/sources` CRUD behavior. - `nu tests/api_pds_tiers.nu` - Tests PDS tier APIs, tier persistence, and custom rate tiers. diff --git a/tests/bench.nu b/tests/bench.nu index b337c23..9eaa7de 100644 --- a/tests/bench.nu +++ b/tests/bench.nu @@ -1,18 +1,31 @@ #!/usr/bin/env nu # criterion benchmarks for car parsing, cid checks, mst walks, the sparse scan and the record -# check (benches/repo.rs), with a regression gate against saved baselines. +# check (benches/repo.rs), with a regression gate. # # usage: -# nu tests/bench.nu fetch # download the bench repos (once, then kept) -# nu tests/bench.nu save # run the benches, save them as baseline -# nu tests/bench.nu compare # run again, exit 1 on a regression against -# nu tests/bench.nu compare-ref # benchmark in a worktree, then compare +# nu tests/bench.nu fetch # download the bench repos (once, then kept) +# nu tests/bench.nu compare-ref [filter] # this tree against : the gate +# nu tests/bench.nu save [criterion args] # run the benches, save baseline +# nu tests/bench.nu compare [criterion args] # run again and compare against # -# extra arguments go to criterion, e.g. a filter: `nu tests/bench.nu compare main parse_car`. # a benchmark looks slower when even the low end of criterion's 95% confidence interval for its -# mean is more than --threshold percent (default 5) above the baseline. those run once more, and -# only the ones slower both times fail the comparison. baselines only compare on the machine -# that saved them and on the same repo snapshots, so each one records their sha256. +# mean change is more than --threshold percent (default 10). the machine drifts: two identical +# binaries run back to back differ by 10-15% on the long benchmarks, the second run slower on +# some and faster on others. so compare-ref builds both sides first, runs each benchmark with +# `ref`, this tree, this tree, `ref`, and judges the geometric mean of the two comparisons, which +# cancels a drift that grows steadily across the four runs. criterion's interval only covers the +# variation within one process: on identical code, one full run still moved single benchmarks by +# up to 6.3%, hence the threshold. all of them take about 15 minutes on a laptop. +# +# save/compare measure minutes apart and are only a rough check within one sitting. baselines +# never carry across machines, and each saved baseline records the sha256 of the repo snapshots +# it ran on. +# +# tight loops are sensitive to where the linker puts them: shifting every function by 8 bytes +# once made parse_commit_car/trust 40% slower with no code change. the bench binary compiles only +# src/car.rs, mst.rs, sparse_mst.rs and allocator.rs, so no other change can move it, but a +# change to one of those can move the others: a benchmark that regressed on a change that does +# not touch its path may be layout rather than the change. const repos = [ { name: "atproto.com", did: "did:plc:ewvi7nxzyoun6zhxrhs64oiz" } @@ -36,6 +49,14 @@ def criterion-dir [] { target-dir | path join "criterion" } +def bench-env [] { + { + CARGO_TARGET_DIR: (target-dir) + HYDRANT_BENCH_DATA: (data-dir) + CRITERION_HOME: (criterion-dir) + } +} + def inputs-file [baseline: string] { criterion-dir | path join $"($baseline).inputs.json" } @@ -51,16 +72,6 @@ def inputs [] { } } -def --wrapped run-benches [dir: string, ...args: string] { - cd $dir - let env_vars = { - CARGO_TARGET_DIR: (target-dir) - HYDRANT_BENCH_DATA: (data-dir) - CRITERION_HOME: (criterion-dir) - } - with-env $env_vars { ^cargo bench --bench repo -- ...$args } -} - def pds-of [did: string] { # served as application/did+ld+json, which nu leaves as text http get --raw $"https://plc.directory/($did)" @@ -71,70 +82,125 @@ def pds-of [did: string] { | get serviceEndpoint } +def --wrapped cargo-bench [...args: string] { + cd (repo-root) + with-env (bench-env) { ^cargo bench --bench repo -- ...$args } +} + +# builds the bench binary of the checkout at `dir` into `target` and returns its path +def bench-binary [dir: string, target: string] { + print $"building the benches in ($dir)..." + cd $dir + let built = (with-env (bench-env | merge {CARGO_TARGET_DIR: $target}) { + ^cargo bench --bench repo --no-run | complete + }) + if $built.exit_code != 0 { + error make {msg: $"building the benches in ($dir) failed:\n($built.stderr)"} + } + let exe = ($built.stderr | parse --regex 'Executable benches/repo\.rs \((?[^)]+)\)') + $dir | path join $exe.0.path | path expand +} + +def bench-ids [binary: string, filter?: string] { + let args = if $filter == null { [] } else { [$filter] } + with-env (bench-env) { ^$binary --bench --list ...$args } + | lines + | where {|line| $line ends-with ": benchmark" } + | each {|line| $line | str replace ": benchmark" "" } +} + +def --wrapped run-binary [binary: string, ...args: string] { + let ran = (with-env (bench-env) { ^$binary --bench ...$args | complete }) + if $ran.exit_code != 0 { + error make {msg: $"($binary) ($args | str join ' ') failed:\n($ran.stdout)($ran.stderr)"} + } +} + +def clear-changes [] { + glob $"(criterion-dir)/**/change" | each {|dir| rm --recursive $dir } | ignore +} + +# a comparison criterion wrote: the relative change in mean time, with its 95% confidence interval +def read-change [file: string] { + let mean = (open $file | get mean) + { + change: $mean.point_estimate + low: $mean.confidence_interval.lower_bound + high: $mean.confidence_interval.upper_bound + } +} + +# the comparisons of the last run, one per benchmark +def changes [] { + glob $"(criterion-dir)/**/change/estimates.json" + | each {|file| + { bench: ($file | path dirname | path dirname | path relative-to (criterion-dir)) } + | merge (read-change $file) + } + | sort-by bench +} + +# a change of `a` against `b` as the change of `b` against `a` +def invert [] { + let row = $in + { + change: (1 / (1 + $row.change) - 1) + low: (1 / (1 + $row.high) - 1) + high: (1 / (1 + $row.low) - 1) + } +} + +# the geometric mean of two changes of the same pair measured in opposite orders +def counterbalance [first: record, second: record] { + let mean = {|a, b| ((1 + $a) * (1 + $b) | math sqrt) - 1 } + { + change: (do $mean $first.change $second.change) + low: (do $mean $first.low $second.low) + high: (do $mean $first.high $second.high) + } +} + def percent [ratio: float] { let rounded = ($ratio * 100 | math round --precision 1) if $rounded > 0 { $"+($rounded)%" } else { $"($rounded)%" } } -# the comparison criterion wrote for each benchmark of the last run -def changes [threshold: float] { +def judge [threshold: float] { + let rows = $in let limit = $threshold / 100 - glob $"(criterion-dir)/**/change/estimates.json" - | each {|file| - let mean = (open $file | get mean) - let low = $mean.confidence_interval.lower_bound - let high = $mean.confidence_interval.upper_bound + $rows | each {|row| # narrow terminals cut columns from the right, so the verdict comes right after the name { - bench: ($file | path dirname | path dirname | path relative-to (criterion-dir)) - verdict: (if $low > $limit { "regressed" } else if $high < (0 - $limit) { "faster" } else { "same" }) - change: (percent $mean.point_estimate) - "95% ci": $"(percent $low) .. (percent $high)" + bench: $row.bench + verdict: (if $row.low > $limit { "regressed" } else if $row.high < (0 - $limit) { "faster" } else { "same" }) + change: (percent $row.change) + "95% ci": $"(percent $row.low) .. (percent $row.high)" } + | merge ($row | reject bench change low high) } - | sort-by bench } -def --wrapped run-compare [baseline: string, threshold: float, ...args: string] { - # only this run's comparisons count - glob $"(criterion-dir)/**/change" | each {|dir| rm --recursive $dir } | ignore - run-benches (repo-root) --baseline-lenient $baseline ...$args - let rows = (changes $threshold) +def show [] { + let rows = $in if ($rows | is-empty) { - error make {msg: $"no benchmark had baseline ($baseline) to compare against"} + error make {msg: "no benchmark had a baseline to compare against"} } print ($rows | table --index false) $rows } -def compare-against [baseline: string, threshold: float, args: list] { - let saved = (inputs-file $baseline) - if not ($saved | path exists) { - error make {msg: $"no baseline ($baseline); run `nu tests/bench.nu save ($baseline)` first"} - } - if (open $saved) != (inputs) { - error make {msg: $"baseline ($baseline) ran on other repo snapshots than (data-dir); save it again"} - } - - let suspects = (run-compare $baseline $threshold ...$args) | where verdict == "regressed" - if ($suspects | is-empty) { - print $"no regression over ($threshold)%" - return - } - # a busy or throttled machine slows whole groups for a while; a real regression reproduces - print $"($suspects | length) benchmarks look slower; running them again..." - let only = $suspects | get bench | each {|bench| $bench | str replace --all "." "\\." } - let confirmed = (run-compare $baseline $threshold $"^\(($only | str join '|')\)$") - | where verdict == "regressed" - if ($confirmed | is-not-empty) { - print $"regressed by more than ($threshold)% in both runs: ($confirmed | get bench | str join ', ')" - exit 1 - } - print "not reproduced on a second run, so no regression" +# benchmark `id` with `saving`, then right away with `comparing`: the change of `comparing` +# against `saving` +def compare-once [saving: string, comparing: string, id: string] { + let change_dir = (criterion-dir | path join $id "change") + rm --recursive --force $change_dir + run-binary $saving --save-baseline compare-ref --exact $id + run-binary $comparing --baseline compare-ref --exact $id + read-change ($change_dir | path join "estimates.json") } def main [] { - print "usage: nu tests/bench.nu (fetch | save | compare | compare-ref )" + print "usage: nu tests/bench.nu (fetch | compare-ref [filter] | save | compare )" } # downloads the bench repos. existing files are kept: a repo grows as its owner posts, so a @@ -158,33 +224,78 @@ def "main fetch" [] { } } -def --wrapped "main save" [baseline: string, ...args: string] { - let snapshot = (inputs) - run-benches (repo-root) --save-baseline $baseline ...$args - $snapshot | save --force (inputs-file $baseline) -} +# the benchmarks of this tree that are slower than at `ref` +def compare-ref [scratch: string, ref: string, filter: any, threshold: float] { + let tree = ($scratch | path join "tree") + ^git -c core.hooksPath=/dev/null worktree add --detach $tree $ref + # never this tree's target directory: cargo decides what is stale by mtime, and a fresh + # checkout is newer than every file here, so a shared one would pass off one side's build + # as the other's. kept between runs so only hydrant itself rebuilds. + let built = try { + { binary: (bench-binary $tree (target-dir | path join "bench-ref")) } + } catch {|err| + { failure: $err.msg } + } + ^git worktree remove --force $tree + if "failure" in $built { + error make {msg: $"building the benches at ($ref) failed: ($built.failure)"} + } + let ref_binary = $built.binary + let tree_binary = (bench-binary (repo-root) (target-dir)) + let ids = (bench-ids $tree_binary $filter) -def --wrapped "main compare" [baseline: string, --threshold: float = 5.0, ...args: string] { - compare-against $baseline $threshold $args + print $"($ids | length) benchmarks, each run with ($ref), this tree, this tree, ($ref):" + let rows = $ids | each {|id| + print $" ($id)" + let ref_first = (compare-once $ref_binary $tree_binary $id) + let tree_first = (compare-once $tree_binary $ref_binary $id | invert) + { bench: $id } + | merge (counterbalance $ref_first $tree_first) + | merge { "ref first": (percent $ref_first.change), "tree first": (percent $tree_first.change) } + } + $rows | sort-by bench | judge $threshold | show | where verdict == "regressed" | get bench } -# saves the benches of `ref` as baseline `ref-` from a temporary worktree sharing this -# target directory, then compares the working tree against it -def --wrapped "main compare-ref" [ref: string, --threshold: float = 5.0, ...args: string] { - let baseline = $"ref-($ref | str replace --all --regex '[^A-Za-z0-9._-]' '-')" - let snapshot = (inputs) - let tree = (mktemp -d -t hydrant_bench.XXXXXX) - ^git -c core.hooksPath=/dev/null worktree add --detach $tree $ref - let failure = try { - run-benches $tree --save-baseline $baseline ...$args - null +# exits 1 when a benchmark of this tree is slower than at `ref` +def "main compare-ref" [ref: string, filter?: string, --threshold: float = 10.0] { + let scratch = (mktemp -d -t hydrant_bench.XXXXXX) + let outcome = try { + { slower: (compare-ref $scratch $ref $filter $threshold) } } catch {|err| - $err.msg + { failure: $err.msg } } - ^git worktree remove --force $tree - if $failure != null { - error make {msg: $"benchmarking ($ref) failed: ($failure)"} + rm --recursive $scratch + if "failure" in $outcome { + error make {msg: $outcome.failure} } + if ($outcome.slower | is-not-empty) { + print $"slower than ($ref) by more than ($threshold)%: ($outcome.slower | str join ', ')" + exit 1 + } + print $"no regression over ($threshold)% against ($ref)" +} + +def --wrapped "main save" [baseline: string, ...args: string] { + let snapshot = (inputs) + cargo-bench --save-baseline $baseline ...$args $snapshot | save --force (inputs-file $baseline) - compare-against $baseline $threshold $args +} + +def --wrapped "main compare" [baseline: string, --threshold: float = 10.0, ...args: string] { + let saved = (inputs-file $baseline) + if not ($saved | path exists) { + error make {msg: $"no baseline ($baseline); run `nu tests/bench.nu save ($baseline)` first"} + } + if (open $saved) != (inputs) { + error make {msg: $"baseline ($baseline) ran on other repo snapshots than (data-dir); save it again"} + } + clear-changes + cargo-bench --baseline-lenient $baseline ...$args + let regressed = (changes | judge $threshold | show | where verdict == "regressed") + if ($regressed | is-not-empty) { + print $"slower than ($baseline) by more than ($threshold)%: ($regressed | get bench | str join ', ')" + print "a saved baseline is minutes old, so machine drift shows up here too; confirm with compare-ref" + exit 1 + } + print $"no regression over ($threshold)% against ($baseline)" } -- 2.51.2