#!/usr/bin/env nu # criterion benchmarks for car parsing, cid checks, mst walks, the sparse scan and the record # check (benches/repo.rs), with a regression gate. # # usage: # nu tests/bench.nu fetch # download the bench repos (once, then kept) # nu tests/bench.nu compare-ref [filter] # this tree against : the gate # nu tests/bench.nu save [criterion args] # run the benches, save baseline # nu tests/bench.nu compare [criterion args] # run again and compare against # # a benchmark looks slower when even the low end of criterion's 95% confidence interval for its # mean change is more than --threshold percent (default 10). the machine drifts: two identical # binaries run back to back differ by 10-15% on the long benchmarks, the second run slower on # some and faster on others. so compare-ref builds both sides first, runs each benchmark with # `ref`, this tree, this tree, `ref`, and judges the geometric mean of the two comparisons, which # cancels a drift that grows steadily across the four runs. criterion's interval only covers the # variation within one process: on identical code, one full run still moved single benchmarks by # up to 6.3%, hence the threshold. all of them take about 15 minutes on a laptop. # # save/compare measure minutes apart and are only a rough check within one sitting. baselines # never carry across machines, and each saved baseline records the sha256 of the repo snapshots # it ran on. # # tight loops are sensitive to where the linker puts them: shifting every function by 8 bytes # once made parse_commit_car/trust 40% slower with no code change. the bench binary compiles only # src/car.rs, mst.rs, sparse_mst.rs and allocator.rs, so no other change can move it, but a # change to one of those can move the others: a benchmark that regressed on a change that does # not touch its path may be layout rather than the change. const repos = [ { name: "atproto.com", did: "did:plc:ewvi7nxzyoun6zhxrhs64oiz" } { name: "jay.bsky.team", did: "did:plc:oky5czdrnfjpqslsw2a5iclo" } { name: "pfrazee.com", did: "did:plc:ragtjsm2j2vknwkz3zp4oxrd" } ] def repo-root [] { $env.FILE_PWD | path join ".." | path expand } def target-dir [] { $env.CARGO_TARGET_DIR? | default (repo-root | path join "target") | path expand } def data-dir [] { $env.HYDRANT_BENCH_DATA? | default (target-dir | path join "bench-data") | path expand } def criterion-dir [] { target-dir | path join "criterion" } def bench-env [] { { CARGO_TARGET_DIR: (target-dir) HYDRANT_BENCH_DATA: (data-dir) CRITERION_HOME: (criterion-dir) } } def inputs-file [baseline: string] { criterion-dir | path join $"($baseline).inputs.json" } # the repo snapshots a run benchmarks def inputs [] { $repos | each {|repo| let file = (data-dir | path join $"($repo.name).car") if not ($file | path exists) { error make {msg: $"missing ($file); run `nu tests/bench.nu fetch` first"} } { name: $repo.name, sha256: (open --raw $file | hash sha256) } } } def pds-of [did: string] { # served as application/did+ld+json, which nu leaves as text http get --raw $"https://plc.directory/($did)" | from json | get service | where id == "#atproto_pds" | first | get serviceEndpoint } def --wrapped cargo-bench [...args: string] { cd (repo-root) with-env (bench-env) { ^cargo bench --bench repo -- ...$args } } # builds the bench binary of the checkout at `dir` into `target` and returns its path def bench-binary [dir: string, target: string] { print $"building the benches in ($dir)..." cd $dir let built = (with-env (bench-env | merge {CARGO_TARGET_DIR: $target}) { ^cargo bench --bench repo --no-run | complete }) if $built.exit_code != 0 { error make {msg: $"building the benches in ($dir) failed:\n($built.stderr)"} } let exe = ($built.stderr | parse --regex 'Executable benches/repo\.rs \((?[^)]+)\)') $dir | path join $exe.0.path | path expand } def bench-ids [binary: string, filter?: string] { let args = if $filter == null { [] } else { [$filter] } with-env (bench-env) { ^$binary --bench --list ...$args } | lines | where {|line| $line ends-with ": benchmark" } | each {|line| $line | str replace ": benchmark" "" } } def --wrapped run-binary [binary: string, ...args: string] { let ran = (with-env (bench-env) { ^$binary --bench ...$args | complete }) if $ran.exit_code != 0 { error make {msg: $"($binary) ($args | str join ' ') failed:\n($ran.stdout)($ran.stderr)"} } } def clear-changes [] { glob $"(criterion-dir)/**/change" | each {|dir| rm --recursive $dir } | ignore } # a comparison criterion wrote: the relative change in mean time, with its 95% confidence interval def read-change [file: string] { let mean = (open $file | get mean) { change: $mean.point_estimate low: $mean.confidence_interval.lower_bound high: $mean.confidence_interval.upper_bound } } # the comparisons of the last run, one per benchmark def changes [] { glob $"(criterion-dir)/**/change/estimates.json" | each {|file| { bench: ($file | path dirname | path dirname | path relative-to (criterion-dir)) } | merge (read-change $file) } | sort-by bench } # a change of `a` against `b` as the change of `b` against `a` def invert [] { let row = $in { change: (1 / (1 + $row.change) - 1) low: (1 / (1 + $row.high) - 1) high: (1 / (1 + $row.low) - 1) } } # the geometric mean of two changes of the same pair measured in opposite orders def counterbalance [first: record, second: record] { let mean = {|a, b| ((1 + $a) * (1 + $b) | math sqrt) - 1 } { change: (do $mean $first.change $second.change) low: (do $mean $first.low $second.low) high: (do $mean $first.high $second.high) } } def percent [ratio: float] { let rounded = ($ratio * 100 | math round --precision 1) if $rounded > 0 { $"+($rounded)%" } else { $"($rounded)%" } } def judge [threshold: float] { let rows = $in let limit = $threshold / 100 $rows | each {|row| # narrow terminals cut columns from the right, so the verdict comes right after the name { bench: $row.bench verdict: (if $row.low > $limit { "regressed" } else if $row.high < (0 - $limit) { "faster" } else { "same" }) change: (percent $row.change) "95% ci": $"(percent $row.low) .. (percent $row.high)" } | merge ($row | reject bench change low high) } } def show [] { let rows = $in if ($rows | is-empty) { error make {msg: "no benchmark had a baseline to compare against"} } print ($rows | table --index false) $rows } # benchmark `id` with `saving`, then right away with `comparing`: the change of `comparing` # against `saving` def compare-once [saving: string, comparing: string, id: string] { let change_dir = (criterion-dir | path join $id "change") rm --recursive --force $change_dir run-binary $saving --save-baseline compare-ref --exact $id run-binary $comparing --baseline compare-ref --exact $id read-change ($change_dir | path join "estimates.json") } def main [] { print "usage: nu tests/bench.nu (fetch | compare-ref [filter] | save | compare )" } # downloads the bench repos. existing files are kept: a repo grows as its owner posts, so a # fresh download is a different benchmark and every baseline saved on the old one goes stale. def "main fetch" [] { let dir = (data-dir) mkdir $dir for repo in $repos { let file = ($dir | path join $"($repo.name).car") if ($file | path exists) { print $"($repo.name): kept ($file)" continue } let pds = (pds-of $repo.did) print $"($repo.name): downloading from ($pds)..." # a cut-off download must not look like a snapshot let partial = $"($file).partial" http get --raw --headers {accept: "application/vnd.ipld.car"} $"($pds)/xrpc/com.atproto.sync.getRepo?did=($repo.did)" | save --raw --force $partial mv $partial $file } } # the benchmarks of this tree that are slower than at `ref` def compare-ref [scratch: string, ref: string, filter: any, threshold: float] { let tree = ($scratch | path join "tree") ^git -c core.hooksPath=/dev/null worktree add --detach $tree $ref # never this tree's target directory: cargo decides what is stale by mtime, and a fresh # checkout is newer than every file here, so a shared one would pass off one side's build # as the other's. kept between runs so only hydrant itself rebuilds. let built = try { { binary: (bench-binary $tree (target-dir | path join "bench-ref")) } } catch {|err| { failure: $err.msg } } ^git worktree remove --force $tree if "failure" in $built { error make {msg: $"building the benches at ($ref) failed: ($built.failure)"} } let ref_binary = $built.binary let tree_binary = (bench-binary (repo-root) (target-dir)) let ids = (bench-ids $tree_binary $filter) print $"($ids | length) benchmarks, each run with ($ref), this tree, this tree, ($ref):" let rows = $ids | each {|id| print $" ($id)" let ref_first = (compare-once $ref_binary $tree_binary $id) let tree_first = (compare-once $tree_binary $ref_binary $id | invert) { bench: $id } | merge (counterbalance $ref_first $tree_first) | merge { "ref first": (percent $ref_first.change), "tree first": (percent $tree_first.change) } } $rows | sort-by bench | judge $threshold | show | where verdict == "regressed" | get bench } # exits 1 when a benchmark of this tree is slower than at `ref` def "main compare-ref" [ref: string, filter?: string, --threshold: float = 10.0] { let scratch = (mktemp -d -t hydrant_bench.XXXXXX) let outcome = try { { slower: (compare-ref $scratch $ref $filter $threshold) } } catch {|err| { failure: $err.msg } } rm --recursive $scratch if "failure" in $outcome { error make {msg: $outcome.failure} } if ($outcome.slower | is-not-empty) { print $"slower than ($ref) by more than ($threshold)%: ($outcome.slower | str join ', ')" exit 1 } print $"no regression over ($threshold)% against ($ref)" } def --wrapped "main save" [baseline: string, ...args: string] { let snapshot = (inputs) cargo-bench --save-baseline $baseline ...$args $snapshot | save --force (inputs-file $baseline) } def --wrapped "main compare" [baseline: string, --threshold: float = 10.0, ...args: string] { let saved = (inputs-file $baseline) if not ($saved | path exists) { error make {msg: $"no baseline ($baseline); run `nu tests/bench.nu save ($baseline)` first"} } if (open $saved) != (inputs) { error make {msg: $"baseline ($baseline) ran on other repo snapshots than (data-dir); save it again"} } clear-changes cargo-bench --baseline-lenient $baseline ...$args let regressed = (changes | judge $threshold | show | where verdict == "regressed") if ($regressed | is-not-empty) { print $"slower than ($baseline) by more than ($threshold)%: ($regressed | get bench | str join ', ')" print "a saved baseline is minutes old, so machine drift shows up here too; confirm with compare-ref" exit 1 } print $"no regression over ($threshold)% against ($baseline)" }