Something went wrong. Try again.
bayes for days
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531"""`scripts/ab.py` decides which performance numbers get believed.
Its statistics and its parsing are therefore the thing to pin: a selector thatquietly matches the wrong row, or a refusal that does not fire, puts a wrongfigure into a PR body with a tool's authority behind it."""
import importlib.utilimport jsonimport osimport statimport subprocessimport sysimport tempfileimport unittestfrom pathlib import Path
REPO = Path(__file__).resolve().parent.parentSCRIPT = REPO / "scripts" / "ab.py"
_spec = importlib.util.spec_from_file_location("ab", SCRIPT)ab = importlib.util.module_from_spec(_spec)_spec.loader.exec_module(ab)
# What `crates/sds-core/examples/perf.rs` actually prints, trimmed.PERF_OUTPUT = """8v8, 544 hexes, 30 features, 8 strategies, single thread
threat map (8 enemies) 14.9 us160 LOS queries, no cache 131.2 us1280 LOS queries, cached 301.7 usstance blend (S^T M) 0.1 usscore 160 candidates 2.7 usfire allocation (48 shots, 8 tgt) 188.4 us shot cache: CacheStats { hits: 4838, misses: 10 }per-location damage, cold cache 19.3 usper-location damage, warm cache 0.6 us20 candidates, per-location 33.4 us location cache: CacheStats { hits: 100, misses: 8 }
observation payload: 28114 bytesparse one observation 291.0 us"""
def runs(before_metric, after_metric, before_control, after_control, pairs=10): """Two builds' worth of runs with fixed figures, for the arithmetic.""" out = [] for pair in range(pairs): out.append(ab.Run("before", pair, before_metric, before_control)) out.append(ab.Run("after", pair, after_metric, after_control)) return out
class Selectors(unittest.TestCase): def test_a_bare_row_name_takes_the_number_after_it(self): selector = ab.compile_selector("score 160 candidates") self.assertAlmostEqual(ab.scrape(PERF_OUTPUT, selector), 2.7)
def test_an_explicit_group_is_used_as_given(self): selector = ab.compile_selector(r"observation payload: ([0-9]+) bytes") self.assertEqual(ab.scrape(PERF_OUTPUT, selector), 28114.0)
def test_a_regex_row_name_still_works(self): selector = ab.compile_selector(r"per-location damage, warm.*") self.assertAlmostEqual(ab.scrape(PERF_OUTPUT, selector), 0.6)
def test_a_row_name_containing_a_number_takes_the_figure_not_the_name(self): # "score 160 candidates" holds two numbers. The figure is the last one. selector = ab.compile_selector("candidates, per-location") self.assertAlmostEqual(ab.scrape(PERF_OUTPUT, selector), 33.4)
def test_a_matching_line_with_no_number_is_an_error(self): selector = ab.compile_selector("headline") with self.assertRaises(ab.AbError): ab.scrape("headline\nsomething 4 us\n", selector)
def test_more_than_one_group_is_refused(self): with self.assertRaises(ab.AbError): ab.compile_selector(r"([a-z]+) ([0-9]+)")
def test_a_broken_regex_is_refused(self): with self.assertRaises(ab.AbError): ab.compile_selector("score (160")
def test_a_selector_matching_nothing_is_an_error(self): selector = ab.compile_selector("best_volley_fast") with self.assertRaises(ab.AbError): ab.scrape(PERF_OUTPUT, selector)
def test_an_ambiguous_selector_is_an_error(self): # Two rows begin "per-location damage". Picking the first silently is # exactly how a run measures something nobody asked about. selector = ab.compile_selector("per-location damage") with self.assertRaises(ab.AbError) as caught: ab.scrape(PERF_OUTPUT, selector) self.assertIn("2 times", str(caught.exception))
def test_a_captured_non_number_is_an_error(self): selector = ab.compile_selector(r"shot cache: (\w+)") with self.assertRaises(ab.AbError): ab.scrape(PERF_OUTPUT, selector)
def test_negative_and_exponent_numbers_parse(self): selector = ab.compile_selector("drift") self.assertEqual(ab.scrape("drift -1.5e-3 s", selector), -1.5e-3)
class Normalisation(unittest.TestCase): def test_a_machine_wide_slowdown_cancels(self): # The "after" build ran while the machine was twice as slow. Both its # figures doubled, so the controlled answer is no change. summary = ab.summarise(runs(100.0, 200.0, 50.0, 100.0)) self.assertAlmostEqual(summary.raw, 2.0) self.assertAlmostEqual(summary.normalised, 1.0)
def test_a_real_speedup_survives_a_slowdown(self): # Metric halves in the code; the machine is 20% slower for the after # runs, which the control sees too. 20% is past the default tolerance, # so this only reads at all with the tolerance widened - but the # arithmetic underneath still recovers the halving. summary = ab.summarise(runs(100.0, 60.0, 50.0, 60.0), tolerance=0.25) self.assertAlmostEqual(summary.normalised, 0.5) self.assertTrue(summary.readable)
def test_the_raw_reading_is_reported_but_flagged_uncontrolled(self): summary = ab.summarise(runs(100.0, 60.0, 50.0, 60.0)) self.assertAlmostEqual(summary.raw, 0.6) self.assertNotAlmostEqual(summary.raw, summary.normalised)
def test_medians_not_means(self): # One run landed under somebody else's build. A mean would carry it. sample = [ ab.Run("before", 0, 100.0, 10.0), ab.Run("before", 1, 100.0, 10.0), ab.Run("before", 2, 5000.0, 500.0), ab.Run("after", 0, 100.0, 10.0), ab.Run("after", 1, 100.0, 10.0), ab.Run("after", 2, 100.0, 10.0), ] summary = ab.summarise(sample) self.assertEqual(summary.before_metric, 100.0)
def test_percent_reads_negative_for_faster(self): self.assertEqual(ab.percent(0.9), "-10.0%") self.assertEqual(ab.percent(1.25), "+25.0%")
class Divergence(unittest.TestCase): """The two statistics disagreeing is itself a reading.
On seven samples one afternoon, raw medians said 7% and within-run normalisation said 24%. Neither was wrong arithmetic; the gap was the sample being too small to carry either. """
def test_agreeing_statistics_raise_nothing(self): summary = ab.summarise(runs(100.0, 80.0, 50.0, 50.0)) self.assertLess(summary.divergence, 0.01) self.assertEqual(summary.warnings, [])
def diverging(self): """Eight pairs whose control moves without the metric moving with it.
The control's median is unchanged, so the refusal does not fire; but which runs the two medians land on differs, and the raw and normalised readings come apart by ten points. """ sample = [] after = [(70.0, 140.0), (70.0, 60.0), (90.0, 100.0), (90.0, 100.0)] * 2 for pair, (metric, control) in enumerate(after): sample.append(ab.Run("before", pair, 100.0, 100.0)) sample.append(ab.Run("after", pair, metric, control)) return sample
def test_disagreeing_statistics_are_called_out(self): summary = ab.summarise(self.diverging()) self.assertTrue(summary.readable) self.assertAlmostEqual(summary.raw, 0.8) self.assertAlmostEqual(summary.normalised, 0.9) self.assertAlmostEqual(summary.divergence, 0.1) self.assertTrue(any("disagree" in w for w in summary.warnings))
def test_the_warning_names_both_figures(self): summary = ab.summarise(self.diverging()) warning = next(w for w in summary.warnings if "disagree" in w) self.assertIn("-20.0%", warning) self.assertIn("-10.0%", warning)
def test_both_figures_are_printed_on_the_result_line(self): args = ab.parse_args(["--before", "b", "--after", "a", "--metric", "m", "--control", "c"]) sample = runs(100.0, 80.0, 50.0, 52.0) text = ab.report(args, sample, ab.summarise(sample), (1.0,) * 3, (1.0,) * 3) self.assertIn("normalised", text) self.assertIn("raw", text)
def test_the_threshold_is_a_flag(self): summary = ab.summarise(self.diverging(), divergence=1.0) self.assertFalse(any("disagree" in w for w in summary.warnings))
class Refusal(unittest.TestCase): def test_a_control_that_moved_refuses_the_whole_result(self): # The A/B where the control itself came out at 1.43. summary = ab.summarise(runs(100.0, 60.0, 50.0, 71.5)) self.assertFalse(summary.readable) self.assertIn("control moved", summary.refusal)
def test_the_refusal_is_symmetric(self): summary = ab.summarise(runs(100.0, 60.0, 50.0, 25.0)) self.assertFalse(summary.readable)
def test_a_steady_control_is_readable(self): summary = ab.summarise(runs(100.0, 60.0, 50.0, 52.0)) self.assertTrue(summary.readable) self.assertIsNone(summary.refusal)
def test_the_tolerance_is_the_boundary(self): drifted = runs(100.0, 100.0, 50.0, 55.0) self.assertFalse(ab.summarise(drifted, tolerance=0.05).readable) self.assertTrue(ab.summarise(drifted, tolerance=0.2).readable)
def test_too_few_pairs_warns_rather_than_refusing(self): summary = ab.summarise(runs(100.0, 90.0, 50.0, 50.0, pairs=3)) self.assertTrue(summary.readable) self.assertTrue(any("3 pairs" in w for w in summary.warnings))
def test_a_noisy_control_warns(self): sample = [] for pair, control in enumerate([10.0, 40.0, 12.0, 60.0, 11.0, 55.0]): sample.append(ab.Run("before", pair, control * 2, control)) sample.append(ab.Run("after", pair, control * 2, control)) summary = ab.summarise(sample) self.assertTrue(any("spread" in w for w in summary.warnings))
def test_one_sided_runs_are_an_error(self): with self.assertRaises(ab.AbError): ab.summarise([ab.Run("before", 0, 1.0, 1.0)])
def test_the_refusal_is_printed_and_the_result_line_is_not(self): args = ab.parse_args(["--before", "b", "--after", "a", "--metric", "m", "--control", "c"]) sample = runs(100.0, 60.0, 50.0, 71.5) text = ab.report(args, sample, ab.summarise(sample), (1.0, 1.0, 1.0), (2.0, 2.0, 2.0)) self.assertIn("REFUSED TO REPORT", text) self.assertNotIn("RESULT", text)
def test_a_readable_run_prints_the_result_and_the_load(self): args = ab.parse_args(["--before", "b", "--after", "a", "--metric", "m", "--control", "c"]) sample = runs(100.0, 60.0, 50.0, 50.0) text = ab.report(args, sample, ab.summarise(sample), (4.0, 1.0, 1.0), (7.5, 2.0, 2.0)) self.assertIn("RESULT", text) self.assertIn("-40.0%", text) self.assertIn("4.00", text) self.assertIn("7.50", text)
class Spread(unittest.TestCase): def test_a_flat_sample_has_no_spread(self): self.assertEqual(ab.spread([5.0] * 8), 0.0)
def test_spread_grows_with_the_scatter(self): tight = ab.spread([9.0, 10.0, 10.0, 11.0]) loose = ab.spread([1.0, 10.0, 10.0, 100.0]) self.assertLess(tight, loose)
def test_a_single_value_is_not_a_spread(self): self.assertEqual(ab.spread([3.0]), 0.0)
class Bootstrap(unittest.TestCase): def test_the_interval_brackets_the_point_estimate(self): before = [1.0, 1.05, 0.95, 1.02, 0.98, 1.01, 0.99, 1.03] after = [0.5, 0.52, 0.49, 0.51, 0.48, 0.5, 0.53, 0.5] lo, hi = ab.bootstrap_interval(before, after) self.assertLess(lo, 0.5) self.assertGreater(hi, 0.5)
def test_the_interval_is_reproducible(self): before = [1.0, 1.2, 0.9, 1.1, 1.05, 0.95] after = [0.8, 0.9, 0.7, 0.85, 0.75, 0.95] self.assertEqual(ab.bootstrap_interval(before, after), ab.bootstrap_interval(before, after))
def test_a_noisier_sample_gives_a_wider_interval(self): tight_lo, tight_hi = ab.bootstrap_interval([1.0] * 10, [0.9, 0.9, 0.91, 0.89] * 3) loose_lo, loose_hi = ab.bootstrap_interval([1.0] * 10, [0.2, 1.6, 0.9, 0.5] * 3) self.assertLess(tight_hi - tight_lo, loose_hi - loose_lo)
def test_too_few_samples_gives_no_interval(self): lo, hi = ab.bootstrap_interval([1.0], [1.0]) self.assertNotEqual(lo, lo) # nan
def fake_benchmark(directory, name, metric, control): """A stand-in binary that prints perf-shaped output under a moving machine.
Every figure it prints is multiplied by the same per-run factor, drawn from a fixed sequence that swings 1.0 to 3.0 - which is what a loaded machine does to a whole run at once, and the thing the within-run normalisation exists to divide out. Deterministic, so the assertions below are exact. """ path = Path(directory) / name path.write_text( "#!/usr/bin/env python3\n" "import pathlib\n" f"counter = pathlib.Path({str(path) + '.count'!r})\n" "n = int(counter.read_text()) if counter.exists() else 0\n" "counter.write_text(str(n + 1))\n" "factor = [1.0, 2.8, 1.2, 2.4, 1.1, 3.0, 1.3, 2.2][n % 8]\n" f"print('score 160 candidates %.4f us' % ({metric} * factor))\n" f"print('parse one observation %.4f us' % ({control} * factor))\n" ) path.chmod(path.stat().st_mode | stat.S_IEXEC) return str(path)
class EndToEnd(unittest.TestCase): """The three cases the tool has to get right, driven by a fake binary.
A live benchmark cannot be made to move its control on demand, so a real run can never check the refusal. These can. """
def ab(self, directory, after_metric, after_control, **extra): before = fake_benchmark(directory, "before", 10.0, 100.0) after = fake_benchmark(directory, "after", after_metric, after_control) argv = [ "--before", before, "--after", after, "--metric", "score 160 candidates", "--control", "parse one observation", "--pairs", "8", "--warmup", "0", "--quiet", ] for key, value in extra.items(): argv += [f"--{key}", str(value)] return ab.main(argv)
def test_a_real_improvement_with_a_steady_control_is_reported(self): with tempfile.TemporaryDirectory() as directory: out = os.path.join(directory, "ab.json") code = self.ab(directory, 8.0, 100.0, json=out) written = json.loads(Path(out).read_text()) self.assertEqual(code, 0) # The machine factor cancels exactly; the raw reading does not. self.assertAlmostEqual(written["summary"]["normalised"], 0.8) self.assertAlmostEqual(written["summary"]["control_drift"], 1.0)
def test_a_control_that_moved_one_and_a_half_times_refuses(self): with tempfile.TemporaryDirectory() as directory: code = self.ab(directory, 8.0, 150.0) self.assertEqual(code, 2)
def test_identical_builds_come_out_at_no_change(self): with tempfile.TemporaryDirectory() as directory: out = os.path.join(directory, "ab.json") code = self.ab(directory, 10.0, 100.0, json=out) written = json.loads(Path(out).read_text()) self.assertEqual(code, 0) self.assertAlmostEqual(written["summary"]["normalised"], 1.0)
def test_a_binary_that_fails_exits_one(self): with tempfile.TemporaryDirectory() as directory: before = fake_benchmark(directory, "before", 10.0, 100.0) code = ab.main( [ "--before", before, "--after", "/nonexistent/binary", "--metric", "score 160 candidates", "--control", "parse one observation", "--pairs", "2", "--warmup", "0", "--quiet", ] ) self.assertEqual(code, 1)
def test_the_builds_alternate_which_goes_first(self): with tempfile.TemporaryDirectory() as directory: before = fake_benchmark(directory, "before", 10.0, 100.0) after = fake_benchmark(directory, "after", 8.0, 100.0) args = ab.parse_args( [ "--before", before, "--after", after, "--metric", "m", "--control", "c", "--pairs", "4", "--warmup", "0", "--quiet", ] ) metric = ab.compile_selector("score 160 candidates") control = ab.compile_selector("parse one observation") order = [r.build for r in ab.collect(args, metric, control)] self.assertEqual( order, ["before", "after", "after", "before", "before", "after", "after", "before"], )
class Reanalysis(unittest.TestCase): """`--json` keeps the outputs, so a different row costs nothing to ask."""
def saved(self, directory): before = fake_benchmark(directory, "before", 10.0, 100.0) after = fake_benchmark(directory, "after", 8.0, 100.0) out = os.path.join(directory, "ab.json") code = ab.main( [ "--before", before, "--after", after, "--metric", "score 160 candidates", "--control", "parse one observation", "--pairs", "8", "--warmup", "0", "--quiet", "--json", out, ] ) self.assertEqual(code, 0) return out
def test_the_saved_run_can_be_re_scraped_the_other_way_round(self): with tempfile.TemporaryDirectory() as directory: out = self.saved(directory) metric = ab.compile_selector("parse one observation") control = ab.compile_selector("score 160 candidates") runs, start, end, names = ab.reanalyse(out, metric, control) summary = ab.summarise(runs) self.assertEqual(len(runs), 16) self.assertTrue(names[0].endswith("before")) # Swapping the two rows inverts the answer, which is the check that the # re-read is reading and not replaying a stored figure. self.assertAlmostEqual(summary.normalised, 1.25)
def test_re_reading_runs_no_binary(self): with tempfile.TemporaryDirectory() as directory: out = self.saved(directory) os.remove(os.path.join(directory, "before")) os.remove(os.path.join(directory, "after")) code = ab.main( [ "--from-json", out, "--metric", "score 160 candidates", "--control", "parse one observation", "--quiet", ] ) self.assertEqual(code, 0)
def test_a_json_without_outputs_says_to_re_run(self): with tempfile.TemporaryDirectory() as directory: out = os.path.join(directory, "old.json") Path(out).write_text( json.dumps({"runs": [{"build": "before", "pair": 0, "metric": 1, "control": 1}]}) ) with self.assertRaises(ab.AbError): ab.reanalyse(out, ab.compile_selector("a"), ab.compile_selector("b"))
class Interface(unittest.TestCase): def test_the_binaries_are_required_unless_re_reading(self): with self.assertRaises(SystemExit): ab.parse_args(["--metric", "m", "--control", "c"]) args = ab.parse_args(["--from-json", "x.json", "--metric", "m", "--control", "c"]) self.assertIsNone(args.before)
def test_arguments_after_a_double_dash_reach_the_binary(self): args = ab.parse_args( [ "--before", "b", "--after", "a", "--metric", "m", "--control", "c", "--", "--iterations", "50", ] ) self.assertEqual(args.rest, ["--iterations", "50"])
def test_the_defaults_are_the_ones_documented(self): args = ab.parse_args(["--before", "b", "--after", "a", "--metric", "m", "--control", "c"]) self.assertEqual(args.pairs, 15) self.assertEqual(args.tolerance, 0.10) self.assertEqual(args.min_pairs, 8)
def test_the_script_runs_as_a_program(self): done = subprocess.run( [sys.executable, str(SCRIPT), "--help"], capture_output=True, text=True ) self.assertEqual(done.returncode, 0) self.assertIn("--control", done.stdout)
if __name__ == "__main__": unittest.main()