From ec03f2f9227b5fd901e8d3558d0b444115c8f00a Mon Sep 17 00:00:00 2001 From: "@permadeath.com" Date: Sat, 29 Aug 2026 02:00:01 -0400 Subject: [PATCH] feat(tactics): measure how precise an agreement figure is `--subsets N` re-measures every pair on disjoint round-robin subsets, one pass over the corpus. The standard error is 0.006 to 0.014, which resolves every nearest-neighbour but leaves Entrench's band decided by which corpus was asked. --- plan/tactics.md | 53 +++++++++++++++ sds/cli.py | 39 +++++++++-- sds/tactics.py | 150 ++++++++++++++++++++++++++++++++++++++++-- tests/test_tactics.py | 63 ++++++++++++++++++ 4 files changed, 292 insertions(+), 13 deletions(-) diff --git a/plan/tactics.md b/plan/tactics.md index 453bbcc..829535c 100644 --- a/plan/tactics.md +++ b/plan/tactics.md @@ -226,6 +226,59 @@ tactics that diverge here are certainly two things; two that agree here might still diverge over a match, and that is the question a run with the layer switched on answers and this one does not. +## How precise is an agreement figure + +The bands sort tactics at 0.4 and 0.8 as though the figures were exact, and the +page prints two corpora side by side and calls the gap between them "at most +0.04 everywhere" without saying whether 0.04 is a lot. Nothing measured it. + +`sds tactics --subsets N` does: it splits a corpus into disjoint +round-robin subsets, re-measures every pair on each, and reports the spread. +One pass over the corpus however many subsets are used. Round-robin and not +contiguous blocks - the logs sort by scenario, so a block is a block of similar +boards and would understate the spread by measuring a narrower question. + +Measured over `tactics-hub-mirrored`, 300 matches and 8214 movement decisions, +six subsets of 50: + +- **Standard error of a full-corpus figure: 0.006 to 0.014**, worst pair + `break/harass`. +- So a difference under **0.054** between two full-corpus figures is not a + difference on the worst pair; per pair the threshold is 0.019 to 0.038. + +Three things follow, and they do not all point the same way. + +**The 'nearest' column stands.** All seven tactics' nearest neighbours are +resolved - the gap to the runner-up beats both figures' error in every case, +narrowly for `regroup` (0.034 against 0.030) and comfortably for the rest. + +**Five of the six published corpus-to-corpus differences are noise**, which is +the reassuring half: the two corpora agreed, and the page's "at most 0.04" +caveat was describing sampling error rather than disagreement. + +**The sixth is not, and it is the one that matters.** `engage`/`entrench` reads +0.38 on the mirrored corpus and 0.41 on the asymmetric one - a real difference +against a 0.019 threshold, and it lands **on opposite sides of the 0.4 +boundary**. So `Entrench` is "a new direction" if you ask the mirrored corpus +and "partly new" if you ask the asymmetric one, and the vocabulary's +classification of it is a property of which corpus was asked. The corpus that +disagreed is the one destroyed on 2026-08-29, so this cannot currently be +resolved either way. + +Two further pairs straddle 0.4 on their own error and are simply unresolved: +`flank`/`harass` at 0.380 with 95% [0.356, 0.404], and `harass`/`regroup` at +0.405 with 95% [0.387, 0.423]. Neither is a headline entry, but both are +printed in the matrix as though they were settled. + +- [x] Measure the sampling error of an agreement figure. `--subsets` +- [ ] Print the interval beside the figure in `docs/TACTICS.md`, and say which + band assignments the corpus does not resolve. A figure quoted to two + decimals inside a band whose boundary sits within its own error is the + shape of number this epic exists to stop +- [ ] `Entrench`'s band is unresolved and needs a second corpus to settle. It is + the concrete cost of the asymmetric corpus's loss - not a missing + confirmation, a missing tiebreak + - [ ] Re-measure once a force actually holds a tactic, and compare. The gap between on-policy and off-policy agreement is itself the measurement of how much a tactic changes the situations it faces diff --git a/sds/cli.py b/sds/cli.py index 36e0532..f806b1e 100644 --- a/sds/cli.py +++ b/sds/cli.py @@ -998,15 +998,28 @@ def cmd_labels(args: argparse.Namespace) -> int: def cmd_tactics(args: argparse.Namespace) -> int: """Re-score a recorded corpus under every tactic and report the matrix.""" - from .tactics import report + from .tactics import TacticsError, precision_report, report - print( - report( - corpus_dir(args.corpus), - phase=args.phase, - matches=args.matches, + try: + if args.subsets: + print( + precision_report( + corpus_dir(args.corpus), + phase=args.phase, + subsets=args.subsets, + ) + ) + return 0 + print( + report( + corpus_dir(args.corpus), + phase=args.phase, + matches=args.matches, + ) ) - ) + except TacticsError as error: + print(error, file=sys.stderr) + return 1 return 0 @@ -2249,6 +2262,18 @@ def main(argv: list[str] | None = None) -> int: "over a thousand decisions is already tighter than the bands " f"(default: {DEFAULT_TACTIC_MATCHES})", ) + tactics_p.add_argument( + "--subsets", + type=int, + default=0, + metavar="N", + help="instead of the matrix, split the corpus into N disjoint subsets, " + "re-measure every pair on each, and report how precise a full-corpus " + "figure is. The page sorts tactics into bands at 0.4 and 0.8 as though " + "the figures were exact; this says by how much they are not. Costs one " + "pass over the corpus however many subsets are used, because the " + "subsets are disjoint", + ) tactics_p.set_defaults(func=cmd_tactics) rank_p = sub.add_parser( diff --git a/sds/tactics.py b/sds/tactics.py index 7f63433..ee2e2c5 100644 --- a/sds/tactics.py +++ b/sds/tactics.py @@ -46,7 +46,10 @@ agreement is a fact about the stand-in. from __future__ import annotations import gzip +import itertools import json +import math +import statistics from collections.abc import Iterator from dataclasses import dataclass, field from pathlib import Path @@ -231,8 +234,18 @@ def sweep( tactics: list[dict], phase: str = "movement", matches: int = DEFAULT_MATCHES, + logs: list[Path] | None = None, ) -> Tally: - """Re-score every logged menu under every expressible tactic.""" + """Re-score every logged menu under every expressible tactic. + + `logs` overrides which decision logs are read, for a caller measuring a + subset of a corpus rather than the whole of it. It is a list of paths and + not a second directory on purpose: building a directory of links to measure + half a corpus puts the links either inside the store, where the next + `rglob` would read them back as part of it, or on another filesystem, where + a hardlink degrades into a symlink into data somebody may delete. Both were + tried; neither is worth a scratch directory when a list will do. + """ if phase not in PHASES: raise TacticsError(f"unknown phase '{phase}'; try one of {', '.join(sorted(PHASES))}") usable = [t for t in tactics if not t.get("gap")] @@ -242,11 +255,11 @@ def sweep( for weights in vectors.values(): wanted |= set(weights) - logs = decision_logs(corpus) - if matches: - logs = logs[:matches] - tally = Tally(phase=phase, matches=len(logs), names=names) - for path in logs: + chosen = list(logs) if logs is not None else decision_logs(corpus) + if matches and logs is None: + chosen = chosen[:matches] + tally = Tally(phase=phase, matches=len(chosen), names=names) + for path in chosen: # The weights the bot actually played with, so the sweep can check that # its own arithmetic reproduces the choice that was recorded. A tool # that scored candidates differently from the bot would report a matrix @@ -303,6 +316,131 @@ def band(value: float) -> str: return "a new direction" +#: 1.96 sigma. The bands this page sorts tactics into are read as exact, so the +#: interval around a figure is worth stating in the same units as the figure. +Z = 1.96 + + +@dataclass +class Precision: + """How much of an agreement figure is the corpus it was measured on. + + Every figure on `docs/TACTICS.md` is a share of decisions over one recorded + corpus, and the page sorts tactics into bands at 0.4 and 0.8 as though the + figures were exact. They are not, and nothing said by how much. + + Measured by splitting one corpus into disjoint subsets and re-measuring + every pair on each, so the whole thing costs **one pass over the corpus** + however many subsets are used. Round-robin rather than contiguous blocks: + the logs sort by scenario, so a block of them is a block of similar boards + and would measure a narrower question than the corpus asks. + """ + + subsets: int + decisions: int + #: `(a, b) -> (mean, lowest, highest, standard deviation, standard error)`. + pairs: dict[tuple[str, str], tuple[float, float, float, float, float]] + + def se(self, first: str, second: str) -> float: + """Standard error of the full-corpus figure for one pair.""" + return self.pairs[Tally.pair(first, second)][4] + + def resolves(self, first: float, second: float, se_first: float, se_second: float) -> bool: + """Whether two agreement figures are far enough apart to be two figures.""" + return abs(first - second) > Z * math.sqrt(se_first**2 + se_second**2) + + +def precision( + corpus: Path, + tactics: list[dict], + phase: str = "movement", + subsets: int = 6, +) -> Precision: + """Re-measure every pair on disjoint subsets, and report the spread.""" + logs = decision_logs(corpus) + if len(logs) < subsets * 2: + raise TacticsError( + f"{corpus} holds {len(logs)} decision log(s), too few to fill {subsets} " + f"subsets meaningfully. A corpus with no logs reports agreement 0.000 " + f"everywhere, which is not a measurement." + ) + tallies: list[Tally] = [] + for index in range(subsets): + tally = sweep(corpus, tactics, phase=phase, logs=logs[index::subsets]) + if tally.read < MIN_DECISIONS: + raise TacticsError( + f"subset {index} yielded {tally.read} {phase} decisions, under the " + f"{MIN_DECISIONS} this needs to be a measurement rather than noise. " + f"Use fewer subsets or a larger corpus." + ) + tallies.append(tally) + + names = tallies[0].names + pairs = {} + for first, second in itertools.combinations(names, 2): + values = [t.rate(first, second) for t in tallies] + mean = statistics.mean(values) + sd = statistics.stdev(values) if len(values) > 1 else 0.0 + pairs[Tally.pair(first, second)] = ( + mean, + min(values), + max(values), + sd, + sd / math.sqrt(len(values)), + ) + return Precision( + subsets=subsets, + decisions=sum(t.read for t in tallies), + pairs=pairs, + ) + + +def precision_report( + corpus: Path, + binary: Path = BOT, + phase: str = "movement", + subsets: int = 6, +) -> str: + """The spread, and what it does to the bands.""" + measured = precision(corpus, vocabulary(binary), phase=phase, subsets=subsets) + lines = [ + f"{measured.decisions} {phase} decisions from {corpus}, " + f"{measured.subsets} disjoint subsets", + "", + f"{'pair':26s} {'mean':>7} {'lowest':>7} {'highest':>8} {'sd':>7} {'se':>8} {'95%':>17}", + ] + ordered = sorted(measured.pairs.items(), key=lambda kv: -(kv[1][2] - kv[1][1])) + for (first, second), (mean, low, high, sd, se) in ordered: + lo, hi = mean - Z * se, mean + Z * se + lines.append( + f"{first + '/' + second:26s} {mean:7.3f} {low:7.3f} {high:8.3f} " + f"{sd:7.3f} {se:8.4f} [{lo:.3f}, {hi:.3f}]" + ) + + worst = max(v[4] for v in measured.pairs.values()) + lines += [ + "", + f"A difference under {Z * math.sqrt(2) * worst:.3f} between two full-corpus " + f"agreement figures is not a difference,", + "on the worst pair. Per pair it is tighter; use `Precision.resolves`.", + "", + ] + + straddling = [ + (a, b, mean, mean - Z * se, mean + Z * se, edge) + for (a, b), (mean, _, _, _, se) in sorted(measured.pairs.items()) + for edge in (NEW_DIRECTION, MOSTLY_A_RENAME) + if mean - Z * se < edge < mean + Z * se + ] + if straddling: + lines.append("Pairs whose band is not resolved by this corpus:") + for a, b, mean, lo, hi, edge in straddling: + lines.append(f" {a}/{b:14s} {mean:.3f} 95% [{lo:.3f}, {hi:.3f}] straddles {edge}") + else: + lines.append("Every pair sits clear of a band boundary.") + return "\n".join(lines) + + def report( corpus: Path, binary: Path = BOT, diff --git a/tests/test_tactics.py b/tests/test_tactics.py index 56c2925..bd77215 100644 --- a/tests/test_tactics.py +++ b/tests/test_tactics.py @@ -22,6 +22,7 @@ from sds.tactics import ( combine, decision_logs, decisions, + precision, report, sweep, ) @@ -229,6 +230,68 @@ class TestTheReport(unittest.TestCase): self.assertIn("1.00 north ~ twin", text) +class TestPrecision(unittest.TestCase): + """How precise a full-corpus agreement figure is, measured on the corpus. + + The bands sort tactics at 0.4 and 0.8 as though the figures were exact. + This is the measurement that says by how much they are not. + """ + + def many_logs(self, count: int, per_log: int = MIN_DECISIONS + 5) -> Path: + """A run directory holding `count` separate match logs.""" + root = Path(tempfile.mkdtemp()) + for index in range(count): + records = [ + decision([{"a": 1.0}, {"a": -1.0}], chosen=index % 2) for _ in range(per_log) + ] + body = "".join(json.dumps(r) + "\n" for r in records) + (root / f"m{index}-sds_North.decisions.jsonl").write_text(body) + return root + + def test_disjoint_subsets_read_the_whole_corpus_once(self): + root = self.many_logs(12) + measured = precision(root, [NORTH, SOUTH], subsets=4) + self.assertEqual(measured.subsets, 4) + # Every decision is read exactly once across the four subsets. + whole = sweep(root, [NORTH, SOUTH], matches=0) + self.assertEqual(measured.decisions, whole.read) + + def test_two_opposed_tactics_never_agree_in_any_subset(self): + root = self.many_logs(12) + measured = precision(root, [NORTH, SOUTH], subsets=4) + mean, low, high, sd, se = measured.pairs[("north", "south")] + self.assertEqual((mean, low, high), (0.0, 0.0, 0.0)) + self.assertEqual(sd, 0.0) + self.assertEqual(se, 0.0) + + def test_identical_tactics_always_agree_in_any_subset(self): + root = self.many_logs(12) + measured = precision(root, [NORTH, TWIN], subsets=4) + mean, low, high, _, se = measured.pairs[("north", "twin")] + self.assertEqual((mean, low, high), (1.0, 1.0, 1.0)) + self.assertEqual(se, 0.0) + + def test_a_corpus_too_small_to_split_is_refused(self): + root = self.many_logs(3) + with self.assertRaises(TacticsError) as caught: + precision(root, [NORTH, SOUTH], subsets=6) + self.assertIn("too few to fill", str(caught.exception)) + + def test_a_subset_under_the_noise_floor_is_refused_not_reported(self): + # The failure this exists to prevent: a table of 0.000 with a standard + # error of 0.0000 reads as a confident measurement that nothing agrees. + root = self.many_logs(12, per_log=1) + with self.assertRaises(TacticsError) as caught: + precision(root, [NORTH, SOUTH], subsets=6) + self.assertIn("noise", str(caught.exception)) + + def test_resolves_needs_the_gap_to_beat_both_errors(self): + root = self.many_logs(12) + measured = precision(root, [NORTH, SOUTH], subsets=4) + self.assertFalse(measured.resolves(0.40, 0.41, 0.01, 0.01)) + self.assertTrue(measured.resolves(0.40, 0.60, 0.01, 0.01)) + + def _binary(*tactics: dict) -> Path: """A stand-in `--print-catalogue`, so the report test needs no build.""" root = Path(tempfile.mkdtemp()) -- 2.51.2