diff --git a/bridge/sds/SdsClient.java b/bridge/sds/SdsClient.java index ef2cd57..d3f6245 100644 --- a/bridge/sds/SdsClient.java +++ b/bridge/sds/SdsClient.java @@ -1043,6 +1043,27 @@ public final class SdsClient extends BotClient { } if (actedOn != Entity.NONE) { line.put("actedOn", actedOn); + // The piloting roll this machine would face right now, from the + // observation the decision was taken on. + // + // **Without it, the only place a unit's piloting base appears + // in a run's output is MegaMek's line for a kick that missed** - + // because that is when the roll happens. Measuring which kicks + // the bot *chose* while damaged then meant measuring the subset + // that missed, which is both about forty samples instead of two + // hundred and fifty and, worse, conditioned on the outcome of + // the very roll under test. + // + // Copied off the request rather than re-read from the game: the + // decision was taken on that observation, and a value fetched + // now would be what the machine looks like after the phase + // moved on. + for (JsonNode unit : request.path("units")) { + if (unit.path("id").asInt(Entity.NONE) == actedOn && unit.has("psrBase")) { + line.put("psrBase", unit.path("psrBase").asInt()); + break; + } + } } line.put("outcome", outcome); line.put("taken", taken); diff --git a/prediction-selfharm-r2.md b/prediction-selfharm-r2.md index d9f98f1..bbf3286 100644 --- a/prediction-selfharm-r2.md +++ b/prediction-selfharm-r2.md @@ -74,3 +74,67 @@ mechanisms on the wire first, and this run leaves it exactly where it was. The `repeat` arm from round 1 is config-identical to `off` twice and is the noise floor for **any** comparison in this design, this round included. It does not need re-running. + +--- + +# Prediction 3, re-registered before the arms + +Amended 2026-08-30 after round 1's floor arm ran and before any round-2 match. +The old threshold is left above rather than edited out. + +## Why the old one is withdrawn + +It read: *the share of missed-kick rolls at base 6 or worse falls from 38.2% to +under 25%* - a 13-point move. + +**The floor arm measured that metric's noise at 23 points.** Config-identical to +`off`, changing nothing, it came out at **61.4%** against `off`'s 38.2% - more +than twice the on-vs-off difference of 10 points, and in the opposite direction. +The base histograms are real and not a parsing artefact: `off` `{2:3,3:4,4:13, +5:1,6:4,7:4,8:5}`, `repeat` `{2:5,3:4,4:8,6:7,7:13,8:7}`. One arm simply had far +more beaten-up machines doing the kicking. + +A 13-point threshold against 23 points of noise is not a test. Running it would +have burned three arms to produce another unresolvable number, and we would not +have known whether the instrument or the feature was at fault. + +**The same floor retires prediction 1's result too.** Registered "under 9%", +treatment 8.9%, and the floor arm - **no feature at all** - came out at 6.8%. +The control passed the threshold better than the treatment did. A prediction that +can be passed by changing nothing is not a prediction, and its "pass" in round 1 +is recorded as false. + +## The instrument that replaces it + +`psrBase` is now written onto every decision in the log, from the observation +the decision was taken on. Verified on one match: 44 of 48 decisions carry it, +values 2, 4, 7, 10 and 12. + +The old metric could only see a machine's piloting base when MegaMek printed it, +which is **when a kick missed** - about forty samples an arm, and conditioned on +the outcome of the very roll under test. The new one sees every decision: about +two hundred and fifty kicks an arm, and it measures the thing the prediction is +about, which is **which kicks the bot chose**, not which ones happened to miss. + +## Prediction 3, restated + +**Among physical-phase decisions in which the bot declared a kick, the share +whose actor had `psrBase` 6 or worse falls by at least a third, relative to the +`off` arm, and by more than the floor arm moves the same figure.** + +The baseline is deliberately **pending**: it is whatever the new `off` arm +reads, on the new base, with #320's retry in it. Writing a number here from the +old base would repeat the error this whole amendment exists to correct. It is +filled in from `off` before `on` and `off` are compared, and the floor arm gives +the noise on the same metric rather than an assumed one. + +**Both conditions must hold.** A third is the effect size; beating the floor is +what makes it a measurement. If the drop is real but inside the floor, that is +reported as underpowered - not as a result, and not as a failure of the feature. + +## Predictions 1 and 2 keep their form, with baselines from the new `off` arm + +Same as registered above - long-odds kicks fall, total kicks barely move - but +every baseline comes from the new `off` arm rather than from 9.3% and 247, which +were measured on a base without #320's refused-move retry. `answered`/`decisions` +is not comparable across that boundary and is reported per round. diff --git a/sds/cli.py b/sds/cli.py index aeb9c9a..1d568b5 100644 --- a/sds/cli.py +++ b/sds/cli.py @@ -388,6 +388,51 @@ def cmd_play(args: argparse.Namespace) -> int: return play_match(spec, _run_dir(Path(args.out), "play")) +def _weights_fingerprint(bot: str | None) -> dict[str, str]: + """Hash every weights file the bot command names. + + **The suite is already fingerprinted and the weights file was not.** + `_suite` hashes every `.mms` so a regenerated suite is incomparable rather + than quietly comparable; the file that decides how every candidate is scored + had no such guard. An arm whose weights changed under it - a rebase, an + edit, a fit written to the same path - would report numbers as though + nothing had happened. + + A running measurement owns its inputs or it does not own its result, and + the two ways to survive a violation are not the same. This **freezes**: + the answer is to invalidate the run, not to carry on, because a benchmark + continued on changed inputs is worse than one that stopped. A build + artefact is the other case and should be rebuilt rather than frozen - see + `plan/harness.md`. + + Best effort by design. A bot command with no `--weights` uses the compiled + default and there is nothing to hash; an unreadable path is left out rather + than failing a run over a guard. + """ + out: dict[str, str] = {} + if not bot: + return out + parts = bot.split() + for flag in ("--weights", "--tactic-weights"): + while flag in parts: + index = parts.index(flag) + parts = parts[index + 1 :] + if not parts: + break + named = parts[0] + # The bot command names container paths; `/work` is this checkout. + local = ( + Path(str(named).replace("/work/", "", 1)) + if named.startswith("/work/") + else Path(named) + ) + try: + out[named] = hashlib.sha256(local.read_bytes()).hexdigest()[:12] + except OSError: + continue + return out + + def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: sds_bot = _with_explore(sds_bot, args) _check_bot(sds_bot) @@ -419,6 +464,13 @@ def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: ) out_dir = _run_dir(Path(args.out), label) + # What the weights were when this run started. Compared again at the end: + # a file that moved mid-run means the matches were not all scored the same + # way, and the run says so rather than reporting a mean over two policies. + weights_before = _weights_fingerprint(sds_bot) + if weights_before: + for named, digest in weights_before.items(): + print(f" weights: {named} {digest}", file=sys.stderr) # Stamp before the first match, so a run that is killed part-way still says # what basis its decisions are in. Never worth failing a run over. try: @@ -545,6 +597,34 @@ def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: # Written before anything else can fail. An hour of matches followed by a # traceback on the way to the baseline used to leave nothing on disk but # the per-match files. + # The other end of the freeze. Loud and last, so it is the final thing on + # the terminal rather than a line scrolled past an hour ago. + weights_after = _weights_fingerprint(sds_bot) + if weights_after != weights_before: + moved = sorted(set(weights_before) | set(weights_after)) + print( + "\nWARNING: a weights file changed while this run was going.\n" + + "\n".join( + f" {name}: {weights_before.get(name, 'absent')}" + f" -> {weights_after.get(name, 'absent')}" + for name in moved + if weights_before.get(name) != weights_after.get(name) + ) + + "\nThe matches were not all scored the same way. Do not compare this run.", + file=sys.stderr, + ) + (out_dir / "inputs.json").write_text( + json.dumps( + { + "weights_before": weights_before, + "weights_after": weights_after, + "scenarios": scenario_label, + }, + indent=2, + sort_keys=True, + ) + + "\n" + ) (out_dir / "summary.txt").write_text(summary.render() + "\n") (out_dir / "results.json").write_text(json.dumps(results, indent=2) + "\n") # Every run leaves the standard report beside it, so which numbers got