From 8494c4cd6a77bba6d60ec62224de017686a661a3 Mon Sep 17 00:00:00 2001 From: "@permadeath.com" Date: Wed, 19 Aug 2026 20:40:05 -0400 Subject: [PATCH] feat(training): label a decision by what happened next The match's final BV differential was given to every decision in it, so a fit's effective sample size was the match count and not the decision count: 15156 candidate comparisons from 20 matches came out at R-squared 0.016 with three weights contradicting their own descriptions. The host now records BV and live units per round, and a decision is labelled by the change over the three rounds after it. A corpus without the round log still fits on the old label, and the report says which was used. --- bridge/sds/SdsHost.java | 33 ++++++++++++++- sds/train.py | 92 +++++++++++++++++++++++++++++++++++++---- tests/test_train.py | 43 +++++++++++++++++++ 3 files changed, 157 insertions(+), 11 deletions(-) diff --git a/bridge/sds/SdsHost.java b/bridge/sds/SdsHost.java index 9cb6345..390913c 100644 --- a/bridge/sds/SdsHost.java +++ b/bridge/sds/SdsHost.java @@ -276,6 +276,7 @@ public final class SdsHost { } }, "sds-host-shutdown")); + ArrayNode roundLog = MAPPER.createArrayNode(); long start = System.currentTimeMillis(); long lastReady = 0L; Set uncompressed = new LinkedHashSet<>(); @@ -353,6 +354,30 @@ public final class SdsHost { if (round != lastRound) { lastRound = round; System.out.printf("[host] round %d%n", round); + // What each side was worth at the top of this round. + // + // The training label was the match's final BV differential, + // given to every decision in it, so a whole match of choices + // shared one number and the effective sample size was the match + // count rather than the decision count. A fit on 15156 candidate + // comparisons from 20 matches came out at R^2 0.016 with three + // weights contradicting their own descriptions. + // + // With a figure per round, a decision can be labelled by what + // happened next instead of by how the match ended. Recorded + // here rather than in the bot because the host already walks the + // rounds and sees both sides, and because a label the bot + // computes for itself is a label it can be wrong about. + ObjectNode tick = roundLog.addObject(); + tick.put("round", round); + ObjectNode bv = tick.putObject("bv"); + ObjectNode alive = tick.putObject("units"); + for (Player p : live.getPlayersList()) { + bv.put(p.getName(), p.getBV()); + alive.put(p.getName(), live instanceof Game rg + ? rg.getLiveEntitiesOwnedBy(p) + : live.getEntitiesOwnedBy(p)); + } } if (a.maxRounds > 0 && round > a.maxRounds) { // Two bots that both refuse to close never finish, and that is @@ -391,7 +416,7 @@ public final class SdsHost { if (state == null) { state = snapshot(server.getGame()); } - report(a, state, outcome, elapsed, sdsSeats); + report(a, state, outcome, elapsed, sdsSeats, roundLog); sdsSeats.values().forEach(SdsClient::stop); server.die(); @@ -516,7 +541,7 @@ public final class SdsHost { } private static void report(Args a, ObjectNode state, String outcome, long elapsedMillis, - Map sdsSeats) throws Exception { + Map sdsSeats, ArrayNode roundLog) throws Exception { ObjectNode root = MAPPER.createObjectNode(); root.put("scenario", a.scenario); root.put("seed", a.seed); @@ -528,6 +553,10 @@ public final class SdsHost { root.put("elapsedMillis", elapsedMillis); root.put("victoryTeam", state.path("victoryTeam").asInt()); root.put("victoryPlayerId", state.path("victoryPlayerId").asInt()); + // One entry per round: what each side was worth at the top of it. This + // is what lets a decision be labelled by what happened next rather than + // by how the whole match ended. + root.set("rounds", roundLog); ArrayNode players = root.putArray("players"); System.out.println("================================================="); diff --git a/sds/train.py b/sds/train.py index e383d2e..bb031b9 100644 --- a/sds/train.py +++ b/sds/train.py @@ -72,6 +72,15 @@ EXPECTED_SIGNS: dict[str, int] = { LABEL_KIND = "final_bv_differential" +# What each label means, for the report. A reader has to know which one a fit +# used: they answer different questions and are not comparable. +LABEL_MEANING = { + "final_bv_differential": ("final BV differential of the match, given to every decision in it"), + "bv_delta_next_3_rounds": ( + "change in BV differential over the three rounds after the decision" + ), +} + class TrainingError(RuntimeError): pass @@ -117,7 +126,8 @@ class Corpus: """Every decision in a run, each with the label of the match it came from.""" decisions: list[Decision] = field(default_factory=list) - labels: dict[tuple[str, str], float] = field(default_factory=dict) + labels: dict[tuple, float] = field(default_factory=dict) + label_kinds: set[str] = field(default_factory=set) matches: set[str] = field(default_factory=set) unlabelled: list[str] = field(default_factory=list) local_features: set[str] = field(default_factory=set) @@ -129,11 +139,12 @@ class Corpus: directory must produce the same numbers, and a directory listing is not a promise. """ - pairs = [ - (d, self.labels[(d.tag, d.seat)]) - for d in self.decisions - if (d.tag, d.seat) in self.labels - ] + pairs = [] + for d in self.decisions: + if (d.tag, d.seat, d.round) in self.labels: + pairs.append((d, self.labels[(d.tag, d.seat, d.round)])) + elif (d.tag, d.seat) in self.labels: + pairs.append((d, self.labels[(d.tag, d.seat)])) pairs.sort(key=lambda p: (p[0].tag, p[0].seat, p[0].round, p[0].phase, p[0].unit)) return pairs @@ -209,6 +220,56 @@ def label_of(result: dict, seat: str) -> float | None: return (ours_left - theirs_left) / float(start) +HORIZON = 3 + + +def differential(tick: dict, seat: str, start: float) -> float | None: + """One round's BV differential from a seat's point of view.""" + bv = tick.get("bv") or {} + if seat not in bv or not start: + return None + ours = float(bv[seat]) + theirs = float(sum(v for k, v in bv.items() if k != seat)) + return (ours - theirs) / start + + +def label_at(result: dict, seat: str, round_: int, horizon: int = HORIZON) -> float | None: + """How much better things got over the `horizon` rounds after a decision. + + The match-wide label gives every decision in a match the same number, so a + fit's effective sample size is the match count and not the decision count: + 15156 candidate comparisons from 20 matches fitted at R^2 0.016, with three + weights contradicting their own descriptions. + + This asks a local question instead - what happened next - which is a label a + single decision can actually be responsible for. Falls back to the last + round recorded when the match ends inside the horizon. + + Returns None when the result carries no per-round log, which is every match + run before that existed. Callers fall back to `label_of`. + """ + ticks = result.get("rounds") or [] + if not ticks: + return None + players = result.get("players") or [] + ours = next((p for p in players if p.get("name") == seat), None) + if ours is None or not ours.get("bvStart"): + return None + start = float(ours["bvStart"]) + + at = {t["round"]: t for t in ticks if "round" in t} + if round_ not in at: + return None + now = differential(at[round_], seat, start) + if now is None: + return None + later_round = max((r for r in at if r <= round_ + horizon), default=round_) + later = differential(at[later_round], seat, start) + if later is None: + return None + return later - now + + def read_corpus(target: Path) -> Corpus: """Every decision log under a run directory, joined to its match result. @@ -249,16 +310,29 @@ def read_corpus(target: Path) -> Corpus: continue corpus.decisions.append(decision) corpus.local_features.update(decision.local) + result = results[tag] + if result is None: + if tag not in corpus.unlabelled: + corpus.unlabelled.append(tag) + continue + # Per round where the result carries one, per match where it does + # not. A corpus recorded before `rounds` existed still fits, with + # the weaker label and a line in the report saying so. + local_label = label_at(result, decision.seat, decision.round) + if local_label is not None: + corpus.labels[(tag, decision.seat, decision.round)] = local_label + corpus.label_kinds.add("bv_delta_next_3_rounds") + continue key = (tag, decision.seat) if key in corpus.labels: continue - result = results[tag] - label = label_of(result, decision.seat) if result else None + label = label_of(result, decision.seat) if label is None: if tag not in corpus.unlabelled: corpus.unlabelled.append(tag) else: corpus.labels[key] = label + corpus.label_kinds.add("final_bv_differential") return corpus @@ -587,7 +661,7 @@ def fit(corpus: Corpus, ridge: float = 1e-3, features: list[str] | None = None) unlabelled_matches=list(corpus.unlabelled), r_squared=r_squared, label_mean=sum(labels) / len(labels) if labels else 0.0, - label_kind=LABEL_KIND, + label_kind="+".join(sorted(corpus.label_kinds)) or LABEL_KIND, ridge=ridge, framing="difference", agreement=_agreement(corpus, names, weights), diff --git a/tests/test_train.py b/tests/test_train.py index 8ced3d7..f86531d 100644 --- a/tests/test_train.py +++ b/tests/test_train.py @@ -15,6 +15,7 @@ from sds.train import ( difference_rows, feature_names, fit, + label_at, label_of, least_squares, load, @@ -445,3 +446,45 @@ class TestDeadColumns(unittest.TestCase): ) with self.assertRaises(TrainingError): fit(corpus) + + +class TestTemporalLabel(unittest.TestCase): + """A decision is labelled by what happened next, not by how the match ended. + + The match-wide label gives every decision in a match the same number, so a + fit's effective sample size is the match count rather than the decision + count. 15156 comparisons from 20 matches fitted at R-squared 0.016. + """ + + def result(self): + return { + "players": [ + {"name": "North", "bvStart": 100, "bvRemaining": 40}, + {"name": "South", "bvStart": 100, "bvRemaining": 90}, + ], + "rounds": [ + {"round": 1, "bv": {"North": 100, "South": 100}}, + {"round": 2, "bv": {"North": 100, "South": 80}}, + {"round": 3, "bv": {"North": 90, "South": 80}}, + {"round": 4, "bv": {"North": 60, "South": 80}}, + {"round": 5, "bv": {"North": 40, "South": 90}}, + ], + } + + def test_a_good_stretch_labels_positive(self): + # Round 1 -> 4: North goes from level to -0.20; South sees +0.20. + self.assertGreater(label_at(self.result(), "South", 1), 0) + self.assertLess(label_at(self.result(), "North", 1), 0) + + def test_it_is_zero_sum_between_the_seats(self): + north = label_at(self.result(), "North", 2) + south = label_at(self.result(), "South", 2) + self.assertAlmostEqual(north, -south, places=6) + + def test_the_horizon_is_clipped_to_the_last_round(self): + # Round 5 is the last, so a 3-round horizon has nowhere to go. + self.assertEqual(label_at(self.result(), "North", 5), 0.0) + + def test_no_round_log_means_no_temporal_label(self): + old = {"players": [{"name": "North", "bvStart": 100, "bvRemaining": 40}]} + self.assertIsNone(label_at(old, "North", 1)) -- 2.51.2