From 7241cf3c5f7a54ff171f9ae7db33067dedea9d05 Mon Sep 17 00:00:00 2001 From: "@permadeath.com" Date: Wed, 19 Aug 2026 20:19:59 -0400 Subject: [PATCH] feat(training): --self-play, so the label has something to vary Against Princess the bot loses every game, so every label is the same number and least squares has nothing to separate. More losses do not help: the problem is the variance, not the count. Seating the bot on both sides gives an even split by construction. The cost is the one plan/training.md names, so weights fitted this way are a starting point to re-measure against Princess, never a reported number. --- sds/cli.py | 25 ++++++++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/sds/cli.py b/sds/cli.py index 8540d3c..78365b5 100644 --- a/sds/cli.py +++ b/sds/cli.py @@ -148,7 +148,24 @@ def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: first = factions[game % 2] second = factions[(game + 1) % 2] seats, roles = [], {} - if sds_bot: + if sds_bot and getattr(args, "self_play", False): + # Both seats, so that one of them wins. + # + # A fit needs the label to vary. Against Princess the bot loses + # every game, so every label is the same number and least squares + # has nothing to separate - more losses do not help, because the + # problem is the variance and not the count. Self-play gives an + # even split by construction and a corpus that spans the range. + # + # The cost is the one `plan/training.md` names: two weak bots teach + # each other to beat a weak bot. Weights fitted this way are a + # starting point to be re-measured against Princess, never the + # number that gets reported. + seats.append(Seat(first, "sds", sds_bot)) + roles[first] = "sds-A" + seats.append(Seat(second, "sds", sds_bot)) + roles[second] = "sds-B" + elif sds_bot: seats.append(Seat(first, "sds", sds_bot)) roles[first] = "sds" seats.append(Seat(second)) @@ -463,6 +480,12 @@ def main(argv: list[str] | None = None) -> int: metavar="NAME", help="compare this run against a recorded baseline and write comparison.md", ) + bench.add_argument( + "--self-play", + action="store_true", + help="seat the bot on both sides; for building a training corpus whose " + "labels vary, not for measuring strength", + ) bench.add_argument("--notes", default="", help="one line, stored with a baseline") bench.add_argument( "--suite", -- 2.51.2