diff --git a/sds/cli.py b/sds/cli.py index 8540d3c..78365b5 100644 --- a/sds/cli.py +++ b/sds/cli.py @@ -148,7 +148,24 @@ def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: first = factions[game % 2] second = factions[(game + 1) % 2] seats, roles = [], {} - if sds_bot: + if sds_bot and getattr(args, "self_play", False): + # Both seats, so that one of them wins. + # + # A fit needs the label to vary. Against Princess the bot loses + # every game, so every label is the same number and least squares + # has nothing to separate - more losses do not help, because the + # problem is the variance and not the count. Self-play gives an + # even split by construction and a corpus that spans the range. + # + # The cost is the one `plan/training.md` names: two weak bots teach + # each other to beat a weak bot. Weights fitted this way are a + # starting point to be re-measured against Princess, never the + # number that gets reported. + seats.append(Seat(first, "sds", sds_bot)) + roles[first] = "sds-A" + seats.append(Seat(second, "sds", sds_bot)) + roles[second] = "sds-B" + elif sds_bot: seats.append(Seat(first, "sds", sds_bot)) roles[first] = "sds" seats.append(Seat(second)) @@ -463,6 +480,12 @@ def main(argv: list[str] | None = None) -> int: metavar="NAME", help="compare this run against a recorded baseline and write comparison.md", ) + bench.add_argument( + "--self-play", + action="store_true", + help="seat the bot on both sides; for building a training corpus whose " + "labels vary, not for measuring strength", + ) bench.add_argument("--notes", default="", help="one line, stored with a baseline") bench.add_argument( "--suite",