diff --git a/sds.sh b/sds.sh new file mode 100755 index 0000000..a6c7bc4 --- /dev/null +++ b/sds.sh @@ -0,0 +1,3 @@ +#!/usr/bin/env bash +# The harness entry point. `./sds.sh bench --games 40` +exec python3 -m sds.cli "$@" diff --git a/sds/__init__.py b/sds/__init__.py new file mode 100644 index 0000000..3c20131 --- /dev/null +++ b/sds/__init__.py @@ -0,0 +1,7 @@ +"""The SDS evaluation harness. + +Runs MegaMek matches headlessly, one bot against another, and says whether the +difference between them is real. +""" + +__all__ = ["match", "stats"] diff --git a/sds/cli.py b/sds/cli.py new file mode 100644 index 0000000..a7bf298 --- /dev/null +++ b/sds/cli.py @@ -0,0 +1,236 @@ +"""sds - run bot matches and say whether the difference is real. + +sds one one match, verbose, for looking at +sds bench many matches, aggregated, for deciding with +sds control the harness's self-test: Princess against itself +""" + +from __future__ import annotations + +import argparse +import concurrent.futures +import json +import os +import signal +import sys +from datetime import UTC, datetime +from pathlib import Path + +from .match import REPO, MatchError, MatchSpec, Seat, kill_all_matches, kill_stragglers, run +from .stats import Summary, games_needed + +DEFAULT_SCENARIO = REPO / "scenarios" / "mirror-lance.mms" + + +def _factions(scenario: Path) -> list[str]: + """The faction names in an .mms, in file order. + + Read rather than configured: a benchmark that needs the seat names spelled + out on the command line is one typo away from silently benchmarking a + Princess mirror and reporting it as an SDS result. + """ + for line in scenario.read_text().splitlines(): + if line.startswith("Factions="): + return [f.strip() for f in line.split("=", 1)[1].split(",")] + raise MatchError(f"no Factions= line in {scenario}") + + +def _run_dir(root: Path, label: str) -> Path: + stamp = datetime.now(UTC).strftime("%Y%m%dT%H%M%SZ") + return root / f"{stamp}-{label}" + + +def _play(spec: MatchSpec, out_dir: Path, roles: dict[str, str]) -> dict: + result = run(spec, out_dir) + # Stamp each player with the role it played this game, so a benchmark that + # swaps sides can still tally by bot. + for player in result.get("players", []): + player["role"] = roles.get(player["name"], player["name"]) + return result + + +def cmd_one(args: argparse.Namespace) -> int: + scenario = Path(args.scenario).resolve() + factions = _factions(scenario) + seats = [] + roles = {} + for name in factions: + if args.sds_seat and name == args.sds_seat: + seats.append(Seat(name, "sds", args.bot)) + roles[name] = "sds" + else: + seats.append(Seat(name)) + roles[name] = "princess" + spec = MatchSpec( + scenario=scenario, + seats=seats, + seed=args.seed, + max_rounds=args.max_rounds, + timeout_ms=args.timeout_ms, + ) + result = _play(spec, _run_dir(Path(args.out), "one"), roles) + print(json.dumps(result, indent=2)) + return 0 if result.get("outcome") == "victory" else 1 + + +def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: + scenario = Path(args.scenario).resolve() + factions = _factions(scenario) + if len(factions) != 2: + raise MatchError(f"{scenario} has {len(factions)} factions; bench wants exactly 2") + + out_dir = _run_dir(Path(args.out), label) + plans = [] + for game in range(args.games): + # Alternate which faction the bot under test plays. No two deployment + # edges are equally good, not even on a mirrored map, and a benchmark + # that never swaps measures the edge as much as the bot. With an odd + # game count one side gets one extra game; the interval already covers + # a bias that small, and forcing an even count would be a worse + # surprise than the imbalance. + first = factions[game % 2] + second = factions[(game + 1) % 2] + seats, roles = [], {} + if sds_bot: + seats.append(Seat(first, "sds", sds_bot)) + roles[first] = "sds" + seats.append(Seat(second)) + roles[second] = "princess" + else: + # The control: two Princesses. Labelled A and B by seat order so the + # two are distinguishable, which is the whole point of running it. + seats.append(Seat(first)) + roles[first] = "A" + seats.append(Seat(second)) + roles[second] = "B" + plans.append( + ( + MatchSpec( + scenario=scenario, + seats=seats, + seed=args.seed + game, + max_rounds=args.max_rounds, + timeout_ms=args.timeout_ms, + ), + roles, + ) + ) + + results: list[dict] = [] + failures = 0 + print(f"{args.games} matches, {args.jobs} at a time -> {out_dir}", file=sys.stderr) + with concurrent.futures.ThreadPoolExecutor(max_workers=args.jobs) as pool: + futures = {pool.submit(_play, spec, out_dir, roles): spec for spec, roles in plans} + for done in concurrent.futures.as_completed(futures): + try: + results.append(done.result()) + except MatchError as error: + failures += 1 + print(f" match failed: {error}", file=sys.stderr) + finished = len(results) + failures + print(f" {finished}/{args.games}", end="\r", file=sys.stderr) + print(file=sys.stderr) + + if not results: + print("every match failed; nothing to report", file=sys.stderr) + return 1 + + summary = Summary(results, key="role") + print(summary.render()) + if failures: + print(f"\n{failures} matches failed and are not in the numbers above") + # The planning number, always: it is the difference between "we measured + # nothing" and "we measured nothing, and here is what it would have taken". + print( + f"\nfor reference: separating a 5-point win-rate difference from noise " + f"takes about {games_needed(0.05)} decided games; 10 points, " + f"about {games_needed(0.10)}." + ) + (out_dir / "summary.txt").write_text(summary.render() + "\n") + (out_dir / "results.json").write_text(json.dumps(results, indent=2) + "\n") + return 0 + + +def cmd_bench(args: argparse.Namespace) -> int: + return _bench(args, "bench", args.bot) + + +def cmd_control(args: argparse.Namespace) -> int: + """Princess against Princess. Should come out 50/50; if it does not, stop.""" + return _bench(args, "control", None) + + +def cmd_clean(args: argparse.Namespace) -> int: + """Kill any match container left running by a harness that died.""" + killed = kill_all_matches() + if killed: + print("killed:\n " + "\n ".join(killed)) + else: + print("no match containers running") + return 0 + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + prog="sds", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + parser.add_argument("--scenario", default=str(DEFAULT_SCENARIO)) + parser.add_argument("--out", default=str(REPO / "runs")) + parser.add_argument("--seed", type=int, default=1) + parser.add_argument("--max-rounds", type=int, default=40) + parser.add_argument("--timeout-ms", type=int, default=10_000) + sub = parser.add_subparsers(dest="command", required=True) + + one = sub.add_parser("one", help="play a single match and print the result") + one.add_argument("--bot", default="python3 /work/bots/random_bot.py") + one.add_argument( + "--sds-seat", default=None, help="faction the bot plays; omit for Princess on both sides" + ) + one.set_defaults(func=cmd_one) + + bench = sub.add_parser("bench", help="play many matches and aggregate") + bench.add_argument("--bot", default="python3 /work/bots/random_bot.py") + bench.add_argument("--games", type=int, default=20) + # Two by default. A match is two thinking bots and a server in one container, + # and this machine is shared with other agents; the sibling repos have all + # learned this the expensive way. Raise it deliberately, having looked. + bench.add_argument("--jobs", type=int, default=int(os.environ.get("SDS_JOBS", "2"))) + bench.set_defaults(func=cmd_bench) + + control = sub.add_parser("control", help="Princess vs Princess: the harness's self-test") + control.add_argument("--games", type=int, default=20) + control.add_argument("--jobs", type=int, default=int(os.environ.get("SDS_JOBS", "2"))) + control.set_defaults(func=cmd_control) + + clean = sub.add_parser( + "clean", + help="kill match containers left by a dead harness - NOT while a run is live", + description="Kills every container named sds-*, including the ones a " + "benchmark running right now is using. For cleaning up after a " + "harness that was killed rather than interrupted; an interrupted " + "one cleans up after itself.", + ) + clean.set_defaults(func=cmd_clean) + + args = parser.parse_args(argv) + + # Ctrl-C has to reach the containers, not just this process. `docker run` is + # a client; killing it leaves the match playing, and a benchmark abandoned + # halfway leaves as many orphans as it had jobs. + def _stop(signum, _frame): + killed = kill_stragglers() + print(f"\ninterrupted; killed {killed} running matches", file=sys.stderr) + sys.exit(130) + + signal.signal(signal.SIGINT, _stop) + signal.signal(signal.SIGTERM, _stop) + + try: + return args.func(args) + except MatchError as error: + print(f"error: {error}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/sds/match.py b/sds/match.py new file mode 100644 index 0000000..a5755dd --- /dev/null +++ b/sds/match.py @@ -0,0 +1,242 @@ +"""Run one match in a container and read the result back.""" + +from __future__ import annotations + +import atexit +import json +import os +import shutil +import subprocess +import threading +import uuid +from dataclasses import dataclass, field +from pathlib import Path + +REPO = Path(__file__).resolve().parent.parent +DEFAULT_IMAGE = os.environ.get("SDS_IMAGE", "lance-blue/sds-runner") +DEFAULT_MM_HOME = Path(os.environ.get("MM_HOME", Path.home() / ".cache" / "mul-build" / "megamek")) + +# MegaMek deserialises a prebuilt units.cache through java.util collections, and +# the module system will not let it reflect into them without being told. +ADD_OPENS = [ + "--add-opens", + "java.base/java.util=ALL-UNNAMED", + "--add-opens", + "java.base/java.util.concurrent=ALL-UNNAMED", +] + + +class MatchError(RuntimeError): + pass + + +# Containers outlive the process that started them. `docker run` is a client: +# kill it and the container keeps playing, which is how the first run of this +# harness left three matches running after its driver was reaped, competing for +# CPU with the run that replaced them. Every container gets a name, the names +# are tracked here, and anything still running when the harness exits is killed. +_LIVE: set[str] = set() +_LIVE_LOCK = threading.Lock() + +# Set when the harness is asked to stop. A benchmark submits every match up +# front, so a thread pool with no way to be told "stop" keeps starting the rest +# after an interrupt: the first attempt at this took a SIGTERM and calmly played +# eighteen more games. Queued matches check this and refuse to start. +STOP = threading.Event() + + +def _track(name: str) -> None: + with _LIVE_LOCK: + _LIVE.add(name) + + +def _untrack(name: str) -> None: + with _LIVE_LOCK: + _LIVE.discard(name) + + +def kill_stragglers() -> int: + """Stop starting matches, and kill any this process already started.""" + STOP.set() + with _LIVE_LOCK: + names = sorted(_LIVE) + _LIVE.clear() + killed = 0 + for name in names: + done = subprocess.run( + ["docker", "kill", name], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=False, + ) + if done.returncode == 0: + killed += 1 + return killed + + +atexit.register(kill_stragglers) + + +def kill_all_matches() -> list[str]: + """Kill every sds match container on this machine, whoever started it. + + The blunt instrument, for when a harness process was killed rather than + interrupted and its containers outlived it - which is exactly what a + reaped background job does. Names are prefixed `sds-`, so this cannot + touch anything else running on a shared machine. + """ + listed = subprocess.run( + ["docker", "ps", "--filter", "name=sds-", "--format", "{{.Names}}"], + capture_output=True, + text=True, + check=False, + ) + names = [n for n in listed.stdout.split() if n.startswith("sds-")] + for name in names: + subprocess.run( + ["docker", "kill", name], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=False, + ) + return names + + +@dataclass +class Seat: + """One faction, and who plays it. + + `name` is the faction name in the .mms file. `kind` is "princess" or "sds"; + an "sds" seat needs `bot`, the command the host runs as a subprocess, with + paths as the container sees them (the repository is mounted at /work). + """ + + name: str + kind: str = "princess" + bot: str | None = None + + +@dataclass +class MatchSpec: + scenario: Path + seats: list[Seat] + seed: int = 1 + max_rounds: int = 40 + timeout_ms: int = 10_000 + wall_clock: int = 900 + behavior: str | None = None + image: str = DEFAULT_IMAGE + mm_home: Path = DEFAULT_MM_HOME + env: dict[str, str] = field(default_factory=dict) + + +def preflight(spec: MatchSpec) -> None: + """Fail before a hundred matches do, not during.""" + if shutil.which("docker") is None: + raise MatchError("docker is not on PATH") + if not (spec.mm_home / "MegaMek.jar").is_file(): + raise MatchError(f"no MegaMek.jar under {spec.mm_home}; set MM_HOME") + if not (REPO / "bridge" / "build" / "sds.jar").is_file(): + raise MatchError("bridge/build/sds.jar is missing; run ./scripts/build.sh") + if not spec.scenario.is_file(): + raise MatchError(f"no scenario at {spec.scenario}") + sds_seats = [s for s in spec.seats if s.kind == "sds"] + for seat in sds_seats: + if not seat.bot: + raise MatchError(f"seat {seat.name} is 'sds' but has no bot command") + + +def run(spec: MatchSpec, out_dir: Path) -> dict: + """Play one match. Returns the result document. + + The container is `--rm` and holds nothing that outlives it: the result is + written into `out_dir`, which is inside the mounted repository, and the raw + log next to it. A match that fails leaves both, because the log is the only + thing that says why. + """ + if STOP.is_set(): + raise MatchError("harness is stopping; match not started") + preflight(spec) + out_dir.mkdir(parents=True, exist_ok=True) + tag = f"{spec.scenario.stem}-seed{spec.seed}-{uuid.uuid4().hex[:8]}" + container = f"sds-{tag}" + result_path = out_dir / f"{tag}.json" + log_path = out_dir / f"{tag}.log" + + classpath = "MegaMek.jar:lib/*:/work/bridge/build/sds.jar" + command = [ + "docker", + "run", + "--rm", + "--name", + container, + "--user", + f"{os.getuid()}:{os.getgid()}", + "-v", + f"{REPO}:/work", + "-v", + f"{spec.mm_home}:/mm", + "-w", + "/mm", + ] + # The bot inherits these. SDS_SEED is what makes a stochastic bot repeatable; + # without it a seeded harness still produces unrepeatable matches and the + # seed is a comforting lie. + env = {"SDS_SEED": str(spec.seed), **spec.env} + for key, value in env.items(): + command += ["-e", f"{key}={value}"] + + command += [ + spec.image, + "java", + "-Dlog4j2.configurationFile=/work/bridge/log4j2-quiet.xml", + "-Dsentry.dsn=", + "-Djava.awt.headless=true", + *ADD_OPENS, + "-cp", + classpath, + "sds.SdsHost", + "--scenario", + f"/work/{spec.scenario.relative_to(REPO)}", + "--out", + f"/work/{result_path.relative_to(REPO)}", + "--seed", + str(spec.seed), + "--max-rounds", + str(spec.max_rounds), + "--timeout-ms", + str(spec.timeout_ms), + "--wall-clock", + str(spec.wall_clock), + ] + if spec.behavior: + command += ["--behavior", spec.behavior] + for seat in spec.seats: + command += ["--seat", f"{seat.name}={seat.kind}"] + if seat.kind == "sds" and seat.bot: + command += ["--bot", seat.bot] + + _track(container) + try: + with log_path.open("wb") as log: + completed = subprocess.run(command, stdout=log, stderr=subprocess.STDOUT, check=False) + finally: + # Whether it finished, failed or the thread was interrupted: the + # container must not survive this call. + subprocess.run( + ["docker", "kill", container], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=False, + ) + _untrack(container) + + if not result_path.is_file(): + raise MatchError( + f"match produced no result (docker exit {completed.returncode}); see {log_path}" + ) + result = json.loads(result_path.read_text()) + result["logPath"] = str(log_path) + result["resultPath"] = str(result_path) + result["exitCode"] = completed.returncode + return result diff --git a/sds/stats.py b/sds/stats.py new file mode 100644 index 0000000..1e894fc --- /dev/null +++ b/sds/stats.py @@ -0,0 +1,180 @@ +"""Turn a pile of match results into a number, with an honest error bar. + +The temptation with a bot benchmark is to run twenty games, see 12-8, and +believe something. Twelve of twenty is a 60% win rate whose 95% interval runs +from 36% to 81%: it is equally consistent with a bot that is worse. Everything +here exists to keep that from happening quietly. +""" + +from __future__ import annotations + +import math +from collections import Counter +from dataclasses import dataclass + + +def wilson(successes: int, trials: int, z: float = 1.96) -> tuple[float, float]: + """A 95% confidence interval for a win rate. + + Wilson rather than the textbook normal interval, because the normal one is + badly wrong exactly where a bot benchmark spends its time - few games, or + rates near 0 and 1 - and can produce bounds outside [0, 1], which reads as + a bug in whatever prints it. + """ + if trials == 0: + return (0.0, 1.0) + p = successes / trials + denominator = 1 + z * z / trials + centre = (p + z * z / (2 * trials)) / denominator + spread = z * math.sqrt(p * (1 - p) / trials + z * z / (4 * trials * trials)) / denominator + return (max(0.0, centre - spread), min(1.0, centre + spread)) + + +def games_needed(effect: float, base: float = 0.5, z: float = 1.96) -> int: + """Roughly how many games it takes to see a win rate `effect` above `base`. + + A planning number, not a promise: it assumes independent games and ignores + draws. Its job is to answer "is 20 games enough" before the 20 games are + run, and the answer is almost always no. + """ + if effect <= 0: + return 0 + variance = base * (1 - base) + return math.ceil((z * z * variance) / (effect * effect)) + + +@dataclass +class SideSummary: + name: str + kind: str + wins: int = 0 + bv_fraction_total: float = 0.0 + units_remaining_total: int = 0 + decisions: int = 0 + answered: int = 0 + passed: int = 0 + illegal: int = 0 + failed: int = 0 + + +class Summary: + """What a run of matches came to.""" + + def __init__(self, results: list[dict], key: str = "name") -> None: + # `key` picks what a "side" is. By default the faction name, which is + # what a single scenario wants. A benchmark that alternates which + # faction a bot plays - and it should, because no two deployment edges + # are equally good - passes "role" instead, so the tally follows the bot + # rather than the corner of the map it started in. + self.key = key + self.results = results + self.outcomes = Counter(r.get("outcome", "unknown") for r in results) + self.rounds = [r["round"] for r in results if "round" in r] + self.elapsed = [r["elapsedMillis"] for r in results if "elapsedMillis" in r] + self.sides: dict[str, SideSummary] = {} + self.decided = 0 + self._tally() + + def _tally(self) -> None: + for result in self.results: + players = result.get("players", []) + for player in players: + label = player.get(self.key, player["name"]) + side = self.sides.setdefault(label, SideSummary(label, player["kind"])) + start = player.get("bvStart") or 0 + side.bv_fraction_total += (player.get("bvRemaining", 0) / start) if start else 0.0 + side.units_remaining_total += player.get("unitsRemaining", 0) + stats = player.get("sds") + if stats: + side.decisions += stats.get("decisions", 0) + side.answered += stats.get("answered", 0) + side.passed += stats.get("passed", 0) + side.illegal += stats.get("illegal", 0) + side.failed += stats.get("failed", 0) + + winner = self._winner(result, self.key) + if winner is not None: + self.decided += 1 + if winner in self.sides: + self.sides[winner].wins += 1 + + @staticmethod + def _winner(result: dict, key: str = "name") -> str | None: + """Who won, as MegaMek recorded it. + + Never inferred from the surviving unit count. A match that hits the + round limit has survivors on both sides and no winner, and calling the + side with more armour left "the winner" would quietly turn a benchmark + of who wins into a benchmark of who hides. + """ + if result.get("outcome") != "victory": + return None + players = result.get("players", []) + team = result.get("victoryTeam", -1) + player_id = result.get("victoryPlayerId", -1) + for player in players: + if player_id >= 0 and player.get("id") == player_id: + return player.get(key, player["name"]) + if team >= 0: + named = [p.get(key, p["name"]) for p in players if p.get("team") == team] + if len(named) == 1: + return named[0] + return None + + def win_rate(self, name: str) -> tuple[float, tuple[float, float]]: + """Win rate among *decided* games, with its interval. + + Decided games only, because a draw is not half a win and pretending it + is makes a passive bot look average instead of looking passive. The + undecided count is reported separately and is itself a finding. + """ + side = self.sides.get(name) + if side is None or self.decided == 0: + return (0.0, (0.0, 1.0)) + return (side.wins / self.decided, wilson(side.wins, self.decided)) + + def render(self) -> str: + lines: list[str] = [] + total = len(self.results) + lines.append(f"matches {total}") + lines.append(f"decided {self.decided}") + for outcome, count in sorted(self.outcomes.items(), key=lambda kv: -kv[1]): + lines.append(f" {outcome:<16} {count}") + if self.rounds: + lines.append(f"rounds mean {sum(self.rounds) / len(self.rounds):.1f}") + if self.elapsed: + lines.append(f"seconds mean {sum(self.elapsed) / len(self.elapsed) / 1000:.1f}") + lines.append("") + header = f"{'side':<14} {'kind':<9} {'wins':>5} {'rate':>7} {'95% CI':>16} {'BV left':>8}" + lines.append(header) + lines.append("-" * len(header)) + for name, side in self.sides.items(): + rate, (low, high) = self.win_rate(name) + bv = side.bv_fraction_total / total if total else 0.0 + lines.append( + f"{name:<14} {side.kind:<9} {side.wins:>5} {rate:>6.1%} " + f"{low:>6.1%}-{high:<6.1%} {bv:>7.1%}" + ) + bridged = [s for s in self.sides.values() if s.decisions] + if bridged: + lines.append("") + lines.append( + f"{'bridge':<14} {'decisions':>10} {'answered':>9} {'passed':>7} " + f"{'illegal':>8} {'failed':>7}" + ) + for side in bridged: + lines.append( + f"{side.name:<14} {side.decisions:>10} {side.answered:>9} " + f"{side.passed:>7} {side.illegal:>8} {side.failed:>7}" + ) + # A bot that answered nothing was Princess wearing a hat, and its win + # rate says nothing about the bot. Worth saying out loud rather than + # leaving in a column. + for side in bridged: + share = side.answered / side.decisions if side.decisions else 0.0 + if share < 0.5: + lines.append( + f" note: {side.name} answered only {share:.0%} of its decisions; " + f"Princess played the rest" + ) + return "\n".join(lines) diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/test_hexes.py b/tests/test_hexes.py new file mode 100644 index 0000000..29a4161 --- /dev/null +++ b/tests/test_hexes.py @@ -0,0 +1,39 @@ +"""Hex geometry, checked against the property that catches the parity bug. + +Offset coordinates with a column shift are easy to port wrong in a way that +works for half the board. Every neighbour of every hex must be distance 1; if +the parity term is wrong, that fails on odd columns only. +""" + +import os +import sys +import unittest + +_REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(_REPO, "bots")) + +from hexes import distance, translated # noqa: E402 + + +class TestHexes(unittest.TestCase): + def test_every_neighbour_is_adjacent(self): + for x in range(12): + for y in range(12): + for direction in range(6): + nx, ny = translated(x, y, direction) + self.assertEqual(distance(x, y, nx, ny), 1, f"({x},{y}) dir {direction}") + + def test_opposite_directions_return(self): + for x in range(12): + for y in range(12): + for direction in range(6): + nx, ny = translated(x, y, direction) + bx, by = translated(nx, ny, (direction + 3) % 6) + self.assertEqual((bx, by), (x, y)) + + def test_straight_line_distance_accumulates(self): + for direction in range(6): + x, y = 6, 6 + for step in range(1, 5): + x, y = translated(x, y, direction) + self.assertEqual(distance(6, 6, x, y), step) diff --git a/tests/test_stats.py b/tests/test_stats.py new file mode 100644 index 0000000..a30d1da --- /dev/null +++ b/tests/test_stats.py @@ -0,0 +1,101 @@ +"""The statistics are load-bearing: they decide what gets believed.""" + +import unittest + +from sds.stats import Summary, games_needed, wilson + + +def result(outcome, winner_team=-1, players=None): + return { + "outcome": outcome, + "round": 10, + "elapsedMillis": 1000, + "victoryTeam": winner_team, + "victoryPlayerId": -1, + "players": players or [], + } + + +def player(name, team, role, units=4, bv=1000, bv_start=1000): + return { + "id": team, + "name": name, + "kind": "princess", + "team": team, + "role": role, + "unitsRemaining": units, + "unitsStart": 4, + "bvRemaining": bv, + "bvStart": bv_start, + } + + +class TestWilson(unittest.TestCase): + def test_small_samples_are_wide(self): + low, high = wilson(12, 20) + self.assertLess(low, 0.5) + self.assertGreater(high, 0.5) + + def test_bounds_stay_in_range(self): + # The textbook normal interval goes outside [0, 1] here, which is the + # reason this is Wilson. + low, high = wilson(20, 20) + self.assertGreaterEqual(low, 0.0) + self.assertLessEqual(high, 1.0) + + def test_no_trials_claims_nothing(self): + self.assertEqual(wilson(0, 0), (0.0, 1.0)) + + +class TestGamesNeeded(unittest.TestCase): + def test_smaller_effects_need_more_games(self): + self.assertGreater(games_needed(0.05), games_needed(0.10)) + + +class TestSummary(unittest.TestCase): + def test_undecided_matches_are_not_half_wins(self): + """A round limit is not a draw split between the sides. + + This is the failure that matters most: counting an undecided game as + half a win makes a bot that refuses to engage look average. + """ + results = [ + result("round-limit", players=[player("N", 1, "sds"), player("S", 2, "princess")]), + result("victory", 1, [player("N", 1, "sds"), player("S", 2, "princess", units=0)]), + ] + summary = Summary(results, key="role") + self.assertEqual(summary.decided, 1) + rate, _ = summary.win_rate("sds") + self.assertEqual(rate, 1.0) + + def test_winner_is_never_inferred_from_survivors(self): + """A side with everything left has not won if MegaMek recorded no win.""" + results = [ + result( + "round-limit", + players=[ + player("N", 1, "sds", units=4, bv=1000), + player("S", 2, "princess", units=1, bv=10), + ], + ) + ] + summary = Summary(results, key="role") + self.assertEqual(summary.decided, 0) + self.assertEqual(summary.sides["sds"].wins, 0) + + def test_roles_follow_the_bot_across_swapped_sides(self): + """Alternating deployment edges must not split a bot's tally in two.""" + results = [ + result("victory", 1, [player("N", 1, "sds"), player("S", 2, "princess")]), + result("victory", 1, [player("N", 1, "princess"), player("S", 2, "sds")]), + ] + summary = Summary(results, key="role") + self.assertEqual(set(summary.sides), {"sds", "princess"}) + self.assertEqual(summary.decided, 2) + # One win each: game one the sds seat was team 1, game two it was team 2. + self.assertEqual(summary.sides["sds"].wins, 1) + self.assertEqual(summary.sides["princess"].wins, 1) + + +if __name__ == "__main__": + unittest.main()