diff --git a/plan/candidates.md b/plan/candidates.md index 2e270f6..7f4eacf 100644 --- a/plan/candidates.md +++ b/plan/candidates.md @@ -793,13 +793,18 @@ vector is a feature change with its own control run, and has not been made. **Rollout** -- [ ] Behind a flag, old generator default, so one match can produce both sets - and diff them. Test: with the flag off, behaviour is unchanged +- [x] Behind a flag, old generator default, so one match can produce both sets + and diff them. Test: with the flag off, behaviour is unchanged. + `SDS_CANDIDATES`, `sds one --candidates`, and + `unit::tests::flag_off_is_the_four_verb_menu` - [ ] **Keep the four verbs only as a transitional instrument, and count how often they win.** They are not a permanent floor: if the scorer keeps preferring them, either the new states are bad or the scorer cannot tell them apart, and both are findings. Remove them once a run shows the - generated states winning, and record the share at which they were removed + generated states winning, and record the share at which they were removed. + Counted: `PlanTally::verb_share`, printed every round. **First real match, + `mirror-lance` seed 1: the four verbs won 26 of 31 movement decisions, 84%.** + The verbs stay until that turns over - [ ] `examples/pathfind.rs`: states/sec and `L x M` pairs/sec, open terrain against heavy forest, recorded in [PERFORMANCE.md](../docs/PERFORMANCE.md) in its own style @@ -808,6 +813,34 @@ vector is a feature change with its own control run, and has not been made. micro-benchmark, deliberately, because a 24-game bench costs twenty minutes and carries +/- 25 points of noise +### First run in a match + +`mirror-lance` seed 1, `sds one --sds-seat North --candidates`, hand-authored +weights, against Princess. It completed: victory on round 9, 60 of 66 decisions +answered, 0 defaulted, 6 illegal. Princess won. + +- **Time is not the problem.** A whole planning cycle - every unit of a lance + swept, scored, surfaced, and the lance reconciled - ran in 2 to 8 ms, mean 4. + The offline 79 ms was six views against enemy *reachable sets*; in a match + every enemy is one `Presence`, so `M = 1` per enemy and the sweep is `L x N`. + Largest single unit seen: 106 states, 848 exchanges. +- **The menu is capped and the cap binds.** 5 to 16 candidates a unit, 11.5 mean. +- **The four verbs won 26 of 31 movement decisions, 84%.** That is the number + the removal step is waiting on. Either the surfaced states are worse than a + straight line at the nearest enemy, or the weight set cannot tell them apart - + the weights were fitted on menus that never contained a generated state, which + makes the second the thing to rule out first. +- **Two of the five chosen generated orders were illegal**, against four of the + twenty-six chosen four-verb orders. The route encoder is exact - see + `candidates::tests::a_route_becomes_a_step_list`, which checks every reached + stand - so the disagreement is between the movement model and MegaMek, not in + the translation. +- **Both illegal generated orders were two different units sent to the same + hex**, `(11, 2)`, in the same round. The sweep blocks on where units *are* and + the force does not deconflict where they are *going*. The four verbs hid this + because two of them rarely name one hex. + + ## Heat: reported, not budgeted `VolleyOutcome` reported expected damage, `p_kill`, `p_mission_kill`, expected diff --git a/sds/cli.py b/sds/cli.py index f77802d..9a61c27 100644 --- a/sds/cli.py +++ b/sds/cli.py @@ -89,6 +89,22 @@ def _play(spec: MatchSpec, out_dir: Path, roles: dict[str, str]) -> dict: return result +def _bot_env(args: argparse.Namespace) -> dict[str, str]: + """What the bot inherits, from the flags that switch a bot behaviour on. + + Both are off unless asked for. `SDS_IMITATE` rebuilds a menu for every + enemy move, which is a second bot's worth of thinking; `SDS_CANDIDATES` + replaces the four hardcoded verbs with the reachable-state sweep, which is + a change to play and must be said out loud on any number it produces. + """ + env: dict[str, str] = {} + if getattr(args, "imitate", False): + env["SDS_IMITATE"] = "1" + if getattr(args, "candidates", False): + env["SDS_CANDIDATES"] = "1" + return env + + def cmd_one(args: argparse.Namespace) -> int: scenario = Path(args.scenario).resolve() factions = _factions(scenario) @@ -109,6 +125,7 @@ def cmd_one(args: argparse.Namespace) -> int: timeout_ms=args.timeout_ms, stall_seconds=args.stall_seconds, bv_destroyed_percent=args.bv_destroyed_percent, + env=_bot_env(args), ) result = _play(spec, _run_dir(Path(args.out), "one"), roles) print(json.dumps(result, indent=2)) @@ -202,7 +219,7 @@ def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: # setting when one seat is not ours: against Princess this # is the whole point, and in self-play it would only # reconstruct moves the bot already logged itself. - env={"SDS_IMITATE": "1"} if getattr(args, "imitate", False) else {}, + env=_bot_env(args), ), roles, ) @@ -613,6 +630,11 @@ def main(argv: list[str] | None = None) -> int: one.add_argument( "--sds-seat", default=None, help="faction the bot plays; omit for Princess on both sides" ) + one.add_argument( + "--candidates", + action="store_true", + help="propose from the reachable-state sweep as well as the four verbs (SDS_CANDIDATES)", + ) one.set_defaults(func=cmd_one) bench = sub.add_parser("bench", help="play many matches and aggregate")