Something went wrong. Try again.
bayes for days
Something went wrong. Try again.
Python
1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401240224032404240524062407240824092410241124122413241424152416241724182419242024212422242324242425242624272428242924302431243224332434243524362437243824392440244124422443244424452446244724482449245024512452245324542455245624572458245924602461246224632464246524662467246824692470247124722473247424752476247724782479248024812482248324842485248624872488248924902491249224932494249524962497249824992500250125022503250425052506250725082509251025112512251325142515251625172518251925202521252225232524252525262527252825292530253125322533253425352536253725382539254025412542254325442545254625472548254925502551255225532554255525562557255825592560256125622563256425652566256725682569257025712572257325742575257625772578257925802581258225832584258525862587258825892590259125922593259425952596259725982599260026012602260326042605260626072608260926102611261226132614261526162617261826192620262126222623262426252626262726282629263026312632263326342635263626372638263926402641264226432644264526462647264826492650265126522653265426552656265726582659266026612662266326642665266626672668266926702671267226732674267526762677267826792680268126822683268426852686268726882689269026912692269326942695269626972698269927002701270227032704270527062707270827092710271127122713271427152716271727182719272027212722272327242725272627272728272927302731273227332734273527362737273827392740274127422743274427452746274727482749275027512752275327542755275627572758275927602761276227632764276527662767276827692770277127722773277427752776277727782779278027812782278327842785278627872788278927902791279227932794279527962797279827992800280128022803280428052806280728082809281028112812281328142815281628172818281928202821282228232824282528262827282828292830283128322833283428352836283728382839284028412842284328442845284628472848284928502851285228532854285528562857285828592860286128622863286428652866286728682869287028712872287328742875287628772878287928802881288228832884288528862887288828892890289128922893289428952896289728982899290029012902290329042905290629072908290929102911291229132914291529162917291829192920292129222923292429252926292729282929293029312932293329342935293629372938293929402941294229432944294529462947294829492950295129522953295429552956295729582959296029612962296329642965296629672968296929702971297229732974297529762977297829792980298129822983298429852986298729882989299029912992299329942995299629972998299930003001300230033004300530063007300830093010301130123013301430153016301730183019302030213022302330243025302630273028302930303031303230333034303530363037303830393040304130423043304430453046304730483049305030513052305330543055305630573058305930603061306230633064306530663067306830693070307130723073307430753076307730783079308030813082308330843085308630873088308930903091309230933094309530963097309830993100310131023103310431053106310731083109311031113112311331143115311631173118311931203121312231233124"""sds - run bot matches and say whether the difference is real.
sds one one match, verbose, for looking atsds play one match with you in a seat, at MegaMek's own clientsds bench many matches, aggregated, for deciding withsds control the harness's self-test: Princess against itselfsds validate every scenario in a suite loads and can deploysds catalogue what the bot reasons with: features, intents, labels, goalssds view what the bot believed and chose, as HTMLsds watch the same, live on a local port, with the map coloured by evaluationsds explain one decision's candidates, feature by featuresds firing legal shots against declared ones, per weaponsds train fit M from a run's decision logssds corpus store a run's decision logs where a worktree removal cannot reach"""
from __future__ import annotations
import argparseimport concurrent.futuresimport hashlibimport jsonimport osimport signalimport sysimport timeimport webbrowserfrom dataclasses import asdictfrom datetime import UTC, datetimefrom pathlib import Path
from . import boardfrom .baseline import Baseline, Incomparable, compare, from_run
# Constants only, so building the parser does not read a corpus.from .counterfactual import DEFAULT_MATCHES as DEFAULT_COUNTERFACTUAL_MATCHESfrom .counterfactual import PHASES as COUNTERFACTUAL_PHASESfrom .explain import ExplainError
# Constants only, so building the parser does not stand up a server. The# module itself is imported inside `cmd_watch`.from .live import DEFAULT_PORT as WATCH_PORTfrom .live import KEEP_DETAILED as WATCH_KEEPfrom .live import MAX_PROPOSALS as WATCH_PROPOSALSfrom .match import ( REPO, MatchError, MatchSpec, Seat, is_sidecar, kill_all_matches, kill_stragglers, run,)from .stats import Summary, games_neededfrom .tactics import DEFAULT_MATCHES as DEFAULT_TACTIC_MATCHESfrom .tactics import PHASES as TACTIC_PHASES
# Constants only, so building the parser does not drag the fit in. Everything# `cmd_train` actually calls stays behind its local import.from .train import CREDIT_DEFAULT as TRAIN_CREDITfrom .train import CREDITS as TRAIN_CREDITSfrom .train import GAMMA as TRAIN_GAMMAfrom .train import LABEL_KIND as TRAIN_LABELfrom .train import LABEL_MEANING as TRAIN_LABELSfrom .train import TIER_DEFAULT as TRAIN_TIERfrom .train import TIERS as TRAIN_TIERSfrom .viewer import ViewerErrorfrom .viewer import write as write_view
DEFAULT_SCENARIO = REPO / "scenarios" / "mirror-lance.mms"
# What `sds play` opens with when nobody says otherwise. A 2v2 rather than the# 1v1: two units a side is enough for the bot's force layer to have something to# do, and few enough that a person's turn is not a chore.DEFAULT_PLAY_SCENARIO = REPO / "scenarios" / "suite" / "2v2-jihad-s102020.mms"
# The bot a seat marked "sds" runs unless told otherwise.## It used to be `bots/random_bot.py`, and that default cost a day. The floor bot# is a deliberate baseline and it answers the protocol perfectly, so a bench run# that forgot `--bot` produced a full run directory, a "sds" column and a# baseline record - measuring the floor while every reader believed it was# measuring the change under test. The floor is still one command away# (`--bot "python3 /work/bots/random_bot.py"`); it is no longer what you get by# saying nothing.## `/work` is the repository, mounted into the match container by `sds.match`.DEFAULT_BOT = "/work/target/release/sds-bot"
# Where DEFAULT_BOT lives on this side of the mount, for the preflight below.BOT_BINARY = REPO / "target" / "release" / "sds-bot"
def _explore_flags(parser: argparse.ArgumentParser) -> None: """The two exploration knobs, on any subcommand that plays matches.""" parser.add_argument( "--explore", type=float, default=0.0, metavar="RATE", help="share of decisions taken off-policy, 0 to 1. It drives BOTH " "movement and firing - one rate, two phases - which the help here used " "to deny. Recorded per row as `policy`, so a fit can tell an explored " "choice from a fault", ) parser.add_argument( "--explore-temperature", type=float, default=None, metavar="POINTS", help="how far down the value scale an explored draw reaches, in value " "points (default 0.25). Not a probability: a candidate this far below " "the best is about a third as likely to be drawn", )
def _with_explore(command: str | None, args: argparse.Namespace) -> str | None: """Put `--explore` and `--explore-temperature` on the bot command.
First-class flags rather than a hand-written `--bot` string. A typo in that string produced a silently non-exploring corpus, which is the same failure as `--bot` once defaulting to the random bot and `SDS_CANDIDATES` once defaulting to off: the run looks fine and measures something else.
Exploration is movement-only today; firing is always on-policy. See `crates/sds-core/src/explore.rs`. """ rate = getattr(args, "explore", 0.0) or 0.0 if rate <= 0.0: return command if not 0.0 < rate <= 1.0: raise MatchError(f"--explore {rate}: must be above 0 and at most 1") bot = command or DEFAULT_BOT if "--explore" in bot: raise MatchError(f"--explore was given both as a flag and inside --bot ({bot!r}); pick one") bot = f"{bot} --explore {rate}" temperature = getattr(args, "explore_temperature", None) if temperature is not None: if temperature <= 0.0: raise MatchError(f"--explore-temperature {temperature}: must be above 0") bot = f"{bot} --explore-temperature {temperature}" return bot
def _seat_answered_nothing(result: dict) -> str | None: """A seat that was asked for decisions and answered none of them.
`_check_bot` covers the default binary only, and deliberately: an arbitrary `--bot` may be a script or a path inside the image. But a wrong one fails the same way the unbuilt binary does - the host cannot start the process, disables the seat, and every decision takes an inert default - and nothing stops it. A relative path in `--bot` cost six matches of exactly that: the log said so once, on one line, and the run played on.
So the run checks its own first match rather than trusting the command. A seat with decisions and no answers has not made any, whatever the reason. """ for player in result.get("players", []): if not isinstance(player, dict): continue tally = player.get("sds") if not isinstance(tally, dict) or not tally: continue if tally.get("decisions", 0) and not tally.get("answered", 0): return str(player.get("name", "a seat")) return None
def _check_bot(command: str | None) -> None: """Refuse to start a run whose bot is not built.
Only for the default binary: an arbitrary `--bot` command is the operator's business and may be a script, an interpreter, or a path inside the image. Without this the match still runs - the host cannot start the process, disables the seat, and every decision takes an inert default - and the run reads as a bot that chose to do nothing, which is the same failure the default itself used to produce. """ if not command or DEFAULT_BOT not in command.split(): return if not BOT_BINARY.exists(): raise MatchError( f"{BOT_BINARY} is not built. `cargo build --release -j 2 -p sds-bot` first, " "or pass --bot with something else." ) # And that it is not older than the code it was built from. A run against a # stale binary is the same failure as a run against the wrong bot: every # number it prints is about a program nobody is looking at. This has already # happened once - a diagnostic batch silently played the binary from forty # minutes earlier - and existence alone would not have caught it. built = BOT_BINARY.stat().st_mtime newer = [path for path in (REPO / "crates").rglob("*.rs") if path.stat().st_mtime > built] newer += [path for path in REPO.glob("crates/**/Cargo.toml") if path.stat().st_mtime > built] if newer: listed = ", ".join(str(p.relative_to(REPO)) for p in sorted(newer)[:3]) more = f" and {len(newer) - 3} more" if len(newer) > 3 else "" raise MatchError( f"{BOT_BINARY} is older than {listed}{more}. " "`cargo build --release -j 2 -p sds-bot` first, or pass --bot with something else." )
def _factions(scenario: Path) -> list[str]: """The faction names in an .mms, in file order.
Read rather than configured: a benchmark that needs the seat names spelled out on the command line is one typo away from silently benchmarking a Princess mirror and reporting it as an SDS result. """ for line in scenario.read_text().splitlines(): if line.startswith("Factions="): return [f.strip() for f in line.split("=", 1)[1].split(",")] raise MatchError(f"no Factions= line in {scenario}")
def _pick(count: int, game: int, games: int) -> int: """Which scenario the nth game of a run plays.
Round-robin rather than random, so a run covers the suite evenly instead of leaving whichever scenarios the seed missed unmeasured.
A plain `game % count` does that only when the run is at least as long as the suite. Shorter than it, `game % count` is the *first* `games` entries of a sorted list, and a suite is sorted by filename with the era in the name: 60 games over a 228-scenario suite played 48 civil-war matches and 12 clan-invasion ones and no others, and read as a measurement over the whole suite. Every short run measured before this was a measurement of its alphabetical prefix.
So a short run strides instead, taking evenly spaced entries across the whole suite. Deterministic either way - a benchmark that samples has to sample the same way twice or two runs are not comparable. """ if games >= count: return game % count return (game * count) // games
def _side(count: int, game: int, games: int) -> int: """Which of the two factions the bot under test plays, this game.
No two deployment edges are equally good, not even on a mirrored map, so a benchmark that never swaps measures the edge as much as the bot. What has to alternate is a *scenario's own* plays: playing map A twice from the north and map B twice from the south balances the seat totals and swaps nothing.
That is what `game % 2` did, and it was right only by accident. With the scenario picked as `game % count`, a scenario's repeats are `count` games apart, so their parities differ only when `count` is odd. `scenarios/wide` has 57 and swapped correctly; `scenarios/suite` has 24 and `scenarios/tonight` has 228, and over both of those every scenario was played twice from the same side. The seat totals still came out level, which is why nothing looked wrong.
So alternate on the repeat rather than on the game, and offset the starting side by the scenario, or an odd number of repeats piles up on one side: three passes over a 24-scenario suite alternate 0,1,0 for every scenario and come out 48 to 24. Offset, half the suite starts on each side and the same run is level.
Below the suite size nothing repeats, and there the game index is what balances the totals. """ if games >= count: return (game // count + _pick(count, game, games)) % 2 return game % 2
def _suite(directory: Path) -> tuple[list[Path], str]: """The scenarios in a suite, and a fingerprint of exactly which they are.
The fingerprint goes into the baseline's scenario field, so a regenerated suite - different maps, different forces, an added size - is *incomparable* with anything measured on the old one rather than quietly comparable. That is the same rule as the victory condition: settings that change what a win rate means travel with the number. """ scenarios = sorted(directory.glob("*.mms")) if not scenarios: raise MatchError(f"no .mms files in {directory}") digest = hashlib.sha256() for path in scenarios: digest.update(path.name.encode()) digest.update(path.read_bytes()) return scenarios, f"suite:{directory.name}:{digest.hexdigest()[:12]}"
def _run_dir(root: Path, label: str) -> Path: stamp = datetime.now(UTC).strftime("%Y%m%dT%H%M%SZ") return root / f"{stamp}-{label}"
def _play(spec: MatchSpec, out_dir: Path, roles: dict[str, str]) -> dict: result = run(spec, out_dir) # Stamp each player with the role it played this game, so a benchmark that # swaps sides can still tally by bot. for player in result.get("players", []): player["role"] = roles.get(player["name"], player["name"]) return result
def _bot_env(args: argparse.Namespace) -> dict[str, str]: """What the bot inherits, from the flags that switch a bot behaviour on.
Off unless asked for. `SDS_IMITATE` rebuilds a menu for every enemy move, which is a second bot's worth of thinking.
There is no candidate switch any more. The reachable-state sweep is the only proposer; the four hardcoded verbs and the `SDS_CANDIDATES` gate that chose between them are gone, because a bench that forgot the flag measured a different, weaker bot and said nothing about it. """ env: dict[str, str] = {} if getattr(args, "imitate", False): env["SDS_IMITATE"] = "1" # How many tokio workers the bot gets, when an experiment is measuring that. # `--pin` otherwise sets it to 1, and this overrides that: the question "is # four workers on one core costing us anything" cannot be asked without it. if os.environ.get("SDS_WORKER_THREADS"): env["SDS_WORKER_THREADS"] = os.environ["SDS_WORKER_THREADS"] return env
def cmd_one(args: argparse.Namespace) -> int: args.bot = _with_explore(args.bot, args) _check_bot(args.bot if args.sds_seat else None) scenario = Path(args.scenario).resolve() factions = _factions(scenario) seats = [] roles = {} for name in factions: if args.sds_seat and name == args.sds_seat: seats.append(Seat(name, "sds", args.bot)) roles[name] = "sds" else: seats.append(Seat(name)) roles[name] = "princess" spec = MatchSpec( scenario=scenario, seats=seats, seed=args.seed, max_rounds=args.max_rounds, timeout_ms=args.timeout_ms, stall_seconds=args.stall_seconds, bv_destroyed_percent=args.bv_destroyed_percent, replay=args.replay, pin=args.pin, env=_bot_env(args), ) result = _play(spec, _run_dir(Path(args.out), "one"), roles) print(json.dumps(result, indent=2)) return 0 if result.get("outcome") == "victory" else 1
def cmd_play(args: argparse.Namespace) -> int: """One match with a person in a seat.""" from .play import PlaySpec, check_display from .play import play as play_match
if args.check_display: return check_display()
_check_bot(args.bot) scenario = Path(args.scenario or DEFAULT_PLAY_SCENARIO).resolve() factions = _factions(scenario) if len(factions) != 2: raise MatchError(f"{scenario} has {len(factions)} factions; play wants exactly 2") # Which side is which. Named by faction when asked for, and otherwise the # first faction in the file is yours - which is also the order the .mms # deployment edges are written in, so "the first one" is a thing you can # look up rather than a coin flip. human = args.side or factions[0] if human not in factions: raise MatchError(f"no faction named '{human}' in {scenario} (has: {factions})") sds = next(name for name in factions if name != human)
spec = PlaySpec( scenario=scenario, human=human, sds=sds, bot=args.bot, seed=args.seed, port=args.port, timeout_ms=args.timeout_ms, bv_destroyed_percent=args.bv_destroyed_percent, max_rounds=args.play_max_rounds, env=_bot_env(args), ) return play_match(spec, _run_dir(Path(args.out), "play"))
def _weights_fingerprint(bot: str | None) -> dict[str, str]: """Hash every weights file the bot command names.
**The suite is already fingerprinted and the weights file was not.** `_suite` hashes every `.mms` so a regenerated suite is incomparable rather than quietly comparable; the file that decides how every candidate is scored had no such guard. An arm whose weights changed under it - a rebase, an edit, a fit written to the same path - would report numbers as though nothing had happened.
A running measurement owns its inputs or it does not own its result, and the two ways to survive a violation are not the same. This **freezes**: the answer is to invalidate the run, not to carry on, because a benchmark continued on changed inputs is worse than one that stopped. A build artefact is the other case and should be rebuilt rather than frozen - see `plan/harness.md`.
Best effort by design. A bot command with no `--weights` uses the compiled default and there is nothing to hash; an unreadable path is left out rather than failing a run over a guard. """ out: dict[str, str] = {} if not bot: return out parts = bot.split() for flag in ("--weights", "--tactic-weights"): while flag in parts: index = parts.index(flag) parts = parts[index + 1 :] if not parts: break named = parts[0] # The bot command names container paths; `/work` is this checkout. local = ( Path(str(named).replace("/work/", "", 1)) if named.startswith("/work/") else Path(named) ) try: out[named] = hashlib.sha256(local.read_bytes()).hexdigest()[:12] except OSError: continue return out
def _check_bot_is_reachable(sds_bot: str | None) -> None: """A custom `--bot` has to be a path the match container can see.
The bot string is handed to the seat process inside the container verbatim, and the container mounts only the repository, at `/work`. A host path outside that mount does not fail: the seat never starts, every decision takes the inert default, and the run completes and reports a win rate.
That is exactly the shape this repository keeps paying for - an absence wearing a legal value. It cost a whole 24-match baseline arm, which came back with 162 decisions of 162 `defaulted` and not one candidate, and the only reason it was caught is that the decision logs were three orders of magnitude too small.
**Why this is its own function rather than a line in `_check_bot`.** That check - the one that refuses a binary older than the code it was built from - does not run for a custom bot at all. So the case where a caller is most likely to hand over something wrong was the one case with no coverage, while the default path, where the binary is derived and can hardly be wrong, had all of it. A guard that stands down exactly where the risk is concentrated is worse than no guard: its presence reads as protection, and nobody looks again. Any check gated on "unless the caller supplied their own" is worth re-reading in that light. """ if sds_bot is None: return binary = sds_bot.split()[0] if sds_bot else sds_bot if binary.startswith("/work/"): return raise MatchError( f"--bot {binary} is not under /work, so the match container cannot see " "it and every decision would silently default. Copy the binary into the " "repository and pass its container path, e.g. " "`--bot /work/target/release/sds-bot-baseline`." )
def _bench(args: argparse.Namespace, label: str, sds_bot: str | None) -> int: sds_bot = _with_explore(sds_bot, args) _check_bot(sds_bot) _check_bot_is_reachable(sds_bot) if getattr(args, "suite", None): scenarios, scenario_label = _suite(Path(args.suite).resolve()) print(f"suite: {len(scenarios)} scenarios, {scenario_label}", file=sys.stderr) else: scenarios = [Path(args.scenario).resolve()] scenario_label = str(scenarios[0]) for path in scenarios: factions = _factions(path) if len(factions) != 2: raise MatchError(f"{path} has {len(factions)} factions; bench wants exactly 2")
# Five seconds against the possibility of an hour. A scenario that cannot # be loaded fails once per game it is drawn for, as "match produced no # result", and a round-robin draws each one many times. if not getattr(args, "no_validate", False): from .match import validate
code, report = validate( scenarios, allow_uncrossable=getattr(args, "allow_uncrossable", False) ) if code != 0: broken = [entry["scenario"] for entry in report if not entry["ok"]] raise MatchError( f"{len(broken)} scenarios cannot be played; nothing was run. " "See results/validate.json" )
out_dir = _run_dir(Path(args.out), label) # What the weights were when this run started. Compared again at the end: # a file that moved mid-run means the matches were not all scored the same # way, and the run says so rather than reporting a mean over two policies. weights_before = _weights_fingerprint(sds_bot) if weights_before: for named, digest in weights_before.items(): print(f" weights: {named} {digest}", file=sys.stderr) # Stamp before the first match, so a run that is killed part-way still says # what basis its decisions are in. Never worth failing a run over. try: from .corpus import stamp
stamped = stamp( out_dir, bot=sds_bot or "", weights=getattr(args, "weights", "") or "", explore_rate=float(getattr(args, "explore", 0.0) or 0.0), ) print(f"basis: {stamped.basis or 'unknown'}", file=sys.stderr) except Exception as error: # noqa: BLE001 print(f"note: could not stamp {out_dir}: {error}", file=sys.stderr) plans = [] for game in range(args.games): scenario = scenarios[_pick(len(scenarios), game, args.games)] factions = _factions(scenario) side = _side(len(scenarios), game, args.games) first = factions[side] second = factions[1 - side] seats, roles = [], {} if sds_bot and getattr(args, "self_play", False): # Both seats, so that one of them wins. # # A fit needs the label to vary. Against Princess the bot loses # every game, so every label is the same number and least squares # has nothing to separate - more losses do not help, because the # problem is the variance and not the count. Self-play gives an # even split by construction and a corpus that spans the range. # # The cost is the one `plan/training.md` names: two weak bots teach # each other to beat a weak bot. Weights fitted this way are a # starting point to be re-measured against Princess, never the # number that gets reported. seats.append(Seat(first, "sds", sds_bot)) roles[first] = "sds-A" seats.append(Seat(second, "sds", sds_bot)) roles[second] = "sds-B" elif sds_bot: seats.append(Seat(first, "sds", sds_bot)) roles[first] = "sds" seats.append(Seat(second)) roles[second] = "princess" else: # The control: two Princesses. Labelled A and B by seat order so the # two are distinguishable, which is the whole point of running it. seats.append(Seat(first)) roles[first] = "A" seats.append(Seat(second)) roles[second] = "B" plans.append( ( MatchSpec( scenario=scenario, seats=seats, seed=args.seed + game, max_rounds=args.max_rounds, timeout_ms=args.timeout_ms, stall_seconds=args.stall_seconds, pin=args.pin, # Watching the other seat move, when asked. Only worth # setting when one seat is not ours: against Princess this # is the whole point, and in self-play it would only # reconstruct moves the bot already logged itself. env=_bot_env(args), ), roles, ) )
results: list[dict] = [] failures = 0 checked_first = False # Which bot, on every run, in the run's own output. The baseline record has # carried this field all along and nothing printed it, so a run of the wrong # bot looked exactly like a run of the right one until somebody opened a JSON # file they had no reason to suspect. jobs = args.jobs if args.pin and jobs != 1: # Every match asks for the same cores by the same rule, so a second match # does not get a second set - it gets the first set again, and both runs # measure contention. print(f"note: --pin runs one match at a time, not {jobs}", file=sys.stderr) jobs = 1 print( f"{args.games} matches, {jobs} at a time -> {out_dir}\n" f" sds bot: {sds_bot or 'princess (no --bot seat)'}" + ("\n pinned: one physical core per seat" if args.pin else ""), file=sys.stderr, ) with concurrent.futures.ThreadPoolExecutor(max_workers=jobs) as pool: futures = {pool.submit(_play, spec, out_dir, roles): spec for spec, roles in plans} for done in concurrent.futures.as_completed(futures): try: result = done.result() results.append(result) if not checked_first and sds_bot: checked_first = True if (dead := _seat_answered_nothing(result)) is not None: pool.shutdown(wait=False, cancel_futures=True) raise MatchError( f"the first match finished with '{dead}' answering none of its " "decisions. The host could not start the bot, " "or the bot answered nothing - either way every seat took an inert " "default and the run would measure the harness rather than the bot. " f"Check the --bot command: {sds_bot}\n" " Paths in --bot are resolved inside the container, where the " "repository is mounted at /work." ) except MatchError as error: failures += 1 print(f" match failed: {error}", file=sys.stderr) finished = len(results) + failures print(f" {finished}/{args.games}", end="\r", file=sys.stderr) print(file=sys.stderr)
if not results: print("every match failed; nothing to report", file=sys.stderr) return 1
summary = Summary(results, key="role") print(summary.render()) # Written before anything else can fail. An hour of matches followed by a # traceback on the way to the baseline used to leave nothing on disk but # the per-match files. # The other end of the freeze. Loud and last, so it is the final thing on # the terminal rather than a line scrolled past an hour ago. weights_after = _weights_fingerprint(sds_bot) if weights_after != weights_before: moved = sorted(set(weights_before) | set(weights_after)) print( "\nWARNING: a weights file changed while this run was going.\n" + "\n".join( f" {name}: {weights_before.get(name, 'absent')}" f" -> {weights_after.get(name, 'absent')}" for name in moved if weights_before.get(name) != weights_after.get(name) ) + "\nThe matches were not all scored the same way. Do not compare this run.", file=sys.stderr, ) (out_dir / "inputs.json").write_text( json.dumps( { "weights_before": weights_before, "weights_after": weights_after, "scenarios": scenario_label, }, indent=2, sort_keys=True, ) + "\n" ) (out_dir / "summary.txt").write_text(summary.render() + "\n") (out_dir / "results.json").write_text(json.dumps(results, indent=2) + "\n") # Every run leaves the standard report beside it, so which numbers got # looked at stops depending on which command somebody remembered. try: from .runreport import write as write_report
print(f"\nreport -> {write_report(out_dir)}", file=sys.stderr) except Exception as error: # noqa: BLE001 - a report is not worth failing a run print(f"\ncould not write the run report: {error}", file=sys.stderr) record = from_run( # `control` has no --save-baseline: it is the harness's self-test, not # a thing to record a bot against. name=getattr(args, "save_baseline", None) or "candidate", bot=sds_bot or "princess", results=results, summary=summary, scenario=scenario_label, bv_destroyed_percent=args.bv_destroyed_percent, max_rounds=args.max_rounds, notes=getattr(args, "notes", "") or "", ) # The record goes in the run directory whether or not it is being saved as a # baseline. It carries the scenario fingerprint and the victory settings, # which are what lets anything read this directory later - `sds train`, a # person - compare it without guessing what it was measured under. (out_dir / "baseline.json").write_text(json.dumps(asdict(record), indent=2) + "\n")
# What each side actually fired. The suites are mirrored, so the two columns # should match: a gap is a bug, not tactics. It catches at once what a win # rate cannot separate - a weapon one side never fires, an ammo choice only # one side makes, a matchup that is not really symmetric, and a bot that # simply shoots less often than the other. try: from . import weaponstats
reports = sorted(out_dir.rglob("*.report.html")) if reports: print("\nweapons, both sides") weaponstats.compare(reports) except Exception as error: # a report is a diagnostic, never a reason to fail a run print(f"\nweapon summary unavailable: {error}")
if failures: print(f"\n{failures} matches failed and are not in the numbers above") # The planning number, always: it is the difference between "we measured # nothing" and "we measured nothing, and here is what it would have taken". print( f"\nfor reference: separating a 5-point win-rate difference from noise " f"takes about {games_needed(0.05)} decided games; 10 points, " f"about {games_needed(0.10)}." ) if getattr(args, "save_baseline", None): path = record.save() print(f"\nrecorded baseline '{record.name}' -> {path}") print( "Commit it: a baseline that lives only on one machine cannot be " "compared against in a review." )
against = getattr(args, "against", None) if against: try: report = compare(record, Baseline.load(against)) except (Incomparable, FileNotFoundError) as error: print(f"\ncannot compare: {error}", file=sys.stderr) return 1 print("\n" + report) (out_dir / "comparison.md").write_text(report + "\n") print(f"\nwritten to {out_dir / 'comparison.md'} - paste it into the PR.")
# **A run that recorded decisions is saved whether or not anybody asked.** # # `runs/` is gitignored and inside the worktree, so a corpus left there is # one `git worktree remove` from gone. That is not hypothetical: it # destroyed a 360-match corpus once, and on 2026-08-29 it destroyed the # asymmetric corpus that every second figure in `docs/TACTICS.md` was # measured against. Both times the run had decision logs and nobody had # passed `--corpus`, so nothing was saved and nothing said so. # # `--corpus` now only chooses the *name*. Whether the work survives is not a # flag, because the person who forgets the flag is the person who needed it. corpus = getattr(args, "corpus", None) from .corpus import CorpusError, save_run
try: manifest, invented = save_run(out_dir, corpus, notes=getattr(args, "notes", "")) except CorpusError as error: if corpus is None: # Not a recording run. Nothing to say. return 0 # Loud, and not fatal: the matches are played and the data is still in # `runs/`. Saying "the run failed" about a naming collision would be # worse than saying what to do about it. print(f"\ncould not save corpus '{corpus}': {error}", file=sys.stderr) print(f"the data is still at {out_dir} - save it before removing this worktree.") return 0 if invented: print( f"\nthis run recorded decisions, so it was saved as '{manifest.name}'.\n" f"Give it a name you will recognise:\n sds corpus save {out_dir} <name>" ) else: print(f"\nsaved corpus '{manifest.name}': {out_dir}") return 0
def cmd_bench(args: argparse.Namespace) -> int: return _bench(args, "bench", args.bot)
def cmd_spread(args: argparse.Namespace) -> int: """What each feature actually reported, over a run's own decisions.""" from .spread import render, scan
run_dir = Path(args.run_dir) if not run_dir.is_dir(): print(f"error: {run_dir} is not a directory", file=sys.stderr) return 2 text = render(scan(run_dir)) print(text) if getattr(args, "out", None): Path(args.out).write_text(text + "\n", encoding="utf-8") print(f"written to {args.out}", file=sys.stderr) return 0
def cmd_report(args: argparse.Namespace) -> int: """The standard post-run report, in one document.""" from .runreport import render, write
run_dir = Path(args.run_dir) if not run_dir.is_dir(): print(f"error: {run_dir} is not a directory", file=sys.stderr) return 2 text = render(run_dir) print(text) if not getattr(args, "no_write", False): print(f"written to {write(run_dir)}", file=sys.stderr) return 0
def cmd_melee(args: argparse.Namespace) -> int: """Physical attacks a run actually landed, from MegaMek's own log.""" from .melee import render, scan
run_dir = Path(args.run_dir) if not run_dir.is_dir(): print(f"error: {run_dir} is not a directory", file=sys.stderr) return 2 print(render(run_dir, scan(run_dir))) return 0
def cmd_stats(args: argparse.Namespace) -> int: """Every counter a run kept, summed into one page.""" from .runstats import render, rollup
run_dir = Path(args.run_dir) if not run_dir.is_dir(): print(f"error: {run_dir} is not a directory", file=sys.stderr) return 2 out = rollup(run_dir) if not out.matches: print(f"error: {run_dir} has no match results in it", file=sys.stderr) return 2 print(render(out)) return 0
def cmd_control(args: argparse.Namespace) -> int: """Princess against Princess. Should come out 50/50; if it does not, stop.""" return _bench(args, "control", None)
def cmd_scenarios(args: argparse.Namespace) -> int: from .scenario import write_suite
sizes = tuple(int(part) for part in args.sizes.split(",")) if args.sizes else None tiers = tuple(int(part) for part in args.bv_tiers.split(",")) if args.bv_tiers else () per_era = int(args.designs_per_era or 0) eras = tuple(part.strip() for part in args.eras.split(",")) if args.eras else None written = write_suite( Path(args.suite_out), per_size=args.per_size, seed=args.suite_seed, **({"sizes": sizes} if sizes else {}), **({"eras": eras} if eras else {}), **({"bv_tiers": tiers} if tiers else {}), **({"designs_per_era": per_era} if per_era else {}), mirrored=not getattr(args, "asymmetric", False), unit_pool=getattr(args, "unit_pool", "mek"), ) print(f"wrote {len(written)} scenarios to {args.suite_out}") for path in written: print(f" {path.name}") return 0
def cmd_validate(args: argparse.Namespace) -> int: """Load every scenario in a suite and report the ones a match could not play.""" from .match import validate
if args.suite: scenarios, label = _suite(Path(args.suite).resolve()) print(f"suite: {len(scenarios)} scenarios, {label}", file=sys.stderr) else: scenarios = [Path(args.scenario or DEFAULT_SCENARIO).resolve()] code, _report = validate(scenarios, allow_uncrossable=args.allow_uncrossable) return code
def cmd_los_dump(args: argparse.Namespace) -> int: """Write the corpus the Rust line of sight is tested against.""" from .match import los_dump
scenarios = [Path(path).resolve() for path in args.scenarios] or [ Path(DEFAULT_SCENARIO).resolve() ] out = Path(args.out).resolve() code = los_dump(scenarios, out, cases=args.cases) if code == 0: print(f"wrote {out}") return code
def cmd_pathfind_dump(args: argparse.Namespace) -> int: """Write the corpus the Rust reachability search is tested against.""" from .match import pathfind_dump
out = Path(args.out).resolve() code = pathfind_dump(out) if code == 0: print(f"wrote {out}") return code
def cmd_routes_check(args: argparse.Namespace) -> int: """Put the routes the generator emits through MegaMek's own legality check.""" from .match import routes_check
routes = Path(args.routes).resolve() out = Path(args.out).resolve() if not routes.is_file(): print(f"no routes at {routes}; run the differential test with", file=sys.stderr) print( f" SDS_ROUTES_OUT={routes} cargo test -j 2 -p sds-core --test pathfind_differential", file=sys.stderr, ) return 2 code = routes_check(routes, out) if code == 0: print(f"wrote {out}") return code
def cmd_arc_dump(args: argparse.Namespace) -> int: """Write the corpus the Rust weapon arcs are tested against.""" from .match import arc_dump
out = Path(args.out).resolve() code = arc_dump(out) if code == 0: print(f"wrote {out}") return code
def cmd_formation_dump(args: argparse.Namespace) -> int: """Write the corpus the Rust formation catalogue is checked against.""" from .match import formation_dump
out = Path(args.out).resolve() code = formation_dump(out) if code == 0: print(f"wrote {out}") return code
def cmd_cluster_dump(args: argparse.Namespace) -> int: """Record how MegaMek groups each weapon's packets into location rolls.""" from .match import cluster_dump
out = Path(args.out).resolve() code = cluster_dump(Path(args.detail).resolve(), out) if code == 0: print(f"wrote {out}") return code
def cmd_motive_damage(args: argparse.Namespace) -> int: """Dump the motive damage result table and the per-arc modifier.""" from .match import motive_damage
out = Path(args.out) code = motive_damage(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_vehicle_livefire(args: argparse.Namespace) -> int: """Fire at a tank from every arc and count where the hits landed.""" from .match import vehicle_livefire
out = Path(args.out) code = vehicle_livefire(Path(args.scenario), out, int(args.shots)) if code == 0: print(f"wrote {out}") return code
def cmd_forces(args: argparse.Namespace) -> int: """Classify the forces a suite fields, and say how often it is anything.
Two halves, because neither side can do it alone: MegaMek parses the scenarios and reads each machine's tonnage, role, movement, armour and weapon classes, and `sds_core::formation` decides what that composition is. """ import subprocess import tempfile
from .match import forces_dump
scenarios = ( sorted(Path(args.suite).glob("*.mms")) if Path(args.suite).is_dir() else [Path(args.suite)] ) if not scenarios: print(f"no scenarios under {args.suite}", file=sys.stderr) return 2 if not BOT_BINARY.is_file(): print( f"{BOT_BINARY} is not built. `cargo build --release -j 2 -p sds-bot` first.", file=sys.stderr, ) return 2
with tempfile.TemporaryDirectory(dir=str(REPO / "runs")) as work: dump = Path(work) / "forces.jsonl" code = forces_dump(scenarios, dump) if code != 0: return code if args.out: out = Path(args.out).resolve() out.parent.mkdir(parents=True, exist_ok=True) out.write_text(dump.read_text()) print(f"wrote {out}") return subprocess.run( [str(BOT_BINARY), "--classify-forces", str(dump)], check=False ).returncode
def cmd_lances(args: argparse.Namespace) -> int: """Draw a corpus of random lances from MegaMek's own force generator.
The corpus is not committed and is not meant to be. It carries every machine's tonnage, armour, movement and weapons, which is a bulk copy of MegaMek's CC BY-NC-SA unit data - the same reason `forces.jsonl` is regenerated rather than shipped. What is committed is this command, so the same seed and the same MegaMek release give the same corpus back. """ from .match import lances_dump
out = Path(args.out).resolve() code = lances_dump( out, int(args.count), int(args.seed), int(args.size), args.scope, args.ratings, ) if code == 0: print(f"wrote {out}") return code
def cmd_cluster_table(args: argparse.Namespace) -> int: """Write the corpus the Rust cluster hits table is tested against.""" from .match import cluster_table
out = Path(args.out).resolve() code = cluster_table(out) if code == 0: print(f"wrote {out}") return code
def cmd_conformance(args: argparse.Namespace) -> int: from .match import conformance
return conformance(Path(args.recording), bool(args.bless))
def cmd_stand_probe(args: argparse.Namespace) -> int: """Record what MegaMek permits a prone, damaged Mek of each chassis.""" from .match import stand_probe
out = Path(args.out).resolve() code = stand_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_club_probe(args: argparse.Namespace) -> int: """Record what MegaMek says a mounted melee weapon does.""" from .match import club_probe
out = Path(args.out).resolve() code = club_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_self_damage_probe(args: argparse.Namespace) -> int: """Record what MegaMek says an attack costs the machine that makes it.""" from .match import self_damage_probe
out = Path(args.out).resolve() code = self_damage_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_dfa_probe(args: argparse.Namespace) -> int: """Record what MegaMek says a death from above costs, and what it refuses.""" from .match import dfa_probe
out = Path(args.out).resolve() code = dfa_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_battle_armor_probe(args: argparse.Namespace) -> int: """Record what MegaMek permits a battle armor squad and its carrier.""" from .match import battle_armor_probe
out = Path(args.out).resolve() code = battle_armor_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_minrange_probe(args: argparse.Namespace) -> int: """Record what MegaMek charges a shot at every range, minimum ranges included.""" from .match import minimum_range_probe
out = Path(args.out).resolve() census = sorted(Path(args.census).glob("*.mms")) if args.census else [] code = minimum_range_probe(Path(args.scenario), census, out) if code == 0: print(f"wrote {out}") return code
def cmd_move_probe(args: argparse.Namespace) -> int: """Record what MegaMek charges each movement mode for terrain, and what it bars.""" from .match import move_probe
out = Path(args.out).resolve() code = move_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_afford_probe(args: argparse.Namespace) -> int: """Record when MegaMek lets a mover spend more than its allowance.""" from .match import afford_probe
out = Path(args.out).resolve() code = afford_probe(out) if code == 0: print(f"wrote {out}") return code
def cmd_step_cost_probe(args: argparse.Namespace) -> int: """Record what one step costs each mover, split by terrain and level.""" from .match import step_cost_probe
out = Path(args.out).resolve() code = step_cost_probe(out) if code == 0: print(f"wrote {out}") return code
def cmd_infantry_cover(args: argparse.Namespace) -> int: """Record what MegaMek charges a platoon for the hex it stands in.""" from .match import infantry_cover
out = Path(args.out) code = infantry_cover(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_cover_probe(args: argparse.Namespace) -> int: """Record what MegaMek charges for the ground between two units.""" from .match import cover_probe
out = Path(args.out).resolve() code = cover_probe(Path(args.scenario), out) if code == 0: print(f"wrote {out}") return code
def cmd_artillery_probe(args: argparse.Namespace) -> int: """Record flight time, ranges and the blast template for every artillery piece.""" from .match import artillery_probe
out = Path(args.out).resolve() code = artillery_probe(out) if code == 0: print(f"wrote {out}") return code
def cmd_catalogue(args: argparse.Namespace) -> int: """Everything the bot reasons with, read out of the code that defines it.""" from .catalogue import CatalogueError, from_bot, goals, labels, render
try: if args.json: document = from_bot() document["labels"] = labels() document["goals"] = goals() print(json.dumps(document, indent=2)) else: print(render()) except CatalogueError as error: print(f"error: {error}", file=sys.stderr) return 1 return 0
def cmd_baselines(args: argparse.Namespace) -> int: from .baseline import BASELINES
found = sorted(BASELINES.glob("*.json")) if not found: print("no baselines recorded. Make one with:") print(" ./sds.sh bench --games 60 --bot ... --save-baseline main") return 0 print(f"{'name':<16} {'bot':<28} {'games':>6} {'decided':>8} {'rate':>7} commit") for path in found: b = Baseline.load(path.stem) print( f"{b.name:<16} {b.bot[:28]:<28} {b.games:>6} {b.decided:>8} {b.rate:>6.1%} {b.commit}" ) _ = args return 0
def cmd_corpus_save(args: argparse.Namespace) -> int: """Copy a run's training data into the durable store and describe it.
The copy is the point. A run directory is in `runs/`, which is gitignored and inside a worktree, and a `git worktree remove --force` took 360 matches with it because the only other reference was a directory of symlinks. """ from .corpus import CorpusError, save from .train import TrainingError
try: manifest = save(Path(args.run_dir), args.name, notes=args.notes or "", replace=args.replace) except (CorpusError, TrainingError) as error: print(f"error: {error}", file=sys.stderr) return 2 print( f"saved corpus '{manifest.name}': {manifest.matches} matches, " f"{manifest.decisions} decisions, {len(manifest.features)} features, " f"epoch {manifest.epoch}" ) print(f" data {manifest.data}") print(f" manifest {manifest.directory / 'manifest.json'}") print( "\nThe data is ignored and the manifest is committed. Commit it from " "the main checkout - the store lives there on purpose, so that removing " "a worktree cannot take a corpus with it." ) return 0
def corpus_dir(name: str) -> Path: """A stored corpus name or a run directory, resolved to a directory.
A bare name means the store, which is where a corpus worth measuring lives; `data` is the directory the results are actually under. """ from .corpus import CorpusError, corpora_root
target = Path(name) if target.exists(): return target stored = corpora_root() / name / "data" if not stored.exists(): raise CorpusError(f"no corpus '{name}', and no directory at {target}") return stored
def cmd_labels(args: argparse.Namespace) -> int: """Correlate every label against every other over a corpus.""" from .labels import report
print(report(corpus_dir(args.corpus))) return 0
def _describe_corpus(name: str) -> None: """Print what a stored corpus is, and anything it should be doubted for.
Best-effort: a run directory has no manifest and is not less usable for it, so an unreadable manifest is silence rather than an error. """ from .corpus import MANIFEST, CorpusError, Manifest, Stale, check, corpora_root
path = corpora_root() / name / MANIFEST if not path.exists(): return try: manifest = Manifest.read(path) except (CorpusError, OSError, ValueError): return print( f"corpus '{manifest.name}': {manifest.matches} matches, " f"{manifest.decisions} decisions, basis {manifest.basis or '?'}, " f"commit {manifest.commit}", file=sys.stderr, ) try: review = check(manifest) except Stale as error: print(f"warning: {error}", file=sys.stderr) return for line in review.warnings + review.notes: print(f"note: {line}", file=sys.stderr)
def cmd_tactics(args: argparse.Namespace) -> int: """Re-score a recorded corpus under every tactic and report the matrix.""" from .tactics import TacticsError, precision_report, report
# **This path read a corpus and validated nothing.** It is the one that # produces the agreement matrices in `sds_core::tactic::MEASURED`, which are # two corpora compared against each other - exactly the comparison a basis # or a commit difference invalidates - and it was the only corpus reader # that never asked. `check` raises on a moved basis and returns the rest as # notes; both are printed to stderr so the numbers on stdout stay pipeable. _describe_corpus(args.corpus) try: if args.subsets: print( precision_report( corpus_dir(args.corpus), phase=args.phase, subsets=args.subsets, ) ) return 0 print( report( corpus_dir(args.corpus), phase=args.phase, matches=args.matches, ) ) except TacticsError as error: print(error, file=sys.stderr) return 1 return 0
def cmd_counterfactual(args: argparse.Namespace) -> int: """Re-score recorded menus with one column ablated, and say what it moved.""" from .counterfactual import CounterfactualError, report
_describe_corpus(args.corpus) try: print( report( corpus_dir(args.corpus), args.column, phase=args.phase, to=args.to, matches=args.matches, ) ) except CounterfactualError as error: print(error, file=sys.stderr) return 1 return 0
def cmd_labeldoc(args: argparse.Namespace) -> int: """Render the label figures, or check the ones on disk are current.""" from .labeldoc import render
root = REPO / "docs" files = render(root) stale = [name for name, body in sorted(files.items()) if _read(root / name) != body] if args.write: (root / "labels").mkdir(parents=True, exist_ok=True) for name, body in sorted(files.items()): (root / name).write_text(body) print(f"wrote {len(files)} file(s) under {root}") return 0 if stale: print(f"{len(stale)} file(s) stale; `sds labeldoc --write` refreshes them", file=sys.stderr) for name in stale: print(f" {name}", file=sys.stderr) return 1 print(f"{len(files)} file(s) current") return 0
def _read(path: Path) -> str | None: try: return path.read_text() except OSError: return None
def cmd_labelrank(args: argparse.Namespace) -> int: """Ask whether each label ranks the winning seat above the losing one.""" from .labelrank import report
print(report(corpus_dir(args.corpus), gamma=args.gamma, tier=args.tier)) return 0
def cmd_progress(args: argparse.Namespace) -> int: """The standings of a run that has not finished yet.
`bench` prints its table once, at the end. A suite run is hours, and the question asked of it long before then is "is this vector even in the fight" - which was being answered by hand-written one-off scripts reading the per-match JSON, three times in one night, once with the wrong denominator.
The same `Summary` the final table is built from, over whatever results exist so far. A partial read is a partial read: the interval it prints is the honest width for the games played, and it is not a reason to stop a run early. """ target = Path(args.run) if not target.is_dir(): print(f"no run directory at {target}", file=sys.stderr) return 1 results = [] for path in sorted(target.glob("*.json")): if is_sidecar(path): continue try: loaded = json.loads(path.read_text()) except (OSError, json.JSONDecodeError): # A match still being written. It will be read on the next call. continue if isinstance(loaded, dict) and "outcome" in loaded: results.append(loaded) if not results: print(f"no finished matches in {target} yet", file=sys.stderr) return 1 print(Summary(results, key="role").render()) return 0
def cmd_paired(args: argparse.Namespace) -> int: """Two runs over the same scenarios and seeds, compared as paired data.
`compare` treats two runs as independent samples, which prices scenario difficulty as noise. It is not noise: on the two runs this was written for, 62 of 108 decided pairs went the same way for both bots. Only the discordant pairs say which vector is better. """ from .stats import mcnemar
def keyed(run: Path) -> dict[tuple[str, int], dict]: """Every match in a run, by scenario and seed.
`results.json` is written when the run ends. A run still going has the per-match documents and nothing else, and pairing a finished run against one in flight is worth doing - it is the same read `progress` makes, and the pairs that exist are as good as the pairs that will exist. """ out: dict[tuple[str, int], dict] = {} document = run / "results.json" if document.exists(): loaded = json.loads(document.read_text()) else: loaded = [] for path in sorted(run.glob("*.json")): if is_sidecar(path): continue try: one = json.loads(path.read_text()) except (OSError, json.JSONDecodeError): continue if isinstance(one, dict) and "outcome" in one: loaded.append(one) for result in loaded: out[(Path(result.get("scenario", "")).name, result.get("seed"))] = result return out
def won(result: dict) -> bool | None: if result.get("outcome") != "victory": return None seats = { player.get("team"): player.get("kind", "") for player in result.get("players", []) if isinstance(player, dict) } return seats.get(result.get("victoryTeam")) == "sds"
left, right = keyed(Path(args.left)), keyed(Path(args.right)) shared = set(left) & set(right) if not shared: print("the two runs share no scenario and seed; they cannot be paired", file=sys.stderr) return 1 only_left = only_right = both_won = both_lost = 0 for key in shared: a, b = won(left[key]), won(right[key]) if a is None or b is None: continue if a and not b: only_left += 1 elif b and not a: only_right += 1 elif a: both_won += 1 else: both_lost += 1 # Split rather than lumped: "both won" and "both lost" are both concordant # and they say opposite things about the suite. A comparison where most # pairs are both-lost is two bots failing the same fights, which is a # different situation from two bots winning the same fights, and neither is # visible in a single agreement count. agreed = both_won + both_lost decided = only_left + only_right + agreed chi, p = mcnemar(only_left, only_right) print(f"{len(shared)} shared scenario+seed pairs, {decided} decided in both runs") print(f" both won {both_won}") print(f" both lost {both_lost}") print(f" {Path(args.left).name} only {only_left}") print(f" {Path(args.right).name} only {only_right}") print() print(f"McNemar chi2 {chi:.3f}, p {p:.3f}") if p < 0.05: better = args.left if only_left > only_right else args.right print(f"The discordant pairs favour {Path(better).name}.") else: print( "The discordant pairs do not separate the two runs. Pairing already " "removes what the scenarios contribute; what is left is the sample." ) return 0
def cmd_correlate(args: argparse.Namespace) -> int: """Win rate broken down by what was in the match.""" from .correlate import report
run, suite = Path(args.run), Path(args.suite) if not run.is_dir(): print(f"no run directory at {run}", file=sys.stderr) return 1 if not suite.is_dir(): print(f"no suite directory at {suite}", file=sys.stderr) return 1 text = report(run, suite) print(text) if args.out: Path(args.out).write_text(text + "\n") print(f"\nwrote {args.out}", file=sys.stderr) return 0
def cmd_moves(args: argparse.Namespace) -> int: """What our units did each round, split by state.""" from .movestates import report
run = Path(args.run) if not run.is_dir(): print(f"no run directory at {run}", file=sys.stderr) return 1 text = report(run) print(text) if args.out: Path(args.out).write_text(text + "\n") print(f"\nwrote {args.out}", file=sys.stderr) return 0
def cmd_corpus_list(args: argparse.Namespace) -> int: from .corpus import corpora_root, listing
found = listing() if not found: print(f"no corpora in {corpora_root()}. Make one with:") print(" ./sds.sh corpus save runs/<timestamp>-bench <name>") return 0 header = ( f"{'name':<20} {'matches':>7} {'decisions':>9} {'epoch':>5} " f"{'features':>8} {'commit':<12} recorded" ) print(header) for m in found: print( f"{m.name[:20]:<20} {m.matches:>7} {m.decisions:>9} {m.epoch:>5} " f"{len(m.features):>8} {m.commit[:12]:<12} {m.recorded}" ) _ = args return 0
def cmd_firing(args: argparse.Namespace) -> int: """Per weapon, how many legal shots existed and how many were declared.
`sds weapons` counts what was fired. This counts what could have been, so "the bot chose a smaller volley" and "the gun was on the floor" stop looking the same. """ from . import firingaudit
firingaudit.report(firingaudit.rows(Path(args.target))) return 0
def cmd_explain(args: argparse.Namespace) -> int: """Print a decision's menu with every candidate's score broken into terms.
The counterpart to `sds view`: that renders a match, this reads one decision closely. Both come out of the same log, and neither recomputes anything the bot did not record. """ from . import explain as ex
try: logs = ex.decision_logs(Path(args.target)) override = Path(args.weights) if args.weights else None shown = 0 for log in logs: weights = ex.weights_for(log, override) rows = ex.read(log, args.phase) if args.seq is not None: rows = [r for r in rows if r["seq"] == args.seq] if args.round is not None: rows = [r for r in rows if r.get("round") == args.round] if args.unit is not None: rows = [r for r in rows if r.get("unit") == args.unit] # Menus of one are the default with nothing to choose between. # Kept in the tally, which is where they belong, and left out of # the tables, where they are a page of zeroes. print(f"== {log.name}") print(f" {ex.tally(rows)}") for row in rows: if len(row["candidates"]) < 2 and not args.all: continue if shown >= args.limit: print(f" ... stopping at --limit {args.limit}") break print() print(ex.render(row, weights, full=args.full)) shown += 1 except ExplainError as error: print(f"error: {error}", file=sys.stderr) return 2 return 0
def cmd_view(args: argparse.Namespace) -> int: """Render a match's decision log to a self-contained HTML file.
`--watch` regenerates the file on an interval and stamps a meta refresh into the page, which is how the view keeps up with a match that is still being played. It is a rewrite rather than a page that polls a data file because a `file://` page cannot fetch its own siblings - see the comment in viewer.py. """ target = Path(args.target) out = Path(args.out_file) if args.out_file else None refresh = args.interval if args.watch else None try: path = write_view(target, out, args.match, refresh) except ViewerError as error: print(f"error: {error}", file=sys.stderr) return 2 print(f"wrote {path}") if not args.watch: return 0
print(f"watching; regenerating every {args.interval}s. Ctrl-C to stop.", file=sys.stderr) while True: time.sleep(args.interval) try: write_view(target, path, args.match, refresh) except ViewerError as error: # A log that vanished or a directory not yet written. Worth saying # once per attempt and not worth ending a watch over: the usual # cause is a match that has not started producing decisions yet. print(f" {error}", file=sys.stderr)
def cmd_watch(args: argparse.Namespace) -> int: """Serve one match's decision log on loopback, tailing it as it is written.
The counterpart to `sds view`, and the reason it exists: a `file://` page cannot fetch its own siblings, so the offline viewer regenerates the whole file. With a server in front of the same log the page can ask only for the decisions it has not seen, which is what makes a hex map redrawn every second affordable while a match is being played.
Nothing is added to the bot to make this work. `SdsClient` already flushes every decision as it writes it, and the run directory is inside the repository the match container mounts. """ from . import live
try: session = live.open_session( Path(args.target), args.match, cap=args.max_proposals, keep=args.keep, mm_home=Path(args.mm_home) if args.mm_home else board.DEFAULT_MM_HOME, # Started beside `sds play`, a run directory exists before the bot # has made its first decision. That is "not yet", not an error. wait=True, ) server = live.serve(session, args.port, args.interval) except (ViewerError, live.LiveError) as error: print(f"error: {error}", file=sys.stderr) return 2
print(f"watching {session.tag}") for tail in session.tails: print(f" {tail.path}") board_note = session.board print(f" board: {board_note['name']} ({board_note['width']}x{board_note['height']} hexes)") # Flushed: the next call blocks forever, and a block-buffered stdout would # hold the URL until the server was killed. url = f"http://127.0.0.1:{args.port}/" print(f"\n {url}\n", flush=True) if not args.no_browser: # Best effort. A headless box, or no browser configured, is not a # reason to refuse to serve - the URL is printed either way. try: webbrowser.open_new_tab(url) except Exception as error: # noqa: BLE001 - any backend failure is fine print(f"could not open a browser ({error}); the URL is above", file=sys.stderr) print("Ctrl-C to stop.", file=sys.stderr) try: server.serve_forever() except KeyboardInterrupt: pass finally: server.server_close() return 0
def _compare_run(target: Path, against: str) -> str: """The run the corpus came from, next to a recorded baseline.
The record a bench writes into its own directory, so the scenario fingerprint, the victory settings and the epoch travel with it and `compare` can refuse a comparison across a rule change the same way it does everywhere else. """ directory = target if target.is_dir() else target.parent record = directory / "baseline.json" if not record.is_file(): raise MatchError( f"{record} does not exist, so there is nothing to compare. Only a " "run played by this version of `bench` has one." ) return compare(Baseline(**json.loads(record.read_text())), Baseline.load(against))
def cmd_train(args: argparse.Namespace) -> int: """Fit a weight vector from a run's decision logs and write it out.
Offline: it reads what a benchmark already produced and plays nothing. `--against` frames the run the corpus came from, which is the bot that produced the decisions and not the bot that learns from them. The comparison that decides whether the fitted M is better is a `bench` of the bot carrying it, `--against` the same baseline. """ from .baseline import head_commit from .corpus import Stale, check, check_basis, check_data, current_basis, resolve from .train import ( Label, TrainingError, dispersion, fit_scan, report, save, scan_run, )
# A saved corpus resolves to its data directory and brings its manifest; # anything else is itself and has none. A run in `runs/` still fits - it # just has nothing on disk saying what recorded it, which is what the note # is for. target, manifest = resolve(args.target) warnings: list[str] = [] if manifest is None: print( f"note: {target} has no manifest, so nothing records what produced " "it and nothing can check it against this checkout. " "`sds corpus save <run-dir> <name>` gives it one.", file=sys.stderr, ) else: try: review = check(manifest, requested=args.feature or None) except Stale as error: if not args.stale_ok: print(f"error: {error}", file=sys.stderr) return 2 print(f"warning: --stale-ok, fitting anyway.\n{error}", file=sys.stderr) review = None if review is not None: warnings = review.warnings for note in review.notes: print(f"note: {note}", file=sys.stderr) for warning in warnings: print(f"warning: {warning}", file=sys.stderr)
try: label = Label(kind=args.label, gamma=args.gamma, credit=args.credit, tier=args.tier) # Streamed rather than read in: a real corpus is tens of gigabytes of # JSON, and the fit needs only the running totals `scan_run` keeps. corpus, normals = scan_run( target, imitation=args.imitation, label=label, drop_displaced=getattr(args, "drop_displaced", False), ) if manifest is not None: try: check_data(manifest, sorted(corpus.phi_names)) check_basis(manifest, current_basis()) except Stale as error: if not args.stale_ok: print(f"error: {error}", file=sys.stderr) return 2 print(f"warning: --stale-ok, fitting anyway.\n{error}", file=sys.stderr) result = fit_scan( target, corpus, normals, imitation=args.imitation, ridge=args.ridge, features=args.feature or None, label=label, ) except TrainingError as error: print(f"error: {error}", file=sys.stderr) return 2
text = report(result) if args.against: try: text += "\n\n## the run this was fitted from\n\n" + _compare_run(target, args.against) except (Incomparable, FileNotFoundError, MatchError) as error: print(f"cannot compare: {error}", file=sys.stderr) return 1 print(text)
out = Path(args.out) save(result.to_document(commit=head_commit()), out) print(f"\nwrote {out}") if args.report: Path(args.report).write_text(text + "\n") print(f"wrote {args.report}") if result.suspect_signs: print( "\nSigns contradict the features' own descriptions. Explain that " "before spending a benchmark on this vector.", file=sys.stderr, ) # Caught here as well as at load time, because this is where somebody # decides whether to spend a benchmark. A fit is unlikely to come out # lopsided on its own - a graft is what does it - but the fit that gets # hand-edited afterwards is written by this command. shape = dispersion({name: result.weights[name] for name in result.features}) if shape is not None and shape.lopsided: print( f"\nwarning: this vector is lopsided. {shape.complaint()}", file=sys.stderr, ) # Again at the end. A warning printed before a hundred lines of report has # been scrolled past by the time anybody decides anything. for warning in warnings: print(f"\nwarning: {warning}", file=sys.stderr) return 0
def cmd_clean(args: argparse.Namespace) -> int: """Kill any match container left running by a harness that died.""" killed = kill_all_matches() if killed: print("killed:\n " + "\n ".join(killed)) else: print("no match containers running") return 0
def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser( prog="sds", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument("--scenario", default=str(DEFAULT_SCENARIO)) parser.add_argument("--out", default=str(REPO / "runs")) parser.add_argument("--seed", type=int, default=1) parser.add_argument("--max-rounds", type=int, default=40) parser.add_argument("--timeout-ms", type=int, default=10_000) parser.add_argument( "--stall-seconds", type=int, default=300, help="give up on a match after this many seconds with no round, phase or turn change", ) parser.add_argument( "--bv-destroyed-percent", type=int, default=70, help="a side is beaten at this %% of its BV destroyed; 0 fights to the last unit", ) parser.add_argument( "--pin", action="store_true", help="give each seat a physical core of its own; for runs whose timings " "will be quoted (forces one match at a time)", ) sub = parser.add_subparsers(dest="command", required=True)
one = sub.add_parser("one", help="play a single match and print the result") one.add_argument("--bot", default=DEFAULT_BOT) one.add_argument( "--sds-seat", default=None, help="faction the bot plays; omit for Princess on both sides" ) _explore_flags(one) # Only here, and never on `bench`. The container starts an X server and # encodes a frame per round, which is fine once and is wrong once a # benchmark is playing hundreds. one.add_argument( "--replay", action="store_true", help="write the match as an animated GIF next to the result", ) one.set_defaults(func=cmd_one)
play = sub.add_parser( "play", help="one match with you in a seat, at MegaMek's own client", description="Starts the server, seats the bot, and opens MegaMek's Swing " "client on your display for the other side. One container, everything " "over its own loopback. The result, the round report and the bot's " "decision log land in the same run directory `sds one` uses, so `sds " "explain` and `sds view` read a game you played exactly as they read a " "benchmarked one.", ) play.add_argument( "--scenario", default=None, help=f"the .mms to play; {DEFAULT_PLAY_SCENARIO.name} by default", ) play.add_argument( "--side", default=None, metavar="FACTION", help="the faction you play; the first in the .mms by default", ) play.add_argument("--bot", default=DEFAULT_BOT, help="the bot on the other side") play.add_argument( "--port", type=int, default=8850, help="server port inside the container; never published" ) play.add_argument( "--play-max-rounds", type=int, default=0, metavar="N", help="stop after N rounds; 0 (the default) plays until somebody wins", ) play.add_argument( "--check-display", action="store_true", help="only ask whether the container can reach your X server, and exit", ) play.set_defaults(func=cmd_play)
bench = sub.add_parser("bench", help="play many matches and aggregate") bench.add_argument("--bot", default=DEFAULT_BOT) bench.add_argument("--games", type=int, default=20) # Two by default. A match is two thinking bots and a server in one container, # and this machine is shared with other agents; the sibling repos have all # learned this the expensive way. Raise it deliberately, having looked. bench.add_argument("--jobs", type=int, default=int(os.environ.get("SDS_JOBS", "2"))) bench.add_argument( "--save-baseline", metavar="NAME", help="record this run as a baseline under baselines/NAME.json", ) bench.add_argument( "--against", metavar="NAME", help="compare this run against a recorded baseline and write comparison.md", ) bench.add_argument( "--self-play", action="store_true", help="seat the bot on both sides; for building a training corpus whose " "labels vary, not for measuring strength", ) bench.add_argument( "--imitate", action="store_true", help="record the opposing seat's moves as a training corpus, beside the " "decision logs; read back with `sds train --imitation`", ) bench.add_argument("--notes", default="", help="one line, stored with a baseline") bench.add_argument( "--corpus", metavar="NAME", help="what to call this run in the corpus store. A run that recorded " "decisions is saved whether or not this is given - `runs/` is " "gitignored and inside the worktree, so a corpus left there is one " "`git worktree remove` from gone, and that has now destroyed two of " "them. Without a name the run's own timestamp is used", ) _explore_flags(bench) bench.add_argument( "--suite", metavar="DIR", help="run games round-robin across every .mms in DIR instead of one scenario", ) bench.add_argument( "--no-validate", action="store_true", help="skip the scenario preflight", ) bench.add_argument( "--allow-uncrossable", action="store_true", help="run scenarios whose deployment zones the bots cannot reach across", ) bench.set_defaults(func=cmd_bench)
control = sub.add_parser("control", help="Princess vs Princess: the harness's self-test") control.add_argument("--games", type=int, default=20) control.add_argument("--jobs", type=int, default=int(os.environ.get("SDS_JOBS", "2"))) control.add_argument("--suite", metavar="DIR", help="control across a whole suite") control.add_argument( "--no-validate", action="store_true", help="skip the scenario preflight", ) control.add_argument( "--allow-uncrossable", action="store_true", help="run scenarios whose deployment zones the bots cannot reach across", ) control.set_defaults(func=cmd_control)
scenarios = sub.add_parser("scenarios", help="generate a benchmark suite") scenarios.add_argument("--out", dest="suite_out", default=str(REPO / "scenarios" / "suite")) scenarios.add_argument("--per-size", type=int, default=2) scenarios.add_argument( "--asymmetric", action="store_true", help="draw the two sides independently. Battle value still matches - the " "machines are random and crew skill is scaled onto the tier - but one side " "may outrange or outrun the other, which a mirrored suite never does and no " "fit has ever seen. Only fair played from both sides, so run it with a game " "count that is a multiple of twice the scenario count.", ) scenarios.add_argument("--suite-seed", type=int, default=1) scenarios.add_argument( "--sizes", default="", help="comma-separated sizes to write, from 1,2,4,8 (default: all four)", ) scenarios.add_argument( "--eras", default="", help="comma-separated eras (default: succession-wars,clan-invasion,jihad; " "civil-war and dark-age also exist)", ) scenarios.add_argument( "--bv-tiers", default="", help="comma-separated battle value tiers, e.g. 4000,6000,8000. Each " "scenario's force is drawn to land on its tier, and every crew's " "gunnery and piloting is rolled - battle value depends on both, so a " "tier is only meaningful if the crews are priced into it. Without this " "every crew is a regular 4/5 and the force is drawn by tonnage.", ) scenarios.add_argument( "--designs-per-era", type=int, default=0, help="draw from a fixed sample of this many designs per era instead of " "the whole pool. The full pool puts about a thousand machines into a " "few hundred scenarios, which says whether the bot generalises and " "cannot say which machines it flies badly - every design appears once. " "A few dozen turns the same matches into tens per machine. Different " "questions, so pick one per run.", ) scenarios.add_argument( "--unit-pool", default="mek", choices=("mek", "vehicle", "mixed"), help="which machines the forces are drawn from. `vehicle` draws canon " "tracked, wheeled and hover combat vehicles - the three motion types " "the movement model has a verified column for. A vehicle suite writes " "its scenarios under names ending `-vehicle`, because a win rate over " "tanks and one over Meks are not comparable numbers and nothing else " "in a run directory would say which was asked. `mixed` draws both into " "one force, which is the only pool that puts a Mek and a tank in front " "of the same gun in the same decision.", ) scenarios.set_defaults(func=cmd_scenarios)
validate = sub.add_parser( "validate", help="load every scenario in a suite and report the broken ones", description="Parses each scenario, builds its game, and checks both " "sides have units with somewhere legal to stand, and that a path the " "bots will walk joins the opposing deployment zones. One JVM for the " "whole suite: a broken map reference costs seconds here rather than an " "hour of benchmark reporting matches that produced no result.", ) validate.add_argument("--suite", metavar="DIR", help="validate every .mms in DIR") validate.add_argument( "--allow-uncrossable", action="store_true", help="report a map the bots cannot cross as a warning instead of a failure", ) validate.set_defaults(func=cmd_validate)
los_dump = sub.add_parser( "los-dump", help="dump MegaMek's LosEffects for one board, as a Rust test corpus", description="Loads a scenario for its board and two units, then walks a " "fixed set of hex pairs through MegaMek's own LosEffects. One JVM, no " "match. The Rust line of sight is checked against the result: a " "reimplementation that disagrees does not fail, it plays badly.", ) los_dump.add_argument( "scenarios", nargs="*", metavar="SCENARIO", help="scenarios to take boards from; one board per scenario", ) los_dump.add_argument( "--out", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "los.jsonl"), ) los_dump.add_argument("--cases", type=int, default=600, help="cases per board") los_dump.set_defaults(func=cmd_los_dump)
pathfind_dump = sub.add_parser( "pathfind-dump", help="dump MegaMek's legal movement end states, as a Rust test corpus", description="Walks MegaMek's own MovePath a step at a time over a fixed " "set of boards, starts, facings and MP budgets, keeping whatever " "isMoveLegal accepts. One JVM, no match. An offline fixture: regenerate " "it deliberately and commit the result, never during a game.", ) pathfind_dump.add_argument( "--out", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "pathfind.jsonl"), ) pathfind_dump.set_defaults(func=cmd_pathfind_dump)
routes_check = sub.add_parser( "routes-check", help="put the generator's own routes through MegaMek's isMoveLegal", description="Replays every step list the Rust route emitter produces " "for the pathfind corpus through MovePath.isMoveLegal and writes one " "verdict line per case. A reachable set can be right while the route " "into it is refused, which is what this catches. An offline fixture: " "regenerate it deliberately and commit the result.", ) routes_check.add_argument( "--routes", default=str(REPO / "target" / "pathfind-routes.jsonl"), help="the routes file the differential test writes under SDS_ROUTES_OUT", ) routes_check.add_argument( "--out", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "pathfind_routes.jsonl"), ) routes_check.set_defaults(func=cmd_routes_check)
arc_dump = sub.add_parser( "arc-dump", help="dump MegaMek's weapon arc tables, as a Rust test corpus", description="Reads ComputeArc and FacingArc out exhaustively: every " "relative offset within ten hexes plus the ring at seventeen, both " "column parities, all six facings, all nine ground arcs. No board and " "no match. An offline fixture, committed once.", ) arc_dump.add_argument( "--out", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "arcs.txt.gz"), ) arc_dump.set_defaults(func=cmd_arc_dump)
formation_dump = sub.add_parser( "formation-dump", help="dump MegaMek's formation definitions, as a Rust test corpus", description="Reads every FormationType back through its own accessors: " "the unit types it admits, its weight bracket, its ideal role and " "faction, and each composition constraint's description and its " "minimum at five force sizes. No board, no scenario and no unit cache. " "An offline fixture, committed - unlike the arc corpus it is small " "enough to read in a pull request.", ) formation_dump.add_argument( "--out", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "formations.txt"), ) formation_dump.set_defaults(func=cmd_formation_dump)
lances = sub.add_parser( "lances", help="draw random lances from MegaMek's force generator, as a corpus", description="Asks megamek.client.ratgenerator - the availability data " "MegaMek draws its own forces from - for lances a faction could field " "in a year, and writes each one in the shape sds_core::formation " "reads. One JVM, no server and no lobby. Four designs picked uniformly " "from the database are not a force anybody fields and cannot produce " "the compositions that matter; a draw from the generator can. The " "corpus is NOT committed - it carries MegaMek's unit data - so this " "command and its seed are what make a measurement repeatable.", ) lances.add_argument("--out", default=str(REPO / "runs" / "lances.jsonl")) lances.add_argument("--count", type=int, default=20000, help="how many lances") lances.add_argument( "--seed", type=int, default=20260904, help="the seed, recorded in the manifest" ) lances.add_argument("--size", type=int, default=4, help="machines per lance") lances.add_argument( "--scope", default="active", choices=["active", "any", "major"], help="which factions may be drawn: those active in the drawn year " "(default), every record, or the active non-minor ones", ) lances.add_argument( "--ratings", default="unspecified", choices=["unspecified", "drawn"], help="leave the equipment rating unset, as MegaMek defaults to, or " "draw one of the faction's rating levels per lance", ) lances.set_defaults(func=cmd_lances)
forces = sub.add_parser( "forces", help="classify the forces a suite fields, and how often one is anything", description="Loads every scenario in a suite, groups its entities by " "the player fielding them, and asks sds_core::formation what each " "force is and what tactic that opens on. Prints a row per force and " "the distribution under it. No matches are played. The number it " "exists for is the last line: how often a force a benchmark actually " "plays opens on something other than engage.", ) forces.add_argument( "--suite", default=str(REPO / "scenarios" / "suite"), help="a suite directory, or one .mms file", ) forces.add_argument("--out", default="", help="also keep the raw dump here") forces.set_defaults(func=cmd_forces)
cluster_dump = sub.add_parser( "cluster-dump", help="record how many packets share one hit-location roll, per weapon", description="Mounts every weapon in MegaMek's equipment tables with " "every round it can take, asks Weapon.getCorrectHandler for the " "handler that would resolve the shot, and reads the grouping off it. " "No board and no match. The summary is committed; the full dump is " "not.", ) cluster_dump.add_argument( "--out", default=str(REPO / "docs" / "CLUSTERS.txt"), ) cluster_dump.add_argument( "--detail", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "clusters.txt"), ) cluster_dump.set_defaults(func=cmd_cluster_dump)
cluster_table = sub.add_parser( "cluster-table", help="dump MegaMek's Cluster Hits Table, as a Rust test corpus", description="Pins the 2d6 and asks Compute.missilesHit for every rack " "and every total, decomposition included. No board and no match. " "Committed, because 40 lines is small enough that the test can always " "run.", ) cluster_table.add_argument( "--out", default=str(REPO / "crates" / "sds-core" / "tests" / "corpus" / "cluster_table.csv"), ) cluster_table.set_defaults(func=cmd_cluster_table)
conformance = sub.add_parser( "conformance", help="re-ask MegaMek the questions already asked, and diff the answers", description=( "One JVM, no server, no match. Every fact this repository has probed " "out of MegaMek and then built on - a class relationship, a sentinel, " "a refusal, a step function, a terrain set - re-asked and compared " "against bridge/conformance.txt. --bless re-records; a commit that " "moves a figure has to say why." ), ) conformance.add_argument( "--recording", default=str(REPO / "bridge" / "conformance.txt"), help="the committed answers to diff against", ) conformance.add_argument( "--bless", action="store_true", help="overwrite the recording with what MegaMek answers now", ) conformance.set_defaults(func=cmd_conformance)
club_probe = sub.add_parser( "club-probe", help="record what MegaMek says a mounted melee weapon does", description="Loads a scenario of hatchet-carriers at three tonnages and " "two machines carrying none, and prints what MegaMek answers for each: " "the club's damage, a kick and a punch beside it as the scale, and what " "getClubs returns when there is no club. No match and no server.", ) club_probe.add_argument( "--scenario", default=str(REPO / "scenarios" / "clubs" / "club-probe.mms"), ) club_probe.add_argument( "--out", default=str(REPO / "scenarios" / "clubs" / "observed.txt"), ) club_probe.set_defaults(func=cmd_club_probe)
self_damage = sub.add_parser( "self-damage-probe", help="record what MegaMek says an attack costs its own attacker", description="Loads a scenario of four attackers and four targets across " "the tonnage range and prints what MegaMek answers for each: a charge's " "self-damage per pair, a death-from-above's per attacker, both as a share " "of the attacker's remaining armour and structure, and every action in " "this build that has a self-damage answer at all. No match and no server.", ) self_damage.add_argument( "--scenario", default=str(REPO / "scenarios" / "self-damage" / "self-damage-probe.mms"), ) self_damage.add_argument( "--out", default=str(REPO / "scenarios" / "self-damage" / "observed.txt"), ) self_damage.set_defaults(func=cmd_self_damage_probe)
dfa = sub.add_parser( "dfa-probe", help="record what MegaMek says a death from above costs and refuses", description="Loads four jumping attackers over four targets and prints what " "MegaMek answers: the damage dealt, the damage taken landing, the damage taken " "falling on a miss, the roll each attacker needs onto each target, and the " "refusal each of the two movement attacks names for the same reversing path. " "No match and no server.", ) dfa.add_argument( "--scenario", default=str(REPO / "scenarios" / "dfa" / "dfa-probe.mms"), ) dfa.add_argument( "--out", default=str(REPO / "scenarios" / "dfa" / "observed.txt"), ) dfa.set_defaults(func=cmd_dfa_probe)
stand_probe = sub.add_parser( "stand-probe", help="record what MegaMek permits a prone, damaged Mek to do", description="Loads a scenario of prone Meks once per cell of a damage " "matrix, destroys that cell's locations through MegaMek's own " "destroyLocation, and prints whether MegaMek then accepts a GET_UP " "path and what piloting roll it asks for. The bot's own canStand is " "printed beside it. No match and no server.", ) stand_probe.add_argument( "--scenario", default=str(REPO / "scenarios" / "stand" / "stand-probe.mms"), ) stand_probe.add_argument( "--out", default=str(REPO / "scenarios" / "stand" / "observed.txt"), ) stand_probe.set_defaults(func=cmd_stand_probe)
ba_probe = sub.add_parser( "ba-probe", help="record what MegaMek permits a battle armor squad and its carrier", description="Loads a scenario of carriers and battle armor once per " "cell, puts the squad at a chosen distance from the carrier through " "MegaMek's own setPosition, and prints what MegaMek then permits: " "whether the carrier can take the squad, what a mount path costs, " "whether a mounted squad still gets a turn of its own, and what a " "dismount costs. No match and no server.", ) ba_probe.add_argument( "--scenario", default=str(REPO / "scenarios" / "ba" / "ba-probe.mms"), ) ba_probe.add_argument( "--out", default=str(REPO / "scenarios" / "ba" / "observed.txt"), ) ba_probe.set_defaults(func=cmd_battle_armor_probe) livefire = sub.add_parser( "vehicle-livefire", help="fire at a tank from every arc and count where the hits landed", description=( "Resolves hit locations against a real deployed tank with real dice, " "one arc at a time, and prints raw counts per location. The check on " "`SdsVehicleHits`, which reads the same tables with the dice pinned: " "this shares no instrument with it, so agreement means something." ), ) livefire.add_argument( "--scenario", default="scenarios/livefire/vehicle-front.mms", help="the probe scenario: one tank, one shooter per arc", ) livefire.add_argument( "--shots", type=int, default=200000, help="hit locations resolved per arc" ) livefire.add_argument( "--out", default="corpora/vehicle-livefire.txt", help="where to write the counts", ) livefire.set_defaults(func=cmd_vehicle_livefire)
motive = sub.add_parser( "motive-damage", help="dump the motive damage result table and the per-arc modifier", description=( "A vehicle's hit table flags four rolls for motive damage, but the " "flag is only the entry to it. This asks MegaMek what a flagged hit " "then does, which is what turns `p_mission_kill` for a vehicle from " "an upper bound into a probability." ), ) motive.add_argument( "--scenario", default="scenarios/livefire/vehicle-front.mms", help="a scenario with a real, crewed, deployed tank in it", ) motive.add_argument( "--out", default="docs/livefire/motive-damage.txt", help="where to write the table", ) motive.set_defaults(func=cmd_motive_damage)
minrange_probe = sub.add_parser( "minrange-probe", help="record what MegaMek charges a shot at every range", description="Loads a scenario of shooters and one target, walks the " "target down a column of the board with no terrain in it, and prints " "WeaponAttackAction.toHit and Compute.getRangeMods at every hex. The " "projection's own arithmetic is printed beside them, so the " "minimum-range term is settled by what MegaMek charges rather than by " "a reading of the rules. No match and no server.", ) minrange_probe.add_argument( "--scenario", default=str(REPO / "scenarios" / "minrange" / "minrange-probe.mms"), ) minrange_probe.add_argument( "--census", default=str(REPO / "scenarios" / "suite"), help="directory of scenarios to count minimum ranges over, or empty to skip", ) minrange_probe.add_argument( "--out", default=str(REPO / "scenarios" / "minrange" / "observed.txt"), ) minrange_probe.set_defaults(func=cmd_minrange_probe)
move_probe = sub.add_parser( "move-probe", help="record what terrain costs each movement mode, and what it bars", description="Loads a scenario for its board and one specimen per " "movement mode, paints a named stack of terrain into the hex ahead of " "each and prints Hex.movementCost, Entity.isLocationProhibited and " "MovePath.isMoveLegal together. The pathfinder's terrain column is one " "branch of MegaMek's - a biped Mek's - and it has no legality gate at " "all; this is where both are read off rather than recited. No match " "and no server.", ) move_probe.add_argument( "--scenario", default=str(REPO / "scenarios" / "move" / "move-probe.mms"), ) move_probe.add_argument( "--out", default=str(REPO / "scenarios" / "move" / "observed.txt"), ) move_probe.set_defaults(func=cmd_move_probe)
afford_probe = sub.add_parser( "afford-probe", help="record when a mover may spend more than its allowance", description="Walks every hex and every direction of the four pathfind " "boards, re-rates each specimen's walk allowance and asks " "MovePath.isMoveLegal about three step lists that all enter the same " "hex. A reference run at an allowance nothing can exhaust separates a " "refusal for terrain from a refusal for cost. No match and no server.", ) afford_probe.add_argument( "--out", default=str(REPO / "scenarios" / "move" / "afford.txt"), ) afford_probe.set_defaults(func=cmd_afford_probe)
step_cost = sub.add_parser( "step-cost", help="record what one step costs each mover, by terrain and level", description="Walks every step of the four pathfind boards with each " "specimen rated past what any of them can charge, and tabulates " "MovePath's own price against the destination's terrain column, its " "depth and the level the mover's feet change by. No match, no server.", ) step_cost.add_argument( "--out", default=str(REPO / "scenarios" / "move" / "stepcost.txt"), ) step_cost.set_defaults(func=cmd_step_cost_probe)
cover_probe = sub.add_parser( "cover-probe", help="record what MegaMek charges for intervening terrain", description="Loads a scenario for its board and two units, clears the " "board, paints a named stack of terrain into the line between them and " "prints LosEffects.losModifiers and WeaponAttackAction.toHit at every " "depth. The ceiling cover_quality normalises by is settled by what " "MegaMek returns rather than by a reading of the rules. No match and no " "server.", ) cover_probe.add_argument( "--scenario", default=str(REPO / "scenarios" / "cover" / "cover-probe.mms"), ) cover_probe.add_argument( "--out", default=str(REPO / "scenarios" / "cover" / "observed.txt"), ) cover_probe.set_defaults(func=cmd_cover_probe)
artillery_probe = sub.add_parser( "artillery-probe", help="record flight time, ranges and blast templates for artillery", description="Walks Compute.turnsTilHit for the bands a shell's flight " "time is constant over, every ArtilleryWeapon and ArtilleryCannonWeapon " "for its rack size and ranges, and AreaEffectHelper for the ring of " "damage each round throws. Offline: no board, no match and no server.", ) artillery_probe.add_argument( "--out", default=str(REPO / "scenarios" / "artillery" / "observed.txt"), ) artillery_probe.set_defaults(func=cmd_artillery_probe) infantry_cover = sub.add_parser( "infantry-cover", help="record what the ground under a platoon is worth", description=( "Paint each terrain type into a platoon's own hex and read " "ServerHelper.infantryInOpen back, walk the infantry damage class " "table with the dice pinned, and run the composition through the " "server's own damage path. The cover set and the class table are " "settled by what MegaMek answers, not by what a rulebook summary " "says." ), ) infantry_cover.add_argument( "--scenario", default=str(REPO / "scenarios" / "infantry" / "infantry-probe.mms"), ) infantry_cover.add_argument( "--out", default=str(REPO / "scenarios" / "infantry" / "observed.txt"), ) infantry_cover.set_defaults(func=cmd_infantry_cover)
train = sub.add_parser( "train", help="fit M from a run's decision logs", description="Reads every decision log in a run directory, joins each " "to its match result, and fits one weight per learnable feature by " "least squares on the difference between the chosen candidate and the " "ones it beat. A `local` feature - min-maxed across one decision's own " "candidates - is refused: a weight fitted against one has learned a " "board. Plays no matches.", ) train.add_argument( "target", help="a stored corpus name, a run directory, or a single .decisions.jsonl", ) train.add_argument( "--out", default=str(REPO / "weights" / "fitted.json"), help="where to write the fitted weights", ) train.add_argument( "--ridge", type=float, default=1e-3, help="ridge penalty; the feature columns are correlated by construction", ) train.add_argument( "--feature", action="append", metavar="NAME", help="fit only these features; naming a local one is an error", ) train.add_argument( "--drop-displaced", action="store_true", help="leave out decisions where the weights' first choice was taken by " "another unit before the force reconciled. On such a row `chosen` is " "what was left rather than what was preferred, so every difference it " "contributes states a preference about a comparison nobody made. It was " "19.6%% of rows on the corpus this was measured against", ) train.add_argument( "--label", default=TRAIN_LABEL, choices=sorted(TRAIN_LABELS), help="what a decision is scored by. The two `bv_*_differential`-shaped " "ones do not move when neither side trades, so declining to fight " "scores what fighting well scores; the one-sided ones count only what " "was taken off the enemy", ) train.add_argument( "--gamma", type=float, default=TRAIN_GAMMA, help="discount per round of delay for the one-sided labels. The default " "spans a typical decided match (median 11 rounds) without paying out " "over the 41-round undecided tail", ) train.add_argument( "--credit", default=TRAIN_CREDIT, choices=sorted(TRAIN_CREDITS), help="`flat` is the plain discounted sum; `ratio` scales each round's " "payoff by how outmatched we were, so a hit that changes the force " "ratio counts for more than one landed while already winning", ) train.add_argument( "--tier", default=TRAIN_TIER, choices=sorted(TRAIN_TIERS), help="whose outcome a decision is scored by. `team` is the whole side, " "which in an 8v8 is seven other machines' luck in every row; `unit` is " "what that machine dealt and took, and `force` is its lance. A label or " "a corpus that cannot carry the tier refuses rather than falling back " "to the team's number", ) train.add_argument( "--against", metavar="NAME", help="also print how the run the corpus came from scored against a baseline", ) train.add_argument( "--imitation", action="store_true", help="also fit the reconstructed enemy moves in *.imitation.jsonl. These " "clone whatever bot played that seat, faults included; the fit's report " "says how many of the rows they are", ) train.add_argument( "--stale-ok", action="store_true", help="fit a corpus whose manifest refuses the fit: a different epoch, a " "--feature it does not have, or data its manifest does not describe. " "Prints what is wrong and carries on; the resulting vector is about a " "different bot than the one it will be measured as", ) train.add_argument("--report", metavar="PATH", help="also write the printed report here") train.set_defaults(func=cmd_train)
catalogue = sub.add_parser( "catalogue", help="every feature, intent, label and goal, with its one-line meaning", description="Reads the list out of the code that defines it: features " "and intents from the bot binary, labels from the trainer, goals from " "plan/. Not a second source of truth - a wrong description is wrong at " "the definition.", ) catalogue.add_argument("--json", action="store_true", help="machine-readable") catalogue.set_defaults(func=cmd_catalogue)
baselines = sub.add_parser("baselines", help="list recorded baselines") baselines.set_defaults(func=cmd_baselines)
melee = sub.add_parser( "melee", help="physical attacks a run resolved, by side", description="Counts the physical attacks MegaMek actually resolved in a " "run's match logs, by side, beside the seat tallies that say whether the " "bot was answering at all. An intention is not an attack: nothing here " "reads a candidate score, an adjacency, or a declaration the host " "refused - only what the game log says happened.", ) melee.add_argument("run_dir", help="the run directory to count, e.g. runs/2026...-bench") melee.set_defaults(func=cmd_melee)
stats = sub.add_parser( "stats", help="sum a run's counters into one page", description="The bot counts a great deal and says none of it out loud: " "the per-phase tallies go into each match result and the cache and " "timing lines go to stderr, which over four hundred matches is " "hundreds of thousands of lines nobody reads. This sums them. The " "tallies come from the results and are as good as the run; the caches " "and timings are parsed out of the match logs, which are a secondary " "source, so the report says how many it could read.", ) stats.add_argument("run_dir", help="the run directory to summarise, e.g. runs/2026...-bench") stats.set_defaults(func=cmd_stats)
report = sub.add_parser( "report", help="the standard post-run report: outcome, faults, speed, effort", description="One document per run, in the order a run is judged in: what " "actually ran, the win rate with an interval, the faults (illegal orders, " "timeouts and defaults, per phase and per round), both sides' thinking time " "measured the same way, and the effort behind it. Written to report.md beside " "the run so it outlives the terminal.", ) spread = sub.add_parser( "spread", help="what each feature reported: range, pinning, and whether it varies", description="A broken feature does not crash - it answers cleanly and " "answers wrong, which is how a clamped denominator, a retired column " "still being measured and a feature measured by nobody all survived " "weeks of runs. This asks four questions per feature per phase: does it " "move, is it pinned to a bound, is it inside 0..1, and does it vary " "*within a decision* - the last being the one that decides whether a " "difference-framed fit can learn from it at all.", ) spread.add_argument("run_dir", help="the run directory to read") spread.add_argument("--out", metavar="PATH", help="also write the table here") spread.set_defaults(func=cmd_spread)
report.add_argument("run_dir", help="the run directory to report on") report.add_argument( "--no-write", action="store_true", help="print only; do not write report.md" ) report.set_defaults(func=cmd_report)
corpus = sub.add_parser( "corpus", help="save a run's decision logs as a durable, described training corpus", description="A corpus is hours of machine time and the only evidence " "behind a weight vector. `save` copies one out of `runs/` - which is " "gitignored, and inside a worktree - into a store beside the main " "checkout, and writes a manifest naming the commit, the epoch, the " "suite, the bot command and every feature in it. The data is ignored; " "the manifest is committed, so a corpus that is gone is still known to " "have existed.", ) corpus_sub = corpus.add_subparsers(dest="corpus_command", required=True) corpus_save = corpus_sub.add_parser("save", help="copy a run directory into the store") corpus_save.add_argument("run_dir", help="the run directory to copy, e.g. runs/2026...-bench") corpus_save.add_argument("name", help="what to call it in the store") corpus_save.add_argument("--notes", help="a sentence on why this corpus was recorded") corpus_save.add_argument( "--replace", action="store_true", help="overwrite a corpus of this name that already exists", ) corpus_save.set_defaults(func=cmd_corpus_save) corpus_list = corpus_sub.add_parser("list", help="one line per stored corpus") corpus_list.set_defaults(func=cmd_corpus_list)
correlate = sub.add_parser( "correlate", help="win rate by tier, era, crew skill, machine and equipment", description="Joins a run's results to the `.meta.json` written beside " "each scenario and reports the win rate within every group. Answers " "what a win rate cannot: which machines the bot flies badly, whether it " "gets more out of a good gunner than Princess does, whether it holds up " "at ten thousand points. Forces are mirrored, so a group won more than " "half the time is one the bot plays better than Princess - not one that " "is stronger. Groups under a dozen matches are summed into a single " "line rather than shown with an interval nobody should read.", ) correlate.add_argument("run", help="a run directory under runs/") correlate.add_argument("suite", help="the suite directory the run played") correlate.add_argument("--out", default="", help="also write the report here") correlate.set_defaults(func=cmd_correlate)
moves = sub.add_parser( "moves", help="what our units did each round, by movement state", description="A unit that did not change hex is not one thing. It chose " "to hold, or it was prone and stayed down, or it was shut down by heat, " "or it was immobile - a destroyed gyro, a dead or unconscious crew - and " "took no decision at all, or the host refused the path it asked for. " "Classifies every unit-round from the per-round entity snapshots, ours " "beside Princess's, and reads refusals off each seat's own movement " "tally. A run recorded before `immobile` reached the snapshot cannot be " "split fully, and the report says so instead of folding those machines " "into a hold they never chose.", ) moves.add_argument("run", help="a run directory under runs/") moves.add_argument("--out", default="", help="also write the report here") moves.set_defaults(func=cmd_moves)
paired = sub.add_parser( "paired", help="compare two finished runs as paired data", description="Two runs over the same suite see the same scenarios with " "the same seeds, which makes them paired rather than independent " "samples. Pairs both bots win, or both lose, say only that some " "scenarios are winnable - it is the discordant pairs that carry the " "comparison. Treating them as independent prices scenario difficulty as " "noise and needs several times the games to see the same effect.", ) paired.add_argument("left", help="a finished run directory") paired.add_argument("right", help="another finished run directory over the same suite") paired.set_defaults(func=cmd_paired)
progress = sub.add_parser( "progress", help="the standings of a run that has not finished", description="Renders the same table `bench` prints at the end, over " "whatever matches have finished so far. For answering 'is this vector " "in the fight' during a suite run that has hours left. The interval is " "the honest width for the games played and gets narrower as they " "arrive; a partial read is not a reason to stop a run early, and " "stopping one because the number looked good is how a benchmark " "becomes a story about when somebody stopped looking.", ) progress.add_argument("run", help="a run directory under runs/") progress.set_defaults(func=cmd_progress)
labels_p = sub.add_parser( "labels", help="how independent each label is from every other", description="Correlates every label against every other over a corpus, " "which is the measurement that says whether a new label is a new " "direction or an old one renamed. Reads result documents only and never " "opens a decision log. A correlation seen on one corpus is a property of " "that corpus until a second one agrees - one pair measured 0.50 on one " "96-match corpus and 0.95 on the next - so run it against two before " "retiring anything.", ) labels_p.add_argument("corpus", help="a stored corpus name, or a run directory") labels_p.set_defaults(func=cmd_labels)
tactics_p = sub.add_parser( "tactics", help="whether two tactics are two things", description="Re-scores every candidate menu in a recorded corpus under " "each tactic's weighting over the named feature basis, and reports the " "share of decisions where each pair picked the same candidate. Needs no " "matches: the decision log already carries every candidate and its phi. " "Agreeing often means nothing - some hexes are simply better - so the " "signal is a pair that never diverges. As with `sds labels`, a figure " "from one corpus is a property of that corpus until a second one " "agrees, so run it against two before retiring anything.", ) tactics_p.add_argument("corpus", help="a stored corpus name, or a run directory") tactics_p.add_argument( "--phase", default="movement", choices=sorted(TACTIC_PHASES), help="which menu to re-score. Movement by default: a firing menu " "carries no positional feature, so tactics that differ only in geometry " "cannot be told apart there.", ) tactics_p.add_argument( "--matches", type=int, default=DEFAULT_TACTIC_MATCHES, help="how many match logs to read, 0 for all. A corpus is tens of " "gigabytes of candidate menus uncompressed and an agreement fraction " "over a thousand decisions is already tighter than the bands " f"(default: {DEFAULT_TACTIC_MATCHES})", ) tactics_p.add_argument( "--subsets", type=int, default=0, metavar="N", help="instead of the matrix, split the corpus into N disjoint subsets, " "re-measure every pair on each, and report how precise a full-corpus " "figure is. The page sorts tactics into bands at 0.4 and 0.8 as though " "the figures were exact; this says by how much they are not. Costs one " "pass over the corpus however many subsets are used, because the " "subsets are disjoint", ) tactics_p.set_defaults(func=cmd_tactics)
counterfactual = sub.add_parser( "counterfactual", help="what one weight column changed, re-scored on decisions that already happened", description="The noise-free half of a single-column control. Re-scores " "every recorded menu with one column ablated and reports how many " "menus offered a real choice on it, how often the argmax moves, and - " "for the menus that did not move - how much bigger the column would " "have to be before one did. That last number is what separates a " "column that is inert from one that is a point short, and a null " "result without it cannot be read. Needs no matches. Not `sds " "control`, which measures harness bias and is a different question.", ) counterfactual.add_argument("corpus", help="a stored corpus name, or a run directory") counterfactual.add_argument("column", help="the feature to ablate; `sds catalogue` lists them") counterfactual.add_argument( "--phase", default="movement", choices=sorted(COUNTERFACTUAL_PHASES), help="which menu to re-score (default: movement)", ) counterfactual.add_argument( "--to", type=float, default=0.0, metavar="W", help="hold the column at this weight rather than at nought, matching " "`sds-bot --ablate <column>=<W>` (default: 0)", ) counterfactual.add_argument( "--matches", type=int, default=DEFAULT_COUNTERFACTUAL_MATCHES, help=f"how many match logs to read, 0 for all (default: {DEFAULT_COUNTERFACTUAL_MATCHES})", ) counterfactual.set_defaults(func=cmd_counterfactual)
rank_p = sub.add_parser( "labelrank", help="whether each label ranks the winner above the loser", description="Asks the question prior to the one `labels` asks: does a " "label point at winning at all. Each comparison is paired inside one " "match - both seats played the same scenario, map, forces and seed, and " "exactly one won - so the scenario is held fixed and what is left is the " "label. Every round is read, because several labels are zero at round 1 " "by construction and reading them at one round reports 50%% and looks " "like a defect. Ties are excluded from the accuracy and reported beside " "it. One corpus is a reading, not a constant: run it against two before " "acting on an ordering.", ) rank_p.add_argument("corpus", help="a stored corpus name, or a run directory") rank_p.add_argument( "--gamma", type=float, default=0.93, help="discount per round of delay (default 0.93)" ) rank_p.add_argument( "--tier", default="team", choices=("team", "unit"), help="whose series to read" ) rank_p.set_defaults(func=cmd_labelrank)
doc_p = sub.add_parser( "labeldoc", help="render docs/LABELS.md and its figures", description="Plots every label over the rounds of one hand-built match, " "both seats on one pair of axes. A label is a window rather than an " "instant - it looks from the decision's round to the end of the match, " "discounts what it finds and collapses it to one number - and none of " "that is visible in the name. Two lines rather than one because the " "question a label answers is which of these two sides was doing better, " "and a label that draws the same line twice cannot answer it. Exits " "non-zero when the files on disk are stale.", ) doc_p.add_argument( "--write", action="store_true", help="refresh the files rather than check them" ) doc_p.set_defaults(func=cmd_labeldoc)
firing = sub.add_parser( "firing", help="legal shots against declared ones, per weapon", description="Reads the firing diagnostic on every FIRING row of a " "decision log and reports, per weapon name, the unit-rounds it was " "alive for, the legal shots the rules priced for it, how many the " "winning candidate declared, and which ladder rungs the rest sat on.", ) firing.add_argument("target", help="a run directory or a single .decisions.jsonl") firing.set_defaults(func=cmd_firing)
explain = sub.add_parser( "explain", help="one decision's candidates, feature by feature", description="Reads a decision log and the weights sidecar the bot wrote " "beside it, and prints each candidate's score broken into its terms: " "feature, raw measured value, weight and product. The whole menu is " "shown, including the default the bot could have taken instead.", ) explain.add_argument( "target", help="a run directory or a single .decisions.jsonl", ) explain.add_argument( "--phase", default="FIRING", help="phase to explain; FIRING by default, MOVEMENT for the other one, " "or an empty string for both", ) explain.add_argument("--seq", type=int, help="one decision, by its sequence number") explain.add_argument("--round", type=int, help="only decisions in this round") explain.add_argument("--unit", type=int, help="only decisions for this unit") explain.add_argument( "--limit", type=int, default=5, help="how many decisions to print (default 5)" ) explain.add_argument( "--all", action="store_true", help="include decisions whose menu was only the default", ) explain.add_argument("--full", action="store_true", help="do not abbreviate candidate labels") explain.add_argument( "--weights", metavar="PATH", help="the weights this run was played with; the sidecar beside the log by default", ) explain.set_defaults(func=cmd_explain)
view = sub.add_parser( "view", help="render a match's decision log as HTML", description="Reads the decision log SdsClient writes next to a match " "result and produces one self-contained HTML file: each force's stance, " "what moved it, every proposal a unit offered and which was taken, and " "the per-phase counters that say whether the seat was answering at all.", ) view.add_argument( "target", help="a run directory, a match result .json, or a single .decisions.jsonl", ) view.add_argument( "--match", metavar="TAG", help="which match in a run directory; the newest by default", ) view.add_argument("--out-file", metavar="PATH", help="where to write; TAG.html by default") view.add_argument( "--watch", action="store_true", help="regenerate on an interval and make the page reload itself", ) view.add_argument("--interval", type=int, default=3, help="seconds between rewrites") view.set_defaults(func=cmd_view)
watch = sub.add_parser( "watch", help="serve a match's decision log live on a local port", description="Tails the decision log by byte offset and serves it to a " "page meant to sit beside the MegaMek window: a hex map of the board " "coloured by the rank of whatever is being ranked - the proposal's " "value, the damage it deals or takes, or any single feature - with the " "terrain drawn over it, plus the chosen candidate against its runner-up " "broken into terms. Works on a match in progress and on a finished run.", ) watch.add_argument( "target", help="a run directory, a match result .json, or a single .decisions.jsonl", ) watch.add_argument( "--match", metavar="TAG", help="which match in a run directory; the newest by default", ) watch.add_argument( "--port", type=int, default=WATCH_PORT, help=f"loopback port to listen on (default {WATCH_PORT})", ) watch.add_argument("--interval", type=int, default=1, help="seconds between page polls") watch.add_argument( "--no-browser", action="store_true", help="print the URL but do not open a browser tab", ) watch.add_argument( "--max-proposals", type=int, default=WATCH_PROPOSALS, help="proposals per unit sent to the page, best first, chosen always kept", ) watch.add_argument( "--keep", type=int, default=WATCH_KEEP, help="how many recent decisions keep their full detail in memory", ) watch.add_argument( "--mm-home", metavar="PATH", help="MegaMek install to read the board from; $MM_HOME by default", ) watch.set_defaults(func=cmd_watch)
clean = sub.add_parser( "clean", help="kill match containers left by a dead harness - NOT while a run is live", description="Kills every container named sds-*, including the ones a " "benchmark running right now is using. For cleaning up after a " "harness that was killed rather than interrupted; an interrupted " "one cleans up after itself.", ) clean.set_defaults(func=cmd_clean)
args = parser.parse_args(argv)
# Ctrl-C has to reach the containers, not just this process. `docker run` is # a client; killing it leaves the match playing, and a benchmark abandoned # halfway leaves as many orphans as it had jobs. def _stop(signum, _frame): killed = kill_stragglers() print(f"\ninterrupted; killed {killed} running matches", file=sys.stderr) sys.exit(130)
signal.signal(signal.SIGINT, _stop) signal.signal(signal.SIGTERM, _stop)
try: return args.func(args) except MatchError as error: print(f"error: {error}", file=sys.stderr) return 2
if __name__ == "__main__": sys.exit(main())