diff --git a/crates/sds-core/examples/perf.rs b/crates/sds-core/examples/perf.rs index b109543..9973ee9 100644 --- a/crates/sds-core/examples/perf.rs +++ b/crates/sds-core/examples/perf.rs @@ -8,6 +8,8 @@ use std::time::Instant; use sds_core::ev::{DamagePmf, EvCache, ShotKey}; +use sds_core::los; +use sds_core::wire::{Board, BoardHex, Coord}; const HEXES: usize = 32 * 17; const ENEMIES: usize = 8; @@ -16,6 +18,59 @@ const CANDIDATES_PER_UNIT: usize = 20; const FEATURES: usize = 30; const STRATEGIES: usize = 8; const WEAPONS_PER_UNIT: usize = 6; +/// 8 shooters x 8 targets x 20 candidate hexes. +const LOS_ASKS_PER_ROUND: usize = 1280; + +/// A two-sheet board with terrain on a fifth of it, which is about what the +/// standard maps carry. +fn board_message() -> Board { + let mut hexes = Vec::new(); + for y in 0..17 { + for x in 0..32 { + let i = y * 32 + x; + if i % 5 == 0 { + hexes.push(BoardHex { + x, + y, + level: (i % 4) - 1, + terrain: [("woods".to_string(), 1), ("foliage_elev".to_string(), 2)] + .into_iter() + .collect(), + }); + } + } + } + Board { + width: 32, + height: 17, + hexes, + } +} + +fn los_board() -> los::LosBoard { + los::LosBoard::new(&board_message()) +} + +/// One shooter's worth: 8 targets from 20 candidate hexes. +fn los_queries(board: &los::LosBoard) -> Vec { + let mut queries = Vec::new(); + for candidate in 0..CANDIDATES_PER_UNIT { + for enemy in 0..ENEMIES { + queries.push(los::AttackInfo::ground( + board, + Coord::new((candidate as i32 * 3) % 32, (candidate as i32) % 17), + 0, + 1, + true, + Coord::new(31 - (enemy as i32 * 3), 16 - enemy as i32), + 0, + 1, + true, + )); + } + } + queries +} fn bench(name: &str, iterations: u32, mut body: impl FnMut()) { // One warm pass so we are not timing the first cache fill. @@ -118,6 +173,25 @@ fn main() { threat_map(&mut distance, &mut damage, &curve) }); + // Line of sight, which `docs/PERFORMANCE.md` expects to dominate once any + // positional feature exists: a round of 8v8 asks ~1280 times. + let board = los_board(); + let queries = los_queries(&board); + bench("160 LOS queries, no cache", 200, || { + for ai in &queries { + std::hint::black_box(los::calculate(&board, ai, los::Rules::default())); + } + }); + let board_wire = board_message(); + bench("1280 LOS queries, cached", 20, || { + let cache = los::LosCache::new(&board_wire, los::Rules::default()); + for _ in 0..LOS_ASKS_PER_ROUND / queries.len() { + for ai in &queries { + std::hint::black_box(cache.get(ai)); + } + } + }); + let m: Vec = (0..STRATEGIES * FEATURES) .map(|i| (i % 7) as f32 * 0.1) .collect(); diff --git a/docs/PERFORMANCE.md b/docs/PERFORMANCE.md index b164778..86409a2 100644 --- a/docs/PERFORMANCE.md +++ b/docs/PERFORMANCE.md @@ -10,6 +10,8 @@ needs sixteen cores is not a number. | operation | cost | |---|---| | threat map, 8 enemies x 544 hexes | 13-17 us | +| 160 LOS queries, no cache | 109-236 us (0.7-1.5 us each) | +| 1280 LOS queries, per-turn cache | 264-715 us (0.2-0.6 us each) | | stance blend `w = S^T M` | 0.1 us | | score 160 candidates x 30 features | 2.4-3.2 us | | fire allocation, 48 shots x 8 targets, greedy PMF convolution | 162-543 us | @@ -33,21 +35,28 @@ Dominated today by 16 observation parses (~4.7 ms). That is a statement about ho little thinking exists yet, not about JSON being slow: scoring 160 candidates takes 3 us because the scoring is one dot product. -Unmeasured, and the real uncertainty: the JVM side. `WeaponAttackAction.toHit` -runs 8 shooters x 6 weapons x 8 targets = **384 LOS-walking calls per firing -phase**. At an estimated 20-200 us each that is 8-77 ms/round, plausibly 10x our -own cost. Instrument it before trusting the range. +The JVM side is now measured rather than estimated. `sds los-dump` times +MegaMek's own `LosEffects.calculateLOS` over 2383 calls on four boards and gets +**95-107 us mean** across runs. That is a cold JVM with no warm-up pass, so +treat it as an upper bound - but it is the right order, and it lands at the top of the 20-200 +us range this file used to guess. `WeaponAttackAction.toHit` runs 8 shooters x 6 +weapons x 8 targets = **384 of those per firing phase**, so ~40 ms/round on the +host: still several times the bot's whole budget. ## What will dominate once the thinking lands -**Line of sight.** `cover_quality`, `arc_exposure`, `crossfire`, `break_los` and -a terrain-aware threat map all need it, and it is O(hexes along the line) per -pair: ~1280 queries a round at 8v8, at maybe 5-20 us each, so **6-25 ms/round**. -That dwarfs everything measured above. +**Line of sight**, and it turned out cheaper than feared. `cover_quality`, +`arc_exposure`, `crossfire`, `break_los` and a terrain-aware threat map all need +it, and it is O(hexes along the line) per pair: ~1280 queries a round at 8v8. +This file predicted 5-20 us each and 6-25 ms/round. Measured, one query is +**0.7-1.5 us** and a whole round is **0.3-0.7 ms** — better than a fortieth of +the guess even taking the high numbers, and about a tenth of what the host +spends answering the same question. -The optimisation that matters is therefore a **per-turn LOS cache shared across -the side**, not anything about the wire. MegaMek does the same internally — the -server passes a `losCache` into `filterEntities`. +The **per-turn LOS cache shared across the side** still earns its place: at 8v8 +it answers seven queries in eight, which is where 1.5 us falls to 0.6 us. +MegaMek does the same internally — the server passes a `losCache` into +`filterEntities`. ## What blows up @@ -80,7 +89,10 @@ must have heard from every unit before it decides. It degrades to sequential cleanly and must not be removed as a pointless optimisation. Where concurrency does pay, even on one CPU, is a compute-once many-waiters LOS -cache: eight units will ask overlapping questions. +cache: eight units will ask overlapping questions. `sds_core::los::LosCache` is +that cache - a `Mutex` around a memo, with a `Condvar` per entry so the second +asker waits rather than computing the same line again. Nothing iterates the map, +so no result depends on its order. Results must not depend on completion order. Reduce in fixed order, budget in work units rather than wall-clock, give each unit its own seeded RNG stream, and diff --git a/plan/README.md b/plan/README.md index fec147d..9341e7d 100644 --- a/plan/README.md +++ b/plan/README.md @@ -69,7 +69,7 @@ for a reason: it fits `M`, and `M` does not exist until `features` and | [protocol](protocol.md) | The observation carries what a bot needs to reason | open | - | | [hierarchy](hierarchy.md) | Units propose, forces decide | open | - | | [ev](ev.md) | Damage as a distribution, not a mean | open | protocol | -| [los](los.md) | Line of sight computed here, checked against MegaMek | blocked | protocol | +| [los](los.md) | Line of sight computed here, checked against MegaMek | open | - | | [features](features.md) | A normalised, named feature basis | open | protocol, los, ev | | [strategies](strategies.md) | Strategy as a blend of features, held with grit | blocked | features | | [beliefs](beliefs.md) | What the bot thinks is happening | blocked | features | diff --git a/plan/los.md b/plan/los.md index bfb44c2..834d69c 100644 --- a/plan/los.md +++ b/plan/los.md @@ -1,8 +1,8 @@ --- id: los title: Line of sight computed here, checked against MegaMek -status: blocked -dependsOn: [protocol] +status: open +dependsOn: [] exitCriterion: > A Rust LOS agrees with MegaMek's LosEffects on a corpus dumped from real matches, and every positional feature uses it. @@ -17,11 +17,37 @@ ourselves regardless of speed. It will also be the dominant cost of any real positional feature: ~1280 LOS queries a round at 8v8, versus microseconds for everything else measured. -- [ ] Rust LOS over the board we already fetch in bulk -- [ ] Partial cover (the eight `COVER_*` variants), foliage, water, buildings, - attacker and target elevation, prone, hull-down, the divided-line case -- [ ] **Differential corpus**: dump every `(attacker, target)` -> `LosEffects` - from real matches and test against it. Divergence here is silent - it - looks like bad play, not an error -- [ ] Per-turn cache shared across the side, compute-once many-waiters -- [ ] Measure MegaMek's `calculateLOS` for comparison rather than assuming +- [x] Rust LOS over the board we already fetch in bulk. `sds_core::los`, ported + from `LosEffects` line for line, including the parts that read like bugs - + the `intervening` split direction rounds radians as if they were facings, + and `X_CONST` is written out as Java's `Math.tan(Math.PI / 6.0)` because + `FRAC_PI_6` is not bit-identical to `PI / 6.0`, and the one bit between + them moves every hexagon +- [x] Partial cover (all ten `COVER_*` values, not eight), foliage, water, + elevation, prone, the divided-line case. Buildings block by their terrain; + what is missing is a building's *identity*, which the wire does not carry, + so a shot traced from inside one is wrong - see [protocol](protocol.md). + Hull-down is not a line of sight question at all: it does not appear + anywhere in `LosEffects`, and is applied by the to-hit code +- [x] **Differential corpus**: `sds los-dump` runs MegaMek's own + `calculateLOS` over a fixed set of hex pairs on four boards and writes + `crates/sds-core/tests/corpus/los.jsonl`. 2383 cases, every field + compared, zero disagreements +- [x] Per-turn cache shared across the side, compute-once many-waiters. + `LosCache`: a memo behind a `Mutex`, a `Condvar` per entry so the second + asker waits instead of recomputing, and nothing iterating it +- [x] Measure MegaMek's `calculateLOS` for comparison rather than assuming. + **95-107 us mean** over 2383 calls against **0.7-1.5 us** for ours - see + `docs/PERFORMANCE.md` + +Still open, and what each waits for: + +- [ ] Every positional feature uses it. The basis in [features](features.md) + names them - `cover_quality`, `arc_exposure`, `crossfire`, `break_los`, + `los_out`, `los_in` - and its positional group is the half still waiting + on this. Wiring them up is that epic's work, not this one's +- [ ] Terrain changes - fire, smoke, cleared woods - are not sent after the + board, so a long match drifts correctly-implemented and wrong. Waits on + [protocol](protocol.md) +- [ ] A corpus dumped from a live match rather than a swept board, which would + also cover a game with the TacOps options on diff --git a/plan/protocol.md b/plan/protocol.md index 934740d..6b6bd9e 100644 --- a/plan/protocol.md +++ b/plan/protocol.md @@ -27,6 +27,14 @@ Adding a field is deliberate, with a reason in the commit message. - [ ] **Scenario objectives** - MMS V2 carries per-player `victory:` conditions and composable triggers. Blocks [goals](goals.md) - [ ] **Minefields** - the bot is blind to them +- [ ] **Buildings as objects** - a building spans hexes and has an identity, and + the board message sends only per-hex terrain. Every terrain attribute + [los](los.md) reads is already on the wire, so building *terrain* blocks + correctly; what cannot be answered is `getBuildingAt` and + `Compute.isInBuilding`, so a shot traced from inside a building, and + infantry protected by one, are both wrong +- [ ] **Entities in an intervening hex** - a grounded DropShip is ten levels of + cover to anything shooting past it, and [los](los.md) cannot see it Fixed already: `WeaponType.getDamage()` returns `-2` for cluster weapons, so the bot valued every missile below nothing. There is now no plain `damage` field.