"""What `sds explain` claims about a decision, tested without the table. The formatting is not the part that can be wrong without anybody noticing. The parts that can are: whether the terms it prints add up to the score the bot actually ranked with, whether it refuses to decompose a log with weights that do not belong to it, and whether "held fire" and "had nothing to shoot at" stay separate - conflating those two is what turned a positioning problem into a scoring problem for a day. """ import json import tempfile import unittest from pathlib import Path from sds.explain import ( ExplainError, decision_logs, read, render, tally, terms, weights_for, ) WEIGHTS = {"expected_damage": 4.0, "overkill": -3.0, "weapon_concentration": 1.5} def candidate(label: str, damage: float, overkill: float, concentration: float) -> dict: phi = { "expected_damage": damage, "overkill": overkill, "weapon_concentration": concentration, } value = sum(phi[name] * WEIGHTS[name] for name in phi) return {"label": label, "phi": phi, "value": value} def firing(seq: int, candidates: list[dict], chosen: int) -> dict: return { "seat": "sds North", "seq": seq, "round": 2, "phase": "FIRING", "unit": 0, "candidates": candidates, "chosen": chosen, } def written(rows: list[dict], weights: dict | None = WEIGHTS) -> tuple[Path, Path]: """A log and its sidecar in a temporary directory, as a run would leave them.""" directory = Path(tempfile.mkdtemp()) log = directory / "tag-sds_North.decisions.jsonl" log.write_text("".join(json.dumps(row) + "\n" for row in rows)) if weights is not None: (directory / "tag-sds_North.weights.json").write_text(json.dumps({"weights": weights})) return directory, log class TermsAddUp(unittest.TestCase): """The property the whole command rests on.""" def test_the_terms_sum_to_the_value_the_bot_recorded(self): one = candidate("1 at Talon", 0.25, 0.05, 1.0) rebuilt = sum(term.product for term in terms(one, WEIGHTS)) self.assertAlmostEqual(rebuilt, one["value"], places=6) def test_a_feature_with_no_weight_contributes_nothing(self): one = candidate("1 at Talon", 0.25, 0.05, 1.0) one["phi"]["some_unweighted_feature"] = 9.0 rebuilt = sum(term.product for term in terms(one, WEIGHTS)) self.assertAlmostEqual(rebuilt, one["value"], places=6) def test_terms_are_ordered_by_how_much_they_moved_the_score(self): one = candidate("1 at Talon", 0.25, 0.9, 1.0) products = [abs(term.product) for term in terms(one, WEIGHTS)] self.assertEqual(products, sorted(products, reverse=True)) class WrongWeightsAreRefused(unittest.TestCase): def test_a_missing_sidecar_says_so_rather_than_guessing(self): _, log = written([firing(1, [candidate("hold fire", 0, 0, 0)], 0)], weights=None) with self.assertRaises(ExplainError): weights_for(log) def test_the_sidecar_is_found_beside_the_log(self): _, log = written([firing(1, [candidate("hold fire", 0, 0, 0)], 0)]) self.assertEqual(weights_for(log), WEIGHTS) def test_weights_that_do_not_rebuild_the_score_are_flagged(self): row = firing( 1, [candidate("hold fire", 0, 0, 0), candidate("1 at Talon", 0.5, 0.1, 1.0)], 1 ) wrong = dict(WEIGHTS, expected_damage=40.0) self.assertIn("WARNING", render(row, wrong)) self.assertNotIn("WARNING", render(row, WEIGHTS)) class TheTallySeparatesHoldingFireFromHavingNoShot(unittest.TestCase): """A menu of one is a unit with no legal shot, not a decision to hold fire.""" def test_a_menu_of_one_is_not_counted_as_taking_the_default(self): rows = [ firing(1, [candidate("hold fire", 0, 0, 0)], 0), firing(3, [candidate("hold fire", 0, 0, 0), candidate("1 at Talon", 0.5, 0.1, 1.0)], 1), ] line = tally(rows) self.assertIn("2 scored decisions", line) self.assertIn("1 had no candidate but the default", line) self.assertIn("0 took the default", line) def test_the_default_being_taken_over_a_real_menu_is_counted(self): rows = [ firing(3, [candidate("hold fire", 0, 0, 0), candidate("1 at Talon", -1.0, 0.9, 0.1)], 0) ] self.assertIn("1 took the default", tally(rows)) class ReadingALog(unittest.TestCase): def test_lines_without_a_menu_are_skipped(self): directory = Path(tempfile.mkdtemp()) log = directory / "tag-sds_North.decisions.jsonl" log.write_text( json.dumps({"seq": 1, "phase": "DEPLOYMENT"}) + "\n" + json.dumps(firing(2, [candidate("hold fire", 0, 0, 0)], 0)) + "\n" ) self.assertEqual([row["seq"] for row in read(log)], [2]) def test_a_phase_filter_is_applied(self): _, log = written([firing(2, [candidate("hold fire", 0, 0, 0)], 0)]) self.assertEqual(read(log, "MOVEMENT"), []) self.assertEqual(len(read(log, "FIRING")), 1) def test_a_directory_finds_its_logs_and_an_empty_one_says_so(self): directory, log = written([firing(2, [candidate("hold fire", 0, 0, 0)], 0)]) self.assertEqual(decision_logs(directory), [log]) with self.assertRaises(ExplainError): decision_logs(Path(tempfile.mkdtemp())) class TheTableNamesWhatWasTaken(unittest.TestCase): def test_the_chosen_column_is_marked_and_every_candidate_is_shown(self): row = firing( 9, [ candidate("hold fire", 0, 0, 0), candidate("1 at Talon", 0.25, 0.05, 1.0), candidate("2 at Talon", 0.50, 0.04, 0.8), ], 2, ) table = render(row, WEIGHTS, full=True) self.assertIn("took [2] '2 at Talon'", table) self.assertIn("hold fire", table) self.assertIn("<= took", table) # A feature every candidate scores at zero is not a row worth printing. row["candidates"][0]["phi"]["ammo_spent"] = 0.0 self.assertNotIn("ammo_spent", render(row, dict(WEIGHTS, ammo_spent=-0.5), full=True)) if __name__ == "__main__": unittest.main()