"""Tests for W3: maintenance templates, candidate authoring, diagnostic rejection, sync/async handling, and activation pinning.""" from __future__ import annotations import ast import json import sqlite3 import tempfile import unittest from pathlib import Path from klbr import behavior, candidate, diagnostic, maintenance, runtime, templates from klbr.behavior import CASConflictError, PublishedRelease, ValidationError from klbr_runtime.context import ( Bridge, ExecutionScope, PURPOSE_INTERACTIVE, PURPOSE_MAINTENANCE, current, ) from peer import ROOT, Peer, behavior_copy class TemplatesTests(unittest.TestCase): """Verify behavior templates installation and programmatic access.""" def test_templates_installed_and_loadable(self): names = templates.list_templates() self.assertIn("standing-responsibility", names) self.assertIn("maintenance-attempt", names) self.assertIn("evaluation-review", names) sr = templates.load_template("standing-responsibility") self.assertIn("Standing responsibility: continued competence", sr) ma = templates.load_template("maintenance-attempt") self.assertIn("Fresh-context maintenance attempt", ma) er = templates.load_template("evaluation-review") self.assertIn("Evaluation review: supported improvement", er) def test_unknown_template_raises_key_error(self): with self.assertRaises(KeyError): templates.load_template("nonexistent-template") def test_module_constants(self): self.assertEqual(templates.STANDING_RESPONSIBILITY, templates.load_template("standing-responsibility")) self.assertEqual(templates.MAINTENANCE_ATTEMPT, templates.load_template("maintenance-attempt")) self.assertEqual(templates.EVALUATION_REVIEW, templates.load_template("evaluation-review")) class RuntimePurposeTests(unittest.TestCase): """Verify purpose definition in execution scope and runtime vocabulary.""" def test_purpose_defaults_to_interactive(self): self.assertEqual(runtime.purpose(), PURPOSE_INTERACTIVE) def test_execution_scope_with_maintenance_purpose(self): class DummyBridge: pass scope = ExecutionScope(id="test-op", bridge=DummyBridge(), purpose=PURPOSE_MAINTENANCE) token = current.set(scope) try: self.assertEqual(runtime.purpose(), PURPOSE_MAINTENANCE) self.assertEqual(scope.purpose, "maintenance") finally: current.reset(token) self.assertEqual(runtime.purpose(), PURPOSE_INTERACTIVE) class CandidateWorkspaceTests(unittest.TestCase): """Verify isolated candidate authoring protecting production source.""" def setUp(self): self.temp_dir = tempfile.TemporaryDirectory(prefix="klbr-test-cand-") self.tmp = Path(self.temp_dir.name) def tearDown(self): self.temp_dir.cleanup() def test_candidate_workspace_isolation(self): base_dir = behavior_copy(self.tmp / "prod_source") base_orig_text = (base_dir / "klbr_hooks/defaults.py").read_text() with candidate.CandidateWorkspace(base_dir) as ws: self.assertEqual(ws.base_path, base_dir.resolve()) self.assertNotEqual(ws.workspace_path, base_dir.resolve()) # Edit file in candidate workspace new_code = base_orig_text + "\n# candidate test addition\n" ws.write("klbr_hooks/defaults.py", new_code) # Candidate file is modified self.assertEqual(ws.read_text("klbr_hooks/defaults.py"), new_code) # Live production source is completely UNTOUCHED self.assertEqual((base_dir / "klbr_hooks/defaults.py").read_text(), base_orig_text) ws.assert_base_unmodified() # Diff captures the change diff = ws.diff() self.assertIn("+", diff) self.assertIn("candidate test addition", diff) def test_prevent_in_place_production_mutation(self): base_dir = behavior_copy(self.tmp / "prod_source") # Creating workspace pointing directly to base dir is forbidden with self.assertRaises(PermissionError): candidate.CandidateWorkspace(base_dir, workspace_path=base_dir) with candidate.CandidateWorkspace(base_dir) as ws: # Writing using absolute path pointing to base_path is rejected target_in_base = base_dir / "klbr_hooks/defaults.py" with self.assertRaises(PermissionError): ws.write(str(target_in_base), "illegal overwrite") def test_candidate_syntax_validation(self): base_dir = behavior_copy(self.tmp / "prod_source") with candidate.CandidateWorkspace(base_dir) as ws: ws.write("klbr_hooks/defaults.py", "def broken_syntax(:\n pass\n") with self.assertRaises(ValidationError): ws.validate_syntax() def test_candidate_validation_produces_validated_candidate(self): base_dir = behavior_copy(self.tmp / "prod_source") with candidate.CandidateWorkspace(base_dir) as ws: ws.validate_syntax() cand = ws.validate(timeout=2.0) self.assertIsInstance(cand, behavior.ValidatedCandidate) self.assertEqual(cand.fingerprint, ws.fingerprint()) class DiagnosticRejectionTests(unittest.TestCase): """Verify rejection of invalid diagnoses and evidence requesting (G15).""" def test_missing_instruction_version_triggers_needs_evidence(self): claim = maintenance.DiagnosticClaim( claim_id="audit-MR-08", category="forbidden_test_suite", description="Audit claims agent ran forbidden test suite", prescribed_rule="ban_all_pytest", ) # Evidence lacks instruction_version and policy_scope evidence = maintenance.MaintenanceEvidence( observation_id="obs-1", executed_commands=({"command": "pytest tests/test_behavior.py", "exit_code": 0},), instruction_version=None, policy_scope=None, ) assessment = maintenance.assess_diagnostic_claim(claim, evidence) self.assertEqual(assessment.verdict, "needs-evidence") self.assertIn("instruction_version", assessment.missing_evidence) self.assertIn("policy_scope", assessment.missing_evidence) self.assertEqual(assessment.target_layer, "none") self.assertEqual(len(assessment.adopted_rules), 0) def test_targeted_test_claim_contradiction_rejects_global_ban(self): claim = maintenance.DiagnosticClaim( claim_id="audit-MR-08", category="forbidden_test_suite", description="Audit calls targeted test a forbidden broad test", prescribed_rule="ban_pytest_globally", ) # Evidence contains targeted test command evidence = maintenance.MaintenanceEvidence( observation_id="obs-2", executed_commands=({"command": "pytest tests/test_behavior.py::BehaviorTypestateTests", "exit_code": 0},), instruction_version="v2.1", policy_scope={"allowed_targeted": ["pytest tests/test_behavior.py"]}, ) assessment = maintenance.assess_diagnostic_claim(claim, evidence) # Must observe: records no-change and rejects adopting the unsupported rule self.assertEqual(assessment.verdict, "no-change") self.assertIn("permitted targeted invocation", assessment.reason) self.assertIn("ban_pytest_globally", assessment.rejected_rules) self.assertEqual(len(assessment.adopted_rules), 0) def test_supported_defect_is_validated(self): claim = maintenance.DiagnosticClaim( claim_id="audit-MR-02", category="signature_mismatch", description="git.diff argument mismatch", target_command="git.diff", ) evidence = maintenance.MaintenanceEvidence( observation_id="obs-3", instruction_version="v2.1", policy_scope={"allowed": True}, error_trace="TypeError: diff() got an unexpected keyword argument 'staged'", ) assessment = maintenance.assess_diagnostic_claim(claim, evidence) self.assertEqual(assessment.verdict, "validated") self.assertEqual(assessment.target_layer, "source") self.assertIsNotNone(assessment.cause) class SyncAsyncHandlingTests(unittest.TestCase): """Verify accidental await of sync functions is handled gracefully without duplicating mutations (G10, G14).""" def test_sync_awaited_mutation_does_not_repeat(self): mutation_counter = 0 def sync_mutation_with_receipt() -> str: nonlocal mutation_counter mutation_counter += 1 return "receipt_effect_123" # Simulating actor code that accidentally awaits synchronous function: # res = await sync_mutation_with_receipt() # In Python, the call executes first, then awaiting the return string raises TypeError. async def _run_mistaken_await(): # In Python, function call executes first, then await fails on the return value val = sync_mutation_with_receipt() await val caught_error: Exception | None = None try: _run_mistaken_await().send(None) except Exception as err: caught_error = err self.assertIsNotNone(caught_error) self.assertEqual(mutation_counter, 1) # Diagnose the await failure diag = diagnostic.diagnose_await_failure( sync_mutation_with_receipt, caught_error, effect_verifier=lambda: mutation_counter > 0, ) self.assertTrue(diag.is_sync_awaited) self.assertTrue(diag.function_executed) self.assertTrue(diag.effect_verified) self.assertFalse(diag.can_replay) self.assertIn("Do not replay", diag.prescribed_action) # Attempting safe recovery with effect verifier verifies effect and refuses replay recovery_result = diagnostic.safe_execute_mutation( sync_mutation_with_receipt, effect_verifier=lambda: mutation_counter > 0, ) self.assertIsNone(recovery_result) # Mutation effect occurred EXACTLY ONCE self.assertEqual(mutation_counter, 1) class HostActivationPinningTests(unittest.TestCase): """Verify host activation pinning and recovery of last-good release (G18).""" def setUp(self): self.temp_dir = tempfile.TemporaryDirectory(prefix="klbr-test-act-") self.tmp = Path(self.temp_dir.name) self.releases_dir = self.tmp / "releases" self.db = sqlite3.connect(":memory:") self.db.execute("PRAGMA foreign_keys=ON") self.db.executescript((ROOT / "klbr-runtime/migrations/001_runtime.sql").read_text()) self.db.execute("INSERT INTO sessions(id) VALUES('pin_session')") self.db.commit() def tearDown(self): self.db.close() self.temp_dir.cleanup() def test_pinning_and_tamper_rejection_preserves_last_good(self): # 1. Base release A activated at epoch 1 d_a = behavior.draft(ROOT / "behavior") rel_a = d_a.test().publish(self.releases_dir) active_a = behavior.activate("pin_session", self.db, expected_epoch=0, release=rel_a) self.assertEqual(active_a.epoch, 1) # 2. Candidate release B activated at epoch 2 d_b = behavior.draft(ROOT / "behavior") d_b.write("klbr_hooks/defaults.py", """ from klbr.hooks import AttentionDecision, AttentionRequest, DeliveryDecision, DeliveryRequest, HookContext def attention(request: AttentionRequest, context: HookContext) -> AttentionDecision: return AttentionDecision.wake("operator") def delivery(request: DeliveryRequest, context: HookContext) -> DeliveryDecision: return DeliveryDecision.allow() """) rel_b = d_b.test().publish(self.releases_dir) active_b = behavior.activate("pin_session", self.db, expected_epoch=1, release=rel_b) self.assertEqual(active_b.epoch, 2) # 3. Create candidate C, but tamper with its content on disk before activation d_c = behavior.draft(ROOT / "behavior") rel_c = d_c.test().publish(self.releases_dir) # Tamper with candidate file on disk tampered_file = rel_c.path / "klbr_hooks/defaults.py" tampered_file.write_text(tampered_file.read_text() + "\n# tampered payload\n") # Activation must fail with ValidationError with self.assertRaises(ValidationError): rel_c.activate(session_id="pin_session", db=self.db, expected_epoch=2) # Verify DB epoch remains at epoch 2 (Release B), last-good retained db_epoch = self.db.execute("SELECT hook_epoch FROM sessions WHERE id='pin_session'").fetchone()[0] self.assertEqual(db_epoch, 2) db_rev = self.db.execute("SELECT hook_revision FROM sessions WHERE id='pin_session'").fetchone()[0] self.assertEqual(db_rev, rel_b.id)