This repository has no description
Something went wrong. Try again.
13 kB · 311 lines
Python
at main
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312"""Tests for W3: maintenance templates, candidate authoring, diagnostic rejection, sync/async handling, and activation pinning."""from __future__ import annotations
import astimport jsonimport sqlite3import tempfileimport unittestfrom pathlib import Path
from klbr import behavior, candidate, diagnostic, maintenance, runtime, templatesfrom klbr.behavior import CASConflictError, PublishedRelease, ValidationErrorfrom klbr_runtime.context import ( Bridge, ExecutionScope, PURPOSE_INTERACTIVE, PURPOSE_MAINTENANCE, current,)from peer import ROOT, Peer, behavior_copy
class TemplatesTests(unittest.TestCase): """Verify behavior templates installation and programmatic access."""
def test_templates_installed_and_loadable(self): names = templates.list_templates() self.assertIn("standing-responsibility", names) self.assertIn("maintenance-attempt", names) self.assertIn("evaluation-review", names)
sr = templates.load_template("standing-responsibility") self.assertIn("Standing responsibility: continued competence", sr)
ma = templates.load_template("maintenance-attempt") self.assertIn("Fresh-context maintenance attempt", ma)
er = templates.load_template("evaluation-review") self.assertIn("Evaluation review: supported improvement", er)
def test_unknown_template_raises_key_error(self): with self.assertRaises(KeyError): templates.load_template("nonexistent-template")
def test_module_constants(self): self.assertEqual(templates.STANDING_RESPONSIBILITY, templates.load_template("standing-responsibility")) self.assertEqual(templates.MAINTENANCE_ATTEMPT, templates.load_template("maintenance-attempt")) self.assertEqual(templates.EVALUATION_REVIEW, templates.load_template("evaluation-review"))
class RuntimePurposeTests(unittest.TestCase): """Verify purpose definition in execution scope and runtime vocabulary."""
def test_purpose_defaults_to_interactive(self): self.assertEqual(runtime.purpose(), PURPOSE_INTERACTIVE)
def test_execution_scope_with_maintenance_purpose(self): class DummyBridge: pass
scope = ExecutionScope(id="test-op", bridge=DummyBridge(), purpose=PURPOSE_MAINTENANCE) token = current.set(scope) try: self.assertEqual(runtime.purpose(), PURPOSE_MAINTENANCE) self.assertEqual(scope.purpose, "maintenance") finally: current.reset(token)
self.assertEqual(runtime.purpose(), PURPOSE_INTERACTIVE)
class CandidateWorkspaceTests(unittest.TestCase): """Verify isolated candidate authoring protecting production source."""
def setUp(self): self.temp_dir = tempfile.TemporaryDirectory(prefix="klbr-test-cand-") self.tmp = Path(self.temp_dir.name)
def tearDown(self): self.temp_dir.cleanup()
def test_candidate_workspace_isolation(self): base_dir = behavior_copy(self.tmp / "prod_source") base_orig_text = (base_dir / "klbr_hooks/defaults.py").read_text()
with candidate.CandidateWorkspace(base_dir) as ws: self.assertEqual(ws.base_path, base_dir.resolve()) self.assertNotEqual(ws.workspace_path, base_dir.resolve())
# Edit file in candidate workspace new_code = base_orig_text + "\n# candidate test addition\n" ws.write("klbr_hooks/defaults.py", new_code)
# Candidate file is modified self.assertEqual(ws.read_text("klbr_hooks/defaults.py"), new_code)
# Live production source is completely UNTOUCHED self.assertEqual((base_dir / "klbr_hooks/defaults.py").read_text(), base_orig_text) ws.assert_base_unmodified()
# Diff captures the change diff = ws.diff() self.assertIn("+", diff) self.assertIn("candidate test addition", diff)
def test_prevent_in_place_production_mutation(self): base_dir = behavior_copy(self.tmp / "prod_source")
# Creating workspace pointing directly to base dir is forbidden with self.assertRaises(PermissionError): candidate.CandidateWorkspace(base_dir, workspace_path=base_dir)
with candidate.CandidateWorkspace(base_dir) as ws: # Writing using absolute path pointing to base_path is rejected target_in_base = base_dir / "klbr_hooks/defaults.py" with self.assertRaises(PermissionError): ws.write(str(target_in_base), "illegal overwrite")
def test_candidate_syntax_validation(self): base_dir = behavior_copy(self.tmp / "prod_source")
with candidate.CandidateWorkspace(base_dir) as ws: ws.write("klbr_hooks/defaults.py", "def broken_syntax(:\n pass\n") with self.assertRaises(ValidationError): ws.validate_syntax()
def test_candidate_validation_produces_validated_candidate(self): base_dir = behavior_copy(self.tmp / "prod_source")
with candidate.CandidateWorkspace(base_dir) as ws: ws.validate_syntax() cand = ws.validate(timeout=2.0) self.assertIsInstance(cand, behavior.ValidatedCandidate) self.assertEqual(cand.fingerprint, ws.fingerprint())
class DiagnosticRejectionTests(unittest.TestCase): """Verify rejection of invalid diagnoses and evidence requesting (G15)."""
def test_missing_instruction_version_triggers_needs_evidence(self): claim = maintenance.DiagnosticClaim( claim_id="audit-MR-08", category="forbidden_test_suite", description="Audit claims agent ran forbidden test suite", prescribed_rule="ban_all_pytest", ) # Evidence lacks instruction_version and policy_scope evidence = maintenance.MaintenanceEvidence( observation_id="obs-1", executed_commands=({"command": "pytest tests/test_behavior.py", "exit_code": 0},), instruction_version=None, policy_scope=None, )
assessment = maintenance.assess_diagnostic_claim(claim, evidence) self.assertEqual(assessment.verdict, "needs-evidence") self.assertIn("instruction_version", assessment.missing_evidence) self.assertIn("policy_scope", assessment.missing_evidence) self.assertEqual(assessment.target_layer, "none") self.assertEqual(len(assessment.adopted_rules), 0)
def test_targeted_test_claim_contradiction_rejects_global_ban(self): claim = maintenance.DiagnosticClaim( claim_id="audit-MR-08", category="forbidden_test_suite", description="Audit calls targeted test a forbidden broad test", prescribed_rule="ban_pytest_globally", ) # Evidence contains targeted test command evidence = maintenance.MaintenanceEvidence( observation_id="obs-2", executed_commands=({"command": "pytest tests/test_behavior.py::BehaviorTypestateTests", "exit_code": 0},), instruction_version="v2.1", policy_scope={"allowed_targeted": ["pytest tests/test_behavior.py"]}, )
assessment = maintenance.assess_diagnostic_claim(claim, evidence) # Must observe: records no-change and rejects adopting the unsupported rule self.assertEqual(assessment.verdict, "no-change") self.assertIn("permitted targeted invocation", assessment.reason) self.assertIn("ban_pytest_globally", assessment.rejected_rules) self.assertEqual(len(assessment.adopted_rules), 0)
def test_supported_defect_is_validated(self): claim = maintenance.DiagnosticClaim( claim_id="audit-MR-02", category="signature_mismatch", description="git.diff argument mismatch", target_command="git.diff", ) evidence = maintenance.MaintenanceEvidence( observation_id="obs-3", instruction_version="v2.1", policy_scope={"allowed": True}, error_trace="TypeError: diff() got an unexpected keyword argument 'staged'", )
assessment = maintenance.assess_diagnostic_claim(claim, evidence) self.assertEqual(assessment.verdict, "validated") self.assertEqual(assessment.target_layer, "source") self.assertIsNotNone(assessment.cause)
class SyncAsyncHandlingTests(unittest.TestCase): """Verify accidental await of sync functions is handled gracefully without duplicating mutations (G10, G14)."""
def test_sync_awaited_mutation_does_not_repeat(self): mutation_counter = 0
def sync_mutation_with_receipt() -> str: nonlocal mutation_counter mutation_counter += 1 return "receipt_effect_123"
# Simulating actor code that accidentally awaits synchronous function: # res = await sync_mutation_with_receipt() # In Python, the call executes first, then awaiting the return string raises TypeError. async def _run_mistaken_await(): # In Python, function call executes first, then await fails on the return value val = sync_mutation_with_receipt() await val
caught_error: Exception | None = None try: _run_mistaken_await().send(None) except Exception as err: caught_error = err
self.assertIsNotNone(caught_error) self.assertEqual(mutation_counter, 1)
# Diagnose the await failure diag = diagnostic.diagnose_await_failure( sync_mutation_with_receipt, caught_error, effect_verifier=lambda: mutation_counter > 0, )
self.assertTrue(diag.is_sync_awaited) self.assertTrue(diag.function_executed) self.assertTrue(diag.effect_verified) self.assertFalse(diag.can_replay) self.assertIn("Do not replay", diag.prescribed_action)
# Attempting safe recovery with effect verifier verifies effect and refuses replay recovery_result = diagnostic.safe_execute_mutation( sync_mutation_with_receipt, effect_verifier=lambda: mutation_counter > 0, ) self.assertIsNone(recovery_result) # Mutation effect occurred EXACTLY ONCE self.assertEqual(mutation_counter, 1)
class HostActivationPinningTests(unittest.TestCase): """Verify host activation pinning and recovery of last-good release (G18)."""
def setUp(self): self.temp_dir = tempfile.TemporaryDirectory(prefix="klbr-test-act-") self.tmp = Path(self.temp_dir.name) self.releases_dir = self.tmp / "releases" self.db = sqlite3.connect(":memory:") self.db.execute("PRAGMA foreign_keys=ON") self.db.executescript((ROOT / "klbr-runtime/migrations/001_runtime.sql").read_text()) self.db.execute("INSERT INTO sessions(id) VALUES('pin_session')") self.db.commit()
def tearDown(self): self.db.close() self.temp_dir.cleanup()
def test_pinning_and_tamper_rejection_preserves_last_good(self): # 1. Base release A activated at epoch 1 d_a = behavior.draft(ROOT / "behavior") rel_a = d_a.test().publish(self.releases_dir) active_a = behavior.activate("pin_session", self.db, expected_epoch=0, release=rel_a) self.assertEqual(active_a.epoch, 1)
# 2. Candidate release B activated at epoch 2 d_b = behavior.draft(ROOT / "behavior") d_b.write("klbr_hooks/defaults.py", """from klbr.hooks import AttentionDecision, AttentionRequest, DeliveryDecision, DeliveryRequest, HookContext
def attention(request: AttentionRequest, context: HookContext) -> AttentionDecision: return AttentionDecision.wake("operator")
def delivery(request: DeliveryRequest, context: HookContext) -> DeliveryDecision: return DeliveryDecision.allow()""") rel_b = d_b.test().publish(self.releases_dir) active_b = behavior.activate("pin_session", self.db, expected_epoch=1, release=rel_b) self.assertEqual(active_b.epoch, 2)
# 3. Create candidate C, but tamper with its content on disk before activation d_c = behavior.draft(ROOT / "behavior") rel_c = d_c.test().publish(self.releases_dir)
# Tamper with candidate file on disk tampered_file = rel_c.path / "klbr_hooks/defaults.py" tampered_file.write_text(tampered_file.read_text() + "\n# tampered payload\n")
# Activation must fail with ValidationError with self.assertRaises(ValidationError): rel_c.activate(session_id="pin_session", db=self.db, expected_epoch=2)
# Verify DB epoch remains at epoch 2 (Release B), last-good retained db_epoch = self.db.execute("SELECT hook_epoch FROM sessions WHERE id='pin_session'").fetchone()[0] self.assertEqual(db_epoch, 2)
db_rev = self.db.execute("SELECT hook_revision FROM sessions WHERE id='pin_session'").fetchone()[0] self.assertEqual(db_rev, rel_b.id)