From 4b417a09ba0d411b8a2b78f57f6cd5bffc0c0a48 Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Sat, 27 Jun 2026 18:32:20 +0300 Subject: [PATCH] add continuous-loop memory benchmark --- .beads/interactions.jsonl | 1 + .beads/issues.jsonl | 2 +- docs/benchmark-issues-report.md | 1 + docs/memory-benches.md | 8 +- docs/memory-implementation-status.md | 4 +- klbr-bench/src/main.rs | 667 ++++++++++++++++++++++++++- klbr-core/src/pipeline.rs | 21 +- 7 files changed, 691 insertions(+), 13 deletions(-) diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index afd6196..53fcc55 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -30,3 +30,4 @@ {"id":"int-12aecbfc","kind":"field_change","created_at":"2026-06-27T15:08:17.198468383Z","actor":"dawn","issue_id":"klbr-b4z","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Regenerated router linear artifact out-router-linear-iter-15 with tool-required metrics, updated benchmark helper default to the fresh model, and filed klbr-mpy for residual tool→memory quality work."}} {"id":"int-61df751e","kind":"field_change","created_at":"2026-06-27T15:12:00.165278848Z","actor":"dawn","issue_id":"klbr-av1","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Graph-only packets are now omitted unless exact/fts/dense corroboration is present; omission traces use graph_only_uncorroborated and regression tests cover omitted and corroborated graph packets."}} {"id":"int-d9f9e698","kind":"field_change","created_at":"2026-06-27T15:17:19.888479202Z","actor":"dawn","issue_id":"klbr-mpy","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Added tool-required training weighting for the linear router and regenerated out-router-linear-iter-16. Test tool→memory improved 0.6913→0.0940, tool-required recall 0.3087→0.9060, and memory false-abstain remained 0.0000 on test/holdout."}} +{"id":"int-0d70493b","kind":"field_change","created_at":"2026-06-27T15:32:03.759629206Z","actor":"dawn","issue_id":"klbr-bqj","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented continuous-loop deterministic benchmark coverage with docs and regression tests"}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 6e71525..d3e3b87 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -31,5 +31,5 @@ {"_type":"issue","id":"klbr-wmz.4","title":"Materialize profile and procedural lanes as first-class notes","description":"The schema and MemoryGarden know about profile and procedural lanes, but the production pipeline observe_session path currently writes raw turns plus episodic notes/memories. Stable preferences, standing instructions, and workflows still depend mostly on tags or model-authored remember calls instead of a first-class markdown-note flow with source refs and update policy.","design":"Build on MemoryGarden and upsert_markdown_note rather than adding a new store. Treat legacy memory tags as routing hints, not the durable source of truth for profile/procedural knowledge.","acceptance_criteria":"There is an explicit writer/reflection path for profile_note and procedural_note artifacts; new notes include source refs, frontmatter, stable paths under profile/ or procedural/, and canonical refs/chunks; updates use supersession or versioning instead of silent overwrite; user-confirmation or policy gates are documented for stable profile changes; tests cover creating and updating one profile note and one procedural note from source turns.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:36Z","created_by":"dawn","updated_at":"2026-06-26T23:42:13Z","started_at":"2026-06-26T23:37:20Z","closed_at":"2026-06-26T23:42:13Z","close_reason":"Added write_memory_note reflection tool for source-grounded profile/procedural markdown notes, ref supersession updates, source/policy frontmatter, stable lane paths, and tests for profile/procedural create/update paths.","labels":["architecture","markdown","memory","profile"],"dependencies":[{"issue_id":"klbr-wmz.4","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:36Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-mpy","title":"Improve tool-required router separability","description":"Fresh router artifact benchmarks/models/router/linear/out-router-linear-iter-15 reports test tool-required false-memory rate 0.6913 and tool-required recall 0.3087 after adding the metric/calibration surface. Improve training data, features, or model shape so tool-required queries stop looking like memory queries without regressing memory false-abstain.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T15:08:07Z","created_by":"dawn","updated_at":"2026-06-27T15:17:20Z","started_at":"2026-06-27T15:12:36Z","closed_at":"2026-06-27T15:17:20Z","close_reason":"Added tool-required training weighting for the linear router and regenerated out-router-linear-iter-16. Test tool→memory improved 0.6913→0.0940, tool-required recall 0.3087→0.9060, and memory false-abstain remained 0.0000 on test/holdout.","dependencies":[{"issue_id":"klbr-mpy","depends_on_id":"klbr-b4z","type":"blocks","created_at":"2026-06-27T18:08:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-b4z","title":"Regenerate router calibration artifacts","description":"After klbr-9yo, rerun the router bench with the calibrated tool-required metrics and commit fresh benchmarks/models/router model/report outputs so future sweeps track tool→memory and tool-required recall from generated artifacts.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T14:56:42Z","created_by":"dawn","updated_at":"2026-06-27T15:08:17Z","started_at":"2026-06-27T15:01:12Z","closed_at":"2026-06-27T15:08:17Z","close_reason":"Regenerated router linear artifact out-router-linear-iter-15 with tool-required metrics, updated benchmark helper default to the fresh model, and filed klbr-mpy for residual tool→memory quality work.","dependencies":[{"issue_id":"klbr-b4z","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T17:56:51Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-bqj","title":"add continuous-loop memory benchmark coverage","description":"benchmark suite mostly tests static query retrieval. add a loop-style harness that simulates multi-session observe/compact/reflect cycles and then evaluates whether reflection notes, markdown sync, and passive recall stay aligned over a longer horizon.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:48Z","created_by":"dawn","updated_at":"2026-06-27T13:15:48Z","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-bqj","title":"add continuous-loop memory benchmark coverage","description":"benchmark suite mostly tests static query retrieval. add a loop-style harness that simulates multi-session observe/compact/reflect cycles and then evaluates whether reflection notes, markdown sync, and passive recall stay aligned over a longer horizon.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:48Z","created_by":"dawn","updated_at":"2026-06-27T15:32:04Z","started_at":"2026-06-27T15:17:51Z","closed_at":"2026-06-27T15:32:04Z","close_reason":"Implemented continuous-loop deterministic benchmark coverage with docs and regression tests","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-6an","title":"unify memory edge tables behind canonical refs","description":"memory.rs still has memory_edges and ref-level edges as separate provenance graphs. design and migrate toward one transactional canonical-ref links graph so packet planning, provenance, supersession, and markdown-note evidence traverse the same substrate.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:44Z","created_by":"dawn","updated_at":"2026-06-27T13:15:44Z","dependency_count":0,"dependent_count":1,"comment_count":0} diff --git a/docs/benchmark-issues-report.md b/docs/benchmark-issues-report.md index 0d099eb..bb18e92 100644 --- a/docs/benchmark-issues-report.md +++ b/docs/benchmark-issues-report.md @@ -67,6 +67,7 @@ the current numbers show a clear mismatch between coarse retrieval success and a ### C. loop-free benchmarking - **symptom**: benches evaluate static query retrieval rather than the continuous agent loop. - **cause**: `klbr-bench` operates on mock inputs rather than simulating multi-session compact/reflect cycles, meaning we cannot benchmark whether reflection helper tools like `write_memory_note` silently degrade recall over a long horizon. +- **status/fix**: `klbr-bench continuous-loop ` now runs a deterministic no-model loop smoke. it observes multiple sessions through `MemoryPipeline`, writes reflection-style markdown notes through the real note api, verifies markdown/fts sync and source edges, checks profile/procedural/episodic/exact-ref recall, and asserts that a negated passive-recall query does not leak into assembled context. --- diff --git a/docs/memory-benches.md b/docs/memory-benches.md index 7a47f8f..d3bd12c 100644 --- a/docs/memory-benches.md +++ b/docs/memory-benches.md @@ -25,6 +25,7 @@ current fast preflight: ```bash rtk cargo test -p klbr-core +rtk cargo run -p klbr-bench -- continuous-loop /tmp/klbr-continuous-loop ``` that suite is the deterministic memory-architecture guardrail: it covers ref @@ -32,8 +33,11 @@ aliasing and reflink expansion, exact ref packet expansion, legacy backfill visibility, tombstone/suppression leakage, supersession, archived provenance, duplicate write avoidance, live-context lane preference, packet trust formatting, dense markdown hits, and packet-level right-session/wrong-chunk -fixtures. it should run before any memory architecture change, with model-backed -longmemeval smoke runs reserved for evidence quality and reader behavior. +fixtures. the `continuous-loop` command adds a no-model loop smoke over +multi-session observe, synthetic reflection notes, markdown-note fts sync, +passive recall suppression, and exact-ref context assembly. both should run +before memory architecture changes, with model-backed longmemeval smoke runs +reserved for evidence quality and reader behavior. longmemeval is a good primary benchmark because it directly targets chat-assistant long-term memory and covers information extraction, multi-session reasoning, temporal reasoning, knowledge updates, and abstention. the official repo exposes the cleaned datasets, evidence labels, qa evaluation scripts, retrieval metrics, and standard `longmemeval_s_cleaned`, `longmemeval_m_cleaned`, and oracle files, which makes it good for comparable results. ([arXiv][1]) diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index 518a7ed..746ac17 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -60,13 +60,15 @@ updated: 2026-06-27 ```bash rtk cargo check rtk cargo test -p klbr-core +rtk cargo run -p klbr-bench -- continuous-loop /tmp/klbr-continuous-loop ``` current result: ```text cargo check -p klbr-bench: 0 errors, 1 pre-existing warning -klbr-core tests: 98 passed, 1 ignored +klbr-core tests: 105 passed, 1 ignored +continuous-loop smoke: passed ``` retrieval benchmark smoke: diff --git a/klbr-bench/src/main.rs b/klbr-bench/src/main.rs index f1728ab..d0d57e5 100644 --- a/klbr-bench/src/main.rs +++ b/klbr-bench/src/main.rs @@ -11,7 +11,7 @@ use std::{ use anyhow::{bail, Context, Result}; use klbr_core::{ context::{format_recalled_memories, RecalledMemory}, - memory::MemoryStore, + memory::{MarkdownNoteInput, MemoryLane, MemoryStore, MemoryStoreStats}, models::{LlmClient, Message, ModelsConfig}, mvp::{ DatasetSplit, EvalObjective, EvalQuery, EvalRouteLabel, ExpectedMemoryAction, @@ -21,6 +21,7 @@ use klbr_core::{ RetrievalExperimentConfig, RetrievalMetrics, RetrievalTrace, VectorSearchMode, WindowAttemptLog, WindowDecision, }, + pipeline::{BenchQuery, BenchSession, BenchTurn, ContextBudget, MemoryPipeline, WriteTrace}, retrieval::{self, RetrievalConfig}, router::{ LinearDecisionParams as RuntimeLinearDecisionParams, @@ -138,6 +139,59 @@ struct SweepReport { results: Vec, } +#[derive(Debug, Clone, Serialize)] +struct ContinuousLoopReport { + benchmark_id: String, + passed: bool, + sessions_observed: usize, + source_ref_count: usize, + reflection_notes: Vec, + stats: MemoryStoreStats, + checks: Vec, + queries: Vec, +} + +#[derive(Debug, Clone, Serialize)] +struct ContinuousLoopNoteTrace { + note_ref: String, + lane: MemoryLane, + path: String, + source_count: usize, + chunk_refs: Vec, +} + +#[derive(Debug, Clone, Serialize)] +struct ContinuousLoopCheck { + name: String, + passed: bool, + details: String, +} + +#[derive(Debug, Clone, Serialize)] +struct ContinuousLoopQueryTrace { + query_id: String, + text: String, + route_reason: String, + candidates: Vec, + packets: Vec, + used_refs: Vec, + omission_reasons: Vec, + expected_terms: Vec, + forbidden_terms: Vec, + passed: bool, +} + +#[derive(Debug, Clone)] +struct ContinuousLoopQueryCase { + query_id: String, + text: String, + expected_terms: Vec, + forbidden_terms: Vec, + expected_omission: Option, + expect_packets: bool, + require_candidates: bool, +} + #[derive(Debug, Clone)] struct SweepQueryRecord { query_id: String, @@ -160,7 +214,7 @@ async fn main() -> Result<()> { let args: Vec = env::args().collect(); if args.len() < 2 { bail!( - "usage:\n cargo run -p klbr-bench -- run --suite longmemeval-s --data --out [--retrieval-only]\n cargo run -p klbr-bench -- retrieval \n cargo run -p klbr-bench -- passive-recall \n cargo run -p klbr-bench -- router \n cargo run -p klbr-bench -- router-multi [dataset2.json ...]\n cargo run -p klbr-bench -- router-multi-linear [dataset2.json ...]\n cargo run -p klbr-bench -- longmem [llm_url]\n cargo run -p klbr-bench -- longmem-retrieval [split.json subset]\n cargo run -p klbr-bench -- dump-tools \n cargo run -p klbr-bench -- sweep [score_start score_end score_step margin_start margin_end margin_step [support_start support_end support_step]]" + "usage:\n cargo run -p klbr-bench -- run --suite longmemeval-s --data --out [--retrieval-only]\n cargo run -p klbr-bench -- retrieval \n cargo run -p klbr-bench -- passive-recall \n cargo run -p klbr-bench -- router \n cargo run -p klbr-bench -- router-multi [dataset2.json ...]\n cargo run -p klbr-bench -- router-multi-linear [dataset2.json ...]\n cargo run -p klbr-bench -- longmem [llm_url]\n cargo run -p klbr-bench -- longmem-retrieval [split.json subset]\n cargo run -p klbr-bench -- continuous-loop \n cargo run -p klbr-bench -- dump-tools \n cargo run -p klbr-bench -- sweep [score_start score_end score_step margin_start margin_end margin_step [support_start support_end support_step]]" ); } @@ -212,6 +266,12 @@ async fn main() -> Result<()> { } run_dump_tools_command(&args[2]).await } + "continuous-loop" => { + if args.len() != 3 { + bail!("usage: cargo run -p klbr-bench -- continuous-loop "); + } + run_continuous_loop_command(&args[2]).await + } "longmemeval" => { longmemeval::run_command(&args).await } @@ -296,7 +356,7 @@ async fn main() -> Result<()> { run_sweep_command(&args[2], &args[3], &args[4], grid).await } other => bail!( - "unknown subcommand '{}'; expected 'retrieval', 'passive-recall', 'router', 'router-multi', 'router-multi-linear', 'dump-tools', or 'sweep'", + "unknown subcommand '{}'; expected 'retrieval', 'passive-recall', 'router', 'router-multi', 'router-multi-linear', 'continuous-loop', 'dump-tools', or 'sweep'", other ), } @@ -315,6 +375,566 @@ async fn run_dump_tools_command(output_path: &str) -> Result<()> { Ok(()) } +async fn run_continuous_loop_command(output_dir: &str) -> Result<()> { + let output_dir = PathBuf::from(output_dir); + fs::create_dir_all(&output_dir)?; + + let report = run_continuous_loop_benchmark().await?; + write_continuous_loop_outputs(&output_dir, &report)?; + + if !report.passed { + bail!( + "continuous-loop benchmark failed; see {}", + output_dir.join("report.md").display() + ); + } + + println!( + "wrote continuous-loop benchmark outputs to {}", + output_dir.display() + ); + Ok(()) +} + +async fn run_continuous_loop_benchmark() -> Result { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open( + tmp.path() + .to_str() + .context("temporary continuous-loop db path was not valid utf-8")?, + 4, + )?; + let mut runtime = ModelsConfig::default(); + runtime.embed_dim = 4; + let llm = LlmClient::new(runtime); + let mut pipeline = MemoryPipeline::new(store, llm, klbr_core::config::MemoryConfig::default()) + .with_profile("klbr-full"); + pipeline.profile.write_episode_notes = true; + pipeline.profile.write_episode_memories = false; + pipeline.profile.lexical = true; + pipeline.profile.dense = false; + pipeline.profile.graph = true; + + let sessions = continuous_loop_sessions(); + let mut writes = Vec::new(); + for session in sessions { + writes.push(pipeline.observe_session(session).await?); + } + + let reflection_notes = write_continuous_loop_reflections(&pipeline, &writes)?; + pipeline.memory.sync_reference_indexes()?; + + let budget = ContextBudget { + max_tokens: 2_000, + top_k: 6, + graph_depth: 1, + }; + let queries = run_continuous_loop_queries(&pipeline, &writes, budget).await?; + let stats = pipeline.memory.stats()?; + let source_ref_count = writes + .iter() + .map(|write| write.source_refs.len()) + .sum::(); + let checks = continuous_loop_checks( + &pipeline, + &writes, + &reflection_notes, + &queries, + &stats, + source_ref_count, + )?; + let passed = checks.iter().all(|check| check.passed); + + Ok(ContinuousLoopReport { + benchmark_id: "continuous-loop-v1".to_string(), + passed, + sessions_observed: writes.len(), + source_ref_count, + reflection_notes, + stats, + checks, + queries, + }) +} + +fn continuous_loop_sessions() -> Vec { + vec![ + BenchSession { + session_id: "voice-session".to_string(), + timestamp: Some(1_000), + turns: vec![ + BenchTurn { + role: "user".to_string(), + content: "project voice note: in klbr docs and comments, keep the lowercase repo voice and avoid officecore phrasing.".to_string(), + timestamp: Some(1_000), + }, + BenchTurn { + role: "assistant".to_string(), + content: "noted; preserve lowercase direct style in klbr work.".to_string(), + timestamp: Some(1_001), + }, + ], + }, + BenchSession { + session_id: "workflow-session".to_string(), + timestamp: Some(2_000), + turns: vec![ + BenchTurn { + role: "user".to_string(), + content: "workflow note: before changing memory retrieval, run rtk cargo test -p klbr-core and the relevant klbr-bench command.".to_string(), + timestamp: Some(2_000), + }, + BenchTurn { + role: "assistant".to_string(), + content: "i will treat that as the retrieval preflight.".to_string(), + timestamp: Some(2_001), + }, + ], + }, + BenchSession { + session_id: "coupon-session".to_string(), + timestamp: Some(3_000), + turns: vec![ + BenchTurn { + role: "user".to_string(), + content: "i redeemed the coffee creamer coupon at Target after checking the grocery budget.".to_string(), + timestamp: Some(3_000), + }, + BenchTurn { + role: "assistant".to_string(), + content: "Target is the store to remember for that coupon.".to_string(), + timestamp: Some(3_001), + }, + ], + }, + BenchSession { + session_id: "dotfiles-klbr-session".to_string(), + timestamp: Some(4_000), + turns: vec![BenchTurn { + role: "user".to_string(), + content: "klbr note: the dotfiles syncer shares the rust cli harness." + .to_string(), + timestamp: Some(4_000), + }], + }, + ] +} + +fn write_continuous_loop_reflections( + pipeline: &MemoryPipeline, + writes: &[WriteTrace], +) -> Result> { + let profile_sources = source_refs_for(writes, "voice-session")?; + let workflow_sources = source_refs_for(writes, "workflow-session")?; + let coupon_sources = source_refs_for(writes, "coupon-session")?; + + Ok(vec![ + upsert_continuous_loop_note( + pipeline, + "nbench_profile_voice", + MemoryLane::Profile, + "profile_note", + "klbr repo voice", + "profile/klbr-repo-voice.md", + "dawn prefers lowercase repo voice for klbr docs and comments. avoid officecore phrasing unless the source text explicitly requires it.", + profile_sources, + &["dawn", "klbr", "repo voice"], + )?, + upsert_continuous_loop_note( + pipeline, + "nbench_proc_retrieval_preflight", + MemoryLane::Procedural, + "procedural_note", + "memory retrieval preflight", + "procedural/memory-retrieval-preflight.md", + "before memory retrieval changes, run rtk cargo test -p klbr-core and the relevant klbr-bench continuous-loop command.", + workflow_sources, + &["klbr", "memory retrieval", "preflight"], + )?, + upsert_continuous_loop_note( + pipeline, + "nbench_semantic_coupon_target", + MemoryLane::Semantic, + "semantic_note", + "coupon store fact", + "semantic/coupon-store-fact.md", + "the coffee creamer coupon was redeemed at Target; this fact came from the coupon session and should stay source-grounded.", + coupon_sources, + &["coupon", "target", "grocery"], + )?, + ]) +} + +#[allow(clippy::too_many_arguments)] +fn upsert_continuous_loop_note( + pipeline: &MemoryPipeline, + note_ref: &str, + lane: MemoryLane, + kind: &str, + title: &str, + path: &str, + body: &str, + sources: Vec, + entities: &[&str], +) -> Result { + let entities = entities + .iter() + .map(|entity| entity.to_string()) + .collect::>(); + let record = pipeline.memory.upsert_markdown_note(&MarkdownNoteInput { + note_ref: Some(note_ref.to_string()), + lane, + kind: kind.to_string(), + title: title.to_string(), + path: Some(path.to_string()), + body: body.to_string(), + sources: sources.clone(), + follow: None, + entities: entities.clone(), + status: "active".to_string(), + frontmatter: serde_json::json!({ + "benchmark": "continuous-loop", + "entities": entities, + "source_refs": sources.clone(), + }), + })?; + + Ok(ContinuousLoopNoteTrace { + note_ref: record.note_ref, + lane, + path: record.path, + source_count: sources.len(), + chunk_refs: record.chunk_refs, + }) +} + +async fn run_continuous_loop_queries( + pipeline: &MemoryPipeline, + writes: &[WriteTrace], + budget: ContextBudget, +) -> Result> { + let coupon_ref = source_refs_for(writes, "coupon-session")? + .first() + .cloned() + .context("coupon session did not produce a source ref")?; + let cases = vec![ + ContinuousLoopQueryCase { + query_id: "profile-voice".to_string(), + text: "what style do i prefer for klbr docs and comments?".to_string(), + expected_terms: vec![ + "lowercase repo voice".to_string(), + "officecore phrasing".to_string(), + ], + forbidden_terms: Vec::new(), + expected_omission: None, + expect_packets: true, + require_candidates: true, + }, + ContinuousLoopQueryCase { + query_id: "procedural-preflight".to_string(), + text: "what workflow should we always run for memory retrieval changes?".to_string(), + expected_terms: vec![ + "rtk cargo test -p klbr-core".to_string(), + "klbr-bench continuous-loop".to_string(), + ], + forbidden_terms: Vec::new(), + expected_omission: None, + expect_packets: true, + require_candidates: true, + }, + ContinuousLoopQueryCase { + query_id: "episode-coupon".to_string(), + text: "where did i redeem the coffee creamer coupon last time?".to_string(), + expected_terms: vec!["coffee creamer coupon".to_string(), "Target".to_string()], + forbidden_terms: Vec::new(), + expected_omission: None, + expect_packets: true, + require_candidates: true, + }, + ContinuousLoopQueryCase { + query_id: "exact-coupon-ref".to_string(), + text: format!("what store is mentioned in [{coupon_ref}]?"), + expected_terms: vec!["Target".to_string()], + forbidden_terms: Vec::new(), + expected_omission: None, + expect_packets: true, + require_candidates: true, + }, + ContinuousLoopQueryCase { + query_id: "negated-project-switch".to_string(), + text: "working on the dotfiles syncer today, not touching klbr".to_string(), + expected_terms: Vec::new(), + forbidden_terms: vec!["rust cli harness".to_string()], + expected_omission: Some("negated_query_term:klbr".to_string()), + expect_packets: false, + require_candidates: true, + }, + ]; + + let mut traces = Vec::new(); + for case in cases { + traces.push(run_continuous_loop_query(pipeline, case, budget).await?); + } + Ok(traces) +} + +async fn run_continuous_loop_query( + pipeline: &MemoryPipeline, + case: ContinuousLoopQueryCase, + budget: ContextBudget, +) -> Result { + let query = BenchQuery { + query_id: case.query_id.clone(), + text: case.text.clone(), + reference_time: Some(5_000), + }; + let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; + let context = pipeline + .assemble_context(query.clone(), &retrieval, budget) + .await?; + let content_lower = context.content.to_lowercase(); + let expected_ok = case + .expected_terms + .iter() + .all(|term| content_lower.contains(&term.to_lowercase())); + let forbidden_ok = case + .forbidden_terms + .iter() + .all(|term| !content_lower.contains(&term.to_lowercase())); + let omission_reasons = retrieval + .packet_omissions + .iter() + .map(|omission| omission.reason.clone()) + .collect::>(); + let omission_ok = case + .expected_omission + .as_ref() + .map(|expected| omission_reasons.iter().any(|reason| reason == expected)) + .unwrap_or(true); + let packets_ok = if case.expect_packets { + !retrieval.packets.is_empty() + } else { + retrieval.packets.is_empty() && context.used_refs.is_empty() + }; + let candidates_ok = !case.require_candidates || !retrieval.candidates.is_empty(); + let passed = expected_ok && forbidden_ok && omission_ok && packets_ok && candidates_ok; + + Ok(ContinuousLoopQueryTrace { + query_id: case.query_id, + text: case.text, + route_reason: retrieval.route.reason, + candidates: retrieval + .candidates + .iter() + .take(12) + .map(|candidate| format!("{}:{}", candidate.source, candidate.ref_id)) + .collect(), + packets: retrieval + .packets + .iter() + .map(|packet| { + format!( + "{}:{}:{}", + packet.packet_kind.as_str(), + packet.anchor_ref, + packet.refs.join(",") + ) + }) + .collect(), + used_refs: context.used_refs, + omission_reasons, + expected_terms: case.expected_terms, + forbidden_terms: case.forbidden_terms, + passed, + }) +} + +fn continuous_loop_checks( + pipeline: &MemoryPipeline, + writes: &[WriteTrace], + reflection_notes: &[ContinuousLoopNoteTrace], + queries: &[ContinuousLoopQueryTrace], + stats: &MemoryStoreStats, + source_ref_count: usize, +) -> Result> { + let notes_resolved = reflection_notes + .iter() + .map(|note| { + pipeline + .memory + .get_resolved_ref(¬e.note_ref) + .map(|resolved| resolved.and_then(|data| data.body).is_some()) + }) + .collect::>>()?; + let query_pass_count = queries.iter().filter(|query| query.passed).count(); + let negated_ok = queries + .iter() + .find(|query| query.query_id == "negated-project-switch") + .is_some_and(|query| query.passed); + let exact_ok = queries + .iter() + .find(|query| query.query_id == "exact-coupon-ref") + .is_some_and(|query| query.passed); + + Ok(vec![ + ContinuousLoopCheck { + name: "sessions_observed".to_string(), + passed: writes.len() == 4 && writes.iter().all(|write| !write.source_refs.is_empty()), + details: format!( + "{} sessions, {} source refs", + writes.len(), + source_ref_count + ), + }, + ContinuousLoopCheck { + name: "reflection_notes_resolved".to_string(), + passed: reflection_notes.len() == 3 && notes_resolved.iter().all(|resolved| *resolved), + details: format!( + "{} reflection notes, {} resolved with promptable bodies", + reflection_notes.len(), + notes_resolved.iter().filter(|resolved| **resolved).count() + ), + }, + ContinuousLoopCheck { + name: "markdown_sync".to_string(), + passed: stats.markdown_notes >= 7 && stats.fts_rows == stats.promptable_refs, + details: format!( + "{} markdown notes, {} promptable refs, {} fts rows", + stats.markdown_notes, stats.promptable_refs, stats.fts_rows + ), + }, + ContinuousLoopCheck { + name: "source_edges".to_string(), + passed: stats.active_edges >= source_ref_count as i64, + details: format!( + "{} active edges for {} source refs", + stats.active_edges, source_ref_count + ), + }, + ContinuousLoopCheck { + name: "query_alignment".to_string(), + passed: query_pass_count == queries.len(), + details: format!("{} / {} queries passed", query_pass_count, queries.len()), + }, + ContinuousLoopCheck { + name: "negated_passive_recall_suppressed".to_string(), + passed: negated_ok, + details: "dotfiles query keeps klbr packet out of assembled context".to_string(), + }, + ContinuousLoopCheck { + name: "exact_ref_resolution".to_string(), + passed: exact_ok, + details: "turn chunk inline ref resolves through packet assembly".to_string(), + }, + ]) +} + +fn source_refs_for(writes: &[WriteTrace], session_id: &str) -> Result> { + writes + .iter() + .find(|write| write.session_id == session_id) + .map(|write| write.source_refs.clone()) + .with_context(|| format!("missing write trace for session {session_id}")) +} + +fn write_continuous_loop_outputs(output_dir: &Path, report: &ContinuousLoopReport) -> Result<()> { + fs::write( + output_dir.join("report.json"), + serde_json::to_vec_pretty(report)?, + )?; + fs::write( + output_dir.join("report.md"), + render_continuous_loop_report(report), + )?; + Ok(()) +} + +fn render_continuous_loop_report(report: &ContinuousLoopReport) -> String { + let checks = report + .checks + .iter() + .map(|check| { + format!( + "| `{}` | `{}` | {} |", + check.name, + yes_no(check.passed), + md_cell(&check.details) + ) + }) + .collect::>() + .join("\n"); + let queries = report + .queries + .iter() + .map(|query| { + format!( + "| `{}` | `{}` | `{}` | `{}` | `{}` | `{}` |", + query.query_id, + yes_no(query.passed), + query.route_reason, + query.candidates.len(), + query.packets.len(), + md_cell(&query.omission_reasons.join(", ")) + ) + }) + .collect::>() + .join("\n"); + + format!( + "# continuous-loop memory benchmark\n\n\ +benchmark: `{}`\n\ +passed: `{}`\n\ +sessions observed: `{}`\n\ +source refs: `{}`\n\ +reflection notes: `{}`\n\n\ +## store stats\n\n\ +- refs: `{}` active `{}`\n\ +- edges: `{}` active `{}`\n\ +- markdown notes: `{}`\n\ +- promptable refs / fts rows: `{}` / `{}`\n\ +- db bytes: `{}`\n\n\ +## checks\n\n\ +| check | passed | details |\n\ +|---|---|---|\n\ +{}\n\n\ +## queries\n\n\ +| query | passed | route | candidates | packets | omissions |\n\ +|---|---|---|---:|---:|---|\n\ +{}\n", + report.benchmark_id, + yes_no(report.passed), + report.sessions_observed, + report.source_ref_count, + report.reflection_notes.len(), + report.stats.refs, + report.stats.active_refs, + report.stats.edges, + report.stats.active_edges, + report.stats.markdown_notes, + report.stats.promptable_refs, + report.stats.fts_rows, + report.stats.db_bytes, + checks, + queries + ) +} + +fn yes_no(value: bool) -> &'static str { + if value { + "yes" + } else { + "no" + } +} + +fn md_cell(value: &str) -> String { + if value.is_empty() { + "(none)".to_string() + } else { + value.replace('|', "\\|").replace('\n', "
") + } +} + async fn run_retrieval_command( dataset_path: &str, config_path: &str, @@ -3778,7 +4398,11 @@ fn ndcg_at_k_from(candidates: &[RetrievalCandidateLog], gold_memory_ids: &[i64], let idcg = (0..ideal_hits) .map(|idx| 1.0 / ((idx as f32 + 2.0).log2())) .sum::(); - if idcg == 0.0 { 0.0 } else { dcg / idcg } + if idcg == 0.0 { + 0.0 + } else { + dcg / idcg + } } fn final_candidates(trace: &RetrievalTrace) -> &[RetrievalCandidateLog] { @@ -5748,7 +6372,11 @@ fn candidates_to_session_ranking(candidates: &[RetrievalCandidateLog]) -> Vec, k: usize) -> f64 { let limit = ranked.len().min(k); let any_hit = ranked[..limit].iter().any(|sid| correct.contains(sid)); - if any_hit { 1.0 } else { 0.0 } + if any_hit { + 1.0 + } else { + 0.0 + } } fn recall_all_at_k(ranked: &[String], correct: &HashSet, k: usize) -> f64 { @@ -5758,7 +6386,11 @@ fn recall_all_at_k(ranked: &[String], correct: &HashSet, k: usize) -> f6 let limit = ranked.len().min(k); let retrieved = ranked[..limit].iter().collect::>(); let all_hit = correct.iter().all(|sid| retrieved.contains(sid)); - if all_hit { 1.0 } else { 0.0 } + if all_hit { + 1.0 + } else { + 0.0 + } } fn retrieval_ndcg_at_k(ranked: &[String], correct: &HashSet, k: usize) -> f64 { @@ -5782,7 +6414,11 @@ fn retrieval_ndcg_at_k(ranked: &[String], correct: &HashSet, k: usize) - .map(|(i, &r)| r / ((i + 2) as f64).log2()) .sum(); - if idcg > 0.0 { dcg / idcg } else { 0.0 } + if idcg > 0.0 { + dcg / idcg + } else { + 0.0 + } } #[cfg(test)] @@ -5792,6 +6428,23 @@ mod tests { EvalInteractionMode, EvalMemoryInput, EvalObjective, MemoryLayer, MemoryStatus, }; + #[tokio::test] + async fn continuous_loop_benchmark_smoke_passes() { + let report = run_continuous_loop_benchmark() + .await + .expect("continuous-loop benchmark should run"); + + assert!(report.passed, "{:#?}", report.checks); + assert!(report + .queries + .iter() + .any(|query| query.query_id == "negated-project-switch" && query.passed)); + assert!(report + .queries + .iter() + .any(|query| query.query_id == "exact-coupon-ref" && query.passed)); + } + #[test] fn parse_holdout_bucket_handles_lane_words_inside_bucket() { assert_eq!( diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index 1fbb5f5..7c492b8 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -375,8 +375,11 @@ impl MemoryPipeline { let mut remaining_tokens = budget.max_tokens; let mut packets = String::new(); packets.push_str("\n"); - let fallback_packets; - let evidence_packets = if retrieved.packets.is_empty() && !retrieved.candidates.is_empty() { + let mut fallback_packets; + let evidence_packets = if retrieved.packets.is_empty() + && !retrieved.candidates.is_empty() + && retrieved.packet_omissions.is_empty() + { fallback_packets = self .evidence_planner() .build_packets( @@ -385,6 +388,8 @@ impl MemoryPipeline { budget.max_tokens.min(900), )? .packets; + let constrained = constrain_packets_for_query(&query.text, fallback_packets); + fallback_packets = constrained.packets; fallback_packets.as_slice() } else { retrieved.packets.as_slice() @@ -1919,6 +1924,18 @@ mod tests { .packet_omissions .iter() .any(|omission| omission.reason == "negated_query_term:klbr")); + let context = pipeline + .assemble_context( + BenchQuery { + query_id: "q-not-klbr".to_string(), + text: "working on the dotfiles syncer today, not touching klbr".to_string(), + reference_time: Some(3), + }, + &retrieval, + budget, + ) + .await?; + assert!(!context.content.contains("rust cli harness")); Ok(()) } -- 2.51.2