diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index 7f68cd2..ae4dc3c 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -18,3 +18,4 @@ {"id":"int-b4471f78","kind":"field_change","created_at":"2026-06-26T23:33:01.356937572Z","actor":"dawn","issue_id":"klbr-wmz.19","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Centralized reader prompt, made packet-local adjacent bodies explicit evidence windows, added prompt contract test, and verified 51a45a95 non-rerank now answers Target with answer-bearing refs in context."}} {"id":"int-50b6a7e3","kind":"field_change","created_at":"2026-06-26T23:36:55.988268642Z","actor":"dawn","issue_id":"klbr-wmz.5","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Episode notes now render as structured episode_event_card artifacts with event metadata, source-grounded timelines, stated facts/decisions, attachment markers, source refs, and promptable retrieval coverage; added core regression."}} {"id":"int-e6260a2f","kind":"field_change","created_at":"2026-06-26T23:42:12.700535629Z","actor":"dawn","issue_id":"klbr-wmz.4","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Added write_memory_note reflection tool for source-grounded profile/procedural markdown notes, ref supersession updates, source/policy frontmatter, stable lane paths, and tests for profile/procedural create/update paths."}} +{"id":"int-e9ba47e4","kind":"field_change","created_at":"2026-06-26T23:44:56.986031797Z","actor":"dawn","issue_id":"klbr-wmz.8","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Added LoCoMo smoke fixture, loader tests for session/haystack/conversation shapes, protocol notes in report/manifest, docs with exact command/result, and verified locomo retrieval-only smoke."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index de74135..3e2e573 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -18,6 +18,6 @@ {"_type":"issue","id":"klbr-djb","title":"update agents.md to reflect latest code state, get rid of anything stale - dont touch rtk / beads instructions","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:27:02Z","created_by":"dawn","updated_at":"2026-06-26T22:31:09Z","started_at":"2026-06-26T22:28:05Z","closed_at":"2026-06-26T22:31:09Z","close_reason":"updated AGENTS.md to match the latest codebase architecture (KDL configs, models.rs, memory lanes, router, evidence, garden, twilight discord)","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.15","title":"Rerank answer-bearing evidence packets instead of raw chunks","description":"docs/evidence-selection.md recommends reranking EvidencePacket objects, not raw chunks or whole sessions. Chunk-level reranking can prefer the query-adjacent setup chunk and still miss the answer-bearing neighbor; whole-session reranking is too broad. Packet reranking should judge whether a budgeted evidence packet contains enough support to answer.","design":"Treat reranking as a tie-breaker until packet-level fixtures show lift. The reranker prompt/input should ask for answer-bearing evidence, not generic relevance.","acceptance_criteria":"The reranker path can score packet bodies that include anchor, local window, episode gist, refs, and source signals; exact-ref packets cannot be suppressed by reranker score alone; traces log pre-rerank and post-rerank packet order; reranker thresholds operate on packets and cannot drop all strong evidence for a gold session without a traceable reason; tests cover a packet that beats its raw anchor chunk because the neighbor contains the answer.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:44Z","created_by":"dawn","updated_at":"2026-06-26T23:19:52Z","started_at":"2026-06-26T23:10:05Z","closed_at":"2026-06-26T23:19:52Z","close_reason":"Implemented packet-level reranking with bounded packet documents, exact-ref protection, traceable threshold fallback, bench flag wiring, tests, and live smoke.","labels":["architecture","context","memory","rerank","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:44Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:39Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.9","title":"Clean up memory architecture docs and dead references","description":"AGENTS.md still links docs/memory_architecture.md, but the current docs are docs/memory-arch.md, docs/memory-benches.md, and docs/memory-implementation-status.md. The architecture doc also reads like a target-state essay while the status doc says several items have since landed. This makes agents start from stale paths and over-file already-completed work.","design":"Do not rewrite the architecture voice into officecore sludge. Make the docs more navigable while preserving the current direct style.","acceptance_criteria":"All repo docs and agent instructions point at the existing memory docs; dead references to docs/memory_architecture.md are removed or redirected; docs/memory-arch.md clearly separates target architecture from already-implemented pieces; docs/memory-implementation-status.md lists open gaps by bead id; rg memory_architecture shows no stale repo references unless intentionally documented as legacy.","status":"closed","priority":2,"issue_type":"chore","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:54:16Z","created_by":"dawn","updated_at":"2026-06-26T23:29:50Z","started_at":"2026-06-26T23:20:28Z","closed_at":"2026-06-26T23:29:50Z","close_reason":"Updated AGENTS and memory docs to current paths/status, separated architecture target-state from landed work, listed open bead ids, and verified stale docs/memory_architecture references are gone.","labels":["architecture","docs","memory"],"dependencies":[{"issue_id":"klbr-wmz.9","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:54:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-wmz.8","title":"Validate the LoCoMo adapter and comparison protocol","description":"docs/memory-benches.md recommends LoCoMo as the secondary public suite for memweaver-style comparison, and docs/memory-implementation-status.md says a flexible locomo adapter exists. Current source parses locomo-ish shapes inside klbr-bench/src/longmemeval.rs, but there is no verified dataset fixture, loader test, scoring protocol, or documented run result proving the adapter matches the intended benchmark semantics.","design":"Keep this as a comparison harness task, not a claim about beating another system. The output should make dataset version, reader model, token budget, and scoring method explicit.","acceptance_criteria":"A small LoCoMo-shaped fixture or documented local dataset path exercises the adapter; loader tests cover supported input shapes and session ordering; the benchmark manifest/report clearly labels LoCoMo runs and avoids claiming comparability without matched reader/budget/scoring; docs include the exact command and current verified result or blocker.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T17:54:09Z","created_by":"dawn","updated_at":"2026-06-26T17:54:09Z","labels":["architecture","benchmarks","locomo","memory"],"dependencies":[{"issue_id":"klbr-wmz.8","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:54:09Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-wmz.8","title":"Validate the LoCoMo adapter and comparison protocol","description":"docs/memory-benches.md recommends LoCoMo as the secondary public suite for memweaver-style comparison, and docs/memory-implementation-status.md says a flexible locomo adapter exists. Current source parses locomo-ish shapes inside klbr-bench/src/longmemeval.rs, but there is no verified dataset fixture, loader test, scoring protocol, or documented run result proving the adapter matches the intended benchmark semantics.","design":"Keep this as a comparison harness task, not a claim about beating another system. The output should make dataset version, reader model, token budget, and scoring method explicit.","acceptance_criteria":"A small LoCoMo-shaped fixture or documented local dataset path exercises the adapter; loader tests cover supported input shapes and session ordering; the benchmark manifest/report clearly labels LoCoMo runs and avoids claiming comparability without matched reader/budget/scoring; docs include the exact command and current verified result or blocker.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:54:09Z","created_by":"dawn","updated_at":"2026-06-26T23:44:57Z","started_at":"2026-06-26T23:42:41Z","closed_at":"2026-06-26T23:44:57Z","close_reason":"Added LoCoMo smoke fixture, loader tests for session/haystack/conversation shapes, protocol notes in report/manifest, docs with exact command/result, and verified locomo retrieval-only smoke.","labels":["architecture","benchmarks","locomo","memory"],"dependencies":[{"issue_id":"klbr-wmz.8","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:54:09Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.5","title":"Upgrade episodic notes from transcript cards to source-grounded event cards","description":"klbr-core/src/pipeline.rs render_episode_card creates an episode note from timestamp, source refs, and truncated role lines. docs/memory-arch.md asks for source-grounded scene memory that preserves who, when, where, what changed, what was said, available attachments, and supporting raw refs. The current implementation is addressable, but still too transcript-shaped for temporal/update reasoning.","design":"Do not synthesize unsupported vivid details. The event card should summarize only what source turns or attachments support, with raw refs retained as the escape hatch.","acceptance_criteria":"Episode artifacts have a stable structured representation for time anchors, participants/entities, changes/decisions, salient quotes or snippets, attachments when present, and source refs; the markdown rendering remains human-editable; retrieval/context packets can expose the structured gist without losing provenance; tests cover an episode with multiple turns and verify source refs, session id resolution, and useful promptable chunks.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:44Z","created_by":"dawn","updated_at":"2026-06-26T23:36:56Z","started_at":"2026-06-26T23:33:59Z","closed_at":"2026-06-26T23:36:56Z","close_reason":"Episode notes now render as structured episode_event_card artifacts with event metadata, source-grounded timelines, stated facts/decisions, attachment markers, source refs, and promptable retrieval coverage; added core regression.","labels":["architecture","episodic","memory","provenance"],"dependencies":[{"issue_id":"klbr-wmz.5","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:44Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.4","title":"Materialize profile and procedural lanes as first-class notes","description":"The schema and MemoryGarden know about profile and procedural lanes, but the production pipeline observe_session path currently writes raw turns plus episodic notes/memories. Stable preferences, standing instructions, and workflows still depend mostly on tags or model-authored remember calls instead of a first-class markdown-note flow with source refs and update policy.","design":"Build on MemoryGarden and upsert_markdown_note rather than adding a new store. Treat legacy memory tags as routing hints, not the durable source of truth for profile/procedural knowledge.","acceptance_criteria":"There is an explicit writer/reflection path for profile_note and procedural_note artifacts; new notes include source refs, frontmatter, stable paths under profile/ or procedural/, and canonical refs/chunks; updates use supersession or versioning instead of silent overwrite; user-confirmation or policy gates are documented for stable profile changes; tests cover creating and updating one profile note and one procedural note from source turns.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:36Z","created_by":"dawn","updated_at":"2026-06-26T23:42:13Z","started_at":"2026-06-26T23:37:20Z","closed_at":"2026-06-26T23:42:13Z","close_reason":"Added write_memory_note reflection tool for source-grounded profile/procedural markdown notes, ref supersession updates, source/policy frontmatter, stable lane paths, and tests for profile/procedural create/update paths.","labels":["architecture","markdown","memory","profile"],"dependencies":[{"issue_id":"klbr-wmz.4","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:36Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} diff --git a/benchmarks/inputs/datasets/locomo_smoke.json b/benchmarks/inputs/datasets/locomo_smoke.json new file mode 100644 index 0000000..909a5a4 --- /dev/null +++ b/benchmarks/inputs/datasets/locomo_smoke.json @@ -0,0 +1,37 @@ +[ + { + "id": "locomo_smoke_1", + "category": "single-hop", + "question": "Where does Alex keep the spare key?", + "answer": "blue planter", + "evidence_session_ids": ["morning_walk"], + "timestamp": "2024/01/03 (Wed) 09:00", + "sessions": [ + { + "session_id": "morning_walk", + "date": "2024/01/01 (Mon) 08:00", + "messages": [ + { + "speaker": "user", + "text": "Alex keeps the spare key in the blue planter by the back door.", + "has_answer": true + }, + { + "speaker": "assistant", + "text": "Noted: spare key, blue planter, back door." + } + ] + }, + { + "session_id": "dinner_plan", + "date": "2024/01/02 (Tue) 18:30", + "messages": [ + { + "speaker": "user", + "text": "Dinner is at the noodle shop tonight." + } + ] + } + ] + } +] diff --git a/docs/memory-benches.md b/docs/memory-benches.md index fb897ef..7a47f8f 100644 --- a/docs/memory-benches.md +++ b/docs/memory-benches.md @@ -332,10 +332,39 @@ for memweaver-style comparison specifically, add: ```bash rtk cargo run -p klbr-bench -- run \ --suite locomo \ - --system klbr \ - --profile klbr-reflink-v2 \ - --reader \ - --budget-read + --data \ + --profile klbr-full \ + --top-k 8 \ + --budget-read \ + --graph-depth 1 \ + --out +``` + +current local adapter smoke: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite locomo \ + --data benchmarks/inputs/datasets/locomo_smoke.json \ + --profile raw-turns/fts-only \ + --top-k 3 \ + --budget-read 1200 \ + --graph-depth 0 \ + --limit 1 \ + --retrieval-only \ + --out /tmp/klbr-locomo-smoke +``` + +verified result: + +```text +evaluated: 1 +answerable: 1 +RecallAny@5: 1.0000 +RecallAll@5: 1.0000 +PacketRecallAny@3: 1.0000 +PacketRecallAll@3: 1.0000 +protocol_note: LoCoMo adapter run. Use for MemWeaver-style comparison only with matched dataset version, reader model, token budget, and scoring protocol. ``` and report both: diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index 9e3d72f..ecdd553 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -44,7 +44,7 @@ updated: 2026-06-27 packets, and falls back with traceable status/message when the reranker fails. - `klbr-bench -- run` now uses the production pipeline instead of manually calling low-level retrieval: - longmemeval-s/m style data - - flexible locomo adapter + - flexible locomo adapter with fixture coverage and protocol-note artifacts - ablation profiles such as `klbr/raw-turns`, `klbr/fts-only`, `klbr/dense-only`, `klbr/no-graph` - `--question-id` - `--diagnostic whenloss` @@ -156,6 +156,33 @@ answer_bearing_ref_in_context: 1.0000 gold_session_plus_neighbor_in_context: 1.0000 ``` +locomo adapter smoke: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite locomo \ + --data benchmarks/inputs/datasets/locomo_smoke.json \ + --profile raw-turns/fts-only \ + --top-k 3 \ + --budget-read 1200 \ + --graph-depth 0 \ + --limit 1 \ + --retrieval-only \ + --out /tmp/klbr-locomo-smoke +``` + +current result: + +```text +evaluated: 1 +answerable: 1 +RecallAny@5: 1.0000 +RecallAll@5: 1.0000 +PacketRecallAny@3: 1.0000 +PacketRecallAll@3: 1.0000 +protocol_note: present in report.json and manifest.json +``` + whenloss numeric diagnostic smoke: ```bash @@ -203,5 +230,4 @@ providing that command/environment. ## open gaps -- `klbr-wmz.8`: validate the LoCoMo adapter and comparison protocol. - `klbr-1yn`: run the real upstream LongMemEval QA evaluator once the command/environment exists. diff --git a/klbr-bench/src/longmemeval.rs b/klbr-bench/src/longmemeval.rs index 5096fbc..1da7a36 100644 --- a/klbr-bench/src/longmemeval.rs +++ b/klbr-bench/src/longmemeval.rs @@ -234,6 +234,104 @@ fn default_question_date() -> String { "9999/01/01 (Fri) 00:00".to_string() } +fn suite_protocol_note(suite: &str) -> Option<&'static str> { + if suite.eq_ignore_ascii_case("locomo") { + Some( + "LoCoMo adapter run. Use for MemWeaver-style comparison only with matched dataset version, reader model, token budget, and scoring protocol.", + ) + } else { + None + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn locomo_loader_parses_session_objects_and_preserves_order() -> Result<()> { + let questions = load_locomo_questions( + br#"{ + "data": [{ + "id": "q1", + "category": "temporal", + "query": "where is the key?", + "answer": "blue planter", + "evidence_session_ids": ["s2"], + "timestamp": "2024/01/03 (Wed) 09:00", + "sessions": [ + { + "conversation_id": "s1", + "timestamp": "2024/01/01 (Mon) 08:00", + "turns": [{"from": "user", "message": "first session"}] + }, + { + "session_id": "s2", + "date": "2024/01/02 (Tue) 08:00", + "messages": [{"speaker": "assistant", "text": "key is in the blue planter", "has_answer": true}] + } + ] + }] + }"#, + )?; + + assert_eq!(questions.len(), 1); + let q = &questions[0]; + assert_eq!(q.question_id, "q1"); + assert_eq!(q.question_type, "temporal"); + assert_eq!(q.question, "where is the key?"); + assert_eq!(q.answer_session_ids, vec!["s2"]); + assert_eq!(q.haystack_session_ids, vec!["s1", "s2"]); + assert_eq!( + q.haystack_dates, + vec!["2024/01/01 (Mon) 08:00", "2024/01/02 (Tue) 08:00"] + ); + assert_eq!(q.haystack_sessions[1][0].role, "assistant"); + assert_eq!(q.haystack_sessions[1][0].content, "key is in the blue planter"); + assert_eq!(q.haystack_sessions[1][0].has_answer, Some(true)); + Ok(()) + } + + #[test] + fn locomo_loader_accepts_legacy_haystack_arrays_and_conversation_shape() -> Result<()> { + let legacy = load_locomo_questions( + br#"[{ + "qa_id": "legacy", + "question": "what snack?", + "answer": "pears", + "haystack_session_ids": ["h1"], + "haystack_dates": ["2024/02/01 (Thu) 12:00"], + "haystack_sessions": [[{"role": "user", "content": "bring pears"}]] + }]"#, + )?; + assert_eq!(legacy[0].question_id, "legacy"); + assert_eq!(legacy[0].haystack_session_ids, vec!["h1"]); + assert_eq!(legacy[0].haystack_sessions[0][0].content, "bring pears"); + + let conversation = load_locomo_questions( + br#"[{ + "id": "conversation-only", + "query": "what color?", + "conversation": [{"speaker": "user", "utterance": "the notebook is green"}] + }]"#, + )?; + assert_eq!(conversation[0].haystack_session_ids, vec!["conversation"]); + assert_eq!( + conversation[0].haystack_sessions[0][0].content, + "the notebook is green" + ); + assert_eq!(conversation[0].question_date, default_question_date()); + Ok(()) + } + + #[test] + fn locomo_protocol_note_is_explicit_about_comparison_limits() { + let note = suite_protocol_note("locomo").unwrap(); + assert!(note.contains("matched dataset version")); + assert!(suite_protocol_note("longmemeval-s").is_none()); + } +} + pub(crate) fn parse_date_to_timestamp(date_str: &str) -> i64 { // format: "YYYY/MM/DD (ddd) HH:MM" // e.g. "2023/05/20 (Sat) 02:21" @@ -842,6 +940,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { let diagnostic = optional_arg(args, "--diagnostic"); let official_eval_cmd = optional_arg(args, "--official-eval-cmd"); let packet_rerank = has_flag(args, "--rerank-packets"); + let protocol_note = suite_protocol_note(&suite); fs::create_dir_all(&out_dir)?; @@ -1109,6 +1208,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { "answerable": answerable, "retrieval_only": retrieval_only, "packet_rerank": config.memory.rerank, + "protocol_note": protocol_note, "diagnostic": diagnostic.clone(), "question_id": question_id_filter.clone(), "top_k": top_k, @@ -1176,6 +1276,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { }, "diagnostic": diagnostic.clone(), "packet_rerank": config.memory.rerank, + "protocol_note": protocol_note, "question_id": question_id_filter.clone(), }))?, )?;