diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index 3f542d7..f7e4697 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -66,3 +66,10 @@ {"id":"int-5e3e4a9a","kind":"field_change","created_at":"2026-06-30T01:39:36.075956451Z","actor":"dawn","issue_id":"klbr-7yo.4","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"EvidencePlanner adds zettel trail_support packet bodies from parent/sibling/child/source refs, including chunk-to-note normalization and regression coverage."}} {"id":"int-851c8c7b","kind":"field_change","created_at":"2026-06-30T01:39:40.647590536Z","actor":"dawn","issue_id":"klbr-7yo.5","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Added folgezettel docs, AGENTS/status updates, exposed-tool tests, mk flow tests, and zettel trail packet regression coverage; klbr-54l remains open for bench validation."}} {"id":"int-01796566","kind":"field_change","created_at":"2026-06-30T01:39:44.868986285Z","actor":"dawn","issue_id":"klbr-7yo","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed folgezettel-native agent memory surface, zettel/fleeting storage, trail-aware packet support, docs, and focused regression tests."}} +{"id":"int-552fc7d3","kind":"field_change","created_at":"2026-06-30T12:55:16.409919653Z","actor":"dawn","issue_id":"klbr-54l.1","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"implemented trace_summary.jsonl/md and trace-diff command; verified against old/current/fixed LongMemEval-S regression rows"}} +{"id":"int-0ae39769","kind":"field_change","created_at":"2026-06-30T12:55:16.831283915Z","actor":"dawn","issue_id":"klbr-54l.2","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"session-first representatives now preserve real turn chunks and turn-window expansion keeps the anchor ref; 6b168ec8 retrieval smoke has answer ref rendered/in context"}} +{"id":"int-aa8165e6","kind":"field_change","created_at":"2026-06-30T12:55:17.248426659Z","actor":"dawn","issue_id":"klbr-54l.3","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"planner parser accepts tolerant op labels and LongMemEval category hints cover update/temporal/preference cases; tests and same-seed sample show update/count labels restored"}} +{"id":"int-ab919770","kind":"field_change","created_at":"2026-06-30T12:55:17.660458585Z","actor":"dawn","issue_id":"klbr-54l.4","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"context packing/render guardrails now keep answer-session packets renderable under packet budgets; traces expose stage/packet/render ranks; targeted rows b5ef892d and gpt4_385a5000 pass answer-ref-in-context"}} +{"id":"int-aa7c2ef1","kind":"field_change","created_at":"2026-06-30T12:55:18.069202784Z","actor":"dawn","issue_id":"klbr-54l.5","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"whenloss diagnostic summaries are emitted into trace_summary rows and docs frame LongMemEval passive QA as best-effort passive recall"}} +{"id":"int-c34adfde","kind":"field_change","created_at":"2026-06-30T12:55:18.472171415Z","actor":"dawn","issue_id":"klbr-54l.6","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately"}} +{"id":"int-e9f364ec","kind":"field_change","created_at":"2026-06-30T12:55:18.89582463Z","actor":"dawn","issue_id":"klbr-54l","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"implemented trace-diff, planner hints/parser hardening, session/packet/render fixes, writer/whenloss diagnostics, docs, and final same-seed retrieval smoke; remaining passive misses are documented as follow-up opportunities"}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 8b38f30..1a4a438 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -1,13 +1,13 @@ -{"_type":"issue","id":"klbr-54l.3","title":"Stabilize op planner labels for LongMemEval update and count cases","description":"The same-seed regression changed planner behavior heavily: old op_plan_counts had update_resolution=10 and lookup=3, while current has update_resolution=0 and lookup=15. That can disable collect/update behavior and shrink packet budgets even when retrieval finds the right area. Harden the structured planner path and fallback behavior for aggregate_count, aggregate_sum, aggregate_avg, order_or_rank, update_resolution, and preference_recommendation without reintroducing hidden English cue lists as the main policy.","acceptance_criteria":"Known klbr-54l rows get stable op labels across reruns or model-parser fallback; update_resolution no longer collapses to lookup on the same 25-row sample; tests cover malformed structured planner output and paraphrased update/count questions; implementation remains language-agnostic rather than relying on hardcoded English cue lists.","status":"open","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:17Z","created_by":"dawn","updated_at":"2026-06-30T11:13:17Z","labels":["bench","longmemeval","memory","planner"],"dependencies":[{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:17Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-54l.2","title":"Expand session-first episode anchors into supporting turn-window packets","description":"The klbr-54l trace research found a concrete failure class: stage_one finds the right session, but session-first emits thin episode_bridge packets from episode/card candidates and the final context drops the actual answer-bearing turn span. For 6b168ec8, the right session is found and a late episode bridge contains the three-bikes evidence, but the rendered context spends budget on irrelevant fts turn windows. In session-first mode, strong episode/session anchors should enrich or produce supporting turn_window packets from the same session's best seed refs instead of relying only on expand_edges from the episode card.","acceptance_criteria":"A production-pipeline fixture covers the candidate-session-hit / thin-episode-bridge-dropped failure; when the answer session is a strong stage_one group, packet planning includes a non-thin supporting turn_window from that session; 6b168ec8 no longer loses the three-bikes evidence before reader time without increasing global read budget.","status":"open","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:10Z","created_by":"dawn","updated_at":"2026-06-30T11:13:10Z","labels":["bench","evidence-packets","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-54l.1","title":"Trace-diff old-pass/current-fail LongMemEval passive QA rows","description":"Build a reusable trace-diff for the same-seed LongMemEval-S 25-row regression in klbr-54l before changing ranking or planner code. Compare baseline benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z against current benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current for old-pass/current-fail ids: 6b168ec8, c14c00dd, b5ef892d, 46a3abf7, 720133ac, gpt4_385a5000, dad224aa. Report per question: op_plan, stage_one answer-session rank, candidate vs packet recall, packet order/kind/session, answer-bearing selected/rendered/in-context, answer_value_visible, critical_fact_row_visible, packet/context tokens, omissions, and reader hypothesis.","acceptance_criteria":"A command, test, or bench report emits a per-question diff for the listed ids; 6b168ec8 clearly shows the candidate-session-hit but answer-packet-not-rendered class; output is durable enough to rerun during klbr-54l without hand-inspecting trace.jsonl.","status":"open","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:02Z","created_by":"dawn","updated_at":"2026-06-30T11:13:02Z","labels":["bench","longmemeval","memory","passive-recall"],"dependencies":[{"issue_id":"klbr-54l.1","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:02Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":6,"comment_count":0} +{"_type":"issue","id":"klbr-54l.3","title":"Stabilize op planner labels for LongMemEval update and count cases","description":"The same-seed regression changed planner behavior heavily: old op_plan_counts had update_resolution=10 and lookup=3, while current has update_resolution=0 and lookup=15. That can disable collect/update behavior and shrink packet budgets even when retrieval finds the right area. Harden the structured planner path and fallback behavior for aggregate_count, aggregate_sum, aggregate_avg, order_or_rank, update_resolution, and preference_recommendation without reintroducing hidden English cue lists as the main policy.","acceptance_criteria":"Known klbr-54l rows get stable op labels across reruns or model-parser fallback; update_resolution no longer collapses to lookup on the same 25-row sample; tests cover malformed structured planner output and paraphrased update/count questions; implementation remains language-agnostic rather than relying on hardcoded English cue lists.","status":"closed","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:17Z","created_by":"dawn","updated_at":"2026-06-30T12:55:17Z","closed_at":"2026-06-30T12:55:17Z","close_reason":"planner parser accepts tolerant op labels and LongMemEval category hints cover update/temporal/preference cases; tests and same-seed sample show update/count labels restored","labels":["bench","longmemeval","memory","planner"],"dependencies":[{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:17Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-54l.2","title":"Expand session-first episode anchors into supporting turn-window packets","description":"The klbr-54l trace research found a concrete failure class: stage_one finds the right session, but session-first emits thin episode_bridge packets from episode/card candidates and the final context drops the actual answer-bearing turn span. For 6b168ec8, the right session is found and a late episode bridge contains the three-bikes evidence, but the rendered context spends budget on irrelevant fts turn windows. In session-first mode, strong episode/session anchors should enrich or produce supporting turn_window packets from the same session's best seed refs instead of relying only on expand_edges from the episode card.","acceptance_criteria":"A production-pipeline fixture covers the candidate-session-hit / thin-episode-bridge-dropped failure; when the answer session is a strong stage_one group, packet planning includes a non-thin supporting turn_window from that session; 6b168ec8 no longer loses the three-bikes evidence before reader time without increasing global read budget.","status":"closed","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:10Z","created_by":"dawn","updated_at":"2026-06-30T12:55:17Z","closed_at":"2026-06-30T12:55:17Z","close_reason":"session-first representatives now preserve real turn chunks and turn-window expansion keeps the anchor ref; 6b168ec8 retrieval smoke has answer ref rendered/in context","labels":["bench","evidence-packets","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-54l.1","title":"Trace-diff old-pass/current-fail LongMemEval passive QA rows","description":"Build a reusable trace-diff for the same-seed LongMemEval-S 25-row regression in klbr-54l before changing ranking or planner code. Compare baseline benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z against current benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current for old-pass/current-fail ids: 6b168ec8, c14c00dd, b5ef892d, 46a3abf7, 720133ac, gpt4_385a5000, dad224aa. Report per question: op_plan, stage_one answer-session rank, candidate vs packet recall, packet order/kind/session, answer-bearing selected/rendered/in-context, answer_value_visible, critical_fact_row_visible, packet/context tokens, omissions, and reader hypothesis.","acceptance_criteria":"A command, test, or bench report emits a per-question diff for the listed ids; 6b168ec8 clearly shows the candidate-session-hit but answer-packet-not-rendered class; output is durable enough to rerun during klbr-54l without hand-inspecting trace.jsonl.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:02Z","created_by":"dawn","updated_at":"2026-06-30T12:55:16Z","started_at":"2026-06-30T11:33:29Z","closed_at":"2026-06-30T12:55:16Z","close_reason":"implemented trace_summary.jsonl/md and trace-diff command; verified against old/current/fixed LongMemEval-S regression rows","labels":["bench","longmemeval","memory","passive-recall"],"dependencies":[{"issue_id":"klbr-54l.1","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:02Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":6,"comment_count":0} {"_type":"issue","id":"klbr-7yo.5","title":"Document folgezettel memory semantics and add regression coverage","description":"Document the new memory ontology for agents: titles are APIs, bodies are attractors, source refs are required for grounded notes, and old memory CRUD tools are not the agent-facing surface. Add focused tests and a small smoke/regression path so the new trail traversal does not silently regress.","acceptance_criteria":"Docs describe mk_* tools and note-writing policy; tests cover exposed tool names, zettel write/follow/recall/revise/link/fleeting flows, and trail packet expansion. If LongMemEval regression klbr-54l is still open, docs should state how the new topology relates to that failure mode without claiming it is solved unless verified.","status":"closed","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T01:21:30Z","created_by":"dawn","updated_at":"2026-06-30T01:39:41Z","closed_at":"2026-06-30T01:39:41Z","close_reason":"Added folgezettel docs, AGENTS/status updates, exposed-tool tests, mk flow tests, and zettel trail packet regression coverage; klbr-54l remains open for bench validation.","labels":["benchmarks","docs","folgezettel","memory","tests","tools"],"dependencies":[{"issue_id":"klbr-7yo.5","depends_on_id":"klbr-7yo","type":"parent-child","created_at":"2026-06-30T04:21:29Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-7yo.5","depends_on_id":"klbr-7yo.2","type":"blocks","created_at":"2026-06-30T04:21:31Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-7yo.1","title":"Add folgezettel note topology to the memory substrate","description":"Represent zettel trails explicitly on top of existing refs/edges/markdown notes. A zettel title is the API handle; the body is a sourced attractor. Need parent/child continuation, branch/cross-link, supersession/revision, and fleeting/unplaced capture without replacing immutable turn/source refs.","acceptance_criteria":"MemoryStore can write zettel/fleeting markdown notes with frontmatter for title, parent/trail metadata, source refs, and status; edges encode follows/branch/source/supersedes relations; index/follow/recall queries can recover roots, children, siblings/path, full note bodies, and source refs.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T01:21:19Z","created_by":"dawn","updated_at":"2026-06-30T01:39:23Z","started_at":"2026-06-30T01:21:44Z","closed_at":"2026-06-30T01:39:23Z","close_reason":"Completed folgezettel note/fleeting storage paths, continuation/source/supersession edges, and trail lookup substrate.","labels":["folgezettel","memory","tools"],"dependencies":[{"issue_id":"klbr-7yo.1","depends_on_id":"klbr-7yo","type":"parent-child","created_at":"2026-06-30T04:21:19Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":2,"comment_count":0} {"_type":"issue","id":"klbr-7yo.2","title":"Implement mk_* folgezettel memory tools","description":"Add the agent-facing mk_* tool family: mk_index, mk_search, mk_follow, mk_recall, mk_remember, mk_revise, mk_link, and mk_fleeting. Tools should make the note move explicit and hide old memory CRUD semantics from the model.","acceptance_criteria":"Each mk_* tool has a focused JSON schema, concise success/error output, source-ref handling where required, and tests. mk_follow returns structure/titles only; mk_recall reads one note fully; mk_remember requires a claim-like title and parent/root placement; mk_fleeting captures unplaced scratch.","status":"closed","priority":1,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-30T01:21:19Z","created_by":"dawn","updated_at":"2026-06-30T01:39:27Z","closed_at":"2026-06-30T01:39:27Z","close_reason":"Implemented mk_index/mk_search/mk_follow/mk_recall/mk_remember/mk_revise/mk_link/mk_fleeting with focused tool-flow coverage.","labels":["folgezettel","memory","tools"],"dependencies":[{"issue_id":"klbr-7yo.2","depends_on_id":"klbr-7yo","type":"parent-child","created_at":"2026-06-30T04:21:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-7yo.2","depends_on_id":"klbr-7yo.1","type":"blocks","created_at":"2026-06-30T04:21:29Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":2,"comment_count":0} {"_type":"issue","id":"klbr-7yo.3","title":"Expose only mk_* memory tools to agents","description":"Swap memory_tools() and normal all_tools() exposure away from remember/recall/context_for/fetch_memories/write_memory_note/edit_memory/list_memories toward the mk_* ontology so agents do not see two competing memory models. Compatibility with old exposed tool names is explicitly not required.","acceptance_criteria":"Runtime agent/reflection registries expose mk_* memory tools only, with old tools either internal-only or removed from the normal registry; tests assert the exposed tool names; docs explain migration and current tool surface.","status":"closed","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T01:21:19Z","created_by":"dawn","updated_at":"2026-06-30T01:39:32Z","closed_at":"2026-06-30T01:39:32Z","close_reason":"memory_tools/all_tools now expose mk_* only for memory; old public memory tool names are removed from the normal registry and asserted in tests.","labels":["agent","folgezettel","memory","tools"],"dependencies":[{"issue_id":"klbr-7yo.3","depends_on_id":"klbr-7yo","type":"parent-child","created_at":"2026-06-30T04:21:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-7yo.3","depends_on_id":"klbr-7yo.2","type":"blocks","created_at":"2026-06-30T04:21:30Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-7yo.4","title":"Teach retrieval and context assembly to walk zettel trails","description":"After retrieval finds a zettel/note entrypoint, packet assembly should expand local folgezettel context: parent/path, previous/next/siblings when available, children/branches under budget, and source refs. This should improve answer-ready context without replacing dense/fts entrypoint search.","acceptance_criteria":"Evidence planner or pipeline can produce trail-aware packets from zettel refs; trace fields expose trail expansion; packets include source bodies near zettel attractor bodies; focused tests cover matched note -\u003e parent/child/source expansion under budget.","status":"closed","priority":1,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-30T01:21:19Z","created_by":"dawn","updated_at":"2026-06-30T01:39:36Z","closed_at":"2026-06-30T01:39:36Z","close_reason":"EvidencePlanner adds zettel trail_support packet bodies from parent/sibling/child/source refs, including chunk-to-note normalization and regression coverage.","labels":["folgezettel","memory","retrieval","tools"],"dependencies":[{"issue_id":"klbr-7yo.4","depends_on_id":"klbr-7yo","type":"parent-child","created_at":"2026-06-30T04:21:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-7yo.4","depends_on_id":"klbr-7yo.1","type":"blocks","created_at":"2026-06-30T04:21:30Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-7yo","title":"Make memory folgezettel-native for agents","description":"Replace the agent-facing memory ontology with folgezettel-style tools and note semantics. Titles are APIs: compact reusable claims. Bodies are attractors: sourced reasoning/provenance that helps the model reconstruct why the title is true. The implementation should keep immutable turns/refs as substrate, but expose mk_* tools instead of the old memory CRUD/search tool set so agents learn to edit trails rather than buckets.","design":"Root idea: retrieval finds entrypoints, folgezettel topology decides local traversal. Compatibility with old agent-facing memory tools is not required; keep old internals only if useful during migration. Expose one ontology: mk_index, mk_search, mk_follow, mk_recall, mk_remember, mk_revise, mk_link, mk_fleeting. Titles are APIs; bodies are attractors grounded in source refs.","acceptance_criteria":"Agent-facing memory tools are mk_* only; zettel notes have claim-like API titles and attractor bodies with source refs; agents can index/search/follow/recall/write/revise/link/fleeting-capture trails; retrieval/context assembly can include local trail packets with source refs; docs and focused tests cover the new ontology and old-tool exposure is removed from normal agent tool registries.","status":"closed","priority":1,"issue_type":"epic","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T01:20:58Z","created_by":"dawn","updated_at":"2026-06-30T01:39:45Z","started_at":"2026-06-30T01:21:44Z","closed_at":"2026-06-30T01:39:45Z","close_reason":"Completed folgezettel-native agent memory surface, zettel/fleeting storage, trail-aware packet support, docs, and focused regression tests.","labels":["folgezettel","memory","tools"],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-54l","title":"Investigate LongMemEval QA regression after memory evidence changes","description":"Same-seed LongMemEval-S 25-sample QA dropped after the recent memory evidence/planner/session-first changes. Baseline run benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z scored 0.6800. Current run benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current scored 0.4400 with the same sample seed 4937553249516211047 and local judge model. Session recall stayed similar, but answer-bearing/rendered evidence and packet/context budget dropped: answer_bearing_ref_in_context 0.80 -\u003e 0.36, answer_bearing_ref_selected 0.84 -\u003e 0.44, packet_tokens_total_mean 4412.64 -\u003e 1811.28. Op plans also changed heavily: old update_resolution=10, current update_resolution=0 and lookup=15. Regressions old-pass/current-fail: 6b168ec8, c14c00dd, b5ef892d, 46a3abf7, 720133ac, gpt4_385a5000, dad224aa. Fix old-fail/current-pass: gpt4_59149c77.","acceptance_criteria":"Identify whether the loss comes from structured model planning, packet selection/rendering budget, answer-bearing ref propagation, or reader prompting; add a focused regression/smoke that prevents the same same-seed sample from losing old passing rows without an explicit expected-metric update.","status":"open","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T01:00:03Z","created_by":"dawn","updated_at":"2026-06-30T01:00:03Z","dependencies":[{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:18Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.2","type":"blocks","created_at":"2026-06-30T14:14:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.3","type":"blocks","created_at":"2026-06-30T14:14:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.4","type":"blocks","created_at":"2026-06-30T14:14:20Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.5","type":"blocks","created_at":"2026-06-30T14:14:20Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.6","type":"blocks","created_at":"2026-06-30T14:14:20Z","created_by":"dawn","metadata":"{}"}],"dependency_count":6,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-54l","title":"Investigate LongMemEval QA regression after memory evidence changes","description":"Same-seed LongMemEval-S 25-sample QA dropped after the recent memory evidence/planner/session-first changes. Baseline run benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z scored 0.6800. Current run benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current scored 0.4400 with the same sample seed 4937553249516211047 and local judge model. Session recall stayed similar, but answer-bearing/rendered evidence and packet/context budget dropped: answer_bearing_ref_in_context 0.80 -\u003e 0.36, answer_bearing_ref_selected 0.84 -\u003e 0.44, packet_tokens_total_mean 4412.64 -\u003e 1811.28. Op plans also changed heavily: old update_resolution=10, current update_resolution=0 and lookup=15. Regressions old-pass/current-fail: 6b168ec8, c14c00dd, b5ef892d, 46a3abf7, 720133ac, gpt4_385a5000, dad224aa. Fix old-fail/current-pass: gpt4_59149c77.","acceptance_criteria":"Identify whether the loss comes from structured model planning, packet selection/rendering budget, answer-bearing ref propagation, or reader prompting; add a focused regression/smoke that prevents the same same-seed sample from losing old passing rows without an explicit expected-metric update.","status":"closed","priority":1,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T01:00:03Z","created_by":"dawn","updated_at":"2026-06-30T12:55:19Z","started_at":"2026-06-30T11:33:29Z","closed_at":"2026-06-30T12:55:19Z","close_reason":"implemented trace-diff, planner hints/parser hardening, session/packet/render fixes, writer/whenloss diagnostics, docs, and final same-seed retrieval smoke; remaining passive misses are documented as follow-up opportunities","dependencies":[{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:18Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.2","type":"blocks","created_at":"2026-06-30T14:14:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.3","type":"blocks","created_at":"2026-06-30T14:14:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.4","type":"blocks","created_at":"2026-06-30T14:14:20Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.5","type":"blocks","created_at":"2026-06-30T14:14:20Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l","depends_on_id":"klbr-54l.6","type":"blocks","created_at":"2026-06-30T14:14:20Z","created_by":"dawn","metadata":"{}"}],"dependency_count":6,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-u0q","title":"Implement session-first lexical-contained memory retrieval","description":"Build the next retrieval architecture from docs/long-term-memory-arch.md: stage-one session/event candidate generation, lexical containment, and language-agnostic operation planning experiments without adding english cue-word hacks.","design":"Keep EvidencePacket as the rank object. Use exact refs, dense episode/session cards, learned sparse or fts side channels, and graph expansion from strong seeds only. Keep QueryOp shape but replace cue-list policy with a structured classifier plus deterministic fallback.","acceptance_criteria":"A bench profile can retrieve from session/episode cards first and use chunk fts/sparse hits as packet enrichment; lexical-only policy paths are removed or contained behind candidate channels; traces expose candidate session recall and packet/rendered evidence metrics; multilingual fixtures cover non-English cue-free retrieval cases.","notes":"2026-06-29 partial: removed hardcoded english cue-list planner, english lane routing, english negation/entity-boundary packet filters, and wh/pronoun neighbor-expansion heuristics. OpPlan now comes from structured model JSON when an llm endpoint is configured, otherwise conservative lookup. Core and bench tests pass.\n2026-06-29 partial: landed initial session-first retrieval profile for klbr-full/dense-only/session-first profiles. Retrieval now groups archival candidates by session_id, prefers episodic/event-card anchors, adds chunk fts/sparse/dense hits as enrichment, serializes stage_one.session_candidates, and reports CandidateSessionRecall metrics in klbr-bench. fts-only now writes episode notes but skips embedded episode-memory rows so lexical ablations do not require the dense embedder at ingest. Verified core/bench tests, continuous-loop, and a one-row session-first/fts-only retrieval-only trace smoke.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T22:06:32Z","created_by":"dawn","updated_at":"2026-06-29T12:04:10Z","started_at":"2026-06-28T22:10:12Z","closed_at":"2026-06-29T12:04:10Z","close_reason":"Completed session/event-first lexical-contained slice: no english cue-list planner/filters, fts remains a candidate channel, trigram fts covers no-space multilingual exact-ish retrieval, traces expose session candidates/packet metrics, and docs clarify fts vs semantic multilingual retrieval.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-jwn","title":"Organize evolving memory architecture docs","description":"Make the memory architecture docs distinguish current truth, implementation status, proposals, research inputs, and archived rationale so future agents do not treat older reports as canonical.","acceptance_criteria":"Docs have a clear index and status taxonomy; current memory architecture direction points at docs/long-term-memory-arch.md; older research reports are marked as historical or supporting; AGENTS.md routes future agents through the doc index.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T22:01:20Z","created_by":"dawn","updated_at":"2026-06-28T22:07:09Z","started_at":"2026-06-28T22:01:23Z","closed_at":"2026-06-28T22:07:09Z","close_reason":"Completed docs index, current architecture review rewrite, status banners, AGENTS routing, and follow-up implementation issue klbr-u0q.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-h4l","title":"Cache bench embeddings locally and run 25-sample QA bench","description":"Make LongMemEval bench embeddings use a project-local gitignored sqlite cache so reruns do not pay for the same embeddings twice. Then run a 25-sample stratified QA bench, inspect failures, try bounded non-overfit tweaks if evidence supports them, and prepare a report packet if results remain weak or tuning would be overfit.","design":"Prefer a global project cache path such as benchmarks/cache/*.db over temp/run-local caches. Keep benchmark polling sparse: start the run, wait for artifacts or process completion, then inspect once.","acceptance_criteria":"Embedding cache lives under the project and is ignored by git; cache hits are reused across bench runs; a 25-sample bench artifact exists with failure analysis; any code changes are tested, committed, and pushed.","notes":"Implemented project-local embedding cache at benchmarks/cache/embeddings.db and wired LongMemEval bench LlmClient construction through it. Ran stratified 25-sample QA on seed 4937553249516211047: current-code run benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z scored official accuracy 0.6800. Report packet written to report_packet.md in that run dir. High-budget targeted reruns fixed only dd2973ad and a1cc6108; remaining failures point to fact/timeline synthesis rather than safe cue tuning.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T20:08:32Z","created_by":"dawn","updated_at":"2026-06-28T21:21:27Z","started_at":"2026-06-28T20:08:35Z","closed_at":"2026-06-28T21:21:27Z","close_reason":"Implemented project-local dense/sparse embedding cache, ran 25-sample QA bench, inspected failures, tried bounded planner/high-budget experiments, and wrote report packet.","dependency_count":0,"dependent_count":0,"comment_count":0} @@ -40,9 +40,9 @@ {"_type":"issue","id":"klbr-wmz.2","title":"Route runtime passive recall through the canonical memory pipeline","description":"Bench retrieval now goes through MemoryPipeline, but runtime passive recall in klbr-core/src/agent.rs still embeds the prompt, calls MemoryStore::get_searchable, runs retrieval::retrieve_exact over legacy memories, then injects Context memory packets. That bypasses fts, exact refs, markdown notes, graph expansion, and the canonical lane/lifecycle path the docs describe.","design":"Avoid duplicating retrieval logic in agent.rs. Either make MemoryPipeline usable by AgentRuntime or extract a shared retrieval facade that both MemoryPipeline and runtime passive recall call.","acceptance_criteria":"Runtime passive recall uses the same lane-aware canonical retrieval and context packet assembly policy as the production pipeline; recalled packets can include fts/exact/dense/graph candidates from refs and markdown notes; archived/tombstoned/suppressed refs do not leak; klbr-core/src/instructions.md matches the actual memory packet format; tests or a focused integration fixture cover passive recall from a markdown note and from an explicit ref.","notes":"Runtime passive recall now calls MemoryPipeline::retrieve_evidence and injects shared EvidencePacket XML via Context::inject_evidence_packets; no-model runtime packet fixture covers turn-window expansion. Remaining acceptance is blocked on klbr-wmz.1 because dense search still starts from legacy memory rows before ref mapping.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:19Z","created_by":"dawn","updated_at":"2026-06-26T19:44:55Z","started_at":"2026-06-26T19:37:32Z","closed_at":"2026-06-26T19:44:55Z","close_reason":"Completed: runtime passive recall now uses MemoryPipeline/EvidencePlanner packets, instructions document current packet XML, and fixtures cover turn-window recall plus explicit markdown refs; dense canonical dependency completed in klbr-wmz.1.","labels":["architecture","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz.1","type":"blocks","created_at":"2026-06-26T22:37:53Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.1","title":"Move dense retrieval onto canonical refs and embedding_items","description":"Current source still has dense retrieval on the legacy memory-row surface: klbr-core/src/pipeline.rs search_dense calls MemoryStore::get_searchable and retrieval::retrieve_exact, while fts/exact retrieval uses canonical refs and promptable_text. The schema already has refs, promptable_text, markdown_note_chunks, and embedding_items, so dense retrieval should not be the odd path out.","design":"Prefer a ref-native embedding index backed by embedding_items. Backfill embeddings from promptable_text, keep memory-id aliases as compatibility aliases, and make klbr/full versus dense-only profiles exercise the same canonical identity layer as fts and exact retrieval.","acceptance_criteria":"Dense candidate generation works over canonical ref ids for memories, turn chunks, markdown note chunks, episode notes, profile notes, and procedural notes; lane and lifecycle filtering come from refs/ref_metadata instead of memory tags alone; benchmark traces return canonical refs for dense hits; regression tests cover a markdown-note-only hit and a tombstoned/suppressed ref not leaking through dense search.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:11Z","created_by":"dawn","updated_at":"2026-06-26T19:43:27Z","started_at":"2026-06-26T19:38:02Z","closed_at":"2026-06-26T19:43:27Z","close_reason":"Completed: dense candidate generation now lazily backfills embedding_items from active promptable refs, scores canonical ref embeddings directly, and tests markdown-note dense hits plus suppressed-ref filtering.","labels":["architecture","memory","refs","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.1","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:11Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz","title":"Finish memory architecture follow-through","description":"Tracks the remaining memory architecture work identified from docs/memory-arch.md, docs/memory-benches.md, docs/memory-implementation-status.md, and current klbr-core/klbr-bench source. Current status says pipeline, typed memory packets, markdown notes, edge mirroring, lifecycle projection, and benchmark runner integration exist; this epic is for gaps still present in source/docs.","acceptance_criteria":"Close when the child issues are complete, docs/memory-implementation-status.md is updated from current verification, and the architecture docs no longer point at missing or stale follow-up work.","status":"closed","priority":1,"issue_type":"epic","owner":"90008@klbr.net","created_at":"2026-06-26T17:52:54Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","closed_at":"2026-06-26T23:45:35Z","close_reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn.","labels":["architecture","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-54l.6","title":"Probe passive writer retention for atomic facts and events","description":"Research on memory systems suggests passive QA can fail because write-time compression loses entity/slot/value/time facts before retrieval ever has a chance. Klbr now writes episode_event_card artifacts and generic fact_rows, but the next passive-recall improvement pass should add writer-side probes for counts, updates, preference constraints, and temporal facts so we can tell whether the passive writer retained the answer-bearing atom before tuning retrieval.","acceptance_criteria":"Fixtures or diagnostics check that answer-bearing entity/slot/value/time atoms exist in stored promptable artifacts before retrieval; failures are reported separately from packet/ranking misses; at least count, update_resolution, temporal order, and preference-constraint cases are covered.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:57Z","created_by":"dawn","updated_at":"2026-06-30T11:13:57Z","labels":["bench","longmemeval","memory","writer"],"dependencies":[{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:56Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:18Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-54l.5","title":"Make WhenLoss-style passive QA diagnostics first-class","description":"The passive QA bench should be treated as best-effort passive recall, not the whole memory-system score. The klbr-bench --diagnostic whenloss mode exists, but regression work needs a first-class joined report that separates write-side loss from retrieval/packet/rendering loss per question. Add a report path for tfc, oracle-evidence, complete-stored-memory, and retrieved-memory scores joined with packet metrics and local/official QA outcomes.","acceptance_criteria":"A normal regression run can emit per-question tfc/oe/csm/rm scores plus write_gap and retrieval_gap; report rows join those scores with op_plan, answer_bearing_ref_selected/rendered/in_context, answer_value_visible, and official/local QA labels; docs explicitly frame LongMemEval passive QA as best-effort passive recall rather than agentic folgezettel memory quality.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:41Z","created_by":"dawn","updated_at":"2026-06-30T11:13:51Z","labels":["bench","diagnostics","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:41Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-54l.4","title":"Add answer-session packet survival guardrail to context packing","description":"Current passive QA can spend the final read budget on unrelated high-scoring fts turn windows while a packet from a strongly supported answer session exists but is too late or too thin to render. Add a generic runtime-safe packing/ranking guardrail: when a session group has strong multi-signal support in stage_one, preserve at least one useful non-thin support packet from that session before unrelated single-signal packets, without using gold labels or answer ids outside benchmark metrics.","acceptance_criteria":"A fixture where stage_one finds the relevant session but context packing omits its useful packet fails before the change and passes after; traces expose first relevant/answer-session packet rank for diagnostics; answer_bearing_ref_rendered improves on old-pass/current-fail rows without packet_tokens_total_mean exploding back into unbounded context dumping.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:26Z","created_by":"dawn","updated_at":"2026-06-30T11:13:26Z","labels":["bench","context-packing","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.4","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:25Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.4","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-54l.6","title":"Probe passive writer retention for atomic facts and events","description":"Research on memory systems suggests passive QA can fail because write-time compression loses entity/slot/value/time facts before retrieval ever has a chance. Klbr now writes episode_event_card artifacts and generic fact_rows, but the next passive-recall improvement pass should add writer-side probes for counts, updates, preference constraints, and temporal facts so we can tell whether the passive writer retained the answer-bearing atom before tuning retrieval.","acceptance_criteria":"Fixtures or diagnostics check that answer-bearing entity/slot/value/time atoms exist in stored promptable artifacts before retrieval; failures are reported separately from packet/ranking misses; at least count, update_resolution, temporal order, and preference-constraint cases are covered.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:57Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately","labels":["bench","longmemeval","memory","writer"],"dependencies":[{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:56Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:18Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-54l.5","title":"Make WhenLoss-style passive QA diagnostics first-class","description":"The passive QA bench should be treated as best-effort passive recall, not the whole memory-system score. The klbr-bench --diagnostic whenloss mode exists, but regression work needs a first-class joined report that separates write-side loss from retrieval/packet/rendering loss per question. Add a report path for tfc, oracle-evidence, complete-stored-memory, and retrieved-memory scores joined with packet metrics and local/official QA outcomes.","acceptance_criteria":"A normal regression run can emit per-question tfc/oe/csm/rm scores plus write_gap and retrieval_gap; report rows join those scores with op_plan, answer_bearing_ref_selected/rendered/in_context, answer_value_visible, and official/local QA labels; docs explicitly frame LongMemEval passive QA as best-effort passive recall rather than agentic folgezettel memory quality.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:41Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"whenloss diagnostic summaries are emitted into trace_summary rows and docs frame LongMemEval passive QA as best-effort passive recall","labels":["bench","diagnostics","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:41Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-54l.4","title":"Add answer-session packet survival guardrail to context packing","description":"Current passive QA can spend the final read budget on unrelated high-scoring fts turn windows while a packet from a strongly supported answer session exists but is too late or too thin to render. Add a generic runtime-safe packing/ranking guardrail: when a session group has strong multi-signal support in stage_one, preserve at least one useful non-thin support packet from that session before unrelated single-signal packets, without using gold labels or answer ids outside benchmark metrics.","acceptance_criteria":"A fixture where stage_one finds the relevant session but context packing omits its useful packet fails before the change and passes after; traces expose first relevant/answer-session packet rank for diagnostics; answer_bearing_ref_rendered improves on old-pass/current-fail rows without packet_tokens_total_mean exploding back into unbounded context dumping.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:26Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"context packing/render guardrails now keep answer-session packets renderable under packet budgets; traces expose stage/packet/render ranks; targeted rows b5ef892d and gpt4_385a5000 pass answer-ref-in-context","labels":["bench","context-packing","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.4","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:25Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.4","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-f0q.5","title":"Add planner-grade benchmark metrics and fixtures","description":"Add rendered-evidence metrics and deterministic production-pipeline fixtures for count, average truncation, ordered temporal sessions, previous-vs-latest, and preference distractor regressions.","design":"Metrics should separate retrieval selection, packet rendering, fact-row visibility, and final answer correctness rather than collapsing them into one recall score.","acceptance_criteria":"Bench reports split selected/rendered/visible evidence metrics; tests cover the concrete failure families from the 20-sample LongMemEval brief; docs list smoke commands for planner/fact-table runs.","notes":"Partial metrics support landed early: LongMemEval reports op_plan_counts plus answer_bearing_ref_selected, answer_bearing_ref_rendered, and answer_value_visible. Remaining: critical_fact_row_visible, collect_mode_fact_group_recall, preference/update-specific correctness metrics, and deterministic fixtures for all failure families.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:47Z","created_by":"dawn","updated_at":"2026-06-29T12:04:05Z","started_at":"2026-06-29T12:04:00Z","closed_at":"2026-06-29T12:04:05Z","close_reason":"Completed split bench metrics for selected/rendered/value/fact-row visibility and collect-mode fact-group recall, plus deterministic tests for count, sum/avg, order, update, preference distractor, lookup fallback, and cjk no-space trigram retrieval.","dependencies":[{"issue_id":"klbr-f0q.5","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.5","depends_on_id":"klbr-f0q.2","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-oba","title":"add stratified random longmemeval sampling","description":"support reproducible random subset runs for the LongMemEval pipeline, stratified by question_type so small incremental qa runs still cover categories.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T16:09:37Z","created_by":"dawn","updated_at":"2026-06-28T16:16:07Z","started_at":"2026-06-28T16:10:01Z","closed_at":"2026-06-28T16:16:07Z","close_reason":"Completed: added reproducible random LongMemEval sampling with question_type stratification.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-8dq","title":"default longmemeval qa output under benchmarks/runs","description":"make the qa bench less annoying to run by defaulting its output directory under benchmarks/runs when --out is not supplied.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T15:55:48Z","created_by":"dawn","updated_at":"2026-06-28T16:16:07Z","started_at":"2026-06-28T15:56:03Z","closed_at":"2026-06-28T16:16:07Z","close_reason":"Completed: LongMemEval pipeline run now defaults --out under benchmarks/runs.","dependency_count":0,"dependent_count":0,"comment_count":0} diff --git a/docs/memory-benches.md b/docs/memory-benches.md index e788b5c..ff25bf4 100644 --- a/docs/memory-benches.md +++ b/docs/memory-benches.md @@ -223,6 +223,10 @@ retrieval: answer-value visible rate critical fact-row visible rate collect-mode fact-group recall + writer-retention probes + stage-one answer-session rank + first answer-session packet rank + first rendered answer-session packet rank memory architecture: ref validity rate @@ -251,6 +255,55 @@ efficiency: cost per answered question ``` +longmemeval passive qa should be read as **best-effort passive recall**, not as +the whole folgezettel memory system. it answers: “if runtime passively retrieves +and assembles evidence before the reader sees the question, did the needed fact +survive into context?” it does not answer whether an agent would actively follow +the zettel tree, call `mk_recall`, or write a better note with a title/body pair +that works as an api. keep a separate agentic folgezettel eval for that. + +the passive qa trace is still valuable because it exposes concrete improvement +points. every `klbr-bench -- run` output now writes: + +```text +trace.jsonl full per-question trace +trace_summary.jsonl compact row per question +trace_summary.md human scan table +``` + +the summary includes operation plan, stage-one answer-session rank, first +answer-session packet rank, first rendered answer-session packet rank, +answer-ref-in-context, packet token totals, omission counts, writer-retention +probes, and a compact whenloss diagnostic summary when `--diagnostic whenloss` +is enabled. + +to compare an old passing run against a current failing run: + +```bash +rtk cargo run -p klbr-bench -- trace-diff \ + benchmarks/runs/longmemeval-s/klbr-full/ \ + benchmarks/runs/longmemeval-s/klbr-full/ \ + 6b168ec8 c14c00dd +``` + +that writes `trace_diff.json` and `trace_diff.md` into the new run directory. +the markdown table is intentionally narrow: official label delta, op plan, +stage rank, packet rank, rendered rank, answer-ref-in-context, and packet-token +delta. + +longmemeval categories may provide a query-op hint to the passive pipeline. +current hints are conservative and dataset-level: + +```text +knowledge-update -> update_resolution +temporal-reasoning -> order_or_rank +single-session-preference -> preference_recommendation +``` + +everything else still uses the structured planner/fallback path. this keeps the +passive bench from depending on brittle english cue words while still testing +the retrieval mode the dataset category actually asks for. + the most useful diagnostic mode is the four-condition setup from WhenLoss: run the same reader under truncated full context, oracle evidence, complete stored memory, and retrieved memory. that separates “the writer threw the fact away” from “the fact exists but retrieval missed it.” WhenLoss defines the write-side gap as oracle evidence → complete stored memory, and the retrieval-side gap as complete stored memory → retrieved memory. ([arXiv][4]) for klbr, that becomes: diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index 25dedbe..84a38c0 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -81,6 +81,16 @@ going next. order, and update-resolution questions. packet planning widens candidate and packet limits, traces fact-group coverage/gap state, and favors distinct session/ref fact groups instead of letting one session monopolize context. +- the structured query planner parser is tolerant of common model label shapes + (`op`, `operation`, `query_op`, `count`, `latest`, `average`, etc.) while + preserving the typed `QueryOp` interface. longmemeval category hints can + directly supply `update_resolution`, `order_or_rank`, or + `preference_recommendation` when the dataset label is already authoritative. +- session-first candidate grouping now keeps an episode/event anchor plus real + turn chunks before spending per-session slots on additional cards. turn-window + expansion also guarantees that the anchor chunk survives `max_refs`, so long + previous turns cannot produce packets whose `anchor_ref` is missing from + `refs`. - evidence packets now carry compact `fact_rows` and `timeline` rows with supporting refs while preserving raw packet bodies as provenance. rendered `` include `` and `` blocks before bodies. @@ -96,6 +106,12 @@ going next. - `--diagnostic whenloss` - `--official-eval-cmd` - manifest/report/store stats output + - `trace_summary.jsonl` / `trace_summary.md` + - writer-retention probes for answer refs and episode cards + - XML-entity-aware answer-value visibility metrics +- `klbr-bench -- trace-diff [question-id ...]` compares + trace summaries, falling back to full trace rows when needed, and writes a + compact json/markdown delta into the new run directory. ## open direction @@ -119,10 +135,11 @@ rtk cargo run -p klbr-bench -- continuous-loop /tmp/klbr-continuous-loop current result: ```text +cargo check -p klbr-core: 0 errors, 1 pre-existing warning cargo check -p klbr-bench: 0 errors, 1 pre-existing warning focused folgezettel tests: mk surface 2 passed; trail-support packet test 1 passed -klbr-core tests: 151 passed, 1 ignored -klbr-bench tests: 17 passed +klbr-core tests: 156 passed, 1 ignored +klbr-bench tests: 18 passed continuous-loop smoke: passed with the pre-existing agent.rs dead-code warning session-first/fts-only trace smoke: evaluated 1, CandidateSessionRecallAny@5 1.0000 ``` @@ -218,6 +235,42 @@ reader behavior on the known hard cases: 58bf7951: reader answers The Glass Menagerie. ``` +same-seed passive retrieval smoke after the trace-diff/render-budget fixes: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned.json \ + --profile klbr-full \ + --top-k 5 \ + --budget-read 5000 \ + --graph-depth 1 \ + --sample 25 \ + --sample-seed 4937553249516211047 \ + --sample-mode stratified \ + --retrieval-only \ + --out /tmp/klbr-lme-sample25-seed4937553249516211047-after-render-budget-fix +``` + +current result: + +```text +evaluated: 25 +RecallAny@5: 0.9600 +RecallAll@5: 0.8400 +answer_bearing_ref_selected: 0.8000 +answer_bearing_ref_in_context: 0.7200 +answer_value_visible: 0.3600 +collect_mode_fact_group_recall: 0.9497 +known remaining passive misses: dd2973ad, 46a3abf7, 57f827a0, gpt4_d6585ce9, dad224aa, 6aeb4375_abs, 561fabcd +``` + +same-seed trace diff against +`benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current` +showed answer-ref-in-context recovery for `6b168ec8`, `720133ac`, +`c14c00dd`, and `gpt4_385a5000`; `46a3abf7` and `dad224aa` remain retrieval +misses. + packet rerank smoke: ```bash diff --git a/klbr-bench/src/longmemeval.rs b/klbr-bench/src/longmemeval.rs index 06f3e61..e5765c9 100644 --- a/klbr-bench/src/longmemeval.rs +++ b/klbr-bench/src/longmemeval.rs @@ -21,6 +21,7 @@ use klbr_core::{ AssembledContext, BenchQuery, BenchRun, BenchSession, BenchTurn, ContextBudget, MemoryPipeline, PipelineRetrievalTrace, WriteTrace, MEMORY_READER_SYSTEM_PROMPT, }, + planner::QueryOp, retrieval::{self, RetrievalConfig}, support::SupportScorer, }; @@ -68,6 +69,49 @@ struct PacketSelectionMetrics { packet_tokens_mean: f64, } +#[derive(Debug, Clone, Serialize, Deserialize)] +struct PacketTraceSummary { + packet_id: String, + packet_kind: String, + session_id: Option, + estimated_tokens: usize, + refs_len: usize, + sources: Vec, + contains_answer_ref: bool, + rendered: bool, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +struct WriterRetentionMetrics { + answer_ref_count: usize, + answer_ref_promptable_count: usize, + answer_value_visible_in_answer_refs: bool, + answer_session_episode_count: usize, + answer_value_visible_in_episode_cards: bool, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +struct TraceSummaryRow { + question_id: String, + question_type: Option, + question: Option, + op_plan: Option, + op_reason: Option, + hypothesis: Option, + local_score: Option, + official_label: Option, + answer_session_ids: Vec, + stage_one_answer_session_rank: Option, + first_answer_session_packet_rank: Option, + first_rendered_answer_session_packet_rank: Option, + packet_metrics: serde_json::Value, + context_tokens: Option, + packet_order: Vec, + omission_counts: BTreeMap, + writer_retention: Option, + diagnostic: Option, +} + #[derive(Debug, Clone, Serialize)] struct OfficialEvalResult { command: String, @@ -310,6 +354,15 @@ fn suite_protocol_note(suite: &str) -> Option<&'static str> { } } +fn op_hint_for_question(item: &LongMemEvalQuestion) -> Option { + match item.question_type.as_str() { + "knowledge-update" => Some(QueryOp::UpdateResolution), + "temporal-reasoning" => Some(QueryOp::OrderOrRank), + "single-session-preference" => Some(QueryOp::PreferenceRecommendation), + _ => None, + } +} + fn default_pipeline_run_dir(suite: &str, profile: &str) -> PathBuf { let timestamp = chrono::Utc::now() .format("%Y-%m-%d_%H%M%S%.6fZ") @@ -555,6 +608,26 @@ mod tests { assert!(suite_protocol_note("longmemeval-s").is_none()); } + #[test] + fn longmemeval_category_hints_cover_update_temporal_and_preference() { + assert_eq!( + op_hint_for_question(&test_question("q-update", "knowledge-update")), + Some(QueryOp::UpdateResolution) + ); + assert_eq!( + op_hint_for_question(&test_question("q-temporal", "temporal-reasoning")), + Some(QueryOp::OrderOrRank) + ); + assert_eq!( + op_hint_for_question(&test_question("q-pref", "single-session-preference")), + Some(QueryOp::PreferenceRecommendation) + ); + assert_eq!( + op_hint_for_question(&test_question("q-multi", "multi-session")), + None + ); + } + #[test] fn answer_value_visible_metric_uses_rendered_context_text() { let mut item = test_question("q-visible", "multi-session"); @@ -564,6 +637,11 @@ mod tests { &item, "the average gpa is 3.83" )); + item.answer = serde_json::Value::String("Trader Joe's".to_string()); + assert!(answer_value_visible_in_context( + &item, + "picked up at Trader Joe's" + )); assert!(!answer_value_visible_in_context( &item, "the packet ref was selected but the value was truncated" @@ -1390,6 +1468,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { } else { None }; + let mut trace_summaries = Vec::::new(); let mut evaluated = 0usize; let mut recall_any_at_5 = 0usize; let mut recall_all_at_5 = 0usize; @@ -1473,6 +1552,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { query_id: item.question_id.clone(), text: item.question.clone(), reference_time: Some(parse_date_to_timestamp(&item.question_date)), + op_hint: op_hint_for_question(item), }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; *op_plan_counts @@ -1513,6 +1593,8 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { last_stats = Some(stats.clone()); let packet_metrics = compute_packet_selection_metrics(item, &retrieval, &context, &answer_ref_ids, top_k); + let writer_retention = + compute_writer_retention_metrics(item, &pipeline, &answer_ref_ids, &write_traces)?; let retrieved_session_ids = retrieval .candidates @@ -1592,25 +1674,26 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { "hypothesis": hypothesis, }))? )?; - writeln!( - traces, - "{}", - serde_json::to_string(&json!({ - "question_id": item.question_id, - "question_type": item.question_type, - "question": item.question, - "answer_session_ids": item.answer_session_ids, - "retrieved_session_ids": retrieved_session_ids, - "retrieval": retrieval, - "context": context, - "answer_bearing_ref_ids": answer_ref_ids, - "packet_metrics": packet_metrics, - "hypothesis": hypothesis, - "write_traces": write_traces, - "diagnostic": diagnostic_trace, - "store_stats": stats, - }))? - )?; + let trace_value = json!({ + "question_id": item.question_id, + "question_type": item.question_type, + "question": item.question, + "answer": item.answer, + "answer_session_ids": item.answer_session_ids, + "retrieved_session_ids": retrieved_session_ids, + "retrieval": retrieval, + "context": context, + "answer_bearing_ref_ids": answer_ref_ids, + "packet_metrics": packet_metrics, + "hypothesis": hypothesis, + "local_score": score_hypothesis_local(item, &hypothesis), + "writer_retention": writer_retention, + "write_traces": write_traces, + "diagnostic": diagnostic_trace, + "store_stats": stats, + }); + trace_summaries.push(trace_summary_from_value(&trace_value)?); + writeln!(traces, "{}", serde_json::to_string(&trace_value)?)?; } hypotheses.flush()?; traces.flush()?; @@ -1627,6 +1710,14 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { } else { None }; + let official_labels = official_eval_result + .as_ref() + .map(official_eval_labels) + .unwrap_or_default(); + for summary in &mut trace_summaries { + summary.official_label = official_labels.get(&summary.question_id).copied(); + } + write_trace_summary_files(&out_dir, &trace_summaries)?; let answerable_denominator = answerable.max(1) as f64; let evaluated_denominator = evaluated.max(1) as f64; @@ -1836,7 +1927,7 @@ fn answer_strings(value: &serde_json::Value) -> Vec { } fn normalize_answer_text(value: &str) -> String { - value + decode_xml_entities(value) .chars() .flat_map(char::to_lowercase) .map(|ch| if ch.is_alphanumeric() { ch } else { ' ' }) @@ -1846,6 +1937,15 @@ fn normalize_answer_text(value: &str) -> String { .join(" ") } +fn decode_xml_entities(value: &str) -> String { + value + .replace("'", "'") + .replace(""", "\"") + .replace("<", "<") + .replace(">", ">") + .replace("&", "&") +} + async fn answer_from_context(llm: &LlmClient, context: &AssembledContext) -> Result { let messages = vec![ Message::system(MEMORY_READER_SYSTEM_PROMPT), @@ -2239,6 +2339,462 @@ fn answer_value_visible_in_context(item: &LongMemEvalQuestion, context: &str) -> .any(|answer| context_norm.contains(&answer)) } +fn compute_writer_retention_metrics( + item: &LongMemEvalQuestion, + pipeline: &MemoryPipeline, + answer_ref_ids: &HashSet, + write_traces: &[WriteTrace], +) -> Result { + let answer_refs = answer_ref_ids.iter().cloned().collect::>(); + let answer_ref_entries = pipeline + .memory + .promptable_entries_for_refs(&answer_refs, "answer_ref_probe")?; + let answer_ref_text = answer_ref_entries + .iter() + .map(|entry| entry.body.as_str()) + .collect::>() + .join("\n"); + + let answer_sessions = item + .answer_session_ids + .iter() + .cloned() + .collect::>(); + let episode_refs = write_traces + .iter() + .filter(|trace| answer_sessions.contains(&trace.session_id)) + .filter_map(|trace| trace.episode_ref.clone()) + .collect::>(); + let episode_entries = pipeline + .memory + .promptable_entries_for_refs(&episode_refs, "episode_probe")?; + let episode_text = episode_entries + .iter() + .map(|entry| entry.body.as_str()) + .collect::>() + .join("\n"); + + Ok(WriterRetentionMetrics { + answer_ref_count: answer_ref_ids.len(), + answer_ref_promptable_count: answer_ref_entries.len(), + answer_value_visible_in_answer_refs: answer_value_visible_in_context( + item, + &answer_ref_text, + ), + answer_session_episode_count: episode_entries.len(), + answer_value_visible_in_episode_cards: answer_value_visible_in_context(item, &episode_text), + }) +} + +fn trace_summary_from_value(value: &serde_json::Value) -> Result { + let question_id = + json_string(value.get("question_id")).context("trace row missing question_id")?; + let answer_session_ids = json_string_vec(value.get("answer_session_ids")); + let answer_sessions = answer_session_ids.iter().cloned().collect::>(); + let answer_refs = json_string_vec(value.get("answer_bearing_ref_ids")); + let answer_refs = answer_refs.iter().cloned().collect::>(); + let used_refs = json_string_vec(value.pointer("/context/used_refs")); + let used_refs = used_refs.iter().cloned().collect::>(); + + let retrieval = value.get("retrieval").unwrap_or(&serde_json::Value::Null); + let stage_one_answer_session_rank = retrieval + .pointer("/stage_one/session_candidates") + .and_then(|items| items.as_array()) + .and_then(|items| { + items.iter().position(|candidate| { + json_string(candidate.get("session_id")) + .as_ref() + .is_some_and(|session_id| answer_sessions.contains(session_id)) + }) + }) + .map(|idx| idx + 1); + + let mut packet_order = Vec::new(); + let mut first_answer_session_packet_rank = None; + let mut first_rendered_answer_session_packet_rank = None; + if let Some(packets) = retrieval.get("packets").and_then(|items| items.as_array()) { + for (idx, packet) in packets.iter().enumerate() { + let refs = json_string_vec(packet.get("refs")); + let sources = json_string_vec(packet.pointer("/signals/sources")); + let session_id = json_string(packet.get("session_id")); + let answer_session_packet = session_id + .as_ref() + .is_some_and(|session_id| answer_sessions.contains(session_id)); + let rendered = refs.iter().any(|ref_id| used_refs.contains(ref_id)); + if answer_session_packet && first_answer_session_packet_rank.is_none() { + first_answer_session_packet_rank = Some(idx + 1); + } + if answer_session_packet + && rendered + && first_rendered_answer_session_packet_rank.is_none() + { + first_rendered_answer_session_packet_rank = Some(idx + 1); + } + packet_order.push(PacketTraceSummary { + packet_id: json_string(packet.get("packet_id")).unwrap_or_default(), + packet_kind: json_string(packet.get("packet_kind")).unwrap_or_default(), + session_id, + estimated_tokens: json_usize(packet.get("estimated_tokens")).unwrap_or_default(), + refs_len: refs.len(), + sources, + contains_answer_ref: refs.iter().any(|ref_id| answer_refs.contains(ref_id)), + rendered, + }); + } + } + + let mut omission_counts = BTreeMap::new(); + if let Some(omissions) = retrieval + .get("packet_omissions") + .and_then(|items| items.as_array()) + { + for omission in omissions { + if let Some(reason) = json_string(omission.get("reason")) { + *omission_counts.entry(reason).or_insert(0) += 1; + } + } + } + + let writer_retention = value + .get("writer_retention") + .filter(|value| !value.is_null()) + .cloned() + .map(serde_json::from_value) + .transpose()?; + + Ok(TraceSummaryRow { + question_id, + question_type: json_string(value.get("question_type")), + question: json_string(value.get("question")), + op_plan: json_string(retrieval.pointer("/op_plan/op")), + op_reason: json_string(retrieval.pointer("/op_plan/reason")), + hypothesis: json_string(value.get("hypothesis")), + local_score: value.get("local_score").and_then(|score| score.as_f64()), + official_label: None, + answer_session_ids, + stage_one_answer_session_rank, + first_answer_session_packet_rank, + first_rendered_answer_session_packet_rank, + packet_metrics: value + .get("packet_metrics") + .cloned() + .unwrap_or_else(|| json!({})), + context_tokens: json_usize(value.pointer("/context/estimated_tokens")), + packet_order, + omission_counts, + writer_retention, + diagnostic: whenloss_summary(value.get("diagnostic")), + }) +} + +fn whenloss_summary(value: Option<&serde_json::Value>) -> Option { + let value = value.filter(|value| !value.is_null())?; + let mut scores = BTreeMap::new(); + let mut context_tokens = BTreeMap::new(); + if let Some(modes) = value.get("modes").and_then(|modes| modes.as_object()) { + for (mode, details) in modes { + if let Some(score) = details.get("score").and_then(|score| score.as_f64()) { + scores.insert(mode.clone(), score); + } + if let Some(tokens) = details + .get("context_tokens") + .and_then(|tokens| tokens.as_u64()) + { + context_tokens.insert(mode.clone(), tokens); + } + } + } + Some(json!({ + "scores": scores, + "context_tokens": context_tokens, + "write_gap": value.pointer("/deltas/write_gap").and_then(|gap| gap.as_f64()), + "retrieval_gap": value.pointer("/deltas/retrieval_gap").and_then(|gap| gap.as_f64()), + })) +} + +fn official_eval_labels(result: &OfficialEvalResult) -> HashMap { + result + .parsed_metrics + .as_ref() + .map(official_eval_labels_from_metrics) + .unwrap_or_default() +} + +fn official_eval_labels_from_metrics(metrics: &serde_json::Value) -> HashMap { + metrics + .get("logs") + .and_then(|logs| logs.as_array()) + .into_iter() + .flatten() + .filter_map(|row| { + let question_id = json_string(row.get("question_id"))?; + let label = row + .pointer("/autoeval_label/label") + .and_then(|value| value.as_bool()) + .or_else(|| row.get("autoeval_label").and_then(|value| value.as_bool()))?; + Some((question_id, label)) + }) + .collect() +} + +pub fn run_trace_diff_command(args: &[String]) -> Result<()> { + if args.len() < 2 { + bail!("usage: cargo run -p klbr-bench -- trace-diff [question_id ...]"); + } + let old_dir = PathBuf::from(&args[0]); + let new_dir = PathBuf::from(&args[1]); + let filters = args[2..].iter().cloned().collect::>(); + let old_rows = load_trace_summary_rows(&old_dir)?; + let new_rows = load_trace_summary_rows(&new_dir)?; + let old_by_id = old_rows + .into_iter() + .map(|row| (row.question_id.clone(), row)) + .collect::>(); + let new_by_id = new_rows + .into_iter() + .map(|row| (row.question_id.clone(), row)) + .collect::>(); + + let mut question_ids = old_by_id + .keys() + .chain(new_by_id.keys()) + .filter(|id| filters.is_empty() || filters.contains(*id)) + .cloned() + .collect::>(); + question_ids.sort(); + question_ids.dedup(); + + let rows = question_ids + .iter() + .map(|question_id| { + let old = old_by_id.get(question_id); + let new = new_by_id.get(question_id); + json!({ + "question_id": question_id, + "old": old, + "new": new, + "delta": trace_summary_delta(old, new), + }) + }) + .collect::>(); + + fs::write( + new_dir.join("trace_diff.json"), + serde_json::to_vec_pretty(&rows)?, + )?; + fs::write( + new_dir.join("trace_diff.md"), + render_trace_diff_markdown(&rows), + )?; + println!( + "wrote trace diff to {} and {}", + new_dir.join("trace_diff.json").display(), + new_dir.join("trace_diff.md").display() + ); + Ok(()) +} + +fn load_trace_summary_rows(run_dir: &Path) -> Result> { + let summary_path = run_dir.join("trace_summary.jsonl"); + let mut rows = if summary_path.exists() { + fs::read_to_string(&summary_path)? + .lines() + .filter(|line| !line.trim().is_empty()) + .map(|line| serde_json::from_str::(line).map_err(Into::into)) + .collect::>>()? + } else { + let trace_path = run_dir.join("trace.jsonl"); + fs::read_to_string(&trace_path) + .with_context(|| format!("failed to read {}", trace_path.display()))? + .lines() + .filter(|line| !line.trim().is_empty()) + .map(|line| { + let value = serde_json::from_str::(line)?; + trace_summary_from_value(&value) + }) + .collect::>>()? + }; + let labels = fs::read(run_dir.join("official_eval.json")) + .ok() + .and_then(|bytes| serde_json::from_slice::(&bytes).ok()) + .map(|metrics| official_eval_labels_from_metrics(&metrics)) + .unwrap_or_default(); + for row in &mut rows { + row.official_label = labels.get(&row.question_id).copied(); + } + Ok(rows) +} + +fn trace_summary_delta( + old: Option<&TraceSummaryRow>, + new: Option<&TraceSummaryRow>, +) -> serde_json::Value { + json!({ + "official_label": { + "old": old.and_then(|row| row.official_label), + "new": new.and_then(|row| row.official_label), + }, + "local_score": { + "old": old.and_then(|row| row.local_score), + "new": new.and_then(|row| row.local_score), + }, + "op_plan": { + "old": old.and_then(|row| row.op_plan.clone()), + "new": new.and_then(|row| row.op_plan.clone()), + }, + "stage_one_answer_session_rank": { + "old": old.and_then(|row| row.stage_one_answer_session_rank), + "new": new.and_then(|row| row.stage_one_answer_session_rank), + }, + "first_answer_session_packet_rank": { + "old": old.and_then(|row| row.first_answer_session_packet_rank), + "new": new.and_then(|row| row.first_answer_session_packet_rank), + }, + "first_rendered_answer_session_packet_rank": { + "old": old.and_then(|row| row.first_rendered_answer_session_packet_rank), + "new": new.and_then(|row| row.first_rendered_answer_session_packet_rank), + }, + "answer_bearing_ref_in_context": { + "old": old.and_then(answer_ref_in_context), + "new": new.and_then(answer_ref_in_context), + }, + "packet_tokens_total": { + "old": old.and_then(packet_tokens_total), + "new": new.and_then(packet_tokens_total), + }, + "context_tokens": { + "old": old.and_then(|row| row.context_tokens), + "new": new.and_then(|row| row.context_tokens), + }, + "writer_answer_value_visible_in_answer_refs": { + "old": old.and_then(|row| row.writer_retention.as_ref()).map(|metrics| metrics.answer_value_visible_in_answer_refs), + "new": new.and_then(|row| row.writer_retention.as_ref()).map(|metrics| metrics.answer_value_visible_in_answer_refs), + }, + }) +} + +fn answer_ref_in_context(row: &TraceSummaryRow) -> Option { + row.packet_metrics + .get("answer_bearing_ref_in_context") + .and_then(|value| value.as_bool()) +} + +fn packet_tokens_total(row: &TraceSummaryRow) -> Option { + row.packet_metrics + .get("packet_tokens_total") + .and_then(|value| value.as_u64()) + .map(|value| value as usize) +} + +fn render_trace_diff_markdown(rows: &[serde_json::Value]) -> String { + let mut out = String::from("| question_id | official | op | stage_rank | packet_rank | rendered_rank | answer_ref_in_context | packet_tokens |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n"); + for row in rows { + let question_id = row + .get("question_id") + .and_then(|value| value.as_str()) + .unwrap_or("unknown"); + let delta = row.get("delta").unwrap_or(&serde_json::Value::Null); + out.push_str(&format!( + "| {} | {} -> {} | {} -> {} | {} -> {} | {} -> {} | {} -> {} | {} -> {} | {} -> {} |\n", + question_id, + display_json(delta.pointer("/official_label/old")), + display_json(delta.pointer("/official_label/new")), + display_json(delta.pointer("/op_plan/old")), + display_json(delta.pointer("/op_plan/new")), + display_json(delta.pointer("/stage_one_answer_session_rank/old")), + display_json(delta.pointer("/stage_one_answer_session_rank/new")), + display_json(delta.pointer("/first_answer_session_packet_rank/old")), + display_json(delta.pointer("/first_answer_session_packet_rank/new")), + display_json(delta.pointer("/first_rendered_answer_session_packet_rank/old")), + display_json(delta.pointer("/first_rendered_answer_session_packet_rank/new")), + display_json(delta.pointer("/answer_bearing_ref_in_context/old")), + display_json(delta.pointer("/answer_bearing_ref_in_context/new")), + display_json(delta.pointer("/packet_tokens_total/old")), + display_json(delta.pointer("/packet_tokens_total/new")), + )); + } + out +} + +fn display_json(value: Option<&serde_json::Value>) -> String { + match value { + Some(serde_json::Value::String(value)) => value.clone(), + Some(serde_json::Value::Number(value)) => value.to_string(), + Some(serde_json::Value::Bool(value)) => value.to_string(), + Some(serde_json::Value::Null) | None => "n/a".to_string(), + Some(value) => value.to_string(), + } +} + +fn write_trace_summary_files(out_dir: &Path, rows: &[TraceSummaryRow]) -> Result<()> { + let mut jsonl = BufWriter::new(File::create(out_dir.join("trace_summary.jsonl"))?); + for row in rows { + writeln!(jsonl, "{}", serde_json::to_string(row)?)?; + } + jsonl.flush()?; + + let mut md = String::from("| question_id | op | stage_rank | packet_rank | rendered_rank | answer_ref_in_context | writer_answer_ref | local | official |\n| --- | --- | ---: | ---: | ---: | --- | --- | ---: | --- |\n"); + for row in rows { + let answer_ref_in_context = row + .packet_metrics + .get("answer_bearing_ref_in_context") + .and_then(|value| value.as_bool()) + .map(|value| value.to_string()) + .unwrap_or_else(|| "n/a".to_string()); + let writer_answer_ref = row + .writer_retention + .as_ref() + .map(|metrics| metrics.answer_value_visible_in_answer_refs.to_string()) + .unwrap_or_else(|| "n/a".to_string()); + md.push_str(&format!( + "| {} | {} | {} | {} | {} | {} | {} | {} | {} |\n", + row.question_id, + row.op_plan.as_deref().unwrap_or("n/a"), + display_optional_usize(row.stage_one_answer_session_rank), + display_optional_usize(row.first_answer_session_packet_rank), + display_optional_usize(row.first_rendered_answer_session_packet_rank), + answer_ref_in_context, + writer_answer_ref, + row.local_score + .map(|score| format!("{score:.1}")) + .unwrap_or_else(|| "n/a".to_string()), + row.official_label + .map(|label| label.to_string()) + .unwrap_or_else(|| "n/a".to_string()), + )); + } + fs::write(out_dir.join("trace_summary.md"), md)?; + Ok(()) +} + +fn display_optional_usize(value: Option) -> String { + value + .map(|value| value.to_string()) + .unwrap_or_else(|| "n/a".to_string()) +} + +fn json_string(value: Option<&serde_json::Value>) -> Option { + value.and_then(|value| value.as_str()).map(str::to_string) +} + +fn json_usize(value: Option<&serde_json::Value>) -> Option { + value + .and_then(|value| value.as_u64()) + .map(|value| value as usize) +} + +fn json_string_vec(value: Option<&serde_json::Value>) -> Vec { + value + .and_then(|value| value.as_array()) + .map(|items| { + items + .iter() + .filter_map(|item| item.as_str().map(str::to_string)) + .collect() + }) + .unwrap_or_default() +} + async fn run_ingest(data_path: &str, db_dir_path: &str) -> Result<()> { let dataset_file = File::open(data_path)?; let dataset: Vec = serde_json::from_reader(dataset_file)?; diff --git a/klbr-bench/src/main.rs b/klbr-bench/src/main.rs index b400137..5ecbc9d 100644 --- a/klbr-bench/src/main.rs +++ b/klbr-bench/src/main.rs @@ -214,7 +214,7 @@ async fn main() -> Result<()> { let args: Vec = env::args().collect(); if args.len() < 2 { bail!( - "usage:\n cargo run -p klbr-bench -- run --suite longmemeval-s --data [--sample ] [--sample-seed ] [--sample-mode stratified|random] [--out ] [--retrieval-only]\n cargo run -p klbr-bench -- retrieval \n cargo run -p klbr-bench -- passive-recall \n cargo run -p klbr-bench -- router \n cargo run -p klbr-bench -- router-multi [dataset2.json ...]\n cargo run -p klbr-bench -- router-multi-linear [dataset2.json ...]\n cargo run -p klbr-bench -- longmem [llm_url]\n cargo run -p klbr-bench -- longmem-retrieval [split.json subset]\n cargo run -p klbr-bench -- continuous-loop \n cargo run -p klbr-bench -- dump-tools \n cargo run -p klbr-bench -- sweep [score_start score_end score_step margin_start margin_end margin_step [support_start support_end support_step]]" + "usage:\n cargo run -p klbr-bench -- run --suite longmemeval-s --data [--sample ] [--sample-seed ] [--sample-mode stratified|random] [--out ] [--retrieval-only]\n cargo run -p klbr-bench -- trace-diff [question_id ...]\n cargo run -p klbr-bench -- retrieval \n cargo run -p klbr-bench -- passive-recall \n cargo run -p klbr-bench -- router \n cargo run -p klbr-bench -- router-multi [dataset2.json ...]\n cargo run -p klbr-bench -- router-multi-linear [dataset2.json ...]\n cargo run -p klbr-bench -- longmem [llm_url]\n cargo run -p klbr-bench -- longmem-retrieval [split.json subset]\n cargo run -p klbr-bench -- continuous-loop \n cargo run -p klbr-bench -- dump-tools \n cargo run -p klbr-bench -- sweep [score_start score_end score_step margin_start margin_end margin_step [support_start support_end support_step]]" ); } @@ -228,6 +228,7 @@ async fn main() -> Result<()> { run_retrieval_command(&args[2], &args[3], &args[4]).await } "run" => longmemeval::run_pipeline_command(&args).await, + "trace-diff" => longmemeval::run_trace_diff_command(&args[2..]), "passive-recall" => { if args.len() != 5 { bail!( @@ -706,6 +707,7 @@ async fn run_continuous_loop_query( query_id: case.query_id.clone(), text: case.text.clone(), reference_time: Some(5_000), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; let context = pipeline diff --git a/klbr-core/src/agent.rs b/klbr-core/src/agent.rs index e2b65b4..96dada7 100644 --- a/klbr-core/src/agent.rs +++ b/klbr-core/src/agent.rs @@ -1454,6 +1454,7 @@ async fn recall_evidence_packets( query_id: "runtime-passive-recall".to_string(), text: query.to_string(), reference_time, + op_hint: None, }, budget, ) diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs index c6bad4a..5e02318 100644 --- a/klbr-core/src/evidence.rs +++ b/klbr-core/src/evidence.rs @@ -486,8 +486,6 @@ pub struct EvidenceCoverageTrace { pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> String { let max_chars = max_tokens.saturating_mul(4); - let body_count = packet.bodies.len().max(1); - let per_body_chars = max_chars / body_count; let lanes = packet .lanes .iter() @@ -507,54 +505,80 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str xml_escape(packet.session_id.as_deref().unwrap_or("")), packet.estimated_tokens.min(max_tokens), ); - if !packet.timeline.is_empty() { - out.push_str(" \n"); - for event in packet.timeline.iter().take(8) { - out.push_str(&format!( - " {}\n", - xml_escape(&event.ref_id), - event.timestamp, - event.ordinal, - xml_escape(event.session_id.as_deref().unwrap_or("")), - xml_escape(&truncate_chars(&event.text, 96)), - )); - } - out.push_str(" \n"); - } if !packet.fact_rows.is_empty() { - out.push_str(" \n"); - for fact in packet.fact_rows.iter().take(8) { - let mut attrs = format!( - "ref=\"{}\" entity_type=\"{}\" ord=\"{}\" session=\"{}\"", - xml_escape(&fact.ref_id), - xml_escape(&fact.entity_type), - fact.ordinal, - xml_escape(fact.session_id.as_deref().unwrap_or("")) - ); - if let Some(timestamp) = fact.timestamp { - attrs.push_str(&format!(" t=\"{}\"", timestamp)); - } - if let Some(value) = &fact.value_text { - attrs.push_str(&format!(" value=\"{}\"", xml_escape(value))); + let open = " \n"; + let close = " \n"; + if out.chars().count() + open.len() + close.len() + " \n".len() <= max_chars { + out.push_str(open); + for fact in packet.fact_rows.iter().take(8) { + let mut attrs = format!( + "ref=\"{}\" entity_type=\"{}\" ord=\"{}\" session=\"{}\"", + xml_escape(&fact.ref_id), + xml_escape(&fact.entity_type), + fact.ordinal, + xml_escape(fact.session_id.as_deref().unwrap_or("")) + ); + if let Some(timestamp) = fact.timestamp { + attrs.push_str(&format!(" t=\"{}\"", timestamp)); + } + if let Some(value) = &fact.value_text { + attrs.push_str(&format!(" value=\"{}\"", xml_escape(value))); + } + if let Some(value) = fact.value_number { + attrs.push_str(&format!(" number=\"{}\"", value)); + } + let line = format!( + " {}\n", + attrs, + xml_escape(&truncate_chars(&fact.text, 120)), + ); + if out.chars().count() + line.chars().count() + close.len() + " \n".len() + > max_chars + { + break; + } + out.push_str(&line); } - if let Some(value) = fact.value_number { - attrs.push_str(&format!(" number=\"{}\"", value)); + out.push_str(close); + } + } + if !packet.timeline.is_empty() { + let open = " \n"; + let close = " \n"; + if out.chars().count() + open.len() + close.len() + " \n".len() <= max_chars { + out.push_str(open); + for event in packet.timeline.iter().take(8) { + let line = format!( + " {}\n", + xml_escape(&event.ref_id), + event.timestamp, + event.ordinal, + xml_escape(event.session_id.as_deref().unwrap_or("")), + xml_escape(&truncate_chars(&event.text, 96)), + ); + if out.chars().count() + line.chars().count() + close.len() + " \n".len() + > max_chars + { + break; + } + out.push_str(&line); } - out.push_str(&format!( - " {}\n", - attrs, - xml_escape(&truncate_chars(&fact.text, 120)), - )); + out.push_str(close); } - out.push_str(" \n"); } + let closing_chars = " \n".len(); for (index, body) in packet.bodies.iter().enumerate() { let ref_id = packet.refs.get(index).unwrap_or(&packet.anchor_ref); - out.push_str(&format!( - " {}\n", - xml_escape(ref_id), - xml_escape(&truncate_chars(body, per_body_chars)), - )); + let prefix = format!(" ", xml_escape(ref_id)); + let suffix = "\n"; + let remaining_chars = max_chars.saturating_sub(out.chars().count() + closing_chars); + if remaining_chars <= prefix.chars().count() + suffix.len() { + break; + } + let body_chars = remaining_chars - prefix.chars().count() - suffix.len(); + out.push_str(&prefix); + out.push_str(&xml_escape(&truncate_chars(body, body_chars))); + out.push_str(suffix); } out.push_str(" \n"); out @@ -1005,7 +1029,10 @@ fn truncate_chars(value: &str, max_chars: usize) -> String { if value.chars().count() <= max_chars { return value.to_string(); } - let mut out = value.chars().take(max_chars).collect::(); + if max_chars <= 3 { + return ".".repeat(max_chars); + } + let mut out = value.chars().take(max_chars - 3).collect::(); out.push_str("..."); out } @@ -1082,6 +1109,58 @@ mod tests { assert!(rendered.contains("")); + assert!(rendered.contains("= max_refs { - return Ok(out); + collected.push(chunk_ref); } } } + let anchor_is_active = collected.iter().any(|chunk_ref| chunk_ref == &canonical); + let mut out = collected.into_iter().take(max_refs).collect::>(); + if anchor_is_active && !out.iter().any(|chunk_ref| chunk_ref == &canonical) { + if out.len() >= max_refs { + out.pop(); + } + out.push(canonical); + } Ok(out) } diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index cd1feb7..0eb9acf 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -52,6 +52,8 @@ pub struct BenchQuery { pub query_id: String, pub text: String, pub reference_time: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub op_hint: Option, } #[derive(Debug, Clone, Copy, serde::Serialize, serde::Deserialize)] @@ -238,7 +240,14 @@ impl MemoryPipeline { self } - async fn plan_query_op(&self, query: &str) -> OpPlan { + async fn plan_query_op(&self, query: &BenchQuery) -> OpPlan { + if let Some(op) = query.op_hint { + return OpPlan::for_op(op, "query_op_hint"); + } + self.plan_query_text(&query.text).await + } + + async fn plan_query_text(&self, query: &str) -> OpPlan { if self.llm.config.llm.url.trim().is_empty() || self.llm.config.llm.model.trim().is_empty() { return OpPlan::default(); @@ -387,7 +396,7 @@ impl MemoryPipeline { .collect::>(); let route = route_memory_query(&query.text, !exact_refs.is_empty()); let routed_lanes = route.lanes.clone(); - let op_plan = self.plan_query_op(&query.text).await; + let op_plan = self.plan_query_op(&query).await; let candidate_limit = op_plan.candidate_limit(budget.top_k); let packet_limit = op_plan.packet_limit(budget.top_k); let packet_token_budget = op_plan.packet_token_budget(budget.max_tokens); @@ -520,7 +529,7 @@ impl MemoryPipeline { ) -> Result { self.memory.sync_reference_indexes()?; let route = route_memory_query(&query.text, false); - let op_plan = self.plan_query_op(&query.text).await; + let op_plan = self.plan_query_op(&query).await; let lanes = route.lanes.clone(); let candidate_limit = op_plan.candidate_limit(budget.top_k); let packet_limit = op_plan.packet_limit(budget.top_k); @@ -960,7 +969,7 @@ fn rank_session_first_candidates( session_candidates.push(trace); } let group_take = if group.session_id.is_some() { 3 } else { 1 }; - for candidate in group.ordered_candidates().into_iter().take(group_take) { + for candidate in group.representative_candidates(group_take) { if seen_refs.insert(candidate.ref_id.clone()) { out.push(candidate); } @@ -1046,13 +1055,43 @@ impl CandidateSessionGroup { } } - fn ordered_candidates(mut self) -> Vec { + fn representative_candidates(mut self, limit: usize) -> Vec { self.candidates.sort_by(|left, right| { candidate_order(left) .cmp(&candidate_order(right)) .then_with(|| left.ref_id.cmp(&right.ref_id)) }); - self.candidates + let mut out = Vec::new(); + let mut seen_refs = HashSet::new(); + + if let Some(anchor) = self + .candidates + .iter() + .find(|candidate| is_session_card_candidate(candidate)) + { + seen_refs.insert(anchor.ref_id.clone()); + out.push(anchor.clone()); + } + let target_turn_chunks = limit.saturating_sub(out.len()).min(2); + for turn_chunk in self + .candidates + .iter() + .filter(|candidate| candidate.entity_type == EntityType::TurnChunk) + .take(target_turn_chunks) + { + if seen_refs.insert(turn_chunk.ref_id.clone()) { + out.push(turn_chunk.clone()); + } + } + for candidate in self.candidates { + if out.len() >= limit { + break; + } + if seen_refs.insert(candidate.ref_id.clone()) { + out.push(candidate); + } + } + out } fn source_names(&self) -> Vec { @@ -1829,6 +1868,109 @@ mod tests { .any(|candidate| candidate.ref_id == "s1_chunk_fts")); } + #[test] + fn session_first_keeps_turn_chunk_when_episode_cards_dominate_group() { + let (ranked, trace) = rank_session_first_candidates( + vec![ + seed_atom( + "s1_episode_a", + Some("s1"), + EntityType::Episode, + RetrievalSource::Dense, + 1, + MemoryLane::Episodic, + ), + seed_atom( + "s1_episode_b", + Some("s1"), + EntityType::Episode, + RetrievalSource::Dense, + 2, + MemoryLane::Episodic, + ), + seed_atom( + "s1_episode_c", + Some("s1"), + EntityType::Episode, + RetrievalSource::Dense, + 3, + MemoryLane::Episodic, + ), + seed_atom( + "s1_context_chunk", + Some("s1"), + EntityType::TurnChunk, + RetrievalSource::Fts, + 1, + MemoryLane::Episodic, + ), + seed_atom( + "s1_answer_chunk", + Some("s1"), + EntityType::TurnChunk, + RetrievalSource::Sparse, + 1, + MemoryLane::Episodic, + ), + ], + 3, + ); + + assert_eq!(trace.session_candidates[0].anchor_ref, "s1_episode_a"); + assert_eq!( + ranked + .iter() + .map(|candidate| candidate.ref_id.as_str()) + .collect::>(), + vec!["s1_episode_a", "s1_answer_chunk", "s1_context_chunk"] + ); + } + + #[tokio::test] + async fn turn_window_keeps_anchor_when_previous_turn_chunks_fill_cap() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 1024)?; + let llm = LlmClient::new(crate::models::ModelsConfig::default()); + let pipeline = MemoryPipeline::new(store.clone(), llm, MemoryConfig::default()) + .with_profile("klbr-full"); + let previous_chunks = (0..12) + .map(|idx| format!("previous shampoo note {idx}")) + .collect::>() + .join("\n\n"); + + let write = pipeline + .observe_session(BenchSession { + session_id: "answer-anchor-window".to_string(), + timestamp: Some(1), + turns: vec![ + BenchTurn { + role: "user".to_string(), + content: previous_chunks, + timestamp: Some(1), + }, + BenchTurn { + role: "assistant".to_string(), + content: "current shampoo brand is Trader Joe's.".to_string(), + timestamp: Some(2), + }, + ], + }) + .await?; + assert!(!write.source_refs.is_empty()); + let canonical_anchor = store + .active_promptable_refs(&[MemoryLane::Episodic], 64)? + .into_iter() + .find(|entry| entry.entity_type == "turn_chunk" && entry.body.contains("Trader Joe's")) + .expect("anchor turn should be promptable") + .ref_id; + + let refs = store.turn_window_ref_ids(&canonical_anchor, 1, 0, 10)?; + + assert_eq!(refs.len(), 10); + assert!(refs.contains(&canonical_anchor)); + Ok(()) + } + #[tokio::test] async fn turn_window_packet_includes_next_turn_answer() -> Result<()> { let tmp = NamedTempFile::new()?; @@ -1866,6 +2008,7 @@ mod tests { query_id: "q-play".to_string(), text: "what play did I recently go to at the community theater?".to_string(), reference_time: Some(3), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; let turn_packet = retrieval @@ -1919,6 +2062,7 @@ mod tests { query_id: "q-coupon".to_string(), text: "what store was the inbox coupon for?".to_string(), reference_time: Some(3), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; let packet = retrieval @@ -1966,6 +2110,7 @@ mod tests { query_id: "q-ref-graph".to_string(), text: format!("what support is linked from [{anchor_alias}]?"), reference_time: Some(3), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; @@ -2078,6 +2223,7 @@ mod tests { query_id: "q-multi".to_string(), text: "cual es el codigo pinata?".to_string(), reference_time: Some(3), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; let context = pipeline.assemble_context(query, &retrieval, budget).await?; @@ -2115,6 +2261,7 @@ mod tests { query_id: "q-cjk".to_string(), text: "図書館裏".to_string(), reference_time: Some(11), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; let context = pipeline.assemble_context(query, &retrieval, budget).await?; @@ -2156,6 +2303,7 @@ mod tests { query_id: "q-lookup".to_string(), text: "where did I buy coffee creamer?".to_string(), reference_time: Some(2), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query, budget).await?; @@ -2165,6 +2313,45 @@ mod tests { Ok(()) } + #[tokio::test] + async fn query_op_hint_activates_collect_mode_without_structured_planner() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + let llm = LlmClient::new(crate::models::ModelsConfig::default()); + let pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) + .with_profile("raw-turns/fts-only"); + let budget = ContextBudget { + max_tokens: 1_000, + top_k: 2, + graph_depth: 0, + }; + + pipeline + .observe_session(BenchSession { + session_id: "hint-session".to_string(), + timestamp: Some(1), + turns: vec![BenchTurn { + role: "user".to_string(), + content: "the old value was 27:45 and the latest value is 26:30".to_string(), + timestamp: Some(1), + }], + }) + .await?; + + let query = BenchQuery { + query_id: "q-hint".to_string(), + text: "which value should I use?".to_string(), + reference_time: Some(2), + op_hint: Some(QueryOp::UpdateResolution), + }; + let retrieval = pipeline.retrieve_evidence(query, budget).await?; + + assert_eq!(retrieval.op_plan.op, QueryOp::UpdateResolution); + assert!(retrieval.op_plan.collect_all); + assert_eq!(retrieval.op_plan.reason, "query_op_hint"); + Ok(()) + } + #[tokio::test] async fn false_recall_adversary_keeps_unmatched_packets_empty() -> Result<()> { let tmp = NamedTempFile::new()?; @@ -2194,6 +2381,7 @@ mod tests { query_id: "q-adversary".to_string(), text: "volcanic obsidian necklace serial number".to_string(), reference_time: Some(3), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query, budget).await?; @@ -2225,6 +2413,7 @@ mod tests { query_id: "q-exact".to_string(), text: format!("what does [{alias}] say?"), reference_time: Some(3), + op_hint: None, }; let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; let context = pipeline.assemble_context(query, &retrieval, budget).await?; diff --git a/klbr-core/src/planner.rs b/klbr-core/src/planner.rs index b395659..b5297ac 100644 --- a/klbr-core/src/planner.rs +++ b/klbr-core/src/planner.rs @@ -102,28 +102,61 @@ impl OpPlan { pub const STRUCTURED_QUERY_PLANNER_PROMPT: &str = r#"classify the user's memory question by the abstract operation needed to answer it. return only a json object with: {"op":"lookup|aggregate_count|aggregate_sum|aggregate_avg|order_or_rank|update_resolution|preference_recommendation|abstain_or_false_premise_check","reason":"short rationale"} -the classifier must work across languages and scripts. do not use keyword matching. do not answer the question."#; -#[derive(Debug, serde::Deserialize)] -struct StructuredPlannerReply { - op: QueryOp, - #[serde(default)] - reason: Option, -} +operation boundaries: +- lookup: one remembered fact is enough. +- aggregate_count/sum/avg: the answer needs a count, total, or average over remembered facts. +- order_or_rank: the answer needs chronology, comparison, first/last ordering, or elapsed-time evidence. +- update_resolution: the answer asks for a current/latest/previous state, habit, preference, or value that may have changed across memories. +- preference_recommendation: the answer should preserve the user's remembered constraints or likes/dislikes. +- abstain_or_false_premise_check: the question may be unsupported or false-premise. + +reason semantically across languages and scripts. examples above define operations, not keyword rules. do not answer the question."#; pub fn parse_structured_op_plan(reply: &str) -> Option { let json = extract_json_object(reply)?; - let parsed = serde_json::from_str::(&json).ok()?; + let parsed = serde_json::from_str::(&json).ok()?; + let op = parsed + .get("op") + .or_else(|| parsed.get("operation")) + .or_else(|| parsed.get("query_op")) + .and_then(|value| value.as_str()) + .and_then(parse_query_op_label)?; let reason = parsed - .reason - .as_deref() + .get("reason") + .or_else(|| parsed.get("rationale")) + .and_then(|value| value.as_str()) .map(sanitize_reason) .filter(|reason| !reason.is_empty()) .unwrap_or_else(|| "model_structured".to_string()); - Some(OpPlan::for_op( - parsed.op, - format!("model_structured:{reason}"), - )) + Some(OpPlan::for_op(op, format!("model_structured:{reason}"))) +} + +fn parse_query_op_label(value: &str) -> Option { + let normalized = value + .chars() + .filter(|ch| ch.is_ascii_alphanumeric()) + .flat_map(char::to_lowercase) + .collect::(); + match normalized.as_str() { + "lookup" | "directlookup" | "direct" => Some(QueryOp::Lookup), + "aggregatecount" | "count" | "countall" => Some(QueryOp::AggregateCount), + "aggregatesum" | "sum" | "total" | "totalcost" => Some(QueryOp::AggregateSum), + "aggregateavg" | "aggregateaverage" | "avg" | "average" => Some(QueryOp::AggregateAvg), + "orderorrank" | "order" | "rank" | "chronology" | "temporalorder" => { + Some(QueryOp::OrderOrRank) + } + "updateresolution" | "update" | "currentstate" | "latest" | "previouslatest" => { + Some(QueryOp::UpdateResolution) + } + "preferencerecommendation" | "preference" | "recommendation" => { + Some(QueryOp::PreferenceRecommendation) + } + "abstainorfalsepremisecheck" | "abstain" | "falsepremise" | "insufficientinfo" => { + Some(QueryOp::AbstainOrFalsePremiseCheck) + } + _ => None, + } } fn extract_json_object(value: &str) -> Option { @@ -185,6 +218,18 @@ mod tests { assert!(plan.reason.starts_with("model_structured:")); } + #[test] + fn parses_tolerant_structured_model_reply() { + let plan = parse_structured_op_plan( + r#"sure: {"operation":"count", "rationale":"needs distinct items"}"#, + ) + .expect("tolerant structured plan should parse"); + + assert_eq!(plan.op, QueryOp::AggregateCount); + assert!(plan.collect_all); + assert!(plan.reason.contains("needs distinct items")); + } + #[test] fn ignores_non_json_model_reply() { assert!(parse_structured_op_plan("lookup probably").is_none());