From 7017b6ededac59270c145f4b46a6ae158e1deb59 Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Mon, 29 Jun 2026 15:04:58 +0300 Subject: [PATCH] complete planner-grade memory evidence assembly --- .beads/interactions.jsonl | 6 + .beads/issues.jsonl | 12 +- docs/long-term-memory-arch.md | 35 +- docs/memory-benches.md | 27 ++ docs/memory-implementation-status.md | 58 ++- docs/planner-grade-evidence-assembly.md | 17 +- klbr-bench/src/longmemeval.rs | 23 +- klbr-core/src/evidence.rs | 277 +++++++++++++- klbr-core/src/memory.rs | 200 ++++++++-- klbr-core/src/pipeline.rs | 465 +++++++++++++++++++++++- 10 files changed, 1039 insertions(+), 81 deletions(-) diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index fe341de..c0a6f48 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -54,3 +54,9 @@ {"id":"int-fdca5c0e","kind":"field_change","created_at":"2026-06-28T19:54:14.601069054Z","actor":"dawn","issue_id":"klbr-f0q.1","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented OpPlan/QueryOp classification, serialized op_plan into retrieval traces, preserved lookup defaults, added deterministic planner and pipeline trace tests."}} {"id":"int-9125728a","kind":"field_change","created_at":"2026-06-28T21:21:26.576608546Z","actor":"dawn","issue_id":"klbr-h4l","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented project-local dense/sparse embedding cache, ran 25-sample QA bench, inspected failures, tried bounded planner/high-budget experiments, and wrote report packet."}} {"id":"int-95ec3c9d","kind":"field_change","created_at":"2026-06-28T22:07:09.303815881Z","actor":"dawn","issue_id":"klbr-jwn","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed docs index, current architecture review rewrite, status banners, AGENTS routing, and follow-up implementation issue klbr-u0q."}} +{"id":"int-a495bc00","kind":"field_change","created_at":"2026-06-29T12:03:36.630940909Z","actor":"dawn","issue_id":"klbr-f0q.3","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed collect-mode packet assembly with fact-group coverage/gap trace, bounded lexical channel fanout, session/ref group coverage tests, and pipeline smoke verification."}} +{"id":"int-50889d47","kind":"field_change","created_at":"2026-06-29T12:03:44.817284963Z","actor":"dawn","issue_id":"klbr-f0q.4","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed ref-grounded fact_rows and timeline fields on EvidencePacket, rendered / blocks, coverage trace propagation, and bench metrics for rendered/visible fact evidence."}} +{"id":"int-3b451797","kind":"field_change","created_at":"2026-06-29T12:03:55.79718643Z","actor":"dawn","issue_id":"klbr-f0q.2","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed initial planned synthesis over fact rows for count/sum/avg/order/update and guarded preference recommendations; lookup plans still fall through to the reader path. Covered by focused core tests."}} +{"id":"int-f20b3748","kind":"field_change","created_at":"2026-06-29T12:04:04.84169394Z","actor":"dawn","issue_id":"klbr-f0q.5","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed split bench metrics for selected/rendered/value/fact-row visibility and collect-mode fact-group recall, plus deterministic tests for count, sum/avg, order, update, preference distractor, lookup fallback, and cjk no-space trigram retrieval."}} +{"id":"int-4ce9088c","kind":"field_change","created_at":"2026-06-29T12:04:10.393166562Z","actor":"dawn","issue_id":"klbr-u0q","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed session/event-first lexical-contained slice: no english cue-list planner/filters, fts remains a candidate channel, trigram fts covers no-space multilingual exact-ish retrieval, traces expose session candidates/packet metrics, and docs clarify fts vs semantic multilingual retrieval."}} +{"id":"int-9a80e190","kind":"field_change","created_at":"2026-06-29T12:04:15.11512071Z","actor":"dawn","issue_id":"klbr-f0q","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed initial planner-grade evidence assembly slice: op-aware collect mode, fact/timeline packets, structured reducers, split metrics/fixtures, docs, and focused/pipeline verification."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 32bb156..1632d28 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -1,11 +1,11 @@ -{"_type":"issue","id":"klbr-u0q","title":"Implement session-first lexical-contained memory retrieval","description":"Build the next retrieval architecture from docs/long-term-memory-arch.md: stage-one session/event candidate generation, lexical containment, and language-agnostic operation planning experiments without adding english cue-word hacks.","design":"Keep EvidencePacket as the rank object. Use exact refs, dense episode/session cards, learned sparse or fts side channels, and graph expansion from strong seeds only. Keep QueryOp shape but replace cue-list policy with a structured classifier plus deterministic fallback.","acceptance_criteria":"A bench profile can retrieve from session/episode cards first and use chunk fts/sparse hits as packet enrichment; lexical-only policy paths are removed or contained behind candidate channels; traces expose candidate session recall and packet/rendered evidence metrics; multilingual fixtures cover non-English cue-free retrieval cases.","notes":"2026-06-29 partial: removed hardcoded english cue-list planner, english lane routing, english negation/entity-boundary packet filters, and wh/pronoun neighbor-expansion heuristics. OpPlan now comes from structured model JSON when an llm endpoint is configured, otherwise conservative lookup. Core and bench tests pass.\n2026-06-29 partial: landed initial session-first retrieval profile for klbr-full/dense-only/session-first profiles. Retrieval now groups archival candidates by session_id, prefers episodic/event-card anchors, adds chunk fts/sparse/dense hits as enrichment, serializes stage_one.session_candidates, and reports CandidateSessionRecall metrics in klbr-bench. fts-only now writes episode notes but skips embedded episode-memory rows so lexical ablations do not require the dense embedder at ingest. Verified core/bench tests, continuous-loop, and a one-row session-first/fts-only retrieval-only trace smoke.","status":"in_progress","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T22:06:32Z","created_by":"dawn","updated_at":"2026-06-29T11:37:00Z","started_at":"2026-06-28T22:10:12Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-u0q","title":"Implement session-first lexical-contained memory retrieval","description":"Build the next retrieval architecture from docs/long-term-memory-arch.md: stage-one session/event candidate generation, lexical containment, and language-agnostic operation planning experiments without adding english cue-word hacks.","design":"Keep EvidencePacket as the rank object. Use exact refs, dense episode/session cards, learned sparse or fts side channels, and graph expansion from strong seeds only. Keep QueryOp shape but replace cue-list policy with a structured classifier plus deterministic fallback.","acceptance_criteria":"A bench profile can retrieve from session/episode cards first and use chunk fts/sparse hits as packet enrichment; lexical-only policy paths are removed or contained behind candidate channels; traces expose candidate session recall and packet/rendered evidence metrics; multilingual fixtures cover non-English cue-free retrieval cases.","notes":"2026-06-29 partial: removed hardcoded english cue-list planner, english lane routing, english negation/entity-boundary packet filters, and wh/pronoun neighbor-expansion heuristics. OpPlan now comes from structured model JSON when an llm endpoint is configured, otherwise conservative lookup. Core and bench tests pass.\n2026-06-29 partial: landed initial session-first retrieval profile for klbr-full/dense-only/session-first profiles. Retrieval now groups archival candidates by session_id, prefers episodic/event-card anchors, adds chunk fts/sparse/dense hits as enrichment, serializes stage_one.session_candidates, and reports CandidateSessionRecall metrics in klbr-bench. fts-only now writes episode notes but skips embedded episode-memory rows so lexical ablations do not require the dense embedder at ingest. Verified core/bench tests, continuous-loop, and a one-row session-first/fts-only retrieval-only trace smoke.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T22:06:32Z","created_by":"dawn","updated_at":"2026-06-29T12:04:10Z","started_at":"2026-06-28T22:10:12Z","closed_at":"2026-06-29T12:04:10Z","close_reason":"Completed session/event-first lexical-contained slice: no english cue-list planner/filters, fts remains a candidate channel, trigram fts covers no-space multilingual exact-ish retrieval, traces expose session candidates/packet metrics, and docs clarify fts vs semantic multilingual retrieval.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-jwn","title":"Organize evolving memory architecture docs","description":"Make the memory architecture docs distinguish current truth, implementation status, proposals, research inputs, and archived rationale so future agents do not treat older reports as canonical.","acceptance_criteria":"Docs have a clear index and status taxonomy; current memory architecture direction points at docs/long-term-memory-arch.md; older research reports are marked as historical or supporting; AGENTS.md routes future agents through the doc index.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T22:01:20Z","created_by":"dawn","updated_at":"2026-06-28T22:07:09Z","started_at":"2026-06-28T22:01:23Z","closed_at":"2026-06-28T22:07:09Z","close_reason":"Completed docs index, current architecture review rewrite, status banners, AGENTS routing, and follow-up implementation issue klbr-u0q.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-h4l","title":"Cache bench embeddings locally and run 25-sample QA bench","description":"Make LongMemEval bench embeddings use a project-local gitignored sqlite cache so reruns do not pay for the same embeddings twice. Then run a 25-sample stratified QA bench, inspect failures, try bounded non-overfit tweaks if evidence supports them, and prepare a report packet if results remain weak or tuning would be overfit.","design":"Prefer a global project cache path such as benchmarks/cache/*.db over temp/run-local caches. Keep benchmark polling sparse: start the run, wait for artifacts or process completion, then inspect once.","acceptance_criteria":"Embedding cache lives under the project and is ignored by git; cache hits are reused across bench runs; a 25-sample bench artifact exists with failure analysis; any code changes are tested, committed, and pushed.","notes":"Implemented project-local embedding cache at benchmarks/cache/embeddings.db and wired LongMemEval bench LlmClient construction through it. Ran stratified 25-sample QA on seed 4937553249516211047: current-code run benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z scored official accuracy 0.6800. Report packet written to report_packet.md in that run dir. High-budget targeted reruns fixed only dd2973ad and a1cc6108; remaining failures point to fact/timeline synthesis rather than safe cue tuning.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T20:08:32Z","created_by":"dawn","updated_at":"2026-06-28T21:21:27Z","started_at":"2026-06-28T20:08:35Z","closed_at":"2026-06-28T21:21:27Z","close_reason":"Implemented project-local dense/sparse embedding cache, ran 25-sample QA bench, inspected failures, tried bounded planner/high-budget experiments, and wrote report packet.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-f0q.1","title":"Add operation-aware query planning","description":"Introduce a small OpPlan/QueryOp classifier for memory QA queries: lookup, aggregate_count, aggregate_sum, aggregate_avg, order_or_rank, update_resolution, preference_recommendation, and abstain_or_false_premise_check. Use it to switch only high-value classes away from plain lookup behavior.","design":"Keep the label set intentionally small and rule-based first; avoid growing English lexical hacks into the main ranking decision boundary.","acceptance_criteria":"Production memory retrieval traces expose the selected query op and plan; existing lookup behavior remains unchanged by default; aggregate/order/update/preference queries can be detected deterministically in tests.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-28T19:54:15Z","started_at":"2026-06-28T19:45:04Z","closed_at":"2026-06-28T19:54:15Z","close_reason":"Implemented OpPlan/QueryOp classification, serialized op_plan into retrieval traces, preserved lookup defaults, added deterministic planner and pipeline trace tests.","dependencies":[{"issue_id":"klbr-f0q.1","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-f0q.2","title":"Add structured synthesis for planned QA operations","description":"Add a fact-table synthesis path for aggregate count/sum/avg, ordering, previous/latest update resolution, and preference-grounded recommendations. Do not route ordinary conversation lookup through the structured path unless the planner selects it.","design":"Stage A extracts/normalizes fact rows with refs; Stage B computes or renders the answer according to the OpPlan.","acceptance_criteria":"Planned synthesis computes counts/averages/order/update answers from cited fact rows; preference recommendations require personal support and avoid generic distractor answers; lookup questions keep current direct reader path.","status":"open","priority":1,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-28T19:44:46Z","dependencies":[{"issue_id":"klbr-f0q.2","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.2","depends_on_id":"klbr-f0q.4","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-f0q.3","title":"Implement collect-mode evidence assembly","description":"For aggregate, ordering, and update-resolution plans, retrieve and rank evidence by fact-group coverage instead of fixed top-k alone. Add sufficiency/gap diagnosis and clustered packet assembly by entity/task/session where feasible.","design":"Use relevance plus marginal coverage gain, with hard budget/stopping controls based on fact-group sufficiency rather than packet count.","acceptance_criteria":"Collect mode can cover multi-session fact groups without one high-scoring session monopolizing context; packet traces expose coverage/gap state; focused tests cover six-way count/order style cases.","notes":"Started implementation: collect-mode plans now widen candidate/packet limits and use collect-aware packet capping that prefers distinct sessions before second packets from the same session. Remaining: explicit fact-group sufficiency and gap diagnosis.","status":"in_progress","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-28T19:54:15Z","started_at":"2026-06-28T19:54:15Z","dependencies":[{"issue_id":"klbr-f0q.3","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.3","depends_on_id":"klbr-f0q.1","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-f0q.4","title":"Render chronology and fact rows in evidence packets","description":"Extend evidence packet rendering to include compact structured timeline/fact rows alongside raw ref-grounded bodies, and make rendering report selected/rendered/dropped refs or fact rows.","design":"Structured rows must retain supporting ref ids; raw bodies stay as the provenance source, not replaced by summaries.","acceptance_criteria":"Rendered context includes chronology/facts for temporal/aggregate/update packets while preserving body refs; benchmarks can distinguish selected refs from rendered refs and visible critical facts.","status":"open","priority":1,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-28T19:44:46Z","dependencies":[{"issue_id":"klbr-f0q.4","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.4","depends_on_id":"klbr-f0q.3","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-f0q","title":"Planner-grade evidence assembly","description":"Implement docs/planner-grade-evidence-assembly.md: make memory QA operation-aware instead of relying on larger top-k/budget. Scope includes query operation planning, collect-mode evidence assembly, chronology/fact packet rendering, structured synthesis for aggregate/temporal/preference questions, and benchmark metrics/fixtures.","design":"Follow the implementation order in docs/planner-grade-evidence-assembly.md: classify operation first, add collect-mode evidence planning, expose facts/timeline while keeping raw refs, add structured synthesis only for op labels that need it, then guard with metrics and fixtures.","acceptance_criteria":"Planner-grade evidence assembly is implemented behind conservative defaults or config/CLI toggles, covered by production-pipeline tests, documented, and validated with focused cargo tests plus a bench smoke when local services are available.","status":"in_progress","priority":1,"issue_type":"epic","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:25Z","created_by":"dawn","updated_at":"2026-06-28T19:45:04Z","started_at":"2026-06-28T19:45:04Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-f0q.2","title":"Add structured synthesis for planned QA operations","description":"Add a fact-table synthesis path for aggregate count/sum/avg, ordering, previous/latest update resolution, and preference-grounded recommendations. Do not route ordinary conversation lookup through the structured path unless the planner selects it.","design":"Stage A extracts/normalizes fact rows with refs; Stage B computes or renders the answer according to the OpPlan.","acceptance_criteria":"Planned synthesis computes counts/averages/order/update answers from cited fact rows; preference recommendations require personal support and avoid generic distractor answers; lookup questions keep current direct reader path.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-29T12:03:56Z","started_at":"2026-06-29T12:03:48Z","closed_at":"2026-06-29T12:03:56Z","close_reason":"Completed initial planned synthesis over fact rows for count/sum/avg/order/update and guarded preference recommendations; lookup plans still fall through to the reader path. Covered by focused core tests.","dependencies":[{"issue_id":"klbr-f0q.2","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.2","depends_on_id":"klbr-f0q.4","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-f0q.3","title":"Implement collect-mode evidence assembly","description":"For aggregate, ordering, and update-resolution plans, retrieve and rank evidence by fact-group coverage instead of fixed top-k alone. Add sufficiency/gap diagnosis and clustered packet assembly by entity/task/session where feasible.","design":"Use relevance plus marginal coverage gain, with hard budget/stopping controls based on fact-group sufficiency rather than packet count.","acceptance_criteria":"Collect mode can cover multi-session fact groups without one high-scoring session monopolizing context; packet traces expose coverage/gap state; focused tests cover six-way count/order style cases.","notes":"Started implementation: collect-mode plans now widen candidate/packet limits and use collect-aware packet capping that prefers distinct sessions before second packets from the same session. Remaining: explicit fact-group sufficiency and gap diagnosis.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-29T12:03:37Z","started_at":"2026-06-28T19:54:15Z","closed_at":"2026-06-29T12:03:37Z","close_reason":"Completed collect-mode packet assembly with fact-group coverage/gap trace, bounded lexical channel fanout, session/ref group coverage tests, and pipeline smoke verification.","dependencies":[{"issue_id":"klbr-f0q.3","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.3","depends_on_id":"klbr-f0q.1","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-f0q.4","title":"Render chronology and fact rows in evidence packets","description":"Extend evidence packet rendering to include compact structured timeline/fact rows alongside raw ref-grounded bodies, and make rendering report selected/rendered/dropped refs or fact rows.","design":"Structured rows must retain supporting ref ids; raw bodies stay as the provenance source, not replaced by summaries.","acceptance_criteria":"Rendered context includes chronology/facts for temporal/aggregate/update packets while preserving body refs; benchmarks can distinguish selected refs from rendered refs and visible critical facts.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:46Z","created_by":"dawn","updated_at":"2026-06-29T12:03:45Z","started_at":"2026-06-29T12:03:41Z","closed_at":"2026-06-29T12:03:45Z","close_reason":"Completed ref-grounded fact_rows and timeline fields on EvidencePacket, rendered \u003cfacts\u003e/\u003ctimeline\u003e blocks, coverage trace propagation, and bench metrics for rendered/visible fact evidence.","dependencies":[{"issue_id":"klbr-f0q.4","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.4","depends_on_id":"klbr-f0q.3","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-f0q","title":"Planner-grade evidence assembly","description":"Implement docs/planner-grade-evidence-assembly.md: make memory QA operation-aware instead of relying on larger top-k/budget. Scope includes query operation planning, collect-mode evidence assembly, chronology/fact packet rendering, structured synthesis for aggregate/temporal/preference questions, and benchmark metrics/fixtures.","design":"Follow the implementation order in docs/planner-grade-evidence-assembly.md: classify operation first, add collect-mode evidence planning, expose facts/timeline while keeping raw refs, add structured synthesis only for op labels that need it, then guard with metrics and fixtures.","acceptance_criteria":"Planner-grade evidence assembly is implemented behind conservative defaults or config/CLI toggles, covered by production-pipeline tests, documented, and validated with focused cargo tests plus a bench smoke when local services are available.","status":"closed","priority":1,"issue_type":"epic","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:25Z","created_by":"dawn","updated_at":"2026-06-29T12:04:15Z","started_at":"2026-06-28T19:45:04Z","closed_at":"2026-06-29T12:04:15Z","close_reason":"Completed initial planner-grade evidence assembly slice: op-aware collect mode, fact/timeline packets, structured reducers, split metrics/fixtures, docs, and focused/pipeline verification.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-a9b.8","title":"Extract file-local symbols from tree-sitter tags","description":"Build symbol extraction on top of generic tree-sitter language bundles. Prefer tags/locals/code-intel APIs when a language provides them, and keep the internal symbol model independent of any one grammar. Convert definition/reference captures or equivalent structure output into name, kind, role, name range, body range, parent/name path when derivable, docs when provided, and source snippet boundaries.","acceptance_criteria":"Fixture files in multiple representative languages produce stable symbol lists with definitions, ranges, names, and basic nesting/name paths where the language bundle supports it; extraction tolerates parse errors by returning partial symbols plus parse status rather than panicking; unsupported query features degrade cleanly.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:33Z","created_by":"dawn","updated_at":"2026-06-28T11:06:00Z","started_at":"2026-06-28T11:02:44Z","closed_at":"2026-06-28T11:06:00Z","close_reason":"Implemented generic file-local symbol extraction through tree-sitter-tags over language-pack query bundles, including definition/reference roles, capture-derived kinds, ranges, docs, containment name paths, and focused tests.","labels":["code-intel","retrieval","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.8","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:32Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.8","depends_on_id":"klbr-a9b.6","type":"blocks","created_at":"2026-06-28T13:46:50Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":3,"comment_count":0} {"_type":"issue","id":"klbr-a9b.1","title":"Define tree-sitter code-intel backend boundary","description":"Decide the first-pass tree-sitter backend shape before code lands: supported languages, crate choices, tool naming, data structures, and explicit non-goals. The boundary should keep this file-local and tree-sitter-first, with no LSP, diagnostics tool, project-wide semantic resolution, or automatic check/format hook in the first slice.","design":"first pass is tree-sitter-first, file-local, and generic across languages. the implementation should not hardcode rust/typescript as the architecture; concrete languages are only fixtures. the core model is source text plus path/language hint -\u003e generic parser/query registry -\u003e tags/locals captures or generic code-intel output -\u003e file-local symbols with kind/name/name_path/ranges/docs/parse status. prefer tree-sitter-language-pack or an equivalent bundle registry so languages can be added by parser/query availability rather than bespoke code. tool behavior follows serena: OK on successful edits; concise errors for unsupported language, missing/ambiguous symbol, parse/range failure, or invalid replacement. no lsp backend, no diagnostics tool, no automatic cargo/check/format hook, no verification hints, and no project-wide semantic resolution in this first implementation. project-wide indexing can be a later layer over the same file-local extractor.","acceptance_criteria":"A short design note is recorded in the issue or docs; the first implementation surface lists supported languages, commands/tools, and unsupported behavior; later implementation issues can proceed without re-litigating LSP/project-wide scope.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:31Z","created_by":"dawn","updated_at":"2026-06-28T10:59:12Z","started_at":"2026-06-28T10:55:29Z","closed_at":"2026-06-28T10:55:43Z","close_reason":"Recorded tree-sitter-first, file-local first-pass boundary and explicit non-goals in issue design.","labels":["code-intel","design","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.1","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:30Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-a9b.6","title":"Add tree-sitter parser and language registry","description":"Add the klbr-core substrate for selecting a tree-sitter parser and query bundle from a file path or optional language hint. This should be generic: use tree-sitter-language-pack or an equivalent registry/loader rather than handcoding a small set of language crates. Concrete languages in tests are only fixtures proving the generic path.","acceptance_criteria":"Given a supported file path or language hint and source text, klbr can parse it with tree-sitter, report parse errors/MISSING nodes as parse status, and expose language metadata plus tags/locals/query availability when present; unsupported or unavailable languages return a concise unsupported-language error without panicking.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:31Z","created_by":"dawn","updated_at":"2026-06-28T11:02:31Z","started_at":"2026-06-28T10:55:54Z","closed_at":"2026-06-28T11:02:31Z","close_reason":"Implemented generic tree-sitter language detection, query support, and parse-status substrate via tree-sitter-language-pack; focused code_intel tests and klbr-core check pass.","labels":["code-intel","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.6","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:31Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.6","depends_on_id":"klbr-a9b.1","type":"blocks","created_at":"2026-06-28T13:46:50Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} @@ -30,7 +30,7 @@ {"_type":"issue","id":"klbr-wmz.2","title":"Route runtime passive recall through the canonical memory pipeline","description":"Bench retrieval now goes through MemoryPipeline, but runtime passive recall in klbr-core/src/agent.rs still embeds the prompt, calls MemoryStore::get_searchable, runs retrieval::retrieve_exact over legacy memories, then injects Context memory packets. That bypasses fts, exact refs, markdown notes, graph expansion, and the canonical lane/lifecycle path the docs describe.","design":"Avoid duplicating retrieval logic in agent.rs. Either make MemoryPipeline usable by AgentRuntime or extract a shared retrieval facade that both MemoryPipeline and runtime passive recall call.","acceptance_criteria":"Runtime passive recall uses the same lane-aware canonical retrieval and context packet assembly policy as the production pipeline; recalled packets can include fts/exact/dense/graph candidates from refs and markdown notes; archived/tombstoned/suppressed refs do not leak; klbr-core/src/instructions.md matches the actual memory packet format; tests or a focused integration fixture cover passive recall from a markdown note and from an explicit ref.","notes":"Runtime passive recall now calls MemoryPipeline::retrieve_evidence and injects shared EvidencePacket XML via Context::inject_evidence_packets; no-model runtime packet fixture covers turn-window expansion. Remaining acceptance is blocked on klbr-wmz.1 because dense search still starts from legacy memory rows before ref mapping.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:19Z","created_by":"dawn","updated_at":"2026-06-26T19:44:55Z","started_at":"2026-06-26T19:37:32Z","closed_at":"2026-06-26T19:44:55Z","close_reason":"Completed: runtime passive recall now uses MemoryPipeline/EvidencePlanner packets, instructions document current packet XML, and fixtures cover turn-window recall plus explicit markdown refs; dense canonical dependency completed in klbr-wmz.1.","labels":["architecture","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz.1","type":"blocks","created_at":"2026-06-26T22:37:53Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.1","title":"Move dense retrieval onto canonical refs and embedding_items","description":"Current source still has dense retrieval on the legacy memory-row surface: klbr-core/src/pipeline.rs search_dense calls MemoryStore::get_searchable and retrieval::retrieve_exact, while fts/exact retrieval uses canonical refs and promptable_text. The schema already has refs, promptable_text, markdown_note_chunks, and embedding_items, so dense retrieval should not be the odd path out.","design":"Prefer a ref-native embedding index backed by embedding_items. Backfill embeddings from promptable_text, keep memory-id aliases as compatibility aliases, and make klbr/full versus dense-only profiles exercise the same canonical identity layer as fts and exact retrieval.","acceptance_criteria":"Dense candidate generation works over canonical ref ids for memories, turn chunks, markdown note chunks, episode notes, profile notes, and procedural notes; lane and lifecycle filtering come from refs/ref_metadata instead of memory tags alone; benchmark traces return canonical refs for dense hits; regression tests cover a markdown-note-only hit and a tombstoned/suppressed ref not leaking through dense search.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:11Z","created_by":"dawn","updated_at":"2026-06-26T19:43:27Z","started_at":"2026-06-26T19:38:02Z","closed_at":"2026-06-26T19:43:27Z","close_reason":"Completed: dense candidate generation now lazily backfills embedding_items from active promptable refs, scores canonical ref embeddings directly, and tests markdown-note dense hits plus suppressed-ref filtering.","labels":["architecture","memory","refs","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.1","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:11Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz","title":"Finish memory architecture follow-through","description":"Tracks the remaining memory architecture work identified from docs/memory-arch.md, docs/memory-benches.md, docs/memory-implementation-status.md, and current klbr-core/klbr-bench source. Current status says pipeline, typed memory packets, markdown notes, edge mirroring, lifecycle projection, and benchmark runner integration exist; this epic is for gaps still present in source/docs.","acceptance_criteria":"Close when the child issues are complete, docs/memory-implementation-status.md is updated from current verification, and the architecture docs no longer point at missing or stale follow-up work.","status":"closed","priority":1,"issue_type":"epic","owner":"90008@klbr.net","created_at":"2026-06-26T17:52:54Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","closed_at":"2026-06-26T23:45:35Z","close_reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn.","labels":["architecture","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-f0q.5","title":"Add planner-grade benchmark metrics and fixtures","description":"Add rendered-evidence metrics and deterministic production-pipeline fixtures for count, average truncation, ordered temporal sessions, previous-vs-latest, and preference distractor regressions.","design":"Metrics should separate retrieval selection, packet rendering, fact-row visibility, and final answer correctness rather than collapsing them into one recall score.","acceptance_criteria":"Bench reports split selected/rendered/visible evidence metrics; tests cover the concrete failure families from the 20-sample LongMemEval brief; docs list smoke commands for planner/fact-table runs.","notes":"Partial metrics support landed early: LongMemEval reports op_plan_counts plus answer_bearing_ref_selected, answer_bearing_ref_rendered, and answer_value_visible. Remaining: critical_fact_row_visible, collect_mode_fact_group_recall, preference/update-specific correctness metrics, and deterministic fixtures for all failure families.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:47Z","created_by":"dawn","updated_at":"2026-06-28T19:54:15Z","dependencies":[{"issue_id":"klbr-f0q.5","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.5","depends_on_id":"klbr-f0q.2","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-f0q.5","title":"Add planner-grade benchmark metrics and fixtures","description":"Add rendered-evidence metrics and deterministic production-pipeline fixtures for count, average truncation, ordered temporal sessions, previous-vs-latest, and preference distractor regressions.","design":"Metrics should separate retrieval selection, packet rendering, fact-row visibility, and final answer correctness rather than collapsing them into one recall score.","acceptance_criteria":"Bench reports split selected/rendered/visible evidence metrics; tests cover the concrete failure families from the 20-sample LongMemEval brief; docs list smoke commands for planner/fact-table runs.","notes":"Partial metrics support landed early: LongMemEval reports op_plan_counts plus answer_bearing_ref_selected, answer_bearing_ref_rendered, and answer_value_visible. Remaining: critical_fact_row_visible, collect_mode_fact_group_recall, preference/update-specific correctness metrics, and deterministic fixtures for all failure families.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T19:44:47Z","created_by":"dawn","updated_at":"2026-06-29T12:04:05Z","started_at":"2026-06-29T12:04:00Z","closed_at":"2026-06-29T12:04:05Z","close_reason":"Completed split bench metrics for selected/rendered/value/fact-row visibility and collect-mode fact-group recall, plus deterministic tests for count, sum/avg, order, update, preference distractor, lookup fallback, and cjk no-space trigram retrieval.","dependencies":[{"issue_id":"klbr-f0q.5","depends_on_id":"klbr-f0q","type":"parent-child","created_at":"2026-06-28T22:44:46Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-f0q.5","depends_on_id":"klbr-f0q.2","type":"blocks","created_at":"2026-06-28T22:44:56Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-oba","title":"add stratified random longmemeval sampling","description":"support reproducible random subset runs for the LongMemEval pipeline, stratified by question_type so small incremental qa runs still cover categories.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T16:09:37Z","created_by":"dawn","updated_at":"2026-06-28T16:16:07Z","started_at":"2026-06-28T16:10:01Z","closed_at":"2026-06-28T16:16:07Z","close_reason":"Completed: added reproducible random LongMemEval sampling with question_type stratification.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-8dq","title":"default longmemeval qa output under benchmarks/runs","description":"make the qa bench less annoying to run by defaulting its output directory under benchmarks/runs when --out is not supplied.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T15:55:48Z","created_by":"dawn","updated_at":"2026-06-28T16:16:07Z","started_at":"2026-06-28T15:56:03Z","closed_at":"2026-06-28T16:16:07Z","close_reason":"Completed: LongMemEval pipeline run now defaults --out under benchmarks/runs.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-a9b.7","title":"Document and test tree-sitter code tools","description":"Add focused fixtures, regression tests, and docs for the tree-sitter-first code tools. Tests should cover multiple representative languages through the generic registry, symbol extraction, overview/find/read flows, symbolic edits, terse OK/error behavior, and the explicitly skipped diagnostics/project-wide/LSP scope.","acceptance_criteria":"Core tests cover supported generic file-local behavior and failure modes; docs explain parser/query-bundle availability, the first-pass boundary, and future expansion path; the epic's verification commands are recorded after implementation lands.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:32Z","created_by":"dawn","updated_at":"2026-06-28T11:20:25Z","started_at":"2026-06-28T11:18:27Z","closed_at":"2026-06-28T11:20:25Z","close_reason":"Added code-intel docs, registration coverage, and recorded focused verification commands.","labels":["code-intel","docs","tests","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:32Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b.2","type":"blocks","created_at":"2026-06-28T13:46:51Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b.3","type":"blocks","created_at":"2026-06-28T13:46:50Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b.5","type":"blocks","created_at":"2026-06-28T13:46:58Z","created_by":"dawn","metadata":"{}"}],"dependency_count":3,"dependent_count":0,"comment_count":0} diff --git a/docs/long-term-memory-arch.md b/docs/long-term-memory-arch.md index 9c4a634..e0f1d1a 100644 --- a/docs/long-term-memory-arch.md +++ b/docs/long-term-memory-arch.md @@ -79,6 +79,14 @@ allowed: substring-heavy identifiers. - lexical features as trace/debug evidence. +multilingual fts rule: sqlite fts is only a lexical side channel. `unicode61` +handles unicode token boundaries/diacritics better than ascii tokenization, and +the trigram side index helps cjk/no-space text and substring-heavy ids, but +neither makes lexical search cross-lingual or semantically equivalent across +languages. semantic multilingual recall should come from dense embeddings or +learned sparse retrieval, with fts/trigram used to add exact-ish anchors into +packets. + not allowed as the main answer: - hand-written english morphology such as plural stripping. @@ -167,18 +175,21 @@ signals, and estimated tokens. keep `QueryOp`, improve structured planner reliability, and preserve the conservative offline fallback. do not reintroduce query-word lists. -3. finish collect-mode fact coverage. - track fact groups for aggregate/order/update questions. expose sufficiency - and gap state in traces. - -4. render fact rows and timelines. - packets should carry enough timestamp/ref structure for the reader to see - previous/current values and ordered event lists. - -5. add multilingual regression fixtures. - include cross-lingual paraphrase, cjk no-space text, turkish suffix - variation, diacritics, transliteration, and ref/id cases where english cue - words are absent. +3. finish collect-mode fact coverage. initial implementation landed: + collect-mode plans widen candidate/packet limits, selected packet traces + expose fact-group coverage and gap state, and packet selection tracks + distinct session/ref fact groups for aggregate/order/update questions. + +4. render fact rows and timelines. initial implementation landed: packets carry + ref-grounded fact rows and timestamped timeline rows, and rendered + `` show those rows before raw bodies. current extraction is + intentionally generic; richer entity/slot/unit extraction is still future + work. + +5. add multilingual regression fixtures. initial no-space cjk trigram coverage + landed. still add cross-lingual paraphrase, turkish suffix variation, + diacritics, transliteration, and ref/id cases where english cue words are + absent. this work is tracked in beads as `klbr-u0q`. diff --git a/docs/memory-benches.md b/docs/memory-benches.md index cf41195..e788b5c 100644 --- a/docs/memory-benches.md +++ b/docs/memory-benches.md @@ -34,6 +34,7 @@ current fast preflight: ```bash rtk cargo test -p klbr-core +rtk cargo test -p klbr-bench rtk cargo run -p klbr-bench -- continuous-loop /tmp/klbr-continuous-loop ``` @@ -48,6 +49,27 @@ passive recall suppression, and exact-ref context assembly. both should run before memory architecture changes, with model-backed longmemeval smoke runs reserved for evidence quality and reader behavior. +planner/fact-table smoke: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --profile session-first/fts-only \ + --top-k 5 \ + --budget-read 1200 \ + --graph-depth 0 \ + --limit 1 \ + --retrieval-only \ + --out /tmp/klbr-fact-row-trace-smoke +``` + +the trace should include `stage_one.session_candidates`, `coverage`, packet +`fact_rows`, and rendered packet metrics. the report should include +`critical_fact_row_visible` and `collect_mode_fact_group_recall`; these metrics +separate selected refs, rendered refs, value visibility, and structured fact-row +visibility instead of folding everything into one recall number. + longmemeval is a good primary benchmark because it directly targets chat-assistant long-term memory and covers information extraction, multi-session reasoning, temporal reasoning, knowledge updates, and abstention. the official repo exposes the cleaned datasets, evidence labels, qa evaluation scripts, retrieval metrics, and standard `longmemeval_s_cleaned`, `longmemeval_m_cleaned`, and oracle files, which makes it good for comparable results. ([arXiv][1]) i would not only use longmemeval because it will not fully test your architecture. it does not directly stress all the parts that make klbr interesting: reflink propagation, deterministic ref expansion, semipassive recall, lifecycle transitions, source provenance, compaction quality, and whether live context is correctly preferred over long-term memory. those need small internal fixtures because public benchmarks will not reliably catch “ref alias silently stopped resolving” or “archived memories still leak into normal retrieval.” @@ -196,6 +218,11 @@ retrieval: nDCG evidence precision@k gold-in-context rate + answer-bearing-ref selected rate + answer-bearing-ref rendered rate + answer-value visible rate + critical fact-row visible rate + collect-mode fact-group recall memory architecture: ref validity rate diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index cc08dab..cd61d23 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -24,6 +24,9 @@ going next. - parent note refs - stable chunk refs - fts rows through `promptable_text` + - unicode-token fts rows in `promptable_text_fts` + - trigram fts rows in `promptable_text_trigram` for no-space, substring, + identifier, and exact-ish candidate generation - `MemoryGarden` can write markdown note files and sync them back into sqlite. - legacy `memory_edges` now mirror into canonical `edges`, `sync_reference_indexes` backfills old rows, and memory-id provenance APIs read canonical ref edges. - memory status changes now project onto canonical refs and rebuild fts visibility. @@ -32,6 +35,8 @@ going next. bounded local embedding text. - fts retrieval stays fts-native in the production pipeline; the old hand-written lexical relevance helpers are not the ranking-critical path. + query cleanup no longer drops english stopwords, and trigram fts is used as a + fallback candidate channel rather than a query-policy rule. - `EvidencePlanner` builds `exact_ref`, `turn_window`, `episode_bridge`, and `graph_bridge` packets; packet fusion replaces raw same-session candidate dedupe with same-session ref-overlap suppression. @@ -61,6 +66,17 @@ going next. `stage_one.session_candidates`. - `fts-only` keeps episode notes for fts but skips embedded episode-memory rows, so lexical ablations no longer require the dense embedder during ingest. +- collect-mode retrieval is wired through `OpPlan::collect_all` for aggregate, + order, and update-resolution questions. packet planning widens candidate and + packet limits, traces fact-group coverage/gap state, and favors distinct + session/ref fact groups instead of letting one session monopolize context. +- evidence packets now carry compact `fact_rows` and `timeline` rows with + supporting refs while preserving raw packet bodies as provenance. rendered + `` include `` and `` blocks before bodies. +- `MemoryPipeline::answer` has an initial planned synthesis path for + aggregate count/sum/avg, chronological ordering, previous/latest update + resolution, and preference recommendations with personal-support guards. + ordinary `lookup` questions still fall through to the reader model. - `klbr-bench -- run` now uses the production pipeline instead of manually calling low-level retrieval: - longmemeval-s/m style data - flexible locomo adapter with fixture coverage and protocol-note artifacts @@ -72,21 +88,20 @@ going next. ## open direction -- `klbr-u0q`: continue session/event-first retrieval beyond the initial profile: - improve fact-group coverage, operation-aware assembly, and multilingual - regression coverage. - harden the structured planner path and add model-backed planner fixtures. -- finish collect-mode fact-group sufficiency and gap diagnosis for aggregate, - order, and update-resolution questions. -- render fact rows/timeline rows for operation-aware synthesis. -- add multilingual retrieval fixtures so english lexical shortcuts cannot become - hidden policy. +- replace the current generic fact-row extraction with richer entity/slot/unit + extraction when the writer/planner path can provide it. +- add deeper multilingual retrieval fixtures so english lexical shortcuts cannot + become hidden policy. current coverage includes no-space cjk trigram fts, but + cross-lingual paraphrase, turkish suffix variation, diacritics, and + transliteration still deserve broader eval coverage. ## verification ```bash rtk cargo check rtk cargo test -p klbr-core +rtk cargo test -p klbr-bench rtk cargo run -p klbr-bench -- continuous-loop /tmp/klbr-continuous-loop ``` @@ -94,12 +109,37 @@ current result: ```text cargo check -p klbr-bench: 0 errors, 1 pre-existing warning -klbr-core tests: 141 passed, 1 ignored +klbr-core tests: 150 passed, 1 ignored klbr-bench tests: 17 passed continuous-loop smoke: passed with the pre-existing agent.rs dead-code warning session-first/fts-only trace smoke: evaluated 1, CandidateSessionRecallAny@5 1.0000 ``` +planner/fact-table trace smoke: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --profile session-first/fts-only \ + --top-k 5 \ + --budget-read 1200 \ + --graph-depth 0 \ + --limit 1 \ + --retrieval-only \ + --out /tmp/klbr-fact-row-trace-smoke +``` + +current result: + +```text +evaluated: 1 +CandidateSessionRecallAny@5: 1.0000 +collect_mode_fact_group_recall: 1.0000 +trace.jsonl includes stage_one.session_candidates, coverage fields, packet +fact_rows, and packet timelines +``` + retrieval benchmark smoke: ```bash diff --git a/docs/planner-grade-evidence-assembly.md b/docs/planner-grade-evidence-assembly.md index 1c8eafa..0a4d4fb 100644 --- a/docs/planner-grade-evidence-assembly.md +++ b/docs/planner-grade-evidence-assembly.md @@ -3,11 +3,24 @@ status: supporting reviewed: 2026-06-29 -this plan is partially implemented under beads epic `klbr-f0q`. it remains -useful for collect-mode, fact-row, and structured-synthesis work, but +this plan is now implemented as the initial planner-grade evidence assembly +slice under beads epic `klbr-f0q`. it remains useful as design background for +future richer entity/slot extraction and iterative gap repair, but [`long-term-memory-arch.md`](./long-term-memory-arch.md) is now the higher-level current architecture direction. +current source status: + +- `OpPlan::collect_all` switches aggregate, order, and update-resolution + questions into collect mode. +- packet traces expose fact-group coverage/gap state. +- evidence packets render compact ref-grounded fact rows and timelines before + raw bodies. +- `MemoryPipeline::answer` deterministically handles aggregate count/sum/avg, + chronological order, previous/latest update resolution, and guarded + preference recommendations when the planner selects those operations. +- ordinary lookup questions still use the existing reader path. + ## diagnosis the good news is that the earlier “right session, wrong chunk” problem looks substantially less central now. your internal benchmark gap report already described the original failure as a selection-granularity bug, where session recall was high but qa accuracy lagged because answer-bearing neighboring turns were getting dropped; that report also notes the move to `EvidencePacket` planning as the production fix. your memory system report, meanwhile, describes `MemoryPipeline` as the shared facade for observing sessions, retrieving evidence, assembling context, and answering through the real stack rather than a benchmark-only shortcut. in other words, klbr is no longer mostly failing because it cannot *find* the right area; it is now more often failing because it does not always *plan the right evidence set and reasoning mode* for the question type. fileciteturn0file0 fileciteturn0file1 diff --git a/klbr-bench/src/longmemeval.rs b/klbr-bench/src/longmemeval.rs index a5931e7..06f3e61 100644 --- a/klbr-bench/src/longmemeval.rs +++ b/klbr-bench/src/longmemeval.rs @@ -57,6 +57,8 @@ struct PacketSelectionMetrics { answer_bearing_ref_rendered: bool, answer_bearing_ref_in_context: bool, answer_value_visible: bool, + critical_fact_row_visible: bool, + collect_mode_fact_group_recall: f64, gold_session_in_context: bool, gold_session_plus_neighbor_in_context: bool, packet_recall_any_at_k: f64, @@ -1397,6 +1399,8 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { let mut answer_bearing_ref_rendered = 0usize; let mut answer_bearing_ref_in_context = 0usize; let mut answer_value_visible = 0usize; + let mut critical_fact_row_visible = 0usize; + let mut collect_mode_fact_group_recall_sum = 0.0f64; let mut candidate_session_recall_any_at_k = 0usize; let mut candidate_session_recall_all_at_k = 0usize; let mut gold_session_in_context = 0usize; @@ -1552,6 +1556,10 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { if packet_metrics.answer_value_visible { answer_value_visible += 1; } + if packet_metrics.critical_fact_row_visible { + critical_fact_row_visible += 1; + } + collect_mode_fact_group_recall_sum += packet_metrics.collect_mode_fact_group_recall; if packet_metrics.candidate_session_recall_any_at_k > 0.0 { candidate_session_recall_any_at_k += 1; } @@ -1654,6 +1662,8 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { "answer_bearing_ref_rendered": if answerable == 0 { 0.0 } else { answer_bearing_ref_rendered as f64 / answerable_denominator }, "answer_bearing_ref_in_context": if answerable == 0 { 0.0 } else { answer_bearing_ref_in_context as f64 / answerable_denominator }, "answer_value_visible": if answerable == 0 { 0.0 } else { answer_value_visible as f64 / answerable_denominator }, + "critical_fact_row_visible": if answerable == 0 { 0.0 } else { critical_fact_row_visible as f64 / answerable_denominator }, + "collect_mode_fact_group_recall": collect_mode_fact_group_recall_sum / evaluated_denominator, "candidate_session_recall_any_at_k": if answerable == 0 { 0.0 } else { candidate_session_recall_any_at_k as f64 / answerable_denominator }, "candidate_session_recall_all_at_k": if answerable == 0 { 0.0 } else { candidate_session_recall_all_at_k as f64 / answerable_denominator }, "gold_session_in_context": if answerable == 0 { 0.0 } else { gold_session_in_context as f64 / answerable_denominator }, @@ -1675,7 +1685,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { fs::write( out_dir.join("report.md"), format!( - "# klbr memory pipeline benchmark\n\n- suite: `{}`\n- profile: `{}`\n- evaluated: `{}`\n- answerable: `{}`\n- sample: `{}`\n- RecallAny@5: `{:.4}`\n- RecallAll@5: `{:.4}`\n- CandidateSessionRecallAny@{}: `{:.4}`\n- CandidateSessionRecallAll@{}: `{:.4}`\n- PacketRecallAny@{}: `{:.4}`\n- PacketRecallAll@{}: `{:.4}`\n- answer_bearing_ref_selected: `{:.4}`\n- answer_bearing_ref_rendered: `{:.4}`\n- answer_bearing_ref_in_context: `{:.4}`\n- answer_value_visible: `{:.4}`\n- gold_session_plus_neighbor_in_context: `{:.4}`\n- same_session_dedupe_suppression_count: `{}`\n- packet_tokens_mean: `{:.2}`\n- op_plan_counts: `{}`\n- retrieval_only: `{}`\n- official_eval: `{}`\n", + "# klbr memory pipeline benchmark\n\n- suite: `{}`\n- profile: `{}`\n- evaluated: `{}`\n- answerable: `{}`\n- sample: `{}`\n- RecallAny@5: `{:.4}`\n- RecallAll@5: `{:.4}`\n- CandidateSessionRecallAny@{}: `{:.4}`\n- CandidateSessionRecallAll@{}: `{:.4}`\n- PacketRecallAny@{}: `{:.4}`\n- PacketRecallAll@{}: `{:.4}`\n- answer_bearing_ref_selected: `{:.4}`\n- answer_bearing_ref_rendered: `{:.4}`\n- answer_bearing_ref_in_context: `{:.4}`\n- answer_value_visible: `{:.4}`\n- critical_fact_row_visible: `{:.4}`\n- collect_mode_fact_group_recall: `{:.4}`\n- gold_session_plus_neighbor_in_context: `{:.4}`\n- same_session_dedupe_suppression_count: `{}`\n- packet_tokens_mean: `{:.2}`\n- op_plan_counts: `{}`\n- retrieval_only: `{}`\n- official_eval: `{}`\n", suite, profile, evaluated, @@ -1695,6 +1705,8 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { if answerable == 0 { 0.0 } else { answer_bearing_ref_rendered as f64 / answerable_denominator }, if answerable == 0 { 0.0 } else { answer_bearing_ref_in_context as f64 / answerable_denominator }, if answerable == 0 { 0.0 } else { answer_value_visible as f64 / answerable_denominator }, + if answerable == 0 { 0.0 } else { critical_fact_row_visible as f64 / answerable_denominator }, + collect_mode_fact_group_recall_sum / evaluated_denominator, if answerable == 0 { 0.0 } else { gold_session_plus_neighbor_in_context as f64 / answerable_denominator }, same_session_dedupe_suppression_count, packet_tokens_mean_sum / evaluated_denominator, @@ -2117,6 +2129,13 @@ fn compute_packet_selection_metrics( .any(|ref_id| used_refs.contains(ref_id)); let answer_bearing_ref_in_context = answer_bearing_ref_rendered; let answer_value_visible = answer_value_visible_in_context(item, &context.content); + let critical_fact_row_visible = answer_value_visible && context.content.contains(" Self { match s { "exact" => Self::Exact, - "fts" => Self::Fts, + "fts" | "fts_trigram" => Self::Fts, "sparse" | "bge_m3_sparse" => Self::Sparse, "dense" => Self::Dense, "graph" => Self::Graph, @@ -177,11 +177,37 @@ pub struct EvidencePacket { pub packet_kind: EvidencePacketKind, pub refs: Vec, pub bodies: Vec, + #[serde(default)] + pub fact_rows: Vec, + #[serde(default)] + pub timeline: Vec, pub signals: EvidencePacketSignals, pub lanes: Vec, pub estimated_tokens: usize, } +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct EvidenceFactRow { + pub row_id: String, + pub ref_id: String, + pub entity_type: String, + pub session_id: Option, + pub timestamp: Option, + pub ordinal: usize, + pub value_text: Option, + pub value_number: Option, + pub text: String, +} + +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct EvidenceTimelineEvent { + pub ref_id: String, + pub session_id: Option, + pub timestamp: i64, + pub ordinal: usize, + pub text: String, +} + #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] pub struct EvidenceOmittedRef { pub packet_id: String, @@ -234,15 +260,18 @@ impl EvidencePlanner { } } + let all_packets = packets.clone(); let ranked = if collect_mode { rank_and_cap_packets_collect_mode(packets, limit) } else { rank_and_cap_packets(packets, limit) }; + let coverage = evidence_coverage_trace(collect_mode, &all_packets, &ranked.packets); omissions.extend(ranked.omissions); Ok(PacketBuildOutcome { packets: ranked.packets, omissions, + coverage, }) } @@ -333,6 +362,20 @@ impl EvidencePlanner { refs.join(",") ); let packet_id = format!("pkt_{}", stable_base36(&packet_seed)); + let fact_rows = + self.fact_rows_for_packet(&refs, &bodies, candidate.session_id.as_deref())?; + let timeline = fact_rows + .iter() + .filter_map(|row| { + row.timestamp.map(|timestamp| EvidenceTimelineEvent { + ref_id: row.ref_id.clone(), + session_id: row.session_id.clone(), + timestamp, + ordinal: row.ordinal, + text: row.text.clone(), + }) + }) + .collect(); let omissions = pending_omissions .into_iter() .map(|(ref_id, reason)| EvidenceOmittedRef { @@ -349,6 +392,8 @@ impl EvidencePlanner { packet_kind, refs, bodies, + fact_rows, + timeline, signals: EvidencePacketSignals { sources, seed_count: 1, @@ -362,12 +407,51 @@ impl EvidencePlanner { omissions, }) } + + fn fact_rows_for_packet( + &self, + refs: &[String], + bodies: &[String], + fallback_session_id: Option<&str>, + ) -> Result> { + refs.iter() + .zip(bodies.iter()) + .enumerate() + .map(|(ordinal, (ref_id, body))| { + let entity_type = self + .memory + .get_resolved_ref(ref_id)? + .map(|resolved| resolved.entity_type) + .unwrap_or_else(|| "unknown".to_string()); + let session_id = self + .memory + .ref_session_id(ref_id)? + .or_else(|| fallback_session_id.map(str::to_string)); + let timestamp = self.memory.ref_timestamp(ref_id)?; + let (value_text, value_number) = first_numeric_value(body); + let text = truncate_chars(body, 180); + let row_seed = format!("{ref_id}:{ordinal}:{}", text); + Ok(EvidenceFactRow { + row_id: format!("fact_{}", stable_base36(&row_seed)), + ref_id: ref_id.clone(), + entity_type, + session_id, + timestamp, + ordinal, + value_text, + value_number, + text, + }) + }) + .collect() + } } #[derive(Debug, Clone)] pub struct PacketBuildOutcome { pub packets: Vec, pub omissions: Vec, + pub coverage: EvidenceCoverageTrace, } struct ExpandedPacket { @@ -380,6 +464,16 @@ pub(crate) struct RankedPacketOutcome { omissions: Vec, } +#[derive(Debug, Clone, Default, serde::Serialize, serde::Deserialize)] +pub struct EvidenceCoverageTrace { + pub collect_mode: bool, + pub available_fact_groups: usize, + pub selected_fact_groups: usize, + pub fact_row_count: usize, + pub gap_count: usize, + pub gap_state: String, +} + pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> String { let max_chars = max_tokens.saturating_mul(4); let body_count = packet.bodies.len().max(1); @@ -403,6 +497,47 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str xml_escape(packet.session_id.as_deref().unwrap_or("")), packet.estimated_tokens.min(max_tokens), ); + if !packet.timeline.is_empty() { + out.push_str(" \n"); + for event in packet.timeline.iter().take(8) { + out.push_str(&format!( + " {}\n", + xml_escape(&event.ref_id), + event.timestamp, + event.ordinal, + xml_escape(event.session_id.as_deref().unwrap_or("")), + xml_escape(&truncate_chars(&event.text, 96)), + )); + } + out.push_str(" \n"); + } + if !packet.fact_rows.is_empty() { + out.push_str(" \n"); + for fact in packet.fact_rows.iter().take(8) { + let mut attrs = format!( + "ref=\"{}\" entity_type=\"{}\" ord=\"{}\" session=\"{}\"", + xml_escape(&fact.ref_id), + xml_escape(&fact.entity_type), + fact.ordinal, + xml_escape(fact.session_id.as_deref().unwrap_or("")) + ); + if let Some(timestamp) = fact.timestamp { + attrs.push_str(&format!(" t=\"{}\"", timestamp)); + } + if let Some(value) = &fact.value_text { + attrs.push_str(&format!(" value=\"{}\"", xml_escape(value))); + } + if let Some(value) = fact.value_number { + attrs.push_str(&format!(" number=\"{}\"", value)); + } + out.push_str(&format!( + " {}\n", + attrs, + xml_escape(&truncate_chars(&fact.text, 120)), + )); + } + out.push_str(" \n"); + } for (index, body) in packet.bodies.iter().enumerate() { let ref_id = packet.refs.get(index).unwrap_or(&packet.anchor_ref); out.push_str(&format!( @@ -415,6 +550,61 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str out } +fn evidence_coverage_trace( + collect_mode: bool, + available_packets: &[EvidencePacket], + selected_packets: &[EvidencePacket], +) -> EvidenceCoverageTrace { + let available = fact_group_keys(available_packets); + let selected = fact_group_keys(selected_packets); + let selected_count = selected.intersection(&available).count(); + let gap_count = available.len().saturating_sub(selected_count); + let fact_row_count = selected_packets + .iter() + .map(|packet| packet.fact_rows.len()) + .sum(); + EvidenceCoverageTrace { + collect_mode, + available_fact_groups: available.len(), + selected_fact_groups: selected_count, + fact_row_count, + gap_count, + gap_state: if !collect_mode { + "not_collect_mode".to_string() + } else if gap_count == 0 { + "covered_available_groups".to_string() + } else { + "partial_available_groups".to_string() + }, + } +} + +fn fact_group_keys(packets: &[EvidencePacket]) -> HashSet { + packets + .iter() + .flat_map(|packet| { + if packet.fact_rows.is_empty() { + vec![packet + .session_id + .as_ref() + .map(|session_id| format!("session:{session_id}")) + .unwrap_or_else(|| format!("ref:{}", packet.anchor_ref))] + } else { + packet + .fact_rows + .iter() + .map(|row| { + row.session_id + .as_ref() + .map(|session_id| format!("session:{session_id}")) + .unwrap_or_else(|| format!("ref:{}", row.ref_id)) + }) + .collect::>() + } + }) + .collect() +} + pub fn render_evidence_packets(packets: &[EvidencePacket], max_packet_tokens: usize) -> String { let mut out = String::from("\n"); for packet in packets { @@ -745,6 +935,38 @@ fn resolved_refs_to_ids(refs: Vec) -> Vec { .collect() } +fn first_numeric_value(text: &str) -> (Option, Option) { + let mut current = String::new(); + let mut has_digit = false; + for ch in text.chars() { + if ch.is_ascii_digit() + || (has_digit && matches!(ch, '.' | ',' | ':' | '/' | '-')) + || (!has_digit && matches!(ch, '+' | '-')) + { + if ch.is_ascii_digit() { + has_digit = true; + } + current.push(ch); + } else if has_digit { + break; + } else { + current.clear(); + } + } + if !has_digit { + return (None, None); + } + let value_text = current + .trim_matches(|ch: char| !ch.is_ascii_digit()) + .to_string(); + if value_text.is_empty() { + return (None, None); + } + let normalized = value_text.replace(',', ""); + let value_number = normalized.parse::().ok(); + (Some(value_text), value_number) +} + pub(crate) fn stable_base36(value: &str) -> String { let mut hash: u64 = 1469598103934665603; for byte in value.as_bytes() { @@ -785,6 +1007,8 @@ mod tests { packet_kind: EvidencePacketKind::TurnWindow, refs: vec!["turn:1:0".to_string()], bodies: vec!["the answer is target".to_string()], + fact_rows: vec![], + timeline: vec![], signals: EvidencePacketSignals { sources: vec!["fts".to_string()], seed_count: 1, @@ -808,6 +1032,37 @@ mod tests { assert!(rendered.contains("the answer is target")); } + #[test] + fn rendered_packet_includes_fact_and_timeline_rows_with_refs() { + let mut packet = test_packet("pkt_fact", "ref_fact", "fts", 1, Some("s1")); + packet.fact_rows = vec![EvidenceFactRow { + row_id: "fact_ref".to_string(), + ref_id: "ref_fact".to_string(), + entity_type: "turn_chunk".to_string(), + session_id: Some("s1".to_string()), + timestamp: Some(42), + ordinal: 0, + value_text: Some("27:45".to_string()), + value_number: None, + text: "personal best was 27:45".to_string(), + }]; + packet.timeline = vec![EvidenceTimelineEvent { + ref_id: "ref_fact".to_string(), + session_id: Some("s1".to_string()), + timestamp: 42, + ordinal: 0, + text: "personal best was 27:45".to_string(), + }]; + + let rendered = render_evidence_packet(&packet, 128); + + assert!(rendered.contains("")); + assert!(rendered.contains("")); + assert!(rendered.contains("ref=\"ref_fact\"")); + assert!(rendered.contains("value=\"27:45\"")); + assert!(rendered.contains(">(); + + let coverage = evidence_coverage_trace(true, &available, &selected); + + assert!(coverage.collect_mode); + assert_eq!(coverage.available_fact_groups, 3); + assert_eq!(coverage.selected_fact_groups, 2); + assert_eq!(coverage.gap_count, 1); + assert_eq!(coverage.gap_state, "partial_available_groups"); + } + fn test_packet( packet_id: &str, ref_id: &str, @@ -966,6 +1239,8 @@ mod tests { .iter() .map(|ref_id| format!("body for {packet_id} {ref_id}")) .collect(), + fact_rows: vec![], + timeline: vec![], signals: EvidencePacketSignals { sources: sources.iter().map(|source| source.to_string()).collect(), seed_count: 1, diff --git a/klbr-core/src/memory.rs b/klbr-core/src/memory.rs index 4bbe20f..5e0e6f8 100644 --- a/klbr-core/src/memory.rs +++ b/klbr-core/src/memory.rs @@ -414,6 +414,13 @@ impl MemoryStore { entity_type UNINDEXED, tokenize = 'unicode61' ); + CREATE VIRTUAL TABLE IF NOT EXISTS promptable_text_trigram USING fts5( + ref_id UNINDEXED, + body, + lane UNINDEXED, + entity_type UNINDEXED, + tokenize = 'trigram' + ); CREATE TABLE IF NOT EXISTS resolution_events ( event_id INTEGER PRIMARY KEY AUTOINCREMENT, @@ -499,6 +506,7 @@ impl MemoryStore { "PRAGMA foreign_keys = OFF; DROP TABLE IF EXISTS embedding_items; DROP TABLE IF EXISTS resolution_events; + DROP TABLE IF EXISTS promptable_text_trigram; DROP TABLE IF EXISTS promptable_text_fts; DROP TABLE IF EXISTS markdown_note_chunks; DROP TABLE IF EXISTS markdown_notes; @@ -598,6 +606,13 @@ impl MemoryStore { entity_type UNINDEXED, tokenize = 'unicode61' ); + CREATE VIRTUAL TABLE IF NOT EXISTS promptable_text_trigram USING fts5( + ref_id UNINDEXED, + body, + lane UNINDEXED, + entity_type UNINDEXED, + tokenize = 'trigram' + ); CREATE TABLE IF NOT EXISTS markdown_notes ( note_id INTEGER PRIMARY KEY AUTOINCREMENT, note_ref TEXT NOT NULL UNIQUE, @@ -1709,6 +1724,70 @@ impl MemoryStore { Ok(results) } + pub fn search_refs_trigram( + &self, + query: &str, + lanes: &[MemoryLane], + limit: usize, + ) -> Result> { + let Some(match_query) = fts_query(query) else { + return Ok(vec![]); + }; + let conn = self.conn.lock().unwrap(); + let lane_filter = lanes.iter().map(|lane| lane.as_str()).collect::>(); + let oversample = limit.saturating_mul(4).max(limit).max(8); + let mut stmt = conn.prepare( + "SELECT + f.ref_id, + ( + SELECT alias FROM ref_aliases a + WHERE a.ref_id = f.ref_id AND a.status = 'active' + ORDER BY CASE a.alias_kind + WHEN 'display' THEN 0 + WHEN 'exact_version' THEN 1 + WHEN 'legacy' THEN 2 + ELSE 3 + END, a.alias ASC + LIMIT 1 + ) AS alias, + r.entity_type, + COALESCE(m.lane, 'semantic') AS lane, + f.body, + COALESCE(p.token_count, length(f.body) / 4) AS token_count, + bm25(promptable_text_trigram) AS score + FROM promptable_text_trigram f + JOIN refs r ON r.ref_id = f.ref_id + LEFT JOIN ref_metadata m ON m.ref_id = f.ref_id + LEFT JOIN promptable_text p ON p.ref_id = f.ref_id + WHERE promptable_text_trigram MATCH ?1 + AND r.status = 'active' + ORDER BY score ASC + LIMIT ?2", + )?; + let mut rows = stmt.query(params![match_query, oversample as i64])?; + let mut results = Vec::new(); + while let Some(row) = rows.next()? { + let lane_raw: String = row.get(3)?; + if !lane_filter.is_empty() && !lane_filter.contains(&lane_raw.as_str()) { + continue; + } + results.push(RefSearchEntry { + ref_id: row.get(0)?, + alias: row.get(1)?, + entity_type: row.get(2)?, + lane: MemoryLane::parse(&lane_raw), + body: row.get(4)?, + token_count: row.get::<_, i64>(5)? as usize, + score: row.get::<_, f64>(6)? as f32, + source: "fts_trigram".to_string(), + }); + if results.len() >= limit { + break; + } + } + Ok(results) + } + pub fn search_refs_sparse( &self, sparse_weights: &std::collections::HashMap, @@ -1979,6 +2058,75 @@ impl MemoryStore { Ok(session_id) } + pub fn ref_timestamp(&self, ref_id: &str) -> Result> { + let conn = self.conn.lock().unwrap(); + + let resolved_id = match conn + .query_row( + "SELECT ref_id FROM ref_aliases WHERE alias = ?1", + params![ref_id], + |row| row.get::<_, String>(0), + ) + .optional()? + { + Some(canonical_id) => canonical_id, + None => ref_id.to_string(), + }; + + let ref_info: Option<(String, i64)> = conn + .query_row( + "SELECT entity_type, entity_id FROM refs WHERE ref_id = ?1", + params![&resolved_id], + |row| Ok((row.get(0)?, row.get(1)?)), + ) + .optional()?; + let Some((entity_type, entity_id)) = ref_info else { + return Ok(None); + }; + + let timestamp = match entity_type.as_str() { + "turn" | "turn_chunk" | "synthetic" => conn + .query_row( + "SELECT ts FROM turns WHERE id = ?1", + params![entity_id], + |row| row.get::<_, i64>(0), + ) + .optional()?, + "memory" => conn + .query_row( + "SELECT event_time FROM memories WHERE id = ?1", + params![entity_id], + |row| row.get::<_, i64>(0), + ) + .optional()?, + "memory_version" => conn + .query_row( + "SELECT m.event_time + FROM memory_versions v + JOIN memories m ON m.id = v.memory_id + WHERE v.version_id = ?1", + params![entity_id], + |row| row.get::<_, i64>(0), + ) + .optional()?, + "episode" | "semantic_note" | "profile_note" | "procedural_note" | "attachment" => conn + .query_row( + "SELECT frontmatter, created_at FROM markdown_notes WHERE note_id = ?1", + params![entity_id], + |row| Ok((row.get::<_, String>(0)?, row.get::<_, i64>(1)?)), + ) + .optional()? + .map(|(frontmatter, created_at)| { + serde_json::from_str::(&frontmatter) + .ok() + .and_then(|value| value.get("timestamp").and_then(|ts| ts.as_i64())) + .unwrap_or(created_at) + }), + _ => None, + }; + Ok(timestamp) + } + pub fn resolve_aliases_batch(&self, aliases: &[String]) -> Result> { let conn = self.conn.lock().unwrap(); let mut results = Vec::new(); @@ -3654,6 +3802,7 @@ fn upsert_promptable_text( fn rebuild_promptable_fts(conn: &Connection) -> Result<()> { conn.execute("DELETE FROM promptable_text_fts", [])?; + conn.execute("DELETE FROM promptable_text_trigram", [])?; conn.execute( "INSERT INTO promptable_text_fts (ref_id, body, lane, entity_type) SELECT p.ref_id, p.body, COALESCE(m.lane, 'semantic'), r.entity_type @@ -3663,6 +3812,15 @@ fn rebuild_promptable_fts(conn: &Connection) -> Result<()> { WHERE r.status = 'active'", [], )?; + conn.execute( + "INSERT INTO promptable_text_trigram (ref_id, body, lane, entity_type) + SELECT p.ref_id, p.body, COALESCE(m.lane, 'semantic'), r.entity_type + FROM promptable_text p + JOIN refs r ON r.ref_id = p.ref_id + LEFT JOIN ref_metadata m ON m.ref_id = p.ref_id + WHERE r.status = 'active'", + [], + )?; Ok(()) } @@ -3843,17 +4001,14 @@ fn fts_query(query: &str) -> Option { .filter_map(|term| { let term = term.trim().to_lowercase(); let len = term.chars().count(); - (!is_fts_stopword(&term) && len >= 3).then_some(term) + (len >= 3).then_some(term) }) .collect::>(); terms.sort(); terms.dedup(); let mut expanded = Vec::new(); - for term in terms.into_iter().filter(|term| { - let len = term.chars().count(); - !is_fts_stopword(term) && len >= 3 - }) { + for term in terms.into_iter().filter(|term| term.chars().count() >= 3) { expanded.push(format!("\"{term}\"")); } expanded.truncate(24); @@ -3864,39 +4019,6 @@ fn fts_query(query: &str) -> Option { } } -fn is_fts_stopword(term: &str) -> bool { - matches!( - term, - "the" - | "and" - | "for" - | "are" - | "but" - | "not" - | "you" - | "your" - | "our" - | "was" - | "were" - | "has" - | "had" - | "his" - | "her" - | "she" - | "him" - | "its" - | "what" - | "who" - | "why" - | "how" - | "did" - | "does" - | "can" - | "could" - | "would" - ) -} - fn memory_exists(conn: &Connection, id: i64) -> Result { let mut stmt = conn.prepare("SELECT 1 FROM memories WHERE id = ?1 LIMIT 1")?; let mut rows = stmt.query(params![id])?; @@ -5147,7 +5269,7 @@ mod tests { let query = fts_query("What is the name of my cat?").unwrap(); assert!(query.contains("\"cat\"")); assert!(query.contains("\"name\"")); - assert!(!query.contains("\"what\"")); + assert!(query.contains("\"what\"")); } #[test] diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index e1dd422..cd1feb7 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -7,8 +7,9 @@ use crate::{ config::MemoryConfig, context::extract_ref_codes, evidence::{ - render_evidence_packet, stable_base36, AnchorKind, EntityType, EvidenceOmittedRef, - EvidencePacket, EvidencePacketKind, EvidencePlanner, RetrievalSource, + render_evidence_packet, stable_base36, AnchorKind, EntityType, EvidenceCoverageTrace, + EvidenceFactRow, EvidenceOmittedRef, EvidencePacket, EvidencePacketKind, EvidencePlanner, + RetrievalSource, }, memory::{ to_base26_suffix, to_base36, MarkdownNoteInput, MemoryLane, MemoryStore, RefSearchEntry, @@ -16,7 +17,7 @@ use crate::{ }, models::{LlmClient, Message, RerankResult}, mvp::{MemoryLayer, MemoryRecordInput, MemoryStatus}, - planner::{parse_structured_op_plan, OpPlan, STRUCTURED_QUERY_PLANNER_PROMPT}, + planner::{parse_structured_op_plan, OpPlan, QueryOp, STRUCTURED_QUERY_PLANNER_PROMPT}, }; pub use crate::evidence::EvidenceAtom; @@ -119,6 +120,8 @@ pub struct PipelineRetrievalTrace { pub packets: Vec, #[serde(default)] pub packet_omissions: Vec, + #[serde(default)] + pub coverage: EvidenceCoverageTrace, #[serde(default, skip_serializing_if = "Option::is_none")] pub packet_rerank: Option, } @@ -442,6 +445,7 @@ impl MemoryPipeline { candidates, packets, packet_omissions: packet_plan.omissions, + coverage: packet_plan.coverage, packet_rerank, }) } @@ -554,6 +558,7 @@ impl MemoryPipeline { candidates, packets, packet_omissions: packet_plan.omissions, + coverage: packet_plan.coverage, packet_rerank, }; self.assemble_context(query, &retrieved, budget).await @@ -564,11 +569,17 @@ impl MemoryPipeline { let context = self .assemble_context(query.clone(), &retrieval, budget) .await?; - let messages = vec![ - Message::system(MEMORY_READER_SYSTEM_PROMPT), - Message::user(context.content.clone()), - ]; - let (hypothesis, _) = self.llm.complete(&messages).await?; + let hypothesis = if let Some(answer) = + synthesize_planned_answer(&retrieval.op_plan, &retrieval.packets) + { + answer + } else { + let messages = vec![ + Message::system(MEMORY_READER_SYSTEM_PROMPT), + Message::user(context.content.clone()), + ]; + self.llm.complete(&messages).await?.0 + }; Ok(AnswerTrace { query_id: query.query_id, hypothesis, @@ -590,9 +601,25 @@ impl MemoryPipeline { lanes: &[MemoryLane], limit: usize, ) -> Result> { - let entries = - self.memory - .search_refs_fts(query, lanes, limit.saturating_mul(200).max(limit))?; + let channel_limit = limit.saturating_mul(4).max(limit).max(16); + let mut entries = self.memory.search_refs_fts(query, lanes, channel_limit)?; + if entries.len() < limit { + let mut seen = entries + .iter() + .map(|entry| entry.ref_id.clone()) + .collect::>(); + for entry in self + .memory + .search_refs_trigram(query, lanes, channel_limit)? + { + if seen.insert(entry.ref_id.clone()) { + entries.push(entry); + } + if entries.len() >= limit { + break; + } + } + } entries .into_iter() .take(limit) @@ -1115,6 +1142,146 @@ fn trace_from_ranked_candidates(strategy: &str, candidates: &[EvidenceAtom]) -> } } +fn synthesize_planned_answer(plan: &OpPlan, packets: &[EvidencePacket]) -> Option { + let facts = packet_fact_rows(packets); + if facts.is_empty() { + return None; + } + match plan.op { + QueryOp::Lookup | QueryOp::AbstainOrFalsePremiseCheck => None, + QueryOp::AggregateCount => Some(distinct_fact_count(&facts).to_string()), + QueryOp::AggregateSum => { + let values = numeric_fact_values(&facts); + (!values.is_empty()).then(|| format_number(values.iter().sum::())) + } + QueryOp::AggregateAvg => { + let values = numeric_fact_values(&facts); + (!values.is_empty()) + .then(|| format_number(values.iter().sum::() / values.len().max(1) as f64)) + } + QueryOp::OrderOrRank => { + let ordered = ordered_fact_rows(&facts); + (!ordered.is_empty()).then(|| { + ordered + .into_iter() + .map(render_fact_brief) + .collect::>() + .join(" -> ") + }) + } + QueryOp::UpdateResolution => { + let ordered = ordered_fact_rows(&facts); + match ordered.as_slice() { + [] => None, + [latest] => Some(render_fact_brief(latest)), + rows => { + let previous = rows.get(rows.len().saturating_sub(2))?; + let latest = rows.last()?; + Some(format!( + "previous: {}; latest: {}", + render_fact_brief(previous), + render_fact_brief(latest) + )) + } + } + } + QueryOp::PreferenceRecommendation => { + if !has_personal_support(packets) { + return Some("I don't know.".to_string()); + } + facts + .iter() + .find(|fact| fact.session_id.is_some()) + .or_else(|| facts.first()) + .map(|fact| render_fact_brief(fact)) + } + } +} + +fn packet_fact_rows(packets: &[EvidencePacket]) -> Vec { + let mut rows = Vec::new(); + let mut seen = HashSet::new(); + for packet in packets { + for fact in &packet.fact_rows { + let key = format!( + "{}:{}", + fact.ref_id, + fact.value_text.as_deref().unwrap_or("") + ); + if seen.insert(key) { + rows.push(fact.clone()); + } + } + } + rows +} + +fn distinct_fact_count(facts: &[EvidenceFactRow]) -> usize { + facts + .iter() + .map(|fact| { + fact.session_id + .as_ref() + .map(|session_id| format!("session:{session_id}")) + .unwrap_or_else(|| format!("ref:{}", fact.ref_id)) + }) + .collect::>() + .len() +} + +fn numeric_fact_values(facts: &[EvidenceFactRow]) -> Vec { + let turn_chunk_values = facts + .iter() + .filter(|fact| fact.entity_type == "turn_chunk") + .filter_map(|fact| fact.value_number) + .filter(|value| value.is_finite()) + .collect::>(); + if !turn_chunk_values.is_empty() { + return turn_chunk_values; + } + facts + .iter() + .filter_map(|fact| fact.value_number) + .filter(|value| value.is_finite()) + .collect() +} + +fn ordered_fact_rows(facts: &[EvidenceFactRow]) -> Vec<&EvidenceFactRow> { + let mut ordered = facts.iter().collect::>(); + ordered.sort_by(|left, right| { + left.timestamp + .unwrap_or(i64::MAX) + .cmp(&right.timestamp.unwrap_or(i64::MAX)) + .then_with(|| left.ordinal.cmp(&right.ordinal)) + .then_with(|| left.ref_id.cmp(&right.ref_id)) + }); + ordered +} + +fn render_fact_brief(fact: &EvidenceFactRow) -> String { + fact.value_text + .clone() + .unwrap_or_else(|| truncate_chars(&fact.text, 120)) +} + +fn has_personal_support(packets: &[EvidencePacket]) -> bool { + packets.iter().any(|packet| { + packet.session_id.is_some() + || packet + .lanes + .iter() + .any(|lane| matches!(lane, MemoryLane::Episodic | MemoryLane::Profile)) + }) +} + +fn format_number(value: f64) -> String { + if (value.fract()).abs() < 0.000_001 { + format!("{}", value as i64) + } else { + format!("{value:.2}") + } +} + struct PacketRerankDecision { packets: Vec, status: String, @@ -1919,6 +2086,47 @@ mod tests { Ok(()) } + #[tokio::test] + async fn cjk_no_space_query_uses_trigram_fts_fallback() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + let llm = LlmClient::new(crate::models::ModelsConfig::default()); + let pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) + .with_profile("raw-turns/fts-only"); + let budget = ContextBudget { + max_tokens: 1_000, + top_k: 5, + graph_depth: 0, + }; + + pipeline + .observe_session(BenchSession { + session_id: "cjk-session".to_string(), + timestamp: Some(10), + turns: vec![BenchTurn { + role: "user".to_string(), + content: "秘密場所は図書館裏です".to_string(), + timestamp: Some(10), + }], + }) + .await?; + + let query = BenchQuery { + query_id: "q-cjk".to_string(), + text: "図書館裏".to_string(), + reference_time: Some(11), + }; + let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; + let context = pipeline.assemble_context(query, &retrieval, budget).await?; + + assert!(retrieval + .candidates + .iter() + .any(|candidate| candidate.body.contains("秘密場所"))); + assert!(context.content.contains("秘密場所")); + Ok(()) + } + #[tokio::test] async fn query_trace_uses_lookup_plan_when_structured_planner_is_unavailable() -> Result<()> { let tmp = NamedTempFile::new()?; @@ -2082,6 +2290,183 @@ mod tests { assert_eq!(decision.packets[0].packet_id, "pkt_neighbor"); } + #[test] + fn planned_synthesis_counts_distinct_fact_groups() { + let packets = (0..6) + .map(|idx| { + fact_packet( + &format!("pkt_{idx}"), + &format!("ref_{idx}"), + Some(&format!("s{idx}")), + Some(100 + idx), + None, + None, + &format!("visited museum {idx}"), + MemoryLane::Episodic, + ) + }) + .collect::>(); + let plan = OpPlan::for_op(QueryOp::AggregateCount, "test"); + + let answer = synthesize_planned_answer(&plan, &packets); + + assert_eq!(answer.as_deref(), Some("6")); + } + + #[test] + fn planned_synthesis_computes_sum_and_average_from_numeric_fact_rows() { + let packets = vec![ + fact_packet( + "pkt_first", + "ref_first", + Some("s1"), + Some(10), + Some("10"), + Some(10.0), + "spent 10", + MemoryLane::Episodic, + ), + fact_packet( + "pkt_second", + "ref_second", + Some("s2"), + Some(20), + Some("20"), + Some(20.0), + "spent 20", + MemoryLane::Episodic, + ), + fact_packet( + "pkt_third", + "ref_third", + Some("s3"), + Some(30), + Some("21.5"), + Some(21.5), + "spent 21.5", + MemoryLane::Episodic, + ), + ]; + + let sum = + synthesize_planned_answer(&OpPlan::for_op(QueryOp::AggregateSum, "test"), &packets); + let avg = + synthesize_planned_answer(&OpPlan::for_op(QueryOp::AggregateAvg, "test"), &packets); + + assert_eq!(sum.as_deref(), Some("51.50")); + assert_eq!(avg.as_deref(), Some("17.17")); + } + + #[test] + fn planned_synthesis_orders_fact_rows_by_timestamp() { + let packets = vec![ + fact_packet( + "pkt_third", + "ref_third", + Some("s3"), + Some(30), + Some("third"), + None, + "third event", + MemoryLane::Episodic, + ), + fact_packet( + "pkt_first", + "ref_first", + Some("s1"), + Some(10), + Some("first"), + None, + "first event", + MemoryLane::Episodic, + ), + fact_packet( + "pkt_second", + "ref_second", + Some("s2"), + Some(20), + Some("second"), + None, + "second event", + MemoryLane::Episodic, + ), + ]; + + let answer = + synthesize_planned_answer(&OpPlan::for_op(QueryOp::OrderOrRank, "test"), &packets); + + assert_eq!(answer.as_deref(), Some("first -> second -> third")); + } + + #[test] + fn planned_synthesis_resolves_previous_and_latest_by_timestamp() { + let packets = vec![ + fact_packet( + "pkt_old", + "ref_old", + Some("s1"), + Some(10), + Some("27:45"), + None, + "personal best 27:45", + MemoryLane::Episodic, + ), + fact_packet( + "pkt_new", + "ref_new", + Some("s2"), + Some(20), + Some("26:30"), + None, + "personal best 26:30", + MemoryLane::Episodic, + ), + ]; + let plan = OpPlan::for_op(QueryOp::UpdateResolution, "test"); + + let answer = synthesize_planned_answer(&plan, &packets).unwrap(); + + assert!(answer.contains("previous: 27:45")); + assert!(answer.contains("latest: 26:30")); + } + + #[test] + fn preference_synthesis_abstains_without_personal_support() { + let packets = vec![fact_packet( + "pkt_generic", + "ref_generic", + None, + None, + None, + None, + "generic travel guide says pick a downtown attraction", + MemoryLane::Semantic, + )]; + let plan = OpPlan::for_op(QueryOp::PreferenceRecommendation, "test"); + + let answer = synthesize_planned_answer(&plan, &packets); + + assert_eq!(answer.as_deref(), Some("I don't know.")); + } + + #[test] + fn lookup_plan_keeps_reader_path_even_when_fact_rows_exist() { + let packets = vec![fact_packet( + "pkt_lookup", + "ref_lookup", + Some("s1"), + Some(10), + Some("library"), + None, + "favorite place is the library", + MemoryLane::Episodic, + )]; + + let answer = synthesize_planned_answer(&OpPlan::for_op(QueryOp::Lookup, "test"), &packets); + + assert_eq!(answer, None); + } + #[test] fn packet_rerank_does_not_demote_exact_ref_packets_below_score_only_hits() { let packets = vec![ @@ -2232,6 +2617,62 @@ mod tests { } } + fn fact_packet( + packet_id: &str, + ref_id: &str, + session_id: Option<&str>, + timestamp: Option, + value_text: Option<&str>, + value_number: Option, + text: &str, + lane: MemoryLane, + ) -> EvidencePacket { + EvidencePacket { + packet_id: packet_id.to_string(), + session_id: session_id.map(str::to_string), + anchor_ref: ref_id.to_string(), + packet_kind: EvidencePacketKind::TurnWindow, + refs: vec![ref_id.to_string()], + bodies: vec![text.to_string()], + fact_rows: vec![EvidenceFactRow { + row_id: format!("fact_{packet_id}"), + ref_id: ref_id.to_string(), + entity_type: "turn_chunk".to_string(), + session_id: session_id.map(str::to_string), + timestamp, + ordinal: 0, + value_text: value_text.map(str::to_string), + value_number, + text: text.to_string(), + }], + timeline: timestamp + .map(|timestamp| { + vec![crate::evidence::EvidenceTimelineEvent { + ref_id: ref_id.to_string(), + session_id: session_id.map(str::to_string), + timestamp, + ordinal: 0, + text: text.to_string(), + }] + }) + .unwrap_or_default(), + signals: EvidencePacketSignals { + sources: vec!["fts".to_string()], + seed_count: 1, + best_source: "fts".to_string(), + best_score: 0.0, + per_source: vec![EvidenceSourceSignal { + source: "fts".to_string(), + ref_id: ref_id.to_string(), + rank: 1, + score: 0.0, + }], + }, + lanes: vec![lane], + estimated_tokens: text.chars().count() / 4, + } + } + fn rerank_test_packet( packet_id: &str, packet_kind: EvidencePacketKind, @@ -2244,6 +2685,8 @@ mod tests { packet_kind, refs: vec![format!("{packet_id}_ref")], bodies: vec![body.to_string()], + fact_rows: vec![], + timeline: vec![], signals: EvidencePacketSignals { sources: vec!["fts".to_string()], seed_count: 1, -- 2.51.2