From 3aa95b69bbcc7fdff480c7d7602c5448428164b7 Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Sat, 27 Jun 2026 16:29:33 +0300 Subject: [PATCH] implement evidence packet constraints --- .beads/interactions.jsonl | 4 + .beads/issues.jsonl | 11 +- docs/evidence-selection.md | 4 +- docs/memory-arch.md | 2 +- docs/memory-implementation-status.md | 8 +- docs/principled-evidence-selection-fusion.md | 651 +++++++++++++++++ klbr-core/src/evidence.rs | 86 ++- klbr-core/src/pipeline.rs | 724 ++++++++++++++++--- 8 files changed, 1344 insertions(+), 146 deletions(-) create mode 100644 docs/principled-evidence-selection-fusion.md diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index fbdb56b..187e0cd 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -22,3 +22,7 @@ {"id":"int-b44eac18","kind":"field_change","created_at":"2026-06-26T23:45:35.019415212Z","actor":"dawn","issue_id":"klbr-1yn","extra":{"field":"status","new_value":"blocked","old_value":"open"}} {"id":"int-818764b6","kind":"field_change","created_at":"2026-06-26T23:45:35.350983219Z","actor":"dawn","issue_id":"klbr-wmz","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn."}} {"id":"int-15002d92","kind":"field_change","created_at":"2026-06-27T01:06:10.488930074Z","actor":"dawn","issue_id":"klbr-f0p","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Closed"}} +{"id":"int-bee7772a","kind":"field_change","created_at":"2026-06-27T13:28:09.963406965Z","actor":"dawn","issue_id":"klbr-9al","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented query constraint pass for fuzzy evidence packets with entity-boundary and negated-term suppression, exact-ref bypass, docs sync, and deterministic core fixtures."}} +{"id":"int-75771327","kind":"field_change","created_at":"2026-06-27T13:28:09.976735493Z","actor":"dawn","issue_id":"klbr-l93","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Replaced fixed two-packet same-session cap with ref-overlap suppression for same-session packets, preserving distinct same-session evidence and updating tests/docs."}} +{"id":"int-27353fbf","kind":"field_change","created_at":"2026-06-27T13:28:58.859954398Z","actor":"dawn","issue_id":"klbr-fdr","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Duplicate/stale: lexical ranking policy was already retired in closed klbr-wmz.14; current code uses FTS-native candidate generation and packet fusion. Optional trigram/BGE sparse work can be filed separately if needed."}} +{"id":"int-e9f123e3","kind":"field_change","created_at":"2026-06-27T13:28:59.01170437Z","actor":"dawn","issue_id":"klbr-g5i","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Duplicate of existing blocked klbr-1yn, which already tracks running the official LongMemEval QA evaluator once the upstream evaluator command/environment is available."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 1fc10af..23d5c97 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -1,3 +1,6 @@ +{"_type":"issue","id":"klbr-9al","title":"add entity and negation constraints to evidence packets","description":"benchmark report shows film-vs-camera and unrelated project bleed failures. add lightweight query/evidence entity boundary checks and negation-aware suppression in klbr-core evidence selection, with deterministic fixtures covering valid answer retention and invalid-query bleed.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:19Z","created_by":"dawn","updated_at":"2026-06-27T13:28:10Z","started_at":"2026-06-27T13:16:10Z","closed_at":"2026-06-27T13:28:10Z","close_reason":"Implemented query constraint pass for fuzzy evidence packets with entity-boundary and negated-term suppression, exact-ref bypass, docs sync, and deterministic core fixtures.","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-u3t","title":"finish principled evidence selection rollout","description":"track the remaining follow-through from docs/principled-evidence-selection-fusion.md and docs/benchmark-issues-report.md after packet planning already landed. completed in this pass: entity/negation constraints and same-session ref-overlap diversity. remaining blockers: graph-only corroboration, router/tools-recall recalibration, canonical graph unification, continuous-loop memory benchmarks, and blocked official LongMemEval QA evaluator parity.","status":"open","priority":1,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:19Z","created_by":"dawn","updated_at":"2026-06-27T13:29:07Z","dependencies":[{"issue_id":"klbr-u3t","depends_on_id":"klbr-1yn","type":"blocks","created_at":"2026-06-27T16:28:58Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-6an","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-9al","type":"blocks","created_at":"2026-06-27T16:15:51Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-av1","type":"blocks","created_at":"2026-06-27T16:15:58Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-bqj","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-fdr","type":"blocks","created_at":"2026-06-27T16:16:01Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-g5i","type":"blocks","created_at":"2026-06-27T16:16:09Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-l93","type":"blocks","created_at":"2026-06-27T16:15:54Z","created_by":"dawn","metadata":"{}"}],"dependency_count":9,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-l93","title":"replace same-session cap with overlap-aware packet diversity","description":"EvidencePlanner currently allows at most two non-exact packets per session. replace that coarse cap with overlap/Jaccard-style fingerprint diversity so distinct answer-bearing packets from one session survive while near-duplicates are still omitted.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:18Z","created_by":"dawn","updated_at":"2026-06-27T13:28:10Z","started_at":"2026-06-27T13:20:16Z","closed_at":"2026-06-27T13:28:10Z","close_reason":"Replaced fixed two-packet same-session cap with ref-overlap suppression for same-session packets, preserving distinct same-session evidence and updating tests/docs.","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz.17","title":"Wire benchmark and runtime recall through the same EvidencePlanner","description":"docs/evidence-selection.md calls out the organizational failure mode: fixing only klbr-bench would leave runtime passive recall on legacy retrieval paths. Existing klbr-wmz.2 already tracks runtime passive recall through the canonical pipeline; this task narrows the implementation around a shared EvidencePlanner and packet renderer.","design":"Avoid parallel planner implementations. If a benchmark-only adapter is needed, it should adapt datasets into shared planner inputs, not duplicate evidence selection logic.","acceptance_criteria":"MemoryPipeline retrieve_evidence, MemoryPipeline assemble_context, and runtime passive recall call the same EvidencePlanner or shared packet-building module; packet rendering and trace schema are shared between benchmark and runtime paths; runtime passive recall can emit the same packet kinds and provenance fields as klbr-bench; tests or smoke fixtures show a benchmark packetization win also appears in runtime-style passive recall.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:50:01Z","created_by":"dawn","updated_at":"2026-06-26T19:37:00Z","started_at":"2026-06-26T19:27:37Z","closed_at":"2026-06-26T19:37:00Z","close_reason":"Completed: extracted EvidencePlanner, routed runtime passive recall through MemoryPipeline packets, shared packet rendering, and added a runtime-style no-model packet recall fixture.","labels":["architecture","benchmarks","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.17","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:50:00Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.17","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:48Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.16","title":"Add packet-level evidence selection metrics and fixtures","description":"docs/evidence-selection.md says RecallAny@5 and RecallAll@5 at session level are insufficient because klbr can retrieve the right session but omit the answer-bearing ref. The benchmark and local regression suite need packet-level metrics and deterministic fixtures for right-session/wrong-chunk failures.","design":"This should extend klbr-wmz.7 rather than replace it: keep fixtures tiny and runnable without external model servers wherever possible, then keep a separate model-backed smoke command.","acceptance_criteria":"Benchmark traces/report include answer_bearing_ref_in_context, gold_session_in_context, gold_session_plus_neighbor_in_context, packet_recall_any_at_k, packet_recall_all_at_k, same_session_dedupe_suppression_count, and packet token cost where applicable; deterministic fixtures cover next-turn answer, previous-turn answer, dual-signal same-session merge, explicit-ref plus graph edge, multilingual lexical stress, and false-recall adversary; LongMemEval-S smoke over the first 5 subset rows checks that Target and The Glass Menagerie are answerable from final packets.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:52Z","created_by":"dawn","updated_at":"2026-06-26T20:05:53Z","started_at":"2026-06-26T19:45:24Z","closed_at":"2026-06-26T20:05:53Z","close_reason":"Completed: report/trace now include packet-level answer-ref/session/neighbor/recall/token metrics and planner same-session cap omissions; deterministic no-model fixtures cover next/previous turn windows, dual-signal fusion, explicit-ref graph support, multilingual FTS, false recall, markdown dense hits, and suppression filtering. First-5 LongMemEval-S smoke confirms packet answerability for Target and Glass Menagerie; Target reader abstention remains tracked in klbr-wmz.10.","labels":["architecture","benchmarks","evaluation","memory","tests"],"dependencies":[{"issue_id":"klbr-wmz.16","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:52Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.16","depends_on_id":"klbr-wmz.11","type":"blocks","created_at":"2026-06-26T21:50:44Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.14","title":"Constrain lexical retrieval to FTS-native signals","description":"docs/evidence-selection.md argues that current hand-written lexical helpers in pipeline.rs are doing too much policy work: lexical_relevance, lexical_entry_score, lexical_terms, term_variants, and cue-list routing are english-ish scaffolding that can suppress dense/graph evidence. Lexical retrieval should remain, but production ranking should lean on FTS-native signals and packet fusion.","design":"Keep lexical retrieval, remove lexical policy hacks. If BGE-M3 sparse retrieval is considered, treat it as a separate future channel rather than a prerequisite for the first fix.","acceptance_criteria":"Production packet ranking no longer depends on hand-written english-ish lexical scoring helpers; SQLite FTS bm25 remains a first-stage signal; any phrase, NEAR, alias, unicode61, trigram, or tokenizer changes are implemented as explicit lexical retrieval features with tests; old lexical helpers are removed, debug-only, or trace-only; multilingual/adversarial fixtures show lexical retrieval does not depend on english plural stripping or cue lists.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:33Z","created_by":"dawn","updated_at":"2026-06-26T19:25:30Z","started_at":"2026-06-26T19:24:10Z","closed_at":"2026-06-26T19:25:30Z","close_reason":"Removed rust-side lexical scoring and plural stemming from production retrieval; lexical hits now flow from SQLite FTS bm25 into packet fusion with regression coverage.","labels":["architecture","lexical","memory","multilingual","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.14","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:32Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.14","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:35Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} @@ -11,10 +14,14 @@ {"_type":"issue","id":"klbr-wmz.2","title":"Route runtime passive recall through the canonical memory pipeline","description":"Bench retrieval now goes through MemoryPipeline, but runtime passive recall in klbr-core/src/agent.rs still embeds the prompt, calls MemoryStore::get_searchable, runs retrieval::retrieve_exact over legacy memories, then injects Context memory packets. That bypasses fts, exact refs, markdown notes, graph expansion, and the canonical lane/lifecycle path the docs describe.","design":"Avoid duplicating retrieval logic in agent.rs. Either make MemoryPipeline usable by AgentRuntime or extract a shared retrieval facade that both MemoryPipeline and runtime passive recall call.","acceptance_criteria":"Runtime passive recall uses the same lane-aware canonical retrieval and context packet assembly policy as the production pipeline; recalled packets can include fts/exact/dense/graph candidates from refs and markdown notes; archived/tombstoned/suppressed refs do not leak; klbr-core/src/instructions.md matches the actual memory packet format; tests or a focused integration fixture cover passive recall from a markdown note and from an explicit ref.","notes":"Runtime passive recall now calls MemoryPipeline::retrieve_evidence and injects shared EvidencePacket XML via Context::inject_evidence_packets; no-model runtime packet fixture covers turn-window expansion. Remaining acceptance is blocked on klbr-wmz.1 because dense search still starts from legacy memory rows before ref mapping.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:19Z","created_by":"dawn","updated_at":"2026-06-26T19:44:55Z","started_at":"2026-06-26T19:37:32Z","closed_at":"2026-06-26T19:44:55Z","close_reason":"Completed: runtime passive recall now uses MemoryPipeline/EvidencePlanner packets, instructions document current packet XML, and fixtures cover turn-window recall plus explicit markdown refs; dense canonical dependency completed in klbr-wmz.1.","labels":["architecture","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz.1","type":"blocks","created_at":"2026-06-26T22:37:53Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.1","title":"Move dense retrieval onto canonical refs and embedding_items","description":"Current source still has dense retrieval on the legacy memory-row surface: klbr-core/src/pipeline.rs search_dense calls MemoryStore::get_searchable and retrieval::retrieve_exact, while fts/exact retrieval uses canonical refs and promptable_text. The schema already has refs, promptable_text, markdown_note_chunks, and embedding_items, so dense retrieval should not be the odd path out.","design":"Prefer a ref-native embedding index backed by embedding_items. Backfill embeddings from promptable_text, keep memory-id aliases as compatibility aliases, and make klbr/full versus dense-only profiles exercise the same canonical identity layer as fts and exact retrieval.","acceptance_criteria":"Dense candidate generation works over canonical ref ids for memories, turn chunks, markdown note chunks, episode notes, profile notes, and procedural notes; lane and lifecycle filtering come from refs/ref_metadata instead of memory tags alone; benchmark traces return canonical refs for dense hits; regression tests cover a markdown-note-only hit and a tombstoned/suppressed ref not leaking through dense search.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:11Z","created_by":"dawn","updated_at":"2026-06-26T19:43:27Z","started_at":"2026-06-26T19:38:02Z","closed_at":"2026-06-26T19:43:27Z","close_reason":"Completed: dense candidate generation now lazily backfills embedding_items from active promptable refs, scores canonical ref embeddings directly, and tests markdown-note dense hits plus suppressed-ref filtering.","labels":["architecture","memory","refs","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.1","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:11Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz","title":"Finish memory architecture follow-through","description":"Tracks the remaining memory architecture work identified from docs/memory-arch.md, docs/memory-benches.md, docs/memory-implementation-status.md, and current klbr-core/klbr-bench source. Current status says pipeline, typed memory packets, markdown notes, edge mirroring, lifecycle projection, and benchmark runner integration exist; this epic is for gaps still present in source/docs.","acceptance_criteria":"Close when the child issues are complete, docs/memory-implementation-status.md is updated from current verification, and the architecture docs no longer point at missing or stale follow-up work.","status":"closed","priority":1,"issue_type":"epic","owner":"90008@klbr.net","created_at":"2026-06-26T17:52:54Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","closed_at":"2026-06-26T23:45:35Z","close_reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn.","labels":["architecture","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-g5i","title":"import official LongMemEval QA evaluator","description":"local LongMemEval answer scoring still uses normalized substring matching. bring in or wrap the official evaluate_qa.py flow so klbr-bench can report upstream QA parity through --official-eval-cmd or a first-class command path.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:41Z","created_by":"dawn","updated_at":"2026-06-27T13:28:59Z","closed_at":"2026-06-27T13:28:59Z","close_reason":"Duplicate of existing blocked klbr-1yn, which already tracks running the official LongMemEval QA evaluator once the upstream evaluator command/environment is available.","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-9yo","title":"recalibrate router for tools recall","description":"tools lane report shows router test precision/recall collapsed for gold tools queries even though the prompt fallback sometimes recovered. update router bench sweeps/datasets so toolish abstain behavior is measured explicitly and thresholds optimize tools recall without regressing memory recall.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:36Z","created_by":"dawn","updated_at":"2026-06-27T13:15:36Z","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-fdr","title":"retire handwritten lexical ranking policy","description":"move lexical retrieval policy toward SQLite FTS/BM25/trigram candidate generation and stop relying on english-ish lexical helpers or cue lists as ranking gates. preserve exact refs and dense candidates from lexical suppression, then add multilingual/identifier fixtures.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:32Z","created_by":"dawn","updated_at":"2026-06-27T13:28:59Z","closed_at":"2026-06-27T13:28:59Z","close_reason":"Duplicate/stale: lexical ranking policy was already retired in closed klbr-wmz.14; current code uses FTS-native candidate generation and packet fusion. Optional trigram/BGE sparse work can be filed separately if needed.","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-av1","title":"gate graph-only packets behind corroborating support","description":"principled fusion doc says graph edges may add support but should not surface graph-only packets as top answer evidence. add corroboration rules and traces so graph-only evidence cannot outrank lexical/dense/exact support without another source.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:18Z","created_by":"dawn","updated_at":"2026-06-27T13:15:18Z","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-f0p","title":"Remove tools lane and refactor router to 2-class setup","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T01:06:06Z","created_by":"dawn","updated_at":"2026-06-27T01:06:10Z","closed_at":"2026-06-27T01:06:10Z","close_reason":"Closed","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.18","title":"Keep packet rerank documents under local reranker batch limits","description":"Current full-profile run for LongMemEval-S question 51a45a95 with --rerank-packets produced packet_rerank.status=error from localhost:8003: input 1184 tokens is too large for physical batch size 1024. Packet rerank documents need a hard budget that samples every body without exceeding the local bge-reranker/llama.cpp batch/context limits, and traces should keep the service error visible.","status":"closed","priority":2,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T23:25:26Z","created_by":"dawn","updated_at":"2026-06-26T23:27:53Z","started_at":"2026-06-26T23:25:36Z","closed_at":"2026-06-26T23:27:53Z","close_reason":"Bounded packet rerank documents, added many-body regression, verified klbr-core tests and 51a45a95 packet rerank smoke now returns scores and Target.","labels":["architecture","benchmarks","memory","rerank"],"dependencies":[{"issue_id":"klbr-wmz.18","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-27T02:25:26Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.19","title":"Stabilize reader answers when evidence packets contain the gold answer","description":"Current source retrieves the answer-bearing ref for LongMemEval-S question 51a45a95: answer_bearing_ref_in_context=true, gold_session_plus_neighbor_in_context=true, but a full-profile non-rerank QA run answered 'I don't know' for gold answer Target. A rerank attempt still failed to run due packet size, and that run answered 'Your email inbox'. Treat this as reader/evidence interpretation work, not retrieval recall: add a focused regression or reader/answer-verifier change so where/action questions prefer supported merchant/store/venue evidence over discovery-channel text when the gold answer is in packet context.","status":"closed","priority":2,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T23:25:26Z","created_by":"dawn","updated_at":"2026-06-26T23:33:01Z","started_at":"2026-06-26T23:30:28Z","closed_at":"2026-06-26T23:33:01Z","close_reason":"Centralized reader prompt, made packet-local adjacent bodies explicit evidence windows, added prompt contract test, and verified 51a45a95 non-rerank now answers Target with answer-bearing refs in context.","labels":["architecture","benchmarks","memory","reader"],"dependencies":[{"issue_id":"klbr-wmz.19","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-27T02:25:26Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-1yn","title":"Run real official LongMemEval QA evaluator","description":"The benchmark harness now captures official evaluator stdout/stderr/status and parses official_eval.json when --official-eval-cmd writes it, but this repo does not include the upstream LongMemEval evaluator environment/command. Provide or install the official evaluator, run a small checked-in/local LongMemEval-S QA smoke with answer generation, and record the resulting official metrics in the run artifacts/docs.","notes":"Blocked on external upstream LongMemEval evaluator command/environment. The repo harness supports --official-eval-cmd and parses official_eval.json, but cannot run a real official QA score until that evaluator is provided or installed.","status":"blocked","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T22:53:31Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-1yn","title":"Run real official LongMemEval QA evaluator","description":"The benchmark harness now captures official evaluator stdout/stderr/status and parses official_eval.json when --official-eval-cmd writes it, but this repo does not include the upstream LongMemEval evaluator environment/command. Provide or install the official evaluator, run a small checked-in/local LongMemEval-S QA smoke with answer generation, and record the resulting official metrics in the run artifacts/docs.","notes":"Blocked on external upstream LongMemEval evaluator command/environment. The repo harness supports --official-eval-cmd and parses official_eval.json, but cannot run a real official QA score until that evaluator is provided or installed.","status":"blocked","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T22:53:31Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-d67","title":"document klbr.kdl and bench.kdl secrets and usage","description":"add documentation to AGENTS.md specifying that klbr.kdl and bench.kdl are config files containing potential secrets, used by klbr and the bench suite respectively","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:40:37Z","created_by":"dawn","updated_at":"2026-06-26T22:41:08Z","started_at":"2026-06-26T22:40:43Z","closed_at":"2026-06-26T22:41:08Z","close_reason":"documented klbr.kdl and bench.kdl config usage and secrets under the config.rs section in AGENTS.md","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-djb","title":"update agents.md to reflect latest code state, get rid of anything stale - dont touch rtk / beads instructions","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:27:02Z","created_by":"dawn","updated_at":"2026-06-26T22:31:09Z","started_at":"2026-06-26T22:28:05Z","closed_at":"2026-06-26T22:31:09Z","close_reason":"updated AGENTS.md to match the latest codebase architecture (KDL configs, models.rs, memory lanes, router, evidence, garden, twilight discord)","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.15","title":"Rerank answer-bearing evidence packets instead of raw chunks","description":"docs/evidence-selection.md recommends reranking EvidencePacket objects, not raw chunks or whole sessions. Chunk-level reranking can prefer the query-adjacent setup chunk and still miss the answer-bearing neighbor; whole-session reranking is too broad. Packet reranking should judge whether a budgeted evidence packet contains enough support to answer.","design":"Treat reranking as a tie-breaker until packet-level fixtures show lift. The reranker prompt/input should ask for answer-bearing evidence, not generic relevance.","acceptance_criteria":"The reranker path can score packet bodies that include anchor, local window, episode gist, refs, and source signals; exact-ref packets cannot be suppressed by reranker score alone; traces log pre-rerank and post-rerank packet order; reranker thresholds operate on packets and cannot drop all strong evidence for a gold session without a traceable reason; tests cover a packet that beats its raw anchor chunk because the neighbor contains the answer.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:44Z","created_by":"dawn","updated_at":"2026-06-26T23:19:52Z","started_at":"2026-06-26T23:10:05Z","closed_at":"2026-06-26T23:19:52Z","close_reason":"Implemented packet-level reranking with bounded packet documents, exact-ref protection, traceable threshold fallback, bench flag wiring, tests, and live smoke.","labels":["architecture","context","memory","rerank","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:44Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:39Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} @@ -22,3 +29,5 @@ {"_type":"issue","id":"klbr-wmz.8","title":"Validate the LoCoMo adapter and comparison protocol","description":"docs/memory-benches.md recommends LoCoMo as the secondary public suite for memweaver-style comparison, and docs/memory-implementation-status.md says a flexible locomo adapter exists. Current source parses locomo-ish shapes inside klbr-bench/src/longmemeval.rs, but there is no verified dataset fixture, loader test, scoring protocol, or documented run result proving the adapter matches the intended benchmark semantics.","design":"Keep this as a comparison harness task, not a claim about beating another system. The output should make dataset version, reader model, token budget, and scoring method explicit.","acceptance_criteria":"A small LoCoMo-shaped fixture or documented local dataset path exercises the adapter; loader tests cover supported input shapes and session ordering; the benchmark manifest/report clearly labels LoCoMo runs and avoids claiming comparability without matched reader/budget/scoring; docs include the exact command and current verified result or blocker.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:54:09Z","created_by":"dawn","updated_at":"2026-06-26T23:44:57Z","started_at":"2026-06-26T23:42:41Z","closed_at":"2026-06-26T23:44:57Z","close_reason":"Added LoCoMo smoke fixture, loader tests for session/haystack/conversation shapes, protocol notes in report/manifest, docs with exact command/result, and verified locomo retrieval-only smoke.","labels":["architecture","benchmarks","locomo","memory"],"dependencies":[{"issue_id":"klbr-wmz.8","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:54:09Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.5","title":"Upgrade episodic notes from transcript cards to source-grounded event cards","description":"klbr-core/src/pipeline.rs render_episode_card creates an episode note from timestamp, source refs, and truncated role lines. docs/memory-arch.md asks for source-grounded scene memory that preserves who, when, where, what changed, what was said, available attachments, and supporting raw refs. The current implementation is addressable, but still too transcript-shaped for temporal/update reasoning.","design":"Do not synthesize unsupported vivid details. The event card should summarize only what source turns or attachments support, with raw refs retained as the escape hatch.","acceptance_criteria":"Episode artifacts have a stable structured representation for time anchors, participants/entities, changes/decisions, salient quotes or snippets, attachments when present, and source refs; the markdown rendering remains human-editable; retrieval/context packets can expose the structured gist without losing provenance; tests cover an episode with multiple turns and verify source refs, session id resolution, and useful promptable chunks.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:44Z","created_by":"dawn","updated_at":"2026-06-26T23:36:56Z","started_at":"2026-06-26T23:33:59Z","closed_at":"2026-06-26T23:36:56Z","close_reason":"Episode notes now render as structured episode_event_card artifacts with event metadata, source-grounded timelines, stated facts/decisions, attachment markers, source refs, and promptable retrieval coverage; added core regression.","labels":["architecture","episodic","memory","provenance"],"dependencies":[{"issue_id":"klbr-wmz.5","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:44Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.4","title":"Materialize profile and procedural lanes as first-class notes","description":"The schema and MemoryGarden know about profile and procedural lanes, but the production pipeline observe_session path currently writes raw turns plus episodic notes/memories. Stable preferences, standing instructions, and workflows still depend mostly on tags or model-authored remember calls instead of a first-class markdown-note flow with source refs and update policy.","design":"Build on MemoryGarden and upsert_markdown_note rather than adding a new store. Treat legacy memory tags as routing hints, not the durable source of truth for profile/procedural knowledge.","acceptance_criteria":"There is an explicit writer/reflection path for profile_note and procedural_note artifacts; new notes include source refs, frontmatter, stable paths under profile/ or procedural/, and canonical refs/chunks; updates use supersession or versioning instead of silent overwrite; user-confirmation or policy gates are documented for stable profile changes; tests cover creating and updating one profile note and one procedural note from source turns.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:36Z","created_by":"dawn","updated_at":"2026-06-26T23:42:13Z","started_at":"2026-06-26T23:37:20Z","closed_at":"2026-06-26T23:42:13Z","close_reason":"Added write_memory_note reflection tool for source-grounded profile/procedural markdown notes, ref supersession updates, source/policy frontmatter, stable lane paths, and tests for profile/procedural create/update paths.","labels":["architecture","markdown","memory","profile"],"dependencies":[{"issue_id":"klbr-wmz.4","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:36Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-bqj","title":"add continuous-loop memory benchmark coverage","description":"benchmark suite mostly tests static query retrieval. add a loop-style harness that simulates multi-session observe/compact/reflect cycles and then evaluates whether reflection notes, markdown sync, and passive recall stay aligned over a longer horizon.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:48Z","created_by":"dawn","updated_at":"2026-06-27T13:15:48Z","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-6an","title":"unify memory edge tables behind canonical refs","description":"memory.rs still has memory_edges and ref-level edges as separate provenance graphs. design and migrate toward one transactional canonical-ref links graph so packet planning, provenance, supersession, and markdown-note evidence traverse the same substrate.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:44Z","created_by":"dawn","updated_at":"2026-06-27T13:15:44Z","dependency_count":0,"dependent_count":1,"comment_count":0} diff --git a/docs/evidence-selection.md b/docs/evidence-selection.md index e520d42..7060180 100644 --- a/docs/evidence-selection.md +++ b/docs/evidence-selection.md @@ -133,7 +133,7 @@ i would define `cluster_by_anchor` this way: - if two seeds resolve to the same canonical ref, merge them. - if two turn-chunk seeds are in the same session and within `±1` logical turn step, merge them. - if one seed is a turn chunk and another is the session episode card, merge the episode gist into the packet instead of choosing one. -- if seeds are in the same session but non-overlapping and far apart, allow more than one packet from that session, up to a cap like `max_packets_per_session = 2`. +- if seeds are in the same session but non-overlapping and far apart, allow more than one packet from that session; drop same-session packets by high ref-overlap rather than by a fixed session-count cap. that gives you a very specific replacement for the lossy session dedupe: @@ -401,4 +401,4 @@ the biggest ranking risk is pretending that raw channel scores are commensurate the final risk is organizational, not algorithmic: if you fix this only in `klbr-bench`, you will get a beautiful benchmark curve and a runtime that still uses the old passive-recall path. your own architecture notes already warn that benchmark behavior and runtime behavior are not yet identical. so the durable fix is a shared planner module, shared packet type, shared packet renderer, and shared trace schema across bench and runtime. otherwise the next person will “optimize” one path and quietly resurrect `klbr-wmz.10` in the other. fileciteturn0file0 -the short version is: keep sqlite, keep refs, keep exact resolution, keep evidence packets, and stop letting same-session dedupe or english-ish lexical scaffolding act like truth. the next hard problem in klbr is not more memory storage. it is **packetizing the right evidence and ranking that packet like it might actually answer the question**. \ No newline at end of file +the short version is: keep sqlite, keep refs, keep exact resolution, keep evidence packets, and stop letting same-session dedupe or english-ish lexical scaffolding act like truth. the next hard problem in klbr is not more memory storage. it is **packetizing the right evidence and ranking that packet like it might actually answer the question**. diff --git a/docs/memory-arch.md b/docs/memory-arch.md index 031da5b..e7bfeae 100644 --- a/docs/memory-arch.md +++ b/docs/memory-arch.md @@ -19,7 +19,7 @@ as of the current source, the big pieces that have landed are: - runtime passive recall routes through the same evidence planner and injects typed `` instead of raw recalled rows. - `EvidencePlanner` builds packet-level context units: `exact_ref`, `turn_window`, `episode_bridge`, and `graph_bridge`. - dense retrieval searches canonical refs through `embedding_items`; fts retrieval stays fts-native instead of relying on hand-written lexical relevance as ranking policy. -- packet fusion, same-session caps, answer-bearing packet metrics, lane route traces, and optional packet-level reranking are implemented and covered by deterministic core tests. +- packet fusion, same-session overlap suppression, answer-bearing packet metrics, lane route traces, lightweight entity/negation packet filters, and optional packet-level reranking are implemented and covered by deterministic core tests. - profile and procedural lanes have an explicit source-grounded markdown note writer via `write_memory_note`, including source refs, policy frontmatter, stable lane paths, and supersession updates. the parts below are still target-state or open work: diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index ecdd553..518a7ed 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -29,7 +29,11 @@ updated: 2026-06-27 hand-written lexical relevance helpers are not the ranking-critical path. - `EvidencePlanner` builds `exact_ref`, `turn_window`, `episode_bridge`, and `graph_bridge` packets; packet fusion replaces raw same-session candidate - dedupe. + dedupe with same-session ref-overlap suppression. +- retrieval applies lightweight query constraints after packet planning: + negated query terms suppress fuzzy packets that mention the excluded target, + and narrow entity-boundary conflicts block sibling-entity bleed such as + film-vs-camera recall. exact-ref packets bypass these filters. - episodic notes written by the pipeline are now `episode_event_card` artifacts with structured event metadata, source-grounded timeline snippets, stated facts/decisions, attachment markers, and source refs. @@ -120,7 +124,7 @@ PacketRecallAny@5: 1.0000 PacketRecallAll@5: 1.0000 answer_bearing_ref_in_context: 1.0000 gold_session_plus_neighbor_in_context: 1.0000 -same_session_dedupe_suppression_count: 65 +same_session_dedupe_suppression_count: 65 # historical same-session-cap smoke before overlap suppression ``` reader behavior on the known hard cases: diff --git a/docs/principled-evidence-selection-fusion.md b/docs/principled-evidence-selection-fusion.md new file mode 100644 index 0000000..094b69f --- /dev/null +++ b/docs/principled-evidence-selection-fusion.md @@ -0,0 +1,651 @@ +# principled evidence selection and fusion for klbr + +## executive summary + +klbr’s next bottleneck is exactly what your bench result suggests: the system is often retrieving the right **session** but assembling the wrong **evidence object**. the current stack already has the right raw ingredients for a better design — canonical refs, aliases, edges, promptable text, FTS, embeddings, and a benchmark-facing `MemoryPipeline` facade with `observe_session`, `retrieve_evidence`, `assemble_context`, `answer`, and `complete_stored_context` — but the remaining gap is selection policy, not storage substrate. the uploaded architecture report also already frames the system as three channels today: passive semantic recall, active memory tools, and deterministic reflink resolution, with the caution that runtime and benchmark behavior are not yet fully identical. fileciteturn0file1 + +the immediate failure mode is the one your own bench report calls out: coarse retrieval succeeds, yet reader accuracy drops because same-session dedupe and chunk-first selection suppress answer-bearing neighboring turns. the report explicitly describes the “right session, wrong chunk” gap and attributes it to same-session deduplication that discarded subsequent turns from the same session, even when the actual answer lived in the next turn. fileciteturn0file0 + +my recommendation is to make **evidence packets** the first-class retrieval object. not chunks alone, not sessions alone, and not dense episode cards alone. a packet should be a small, budgeted, ref-native bundle built around an anchor hit and expanded with deterministic neighbors, session gist, and provenance edges only when they increase support density. fusion should happen twice: first at the **anchor level** using safe rank-based fusion across exact, FTS/BM25, dense, and graph channels; then at the **packet level** using grouped reranking and budget-aware packet selection. this design is much closer to what LongMemEval and LoCoMo actually stress — temporal, multi-session, answer-bearing conversational memory — and also matches MemWeaver’s dual-channel insight that structured memory alone is not enough unless it is retrieved together with supporting evidence. citeturn0academia1turn12academia2turn0academia0turn4academia0 + +the practical decisions are: + +1. **replace same-session dedupe with same-packet overlap dedupe**; +2. **keep lexical retrieval, but constrain it** to candidate generation and bounded packet evidence, not policy-level suppression; +3. **rerank mixed evidence packets**, not isolated chunks or whole sessions; +4. **expand chunk hits to neighbors deterministically** with budget caps and query-aware stop conditions; +5. **unify benchmark and runtime passive recall** behind the same packet planner so klbr-bench numbers are not measuring a nicer path than the live agent uses. fileciteturn0file0turn0file1 + +## current failure pattern in klbr + +the primary architectural fact to preserve is that klbr is already ref-native enough to do this cleanly. the memory store schema described in the uploaded report includes canonical `refs`, `ref_aliases`, `edges`, `promptable_text`, `promptable_text_fts`, `markdown_notes`, `markdown_note_chunks`, and `embedding_items`; prompt assembly already treats resolved references as **evidence, not instructions**; and the production-facing pipeline already exposes the right evaluation surface. that means the problem is not “you need a graph db” or “you need a benchmark-only pipeline.” it is “the fusion boundary is in the wrong place.” fileciteturn0file1 + +the observed gap is also exactly the kind LongMemEval was designed to expose. LongMemEval’s question set stresses information extraction, multi-session reasoning, temporal reasoning, knowledge updates, and abstention. those tasks are not solved by session recall alone; they require the final context to contain the specific supporting evidence that the reader needs. WhenLoss makes the same point more explicitly: end-to-end failure under a fixed context budget can come either from write-time loss or retrieval-time loss, and those need to be diagnosed separately with oracle evidence, complete stored memory, and retrieved memory conditions. citeturn0academia1turn4academia0 + +for klbr, the important inference is simple: `RecallAny@5 = 1.0` at the session level is a necessary but very non-sufficient success condition. if your final context contains “i redeemed a coupon on coffee creamer...” but not the later turn that names `Target`, or “i went to a play at the local community theater...” but not the later turn that names `The Glass Menagerie`, then the failure is downstream evidence selection, not upstream session retrieval. your benchmark issues report already documents that exact mode. fileciteturn0file0 + +the live-agent wrinkle matters too. according to the uploaded system report, runtime passive recall still uses the older exact-retrieval lane over searchable `L1MemoryRecord`s in `agent.rs`, while reflink resolution is always assembled separately, and the benchmark harness mostly evaluates the memory retrieval policy rather than the full continuous agent loop. that means a packet planner added only to `klbr-bench` would make the benchmark prettier without actually fixing passive recall in production. the shared abstraction should therefore live under `klbr-core`, then be called by both `pipeline.rs` and the passive-recall path in `agent.rs`. fileciteturn0file1 + +## recommended retrieval and fusion architecture + +the right unit of fusion for klbr is a **packet**. chunks are too brittle, sessions are too coarse, and episode cards alone are too lossy. the packet should be the smallest budgetable object that can still carry an answer-bearing scene: one anchor ref, optional immediate neighbors, optional episode gist, optional one-hop provenance, and the metadata needed for safe rendering. this keeps exact refs and provenance first-class while preventing chunk-level suppression from collapsing the answer. that recommendation also aligns with the direction of recent long-horizon systems: MemWeaver explicitly combines structured memory with supporting evidence, and LongMemEval itself found that indexing granularity and retrieval scope are major design choices affecting downstream QA. citeturn0academia0turn0academia1 + +### fusion should happen at the packet layer + +the retrieval flow should be: + +```mermaid +flowchart TD + A[query] --> B[exact ref resolution] + A --> C[fts or bm25 lexical retrieval] + A --> D[dense retrieval] + A --> E[graph expansion from explicit refs and strong anchors] + + B --> F[anchor atoms] + C --> F + D --> F + E --> F + + F --> G[group by canonical anchor and session] + G --> H[packet planning] + H --> I[packet-level fusion] + I --> J[top packet rerank] + J --> K[budgeted packet selection] + K --> L[context assembly] + L --> M[reader answer] +``` + +the anchor stage should convert every raw hit into a common `EvidenceAtom`: + +```rust +struct EvidenceAtom { + ref_id: String, + session_id: Option, + lane: MemoryLane, + source: RetrievalSource, // Exact | Fts | Dense | Graph + rank_in_source: usize, + raw_score: f32, + token_count: usize, + body: String, + entity_type: EntityType, // TurnChunk | Turn | Episode | Memory | NoteChunk | ... + anchor_kind: AnchorKind, // QueryMatch | ExplicitRef | Neighbor | Provenance +} +``` + +then build `EvidencePacket`s around anchors, not around sessions: + +```rust +struct EvidencePacket { + packet_id: String, + session_id: Option, + anchor_ref: String, + atoms: Vec, + est_tokens: usize, + source_support: BTreeSet, + fused_score: f32, + rerank_score: Option, + overlap_fingerprint: BTreeSet, // canonical refs included +} +``` + +### scoring formulas and deterministic rules + +i would use **rank-based outer fusion** and **score-based inner reranking**. specifically: + +\[ +\text{rrf}(p) = \sum_{s \in S} \frac{w_s}{k_{\text{rrf}} + \text{rank}_s(p)} +\] + +where `S = {exact, fts, dense, graph}` and `rank_s(p)` is the best rank among atoms in packet `p` for source `s`. a good starting point is: + +- `w_exact = 3.0` +- `w_dense = 1.2` +- `w_fts = 1.0` +- `w_graph = 0.6` +- `k_rrf = 20` + +then add deterministic bonuses: + +\[ +\text{packet\_score}(p) = +\text{rrf}(p) ++ \beta_1 \cdot \log(1 + |\text{source\_support}(p)|) ++ \beta_2 \cdot \mathbf{1}[p\ \text{contains explicit ref}] ++ \beta_3 \cdot \mathbf{1}[p\ \text{contains anchor+neighbor pair}] +- \lambda \cdot \log(1 + \text{est\_tokens}(p)) +\] + +where the anchor-plus-neighbor bonus exists specifically to reward answer-bearing conversational adjacency instead of query-adjacent orphan chunks. this is the exact failure the current bench surfaced. fileciteturn0file0 + +the hard deterministic rules should be: + +1. exact refs are always seeds and are never suppressed by lexical or dense hits; +2. same-session evidence is allowed to contribute more than once if it forms distinct packets with low overlap; +3. graph edges can **add** support but cannot remove a lexical or dense packet from consideration; +4. dense and lexical channels may nominate different anchors from the same session, but dedupe should happen on **overlap of included refs**, not on `session_id`; +5. packet selection should be budget-aware and diversity-aware, with per-session soft caps but no session-level hard ban. fileciteturn0file1 + +### fusion strategy comparison + +the table below compares the main fusion choices you asked about. the recommendation is the last row: packet-level grouped fusion followed by grouped reranking. + +| strategy | recall | precision | context cost | implementation complexity | fit for klbr | +|---|---:|---:|---:|---:|---| +| reciprocal-rank fusion over raw chunks | high candidate recall | medium | medium-high | low | good as outer fusion, bad as final selection because it still overvalues query-adjacent chunks | +| weighted raw-score fusion over chunks | medium-high if calibrated | medium | medium-high | medium-high | risky because bm25, dense distance, and graph scores are not naturally commensurate | +| per-session grouping before ranking | high session recall | low-medium | high | low-medium | too coarse; reproduces the current “right session, wrong chunk” problem | +| two-stage grouped rerank over evidence packets | high | high | medium | medium-high | best fit; preserves neighbors, exact refs, and budgetable evidence units | +| episode-only dual-channel fusion | medium | low-medium | low | medium | good for summaries, bad for exact answer-bearing turns | + +rank-based fusion is a safer outer layer than weighted raw-score fusion because it avoids pretending bm25, dense cosine distance, and graph edge counts live on the same numeric scale. hybrid retrieval work keeps finding that multi-stage fusion plus reranking is more robust than single-stage score mixing, and conversational retrieval work has shown RRF to be effective when combining heterogeneous retrieval routes. citeturn9academia2turn13academia2turn13academia1 + +## lexical scoring and reranking + +### lexical scoring should be kept, but sharply constrained + +my position is **keep lexical retrieval, remove handwritten lexical policy**. + +the current `lexical_relevance`, `lexical_entry_score`, `lexical_terms`, `term_variants`, and cue-list lane routing are useful development scaffolding, but they should not be allowed to become the architecture’s real decision boundary. the reason is not just aesthetics. your current lexical helpers encode English-specific assumptions: ASCII-ish token splitting, a hand-written stopword list, plural stripping by trimming trailing `s`, and route cues such as “that / this / it / yes / do it / same”. those are fine for debugging, but they are not principled multilingual retrieval. the uploaded report also already flags lexical false alarms and cross-project bleed as an issue class in passive recall. fileciteturn0file0turn0file1 + +the replacement hierarchy i recommend is: + +1. **SQLite FTS5 BM25 as the default sparse lexical channel** over `promptable_text`; +2. **locale-aware tokenizer support** when locale is known; +3. **trigram fallback** for scripts or identifiers where word segmentation is weak; +4. **optional BGE-M3 sparse retrieval** if your embedder service can expose sparse weights; +5. keep hand-written cue lists only as tiny priors or debug toggles, never as gating logic. + +this fits both the codebase and the literature. SQLite FTS5 already supports BM25 ranking, the default `unicode61` tokenizer, a locale hook for custom tokenizers, and a trigram tokenizer for substring matching; its own docs explicitly note that the Porter tokenizer is English-only and may not improve other languages. BGE-M3 is even more aligned with your current stack, because it was designed as a single multilingual retrieval model supporting **dense**, **sparse**, and **multi-vector** retrieval across more than 100 languages. citeturn7view0turn8view0turn8view1turn8view2turn8view3turn5academia0 + +that gives klbr a clean position: + +- **remove** `lexical_relevance` and `term_variants` from ranking policy; +- **keep** FTS/BM25 as a lexical channel; +- **optionally add** a second sparse channel from BGE-M3 if the embedder supports it; +- **never let lexical alone suppress dense or exact evidence from the same session**. + +in practice: + +```rust +enum LexicalMode { + FtsBm25, + FtsBm25PlusTrigram, + BgeM3Sparse, + BgeM3SparsePlusFts, +} +``` + +and the lane router should stop being a keyword classifier. instead: + +- if there is an explicit ref, route exact-first; +- if the query is discourse-local and referential, prefer live-context resolution first; +- otherwise fan out across lanes with priors, not hard exclusions. + +### the reranker should rank mixed evidence packets + +the reranker should not rank isolated chunks. that simply recreates the existing failure, because a chunk-level reranker can still prefer the query-adjacent mention over the answer-bearing next turn. it also should not rank whole sessions, because session-level reranking spends model capacity deciding among overly broad objects. the right rerank object is a **mixed packet**: + +- anchor chunk or note paragraph, +- up to two neighboring turns, +- optional episode gist, +- optional provenance source refs, +- lane and session metadata, +- explicit marker of which refs are anchor vs neighbors. + +that matches the real task the reader faces: “does this packet contain enough grounded evidence to answer the query?” not “is this singleton chunk semantically similar?” and not “is this whole session vaguely about the same topic?” the most relevant public analogues are systems like MemWeaver, which retrieve structured memory jointly with supporting passages, and general multi-stage ranking pipelines where neural reranking is applied after a broader hybrid retrieval stage. citeturn0academia0turn13academia2 + +a practical reranker input format is: + +```text +[query] +where did i redeem a $5 coupon on coffee creamer? + +[packet] +session: s17 +lane: episodic +anchor: [d1b] "i redeemed a $5 coupon on coffee creamer..." +prev: [d1a] "i was at the store picking up groceries..." +next: [d1c] "i got it at target while i was there." +gist: [m4] "episode s17 ... grocery shopping ..." +sources: [d1a] [d1b] [d1c] [m4] +``` + +the model output should be a support score and a binary answerability judgment: + +```json +{ + "support_score": 0.94, + "contains_direct_answer": true, + "answer_span_refs": ["d1c"] +} +``` + +for training, the clean objective is **pairwise packet ranking** with positives defined as packets containing any gold answer-bearing ref and negatives as packets from the same candidate set that do not. pairwise ranking losses are a natural fit when the main goal is relative ordering within a candidate list rather than calibrated global probability. citeturn11academia0 + +a simple loss is: + +\[ +\mathcal{L}_{pair} = \log(1 + \exp(-(s_{pos} - s_{neg}))) +\] + +with optional auxiliary heads for: + +- `contains_direct_answer` +- `needs_more_neighbors` +- `answer_span_ref` + +if you do not want to train yet, an LLM reranker can emulate the same contract deterministically at temperature 0, then you can distill those judgments into a smaller reranker later. + +## expansion and pipeline changes + +### expansion policy + +the expansion rules should be deterministic enough that you can reason about them in a trace, but flexible enough to avoid dumping whole sessions. + +a good starting policy is: + +| anchor type | mandatory expansion | optional expansion | hard stop | +|---|---|---|---| +| explicit ref to turn chunk | same turn + immediate `±1` turn window | second-hop neighbor if packet still under `min_support_tokens` | stop at `max_packet_tokens` or two speaker alternations | +| FTS hit on turn chunk | same turn + best adjacent turn by query-aware support | episode gist if present and cheap | no full-session dump | +| dense hit on episode card or memory | include episode gist + top 1–2 `derived_from` source turns | neighboring turn if source turn is itself chunked | do not include more than one gist per session | +| graph hit only | include only if corroborated by exact, lexical, or dense support | follow one-hop supports | never surface graph-only packet as top answer packet | +| note paragraph | parent note title + one sibling paragraph before/after | source refs if packet is still thin | no entire note by default | + +the query-aware choice for optional neighbor expansion can stay cheap. it does not need to be a second model call. score each candidate neighbor with: + +\[ +\text{neighbor\_gain}(t) = +\alpha \cdot \text{dense\_sim}(q, t) ++ \gamma \cdot \text{fts\_match}(q, t) ++ \delta \cdot \mathbf{1}[\text{same entity mentions as anchor}] +\] + +and only add it if its marginal gain per token exceeds a threshold. + +this is the right place to include the episode card, too. a good rule is: **if any turn packet from a session is selected, allow at most one short episode gist from that same session if it increases cross-source support and stays under the per-session budget cap**. this mirrors the “structured memory plus evidence” idea from MemWeaver, but preserves your existing ref-native provenance and packet policy. citeturn0academia0 + +### concrete replacement for same-session dedupe in `merge_candidates` + +the replacement should be “merge by canonical ref, then plan packets, then dedupe by packet overlap,” not “drop anything after the first accepted hit from a session.” + +```rust +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)] +enum RetrievalSource { + Exact, + Fts, + Dense, + Graph, +} + +#[derive(Debug, Clone)] +struct PacketParams { + rrf_k: usize, + max_packets: usize, + max_packet_tokens: usize, + min_packet_tokens: usize, + max_packets_per_session: usize, + max_tokens_total: usize, + neighbor_window: usize, + overlap_jaccard_drop: f32, +} + +fn merge_candidates_v2( + query: &BenchQuery, + raw: Vec, + memory: &MemoryStore, + params: &PacketParams, +) -> anyhow::Result> { + let atoms = canonicalize_atoms(raw, memory)?; + let anchors = select_anchor_atoms(&atoms); + let mut packets = Vec::new(); + + for anchor in anchors { + let packet = plan_packet(query, &anchor, &atoms, memory, params)?; + if !packet.atoms.is_empty() { + packets.push(packet); + } + } + + for packet in &mut packets { + packet.fused_score = fused_packet_score(packet, params); + } + + packets.sort_by(score_desc_then_tokens_asc); + + let top_for_rerank = packets.iter().take(params.max_packets * 3).cloned().collect::>(); + let reranked = rerank_packets(query, top_for_rerank)?; + + Ok(select_budgeted_packets(reranked, params)) +} +``` + +the packet planning itself should preserve neighbors: + +```rust +fn plan_packet( + query: &BenchQuery, + anchor: &EvidenceAtom, + all_atoms: &[EvidenceAtom], + memory: &MemoryStore, + params: &PacketParams, +) -> anyhow::Result { + let mut atoms = Vec::new(); + + // anchor is always included + atoms.push(anchor.clone()); + + // exact refs get highest trust and may pull one-hop supports + if anchor.source == RetrievalSource::Exact { + atoms.extend(one_hop_provenance(anchor, memory, params)?); + } + + // conversational neighbor window for turn-like anchors + if matches!(anchor.entity_type, EntityType::Turn | EntityType::TurnChunk) { + atoms.extend(select_turn_neighbors(query, anchor, all_atoms, params)); + } + + // short episode gist if same session and cheap + if let Some(sess) = &anchor.session_id { + if let Some(gist) = best_episode_gist(sess, all_atoms) { + atoms.push(gist); + } + } + + atoms = dedupe_atoms_by_ref(atoms); + atoms.sort_by(anchor_first_then_ord); + + let est_tokens = atoms.iter().map(|a| a.token_count).sum(); + let atoms = trim_packet_to_budget(atoms, params.max_packet_tokens, params.min_packet_tokens); + + Ok(EvidencePacket { + packet_id: format!("pkt:{}", anchor.ref_id), + session_id: anchor.session_id.clone(), + anchor_ref: anchor.ref_id.clone(), + overlap_fingerprint: atoms.iter().map(|a| a.ref_id.clone()).collect(), + source_support: atoms.iter().map(|a| a.source.clone()).collect(), + est_tokens, + fused_score: 0.0, + rerank_score: None, + atoms, + }) +} +``` + +the key selection change is here: + +```rust +fn select_budgeted_packets( + mut packets: Vec, + params: &PacketParams, +) -> Vec { + packets.sort_by(final_packet_order); + + let mut chosen = Vec::new(); + let mut total_tokens = 0usize; + let mut session_counts: HashMap = HashMap::new(); + + 'outer: for packet in packets { + if chosen.len() >= params.max_packets { + break; + } + + if total_tokens + packet.est_tokens > params.max_tokens_total { + continue; + } + + if let Some(session_id) = &packet.session_id { + let used = session_counts.get(session_id).copied().unwrap_or(0); + if used >= params.max_packets_per_session { + continue; + } + } + + for prev in &chosen { + let j = overlap_jaccard(&packet.overlap_fingerprint, &prev.overlap_fingerprint); + if j >= params.overlap_jaccard_drop { + continue 'outer; + } + } + + total_tokens += packet.est_tokens; + if let Some(session_id) = &packet.session_id { + *session_counts.entry(session_id.clone()).or_insert(0) += 1; + } + chosen.push(packet); + } + + chosen +} +``` + +that gives you what the current logic does not: multiple packets from the same session are allowed when they bring **new refs**, while near-duplicates are still suppressed. + +### pipeline pseudocode + +the clean implementation path is to make packet planning a shared core module and thread it through both benchmark and runtime paths. + +```rust +pub async fn retrieve_evidence(&self, query: BenchQuery, budget: ContextBudget) + -> Result +{ + self.memory.sync_reference_indexes()?; + + let exact_atoms = self.resolve_exact_atoms(&query)?; + let fts_atoms = self.retrieve_fts_atoms(&query, budget.top_k)?; + let dense_atoms = self.retrieve_dense_atoms(&query, budget.top_k).await?; + let graph_atoms = self.expand_graph_atoms(&exact_atoms, &fts_atoms, &dense_atoms, budget.graph_depth)?; + + let raw_atoms = concat![exact_atoms, fts_atoms, dense_atoms, graph_atoms]; + let packets = merge_candidates_v2(&query, raw_atoms, &self.memory, &self.packet_params())?; + + Ok(PipelineRetrievalTrace { + query_id: query.query_id, + routed_lanes: routed_lanes(&query), + exact_refs: explicit_refs(&query), + candidates: flatten_packets_for_trace(&packets), + packets, + }) +} +``` + +```rust +pub async fn assemble_context( + &self, + query: BenchQuery, + retrieved: &PipelineRetrievalTrace, + budget: ContextBudget, +) -> Result { + let packets = budgeted_packet_render_plan(&retrieved.packets, budget.max_tokens); + + let mut xml = String::new(); + xml.push_str("\n"); + for packet in packets { + xml.push_str(&render_packet_xml(&packet)); + } + xml.push_str(""); + + Ok(AssembledContext { + query_id: query.query_id, + content: format!( + "\n{}\n\n\n\n{}\n", + xml, + xml_escape(&query.text) + ), + used_refs: packets.iter().flat_map(|p| p.atoms.iter().map(|a| a.ref_id.clone())).collect(), + estimated_tokens: estimate_tokens(&xml), + }) +} +``` + +runtime passive recall should call the same packet planner: + +```rust +async fn passive_recall_packets( + &self, + interrupt: &Interrupt, + ctx: &Context, +) -> Result> { + let query = BenchQuery::from_interrupt(interrupt); + let trace = self.pipeline.retrieve_evidence(query, self.passive_budget()).await?; + Ok(trace.packets) +} +``` + +### canonical ref-native trace example + +```json +{ + "question_id": "51a45a95", + "query": "where did i redeem a $5 coupon on coffee creamer?", + "retrieval_mode": "exact+fts+dense+graph+packet-rerank", + "seed_refs": [], + "candidate_packets": [ + { + "packet_id": "pkt:d1b", + "session_id": "s17", + "anchor_ref": "d1b", + "sources": ["fts", "dense"], + "fused_score": 0.841, + "rerank_score": 0.972, + "atoms": [ + {"ref_id":"d1b","role":"anchor","entity_type":"turn_chunk"}, + {"ref_id":"d1c","role":"neighbor","entity_type":"turn_chunk"}, + {"ref_id":"m4","role":"episode_gist","entity_type":"memory"} + ] + } + ], + "final_used_refs": ["d1b", "d1c", "m4"], + "answer_span_refs": ["d1c"], + "context_tokens": 412, + "latency_ms": { + "fts": 9, + "dense": 37, + "graph": 2, + "packet_plan": 4, + "rerank": 28, + "assemble": 3 + } +} +``` + +## evaluation and rollout + +### what should be measured permanently + +the current benchmark reports already prove that session-level recall can look nearly perfect while answer generation still fails, so the permanent guardrail metrics need to sit at the **final context** level, not just at the candidate-session level. your benchmark issues report already identifies this exact mismatch, and WhenLoss provides the right diagnostic framing for separating write-side from retrieval-side failures. fileciteturn0file0 citeturn4academia0 + +the minimum permanent metric set should be: + +| metric | what it catches | +|---|---| +| `answer_bearing_ref_in_context@k` | whether the final assembled context actually contains the gold ref | +| `gold_neighbor_in_context@k` | whether the system kept the adjacent answer-bearing turn when the anchor was query-adjacent | +| `session_recall@k` | whether the right session was found at all | +| `packet_support_precision@k` | how many selected packets actually contain answer-bearing evidence | +| `same_session_dedupe_suppression_count` | how often a candidate from a gold session was dropped solely due to session dedupe | +| `fts_only / dense_only / graph_only / exact_only / full` delta | which channel is rescuing or hurting recall | +| `false_recall_rate_negated_queries` | passive-recall lexical false alarms | +| `multilingual_fixture_accuracy` | tokenizer and lexical fairness across scripts | +| `tokens_per_answered_question` | whether neighbor expansion is exploding context cost | + +the best public eval stack for this is still **LongMemEval as the main scoreboard**, **LoCoMo as the comparison point for systems like MemWeaver**, and **WhenLoss diagnosis** as the internal debugging lens. LongMemEval is tuned to long-term assistant memory; LoCoMo stresses very long conversations, temporal and causal dynamics; and MemWeaver specifically reports gains on LoCoMo with a dual-channel structured-plus-evidence retrieval design. citeturn0academia1turn12academia2turn0academia0turn4academia0 + +### deterministic fixtures + +these fixtures should live in-repo and run with the actual `MemoryPipeline`, not a benchmark-only retrieval shim. + +| fixture | input shape | expected refs in final packet/context | failure it guards | +|---|---|---|---| +| same-session neighbor rescue | one session; lexical anchor in turn `d1b`, answer in next turn `d1c` | `d1b` and `d1c` | current right-session-wrong-chunk gap | +| dense-episode-to-source rescue | paraphrased query matches episode card `m4`; answer is in source turn `d2f` | `m4` and `d2f` | dense-only episode hit that never surfaces a source turn | +| explicit-ref supremacy | query includes `[d7c]` and asks follow-up | `d7c` plus one-hop support refs | exact refs being diluted by lexical or dense fusion | +| negation false-recall | prior memory says dotfiles; query says “not touching klbr / not about dotfiles” | no memory packet or explicit abstain packet | lexical bleed on negated contexts | +| multilingual lexical | session text in japanese, spanish, or mixed script; answer in turn neighbor | correct turn refs in same script session | english-ish stemming and stopword hacks | +| supersession update | old value archived, new value active with `supersedes` edge | only current active ref plus provenance note | stale memory leakage and update confusion | + +if you want one more fixture, add a seventh: + +| fixture | input shape | expected refs in final packet/context | failure it guards | +|---|---|---|---| +| graph corroboration boundary | graph edge exists but lexical and dense disagree | graph-only packet should not be top answer packet | false recall from overly eager graph expansion | + +### longmemeval smoke command + +use one smoke command that exercises the actual packetized production path: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --profile klbr-packet-v1 \ + --top-k 8 \ + --budget-read 5000 \ + --graph-depth 1 \ + --limit 30 \ + --diagnostic whenloss \ + --out /tmp/klbr-lme-s-packet-v1 +``` + +for pure retrieval debugging on the exact bug class you described, keep a shorter loop too: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --profile klbr-packet-v1 \ + --top-k 8 \ + --budget-read 5000 \ + --graph-depth 1 \ + --retrieval-only \ + --out /tmp/klbr-lme-s-ret-packet-v1 +``` + +the current smoke in your implementation-status report already uses the production `run` path and achieved perfect session retrieval on the subset30 retrieval-only run, which makes it a good baseline for checking whether packet selection improves answer-bearing context without regressing recall. fileciteturn0file1turn0file0 + +### rollout timeline + +```mermaid +gantt + title klbr evidence-selection rollout + dateFormat YYYY-MM-DD + section core retrieval + add EvidenceAtom and EvidencePacket types :a1, 2026-06-28, 3d + replace merge_candidates with packet planner :a2, after a1, 4d + move packet planner into shared klbr-core path :a3, after a2, 2d + + section lexical + remove handwritten lexical ranking policy :b1, 2026-07-02, 2d + add bm25 plus trigram fallback :b2, after b1, 3d + optional bge-m3 sparse integration :b3, after b2, 5d + + section rerank and eval + packet-level reranker prompt or model :c1, 2026-07-08, 4d + add deterministic fixtures and trace metrics :c2, after c1, 3d + run longmemeval plus whenloss diagnostics :c3, after c2, 3d + + section production + wire runtime passive recall to packet planner :d1, 2026-07-15, 3d + shadow mode with trace-only logging :d2, after d1, 4d + full rollout with rollback guard :d3, after d2, 2d +``` + +## risks and failure modes + +the biggest risk is **multilingual lexical brittleness**. SQLite’s default `unicode61` tokenizer is much saner than a hand-written English stopword list, but it is still just a tokenizer, not a full multilingual retrieval policy. SQLite’s docs are very explicit that Porter stemming is English-only, and its locale option only becomes meaningful when you implement a custom tokenizer. that means klbr should not ship any more English plural stripping or hard-coded intent words as policy. if you need a language-agnostic sparse signal, BGE-M3’s sparse mode is a much more principled route because it was built for multilingual dense+sparse retrieval in one model family. citeturn8view2turn8view3turn5academia0 + +the second risk is **false recall from packet expansion**. neighbor windows fix your current failures, but a sloppy expansion rule can also amplify irrelevant but nearby turns. that is why expansion should be anchor-typed, capped, and query-aware, with graph-only packets barred from becoming the top answer packet unless corroborated by another channel. this is also the place where negation fixtures matter, because your uploaded issues report already shows lexical false alarms and entity bleed in passive recall. fileciteturn0file0 + +the third risk is **benchmark/runtime drift**. if packet selection lands only in `MemoryPipeline` and not in `agent.rs`, klbr-bench will tell you that the system is fixed while the live agent still runs the old passive-recall path. the uploaded system report explicitly warns that runtime passive recall still takes its own route today, so the rollout must end with shared `klbr-core` packet planning used in both places. fileciteturn0file1 + +the fourth risk is **overfitting to LongMemEval wording**. LongMemEval is the correct main benchmark for assistant memory, but it is still one benchmark. LoCoMo and newer evaluations such as LoCoMo-Plus show that “memory” failures also occur when the later cue does not look lexically similar to the original evidence and when responses depend on latent constraints rather than direct factual repetition. that is another reason to demote hand-written lexical policy and promote packet reranking over actual support evidence. citeturn0academia1turn12academia2turn0academia2 + +the safest rollout plan is therefore: + +- ship packet planning in **shadow mode** first, writing traces but not changing the final assembled context; +- compare `answer_bearing_ref_in_context`, `same_session_dedupe_suppression_count`, and `tokens_per_answered_question` against the current profile; +- then switch `klbr-bench -- run` to the new planner; +- then wire runtime passive recall to it behind a config flag; +- finally remove the old same-session dedupe path once the deterministic fixtures and LongMemEval smoke stay green. fileciteturn0file0turn0file1 + +the short version is: **klbr should stop fusing raw hits and start selecting budgeted evidence packets**. that one move contains the same-session dedupe bug, gives exact refs a principled home, keeps SQLite and canonical refs, constrains lexical policy, and makes both benchmarking and runtime behavior meaningfully more truthful to what the agent actually needs to answer. \ No newline at end of file diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs index 9dacab8..7fae846 100644 --- a/klbr-core/src/evidence.rs +++ b/klbr-core/src/evidence.rs @@ -2,7 +2,7 @@ use std::collections::{HashMap, HashSet}; use anyhow::Result; -use crate::memory::{MemoryLane, MemoryStore, ResolvedRef, to_base36}; +use crate::memory::{to_base36, MemoryLane, MemoryStore, ResolvedRef}; #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] pub struct RetrievedRef { @@ -95,7 +95,8 @@ impl EvidencePlanner { let mut omissions = Vec::new(); for (index, candidate) in candidates.iter().enumerate() { - let expanded = self.expand_candidate_packet(candidate, index + 1, packet_token_budget)?; + let expanded = + self.expand_candidate_packet(candidate, index + 1, packet_token_budget)?; let packet_id = expanded.packet.packet_id.clone(); omissions.extend(expanded.omissions); if let Some(position) = packet_positions.get(&packet_id).copied() { @@ -131,12 +132,15 @@ impl EvidencePlanner { score: candidate.score, }]; let mut sources = vec![candidate.source.clone()]; - let mut estimated_tokens = candidate.token_count.max(candidate.body.chars().count() / 4); + let mut estimated_tokens = candidate + .token_count + .max(candidate.body.chars().count() / 4); let mut pending_omissions = Vec::new(); let expansion_refs = match packet_kind { EvidencePacketKind::TurnWindow => { - self.memory.turn_window_ref_ids(&candidate.ref_id, 2, 2, 18)? + self.memory + .turn_window_ref_ids(&candidate.ref_id, 2, 2, 18)? } EvidencePacketKind::ExactRef | EvidencePacketKind::EpisodeBridge @@ -303,7 +307,6 @@ pub(crate) fn rank_and_cap_packets( let mut out = Vec::new(); let mut seen_bodies = HashSet::new(); - let mut session_counts: HashMap = HashMap::new(); let mut omissions = Vec::new(); for packet in packets { if out.len() >= limit { @@ -330,15 +333,18 @@ pub(crate) fn rank_and_cap_packets( omissions.extend(omit_packet("duplicate_body", &packet)); continue; } - if packet.packet_kind != EvidencePacketKind::ExactRef { - if let Some(session_id) = &packet.session_id { - let count = session_counts.entry(session_id.clone()).or_default(); - if *count >= 2 { - omissions.extend(omit_packet("same_session_cap", &packet)); - continue; - } - *count += 1; - } + if packet.packet_kind != EvidencePacketKind::ExactRef + && packet.session_id.is_some() + && out + .iter() + .filter(|previous: &&EvidencePacket| { + previous.packet_kind != EvidencePacketKind::ExactRef + && previous.session_id == packet.session_id + }) + .any(|previous| packet_ref_overlap(previous, &packet) >= 0.6) + { + omissions.extend(omit_packet("same_session_overlap", &packet)); + continue; } out.push(packet); } @@ -365,7 +371,8 @@ pub(crate) fn merge_packet_signals(existing: &mut EvidencePacket, incoming: Evid } } existing.signals.seed_count += incoming.signals.seed_count; - if source_priority(&incoming.signals.best_source) < source_priority(&existing.signals.best_source) + if source_priority(&incoming.signals.best_source) + < source_priority(&existing.signals.best_source) || (source_priority(&incoming.signals.best_source) == source_priority(&existing.signals.best_source) && incoming.signals.best_score < existing.signals.best_score) @@ -392,6 +399,17 @@ fn packet_fusion_score(packet: &EvidencePacket) -> f32 { exact_bonus + signal_score + agreement_bonus + support_bonus } +fn packet_ref_overlap(left: &EvidencePacket, right: &EvidencePacket) -> f32 { + let left_refs = left.refs.iter().collect::>(); + let right_refs = right.refs.iter().collect::>(); + let union = left_refs.union(&right_refs).count(); + if union == 0 { + return 0.0; + } + let intersection = left_refs.intersection(&right_refs).count(); + intersection as f32 / union as f32 +} + fn source_weight(source: &str) -> f32 { match source { "exact" => 10.0, @@ -517,12 +535,12 @@ mod tests { } #[test] - fn packet_diversity_cap_happens_after_two_same_session_packets() { + fn packet_diversity_drops_same_session_overlap_not_distinct_packets() { let ranked = rank_and_cap_packets( vec![ - test_packet("pkt_1", "ref_1", "fts", 1, Some("s1")), - test_packet("pkt_2", "ref_2", "fts", 2, Some("s1")), - test_packet("pkt_3", "ref_3", "fts", 3, Some("s1")), + test_packet_with_refs("pkt_1", &["ref_1", "ref_2"], "fts", 1, Some("s1")), + test_packet_with_refs("pkt_2", &["ref_1", "ref_2"], "fts", 2, Some("s1")), + test_packet_with_refs("pkt_3", &["ref_3", "ref_4"], "fts", 3, Some("s1")), ], 5, ); @@ -532,14 +550,14 @@ mod tests { ranked .packets .iter() - .filter(|packet| packet.session_id.as_deref() == Some("s1")) - .count(), - 2 + .map(|packet| packet.packet_id.as_str()) + .collect::>(), + vec!["pkt_1", "pkt_3"] ); assert!(ranked .omissions .iter() - .any(|omission| omission.reason == "same_session_cap")); + .any(|omission| omission.reason == "same_session_overlap")); } fn test_packet( @@ -549,13 +567,27 @@ mod tests { rank: usize, session_id: Option<&str>, ) -> EvidencePacket { + test_packet_with_refs(packet_id, &[ref_id], source, rank, session_id) + } + + fn test_packet_with_refs( + packet_id: &str, + refs: &[&str], + source: &str, + rank: usize, + session_id: Option<&str>, + ) -> EvidencePacket { + let anchor_ref = refs.first().copied().unwrap_or("ref_empty"); EvidencePacket { packet_id: packet_id.to_string(), session_id: session_id.map(str::to_string), - anchor_ref: ref_id.to_string(), + anchor_ref: anchor_ref.to_string(), packet_kind: EvidencePacketKind::TurnWindow, - refs: vec![ref_id.to_string()], - bodies: vec![format!("body for {ref_id}")], + refs: refs.iter().map(|ref_id| ref_id.to_string()).collect(), + bodies: refs + .iter() + .map(|ref_id| format!("body for {packet_id} {ref_id}")) + .collect(), signals: EvidencePacketSignals { sources: vec![source.to_string()], seed_count: 1, @@ -563,7 +595,7 @@ mod tests { best_score: 0.0, per_source: vec![EvidenceSourceSignal { source: source.to_string(), - ref_id: ref_id.to_string(), + ref_id: anchor_ref.to_string(), rank, score: 0.0, }], diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index d1a0be2..1fbb5f5 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -1,4 +1,4 @@ -use std::collections::{HashMap, VecDeque}; +use std::collections::{HashMap, HashSet, VecDeque}; use std::time::Duration; use anyhow::Result; @@ -7,12 +7,12 @@ use crate::{ config::MemoryConfig, context::extract_ref_codes, evidence::{ - EvidenceOmittedRef, EvidencePacket, EvidencePacketKind, EvidencePlanner, - render_evidence_packet, stable_base36, + render_evidence_packet, stable_base36, EvidenceOmittedRef, EvidencePacket, + EvidencePacketKind, EvidencePlanner, }, memory::{ - MarkdownNoteInput, MemoryLane, MemoryStore, RefSearchEntry, ResolvedRef, to_base26_suffix, - to_base36, + to_base26_suffix, to_base36, MarkdownNoteInput, MemoryLane, MemoryStore, RefSearchEntry, + ResolvedRef, }, models::{LlmClient, Message, RerankResult}, mvp::{MemoryLayer, MemoryRecordInput, MemoryStatus}, @@ -89,7 +89,11 @@ pub struct LaneRoute { impl Default for LaneRoute { fn default() -> Self { Self { - lanes: vec![MemoryLane::Semantic, MemoryLane::Episodic, MemoryLane::Profile], + lanes: vec![ + MemoryLane::Semantic, + MemoryLane::Episodic, + MemoryLane::Profile, + ], reason: "semantic_default".to_string(), archival_allowed: true, } @@ -164,9 +168,8 @@ impl PipelineProfile { let raw_turns = lower.contains("raw-turns"); let fts_only = lower.contains("fts-only"); let dense_only = lower.contains("dense-only"); - let no_graph = lower.contains("no-graph") - || lower.ends_with("/semantic") - || lower == "semantic"; + let no_graph = + lower.contains("no-graph") || lower.ends_with("/semantic") || lower == "semantic"; Self { name: name.to_string(), write_episode_memories: !raw_turns, @@ -221,15 +224,13 @@ impl MemoryPipeline { "benchmark_session": true, "turn_index": turn_idx, }); - let entry = self - .memory - .log_turn_with_metadata_at( - role, - &turn.content, - None, - &metadata, - turn.timestamp.unwrap_or(timestamp), - )?; + let entry = self.memory.log_turn_with_metadata_at( + role, + &turn.content, + None, + &metadata, + turn.timestamp.unwrap_or(timestamp), + )?; turn_ids.push(entry.id); let turn_alias = format!("d{}", to_base36(entry.id as u64)); @@ -248,7 +249,10 @@ impl MemoryPipeline { lane: MemoryLane::Episodic, kind: "episode_event_card".to_string(), title: format!("episode {}", session.session_id), - path: Some(format!("episodes/e{}.md", stable_base36(&session.session_id))), + path: Some(format!( + "episodes/e{}.md", + stable_base36(&session.session_id) + )), body: episode_body.clone(), sources: source_refs.clone(), follow: None, @@ -258,38 +262,39 @@ impl MemoryPipeline { }); } - let episode_memory_id = if episode_body.trim().is_empty() || !self.profile.write_episode_memories { - None - } else { - let emb = self.llm.embed(&episode_body).await?; - let id = self.memory.store_with_metadata(&MemoryRecordInput { - memory_id: None, - namespace: "default".to_string(), - layer: MemoryLayer::L1, - text: episode_body, - event_time: timestamp, - ingest_time: unix_timestamp(), - embedding_model: self.llm.config.embedder.model.clone(), - embedding_dim: emb.len(), - embedding_version: "pipeline".to_string(), - status: MemoryStatus::Active, - source_ref: Some(format!("session:{}", session.session_id)), - tags: vec![ - "lane:episodic".to_string(), - format!("session:{}", session.session_id), - ], - pinned: false, - embedding: emb, - })?; - - let episode_alias = format!("m{}", to_base36(id as u64)); - for source_ref in &source_refs { - let _ = self - .memory - .add_reflink_edge(&episode_alias, source_ref, "derived_from"); - } - Some(id) - }; + let episode_memory_id = + if episode_body.trim().is_empty() || !self.profile.write_episode_memories { + None + } else { + let emb = self.llm.embed(&episode_body).await?; + let id = self.memory.store_with_metadata(&MemoryRecordInput { + memory_id: None, + namespace: "default".to_string(), + layer: MemoryLayer::L1, + text: episode_body, + event_time: timestamp, + ingest_time: unix_timestamp(), + embedding_model: self.llm.config.embedder.model.clone(), + embedding_dim: emb.len(), + embedding_version: "pipeline".to_string(), + status: MemoryStatus::Active, + source_ref: Some(format!("session:{}", session.session_id)), + tags: vec![ + "lane:episodic".to_string(), + format!("session:{}", session.session_id), + ], + pinned: false, + embedding: emb, + })?; + + let episode_alias = format!("m{}", to_base36(id as u64)); + for source_ref in &source_refs { + let _ = + self.memory + .add_reflink_edge(&episode_alias, source_ref, "derived_from"); + } + Some(id) + }; let episode_ref = episode_memory_id.map(|id| format!("m{}", to_base36(id as u64))); Ok(WriteTrace { @@ -308,7 +313,9 @@ impl MemoryPipeline { budget: ContextBudget, ) -> Result { self.memory.sync_reference_indexes()?; - let exact_refs = extract_ref_codes(&query.text).into_iter().collect::>(); + let exact_refs = extract_ref_codes(&query.text) + .into_iter() + .collect::>(); let route = route_memory_query(&query.text, !exact_refs.is_empty()); let routed_lanes = route.lanes.clone(); @@ -319,8 +326,13 @@ impl MemoryPipeline { } if route.archival_allowed && self.profile.dense { candidates.extend( - self.search_dense(&query.text, query.reference_time, &routed_lanes, budget.top_k) - .await?, + self.search_dense( + &query.text, + query.reference_time, + &routed_lanes, + budget.top_k, + ) + .await?, ); } @@ -329,12 +341,17 @@ impl MemoryPipeline { candidates.extend(self.expand_graph(&seed_refs, budget.top_k)?); } - let candidates = rank_seed_candidates(candidates, budget.top_k.saturating_mul(4).max(budget.top_k)); - let packet_plan = self - .evidence_planner() - .build_packets(&candidates, budget.top_k, budget.max_tokens.min(900))?; + let candidates = + rank_seed_candidates(candidates, budget.top_k.saturating_mul(4).max(budget.top_k)); + let mut packet_plan = self.evidence_planner().build_packets( + &candidates, + budget.top_k, + budget.max_tokens.min(900), + )?; + let constrained = constrain_packets_for_query(&query.text, packet_plan.packets); + packet_plan.omissions.extend(constrained.omissions); let (packets, packet_rerank) = self - .rerank_packets_if_enabled(&query.text, packet_plan.packets) + .rerank_packets_if_enabled(&query.text, constrained.packets) .await; Ok(PipelineRetrievalTrace { query_id: query.query_id, @@ -362,7 +379,11 @@ impl MemoryPipeline { let evidence_packets = if retrieved.packets.is_empty() && !retrieved.candidates.is_empty() { fallback_packets = self .evidence_planner() - .build_packets(&retrieved.candidates, budget.top_k, budget.max_tokens.min(900))? + .build_packets( + &retrieved.candidates, + budget.top_k, + budget.max_tokens.min(900), + )? .packets; fallback_packets.as_slice() } else { @@ -420,11 +441,15 @@ impl MemoryPipeline { .into_iter() .map(|entry| self.ref_search_entry_to_retrieved(entry)) .collect::>>()?; - let packet_plan = self - .evidence_planner() - .build_packets(&candidates, budget.top_k, budget.max_tokens.min(900))?; + let mut packet_plan = self.evidence_planner().build_packets( + &candidates, + budget.top_k, + budget.max_tokens.min(900), + )?; + let constrained = constrain_packets_for_query(&query.text, packet_plan.packets); + packet_plan.omissions.extend(constrained.omissions); let (packets, packet_rerank) = self - .rerank_packets_if_enabled(&query.text, packet_plan.packets) + .rerank_packets_if_enabled(&query.text, constrained.packets) .await; let retrieved = PipelineRetrievalTrace { query_id: query.query_id.clone(), @@ -439,11 +464,7 @@ impl MemoryPipeline { self.assemble_context(query, &retrieved, budget).await } - pub async fn answer( - &self, - query: BenchQuery, - budget: ContextBudget, - ) -> Result { + pub async fn answer(&self, query: BenchQuery, budget: ContextBudget) -> Result { let retrieval = self.retrieve_evidence(query.clone(), budget).await?; let context = self .assemble_context(query.clone(), &retrieval, budget) @@ -474,9 +495,9 @@ impl MemoryPipeline { lanes: &[MemoryLane], limit: usize, ) -> Result> { - let entries = self - .memory - .search_refs_fts(query, lanes, limit.saturating_mul(200).max(limit))?; + let entries = + self.memory + .search_refs_fts(query, lanes, limit.saturating_mul(200).max(limit))?; entries .into_iter() .take(limit) @@ -738,11 +759,8 @@ fn apply_packet_rerank( .map(|result| result.score) .filter(|score| score.is_finite()) .collect::>(); - sorted_scores.sort_by(|left, right| { - right - .partial_cmp(left) - .unwrap_or(std::cmp::Ordering::Equal) - }); + sorted_scores + .sort_by(|left, right| right.partial_cmp(left).unwrap_or(std::cmp::Ordering::Equal)); let Some(top_score) = sorted_scores.first().copied() else { return PacketRerankDecision { packets, @@ -783,8 +801,9 @@ fn apply_packet_rerank( packet_exact_priority(right) .cmp(&packet_exact_priority(left)) .then_with(|| { - packet_promotion_priority(right, &score_by_packet, top_score, config) - .cmp(&packet_promotion_priority(left, &score_by_packet, top_score, config)) + packet_promotion_priority(right, &score_by_packet, top_score, config).cmp( + &packet_promotion_priority(left, &score_by_packet, top_score, config), + ) }) .then_with(|| { packet_score(right, &score_by_packet) @@ -833,7 +852,10 @@ fn packet_promotion_priority( } fn packet_ids(packets: &[EvidencePacket]) -> Vec { - packets.iter().map(|packet| packet.packet_id.clone()).collect() + packets + .iter() + .map(|packet| packet.packet_id.clone()) + .collect() } fn packet_rerank_scores( @@ -920,6 +942,288 @@ fn drain_source( } } +struct PacketConstraintOutcome { + packets: Vec, + omissions: Vec, +} + +#[derive(Debug, Clone)] +struct QueryEvidenceConstraints { + query_terms: HashSet, + negated_terms: Vec, +} + +impl QueryEvidenceConstraints { + fn from_query(query: &str) -> Self { + let negated_terms = negated_constraint_terms(query); + let negated = negated_terms.iter().cloned().collect::>(); + let query_terms = normalized_constraint_terms(query) + .into_iter() + .filter(|term| !negated.contains(term)) + .collect::>(); + Self { + query_terms, + negated_terms, + } + } + + fn is_empty(&self) -> bool { + self.negated_terms.is_empty() + && entity_boundary_classes_for_query(&self.query_terms).is_empty() + } + + fn rejection_reason(&self, packet: &EvidencePacket) -> Option { + if packet.packet_kind == EvidencePacketKind::ExactRef || self.is_empty() { + return None; + } + + let packet_terms = packet_terms(packet); + if let Some(term) = self + .negated_terms + .iter() + .find(|term| packet_terms.contains(term.as_str())) + { + return Some(format!("negated_query_term:{term}")); + } + + entity_boundary_mismatch(&self.query_terms, &packet_terms) + } +} + +fn constrain_packets_for_query( + query: &str, + packets: Vec, +) -> PacketConstraintOutcome { + let constraints = QueryEvidenceConstraints::from_query(query); + if constraints.is_empty() { + return PacketConstraintOutcome { + packets, + omissions: vec![], + }; + } + + let mut kept = Vec::new(); + let mut omissions = Vec::new(); + for packet in packets { + if let Some(reason) = constraints.rejection_reason(&packet) { + omissions.extend(packet.refs.iter().map(|ref_id| EvidenceOmittedRef { + packet_id: packet.packet_id.clone(), + ref_id: ref_id.clone(), + reason: reason.clone(), + })); + } else { + kept.push(packet); + } + } + + PacketConstraintOutcome { + packets: kept, + omissions, + } +} + +fn negated_constraint_terms(query: &str) -> Vec { + const NEGATION_MARKERS: &[&str] = &[ + "not touching", + "not about", + "not working on", + "not using", + "not related to", + "unrelated to", + "without", + "excluding", + "exclude", + ]; + + let lower = query + .chars() + .flat_map(char::to_lowercase) + .collect::(); + let mut out = Vec::new(); + for marker in NEGATION_MARKERS { + let mut cursor = 0usize; + while let Some(offset) = lower[cursor..].find(marker) { + let after_marker = cursor + offset + marker.len(); + for term in constraint_terms_until_boundary(&lower[after_marker..], 4) { + if !out.contains(&term) { + out.push(term); + } + } + cursor = after_marker; + } + } + out +} + +fn constraint_terms_until_boundary(value: &str, limit: usize) -> Vec { + let local = value + .split(|ch| [',', '.', ';', ':', '!', '?', '\n'].contains(&ch)) + .next() + .unwrap_or_default(); + normalized_constraint_terms(local) + .into_iter() + .take(limit) + .collect() +} + +fn packet_terms(packet: &EvidencePacket) -> HashSet { + packet + .bodies + .iter() + .flat_map(|body| normalized_constraint_terms(body)) + .collect() +} + +fn normalized_constraint_terms(value: &str) -> Vec { + value + .split(|ch: char| !ch.is_alphanumeric() && ch != '_') + .filter_map(|term| { + let term = term + .chars() + .flat_map(char::to_lowercase) + .collect::(); + let len = term.chars().count(); + (!constraint_stopword(&term) && len >= 3).then_some(term) + }) + .collect() +} + +fn constraint_stopword(term: &str) -> bool { + matches!( + term, + "the" + | "and" + | "for" + | "are" + | "but" + | "not" + | "you" + | "your" + | "our" + | "was" + | "were" + | "has" + | "had" + | "his" + | "her" + | "she" + | "him" + | "its" + | "what" + | "who" + | "why" + | "how" + | "did" + | "does" + | "can" + | "could" + | "would" + | "with" + | "from" + | "that" + | "this" + | "there" + | "today" + | "current" + | "currently" + | "working" + | "touching" + | "using" + | "about" + | "related" + | "unrelated" + | "without" + | "exclude" + | "excluding" + ) +} + +const ENTITY_BOUNDARY_CLASSES: &[&[&[&str]]] = &[ + &[ + &["film", "films", "movie", "movies"], + &[ + "camera", + "cameras", + "photo", + "photos", + "photograph", + "photographs", + ], + ], + &[ + &["uncle"], + &["aunt"], + &["niece"], + &["nephew"], + &["cousin"], + &["mother", "mom"], + &["father", "dad"], + &["sister"], + &["brother"], + ], + &[&["party", "parties"], &["cake", "cakes"]], +]; + +fn entity_boundary_classes_for_query(query_terms: &HashSet) -> Vec> { + ENTITY_BOUNDARY_CLASSES + .iter() + .map(|classes| { + classes + .iter() + .enumerate() + .filter_map(|(index, variants)| { + variants + .iter() + .any(|term| query_terms.contains(*term)) + .then_some(index) + }) + .collect::>() + }) + .filter(|indices| !indices.is_empty()) + .collect() +} + +fn entity_boundary_mismatch( + query_terms: &HashSet, + packet_terms: &HashSet, +) -> Option { + for classes in ENTITY_BOUNDARY_CLASSES { + let query_indices = classes + .iter() + .enumerate() + .filter_map(|(index, variants)| { + variants + .iter() + .any(|term| query_terms.contains(*term)) + .then_some(index) + }) + .collect::>(); + if query_indices.is_empty() { + continue; + } + + let packet_has_query_class = query_indices.iter().any(|index| { + classes[*index] + .iter() + .any(|term| packet_terms.contains(*term)) + }); + let conflicting_class = classes.iter().enumerate().find(|(index, variants)| { + !query_indices.contains(index) + && variants.iter().any(|term| packet_terms.contains(*term)) + }); + + if !packet_has_query_class { + if let Some((_, variants)) = conflicting_class { + return Some(format!( + "entity_boundary_mismatch:{}:{}", + classes[query_indices[0]][0], variants[0] + )); + } + } + } + None +} + fn xml_escape(value: &str) -> String { value .replace('&', "&") @@ -953,7 +1257,11 @@ fn chunk_count_for_role(role: &str, content: &str) -> usize { } fn render_episode_card(session: &BenchSession, timestamp: i64, source_refs: &[String]) -> String { - if session.turns.iter().all(|turn| turn.content.trim().is_empty()) { + if session + .turns + .iter() + .all(|turn| turn.content.trim().is_empty()) + { return String::new(); } let participants = participant_roles(session).join(", "); @@ -964,7 +1272,10 @@ fn render_episode_card(session: &BenchSession, timestamp: i64, source_refs: &[St .map(|source| format!("[{source}]")) .collect::>(); if source_refs.len() > source_labels.len() { - source_labels.push(format!("[+{} more refs]", source_refs.len() - source_labels.len())); + source_labels.push(format!( + "[+{} more refs]", + source_refs.len() - source_labels.len() + )); } let sources = source_labels.join(" "); let timeline = session @@ -999,11 +1310,16 @@ fn render_episode_card(session: &BenchSession, timestamp: i64, source_refs: &[St .and_then(|value| value.as_ref()) .map(|value| format!("[{value}]")) .unwrap_or_else(|| "[unref]".to_string()); - format!("- {} {}", ref_label, truncate_chars(turn.content.trim(), 180)) + format!( + "- {} {}", + ref_label, + truncate_chars(turn.content.trim(), 180) + ) }) .collect::>(); let stated_facts = if stated_facts.is_empty() { - "- none explicitly stated by the user; inspect the source-grounded timeline above.".to_string() + "- none explicitly stated by the user; inspect the source-grounded timeline above." + .to_string() } else { stated_facts.join("\n") }; @@ -1143,7 +1459,14 @@ fn route_memory_query(query: &str, has_explicit_refs: bool) -> LaneRoute { }; } if [ - "when", "before", "after", "last", "yesterday", "session", "earlier", "again", + "when", + "before", + "after", + "last", + "yesterday", + "session", + "earlier", + "again", "previous", ] .iter() @@ -1156,23 +1479,44 @@ fn route_memory_query(query: &str, has_explicit_refs: bool) -> LaneRoute { }; } if [ - "prefer", "preference", "like", "dislike", "favorite", "profile", "who am i", + "prefer", + "preference", + "like", + "dislike", + "favorite", + "profile", + "who am i", ] .iter() .any(|cue| lower.contains(cue)) { return LaneRoute { - lanes: vec![MemoryLane::Profile, MemoryLane::Semantic, MemoryLane::Episodic], + lanes: vec![ + MemoryLane::Profile, + MemoryLane::Semantic, + MemoryLane::Episodic, + ], reason: "profile_query".to_string(), archival_allowed: true, }; } - if ["how do we", "workflow", "procedure", "policy", "always", "habit"] - .iter() - .any(|cue| lower.contains(cue)) + if [ + "how do we", + "workflow", + "procedure", + "policy", + "always", + "habit", + ] + .iter() + .any(|cue| lower.contains(cue)) { return LaneRoute { - lanes: vec![MemoryLane::Procedural, MemoryLane::Semantic, MemoryLane::Profile], + lanes: vec![ + MemoryLane::Procedural, + MemoryLane::Semantic, + MemoryLane::Profile, + ], reason: "procedural_query".to_string(), archival_allowed: true, }; @@ -1313,15 +1657,23 @@ mod tests { async fn exact_ref_packet_includes_graph_support() -> Result<()> { let tmp = NamedTempFile::new()?; let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; - let anchor_id = store.store("anchor memory mentions project atlas", &[1.0, 0.0, 0.0, 0.0], &[])?; - let support_id = store.store("support memory says atlas uses blue quartz", &[0.0, 1.0, 0.0, 0.0], &[])?; + let anchor_id = store.store( + "anchor memory mentions project atlas", + &[1.0, 0.0, 0.0, 0.0], + &[], + )?; + let support_id = store.store( + "support memory says atlas uses blue quartz", + &[0.0, 1.0, 0.0, 0.0], + &[], + )?; let anchor_alias = format!("m{}", crate::memory::to_base36(anchor_id as u64)); let support_alias = format!("m{}", crate::memory::to_base36(support_id as u64)); store.add_reflink_edge(&anchor_alias, &support_alias, "supports")?; let llm = LlmClient::new(crate::models::ModelsConfig::default()); - let pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) - .with_profile("fts-only"); + let pipeline = + MemoryPipeline::new(store, llm, MemoryConfig::default()).with_profile("fts-only"); let budget = ContextBudget { max_tokens: 1_000, top_k: 5, @@ -1349,8 +1701,8 @@ mod tests { let tmp = NamedTempFile::new()?; let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; let llm = LlmClient::new(crate::models::ModelsConfig::default()); - let mut pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) - .with_profile("klbr-full"); + let mut pipeline = + MemoryPipeline::new(store, llm, MemoryConfig::default()).with_profile("klbr-full"); pipeline.profile.write_episode_notes = true; pipeline.profile.write_episode_memories = false; @@ -1397,9 +1749,10 @@ mod tests { Some("event-session".to_string()) ); - let hits = pipeline - .memory - .search_refs_fts("coffee creamer Target", &[MemoryLane::Episodic], 8)?; + let hits = + pipeline + .memory + .search_refs_fts("coffee creamer Target", &[MemoryLane::Episodic], 8)?; assert!(hits.iter().any(|entry| { entry.body.contains("coffee creamer") && pipeline @@ -1487,6 +1840,123 @@ mod tests { Ok(()) } + #[tokio::test] + async fn entity_boundary_filter_blocks_sibling_entity_bleed() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + let llm = LlmClient::new(crate::models::ModelsConfig::default()); + let pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) + .with_profile("raw-turns/fts-only"); + let budget = ContextBudget { + max_tokens: 1_000, + top_k: 5, + graph_depth: 0, + }; + + pipeline + .observe_session(BenchSession { + session_id: "camera-session".to_string(), + timestamp: Some(1), + turns: vec![BenchTurn { + role: "user".to_string(), + content: "I have been collecting vintage cameras for three months.".to_string(), + timestamp: Some(1), + }], + }) + .await?; + + let query = BenchQuery { + query_id: "q-vintage-films".to_string(), + text: "how long have I been collecting vintage films?".to_string(), + reference_time: Some(3), + }; + let retrieval = pipeline.retrieve_evidence(query, budget).await?; + + assert!(!retrieval.candidates.is_empty()); + assert!(retrieval.packets.is_empty()); + assert!(retrieval.packet_omissions.iter().any(|omission| omission + .reason + .starts_with("entity_boundary_mismatch:film:camera"))); + Ok(()) + } + + #[tokio::test] + async fn negated_query_term_filter_blocks_excluded_project_bleed() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + let llm = LlmClient::new(crate::models::ModelsConfig::default()); + let pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) + .with_profile("raw-turns/fts-only"); + let budget = ContextBudget { + max_tokens: 1_000, + top_k: 5, + graph_depth: 0, + }; + + pipeline + .observe_session(BenchSession { + session_id: "dotfiles-klbr-session".to_string(), + timestamp: Some(1), + turns: vec![BenchTurn { + role: "user".to_string(), + content: "klbr note: the dotfiles syncer shares the rust cli harness." + .to_string(), + timestamp: Some(1), + }], + }) + .await?; + + let query = BenchQuery { + query_id: "q-not-klbr".to_string(), + text: "working on the dotfiles syncer today, not touching klbr".to_string(), + reference_time: Some(3), + }; + let retrieval = pipeline.retrieve_evidence(query, budget).await?; + + assert!(!retrieval.candidates.is_empty()); + assert!(retrieval.packets.is_empty()); + assert!(retrieval + .packet_omissions + .iter() + .any(|omission| omission.reason == "negated_query_term:klbr")); + Ok(()) + } + + #[tokio::test] + async fn exact_ref_packets_bypass_query_constraint_filters() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + let memory_id = store.store( + "klbr exact note: the dotfiles syncer borrowed the rust cli harness.", + &[1.0, 0.0, 0.0, 0.0], + &[], + )?; + let alias = format!("m{}", crate::memory::to_base36(memory_id as u64)); + let llm = LlmClient::new(crate::models::ModelsConfig::default()); + let pipeline = MemoryPipeline::new(store, llm, MemoryConfig::default()) + .with_profile("fts-only/no-graph"); + let budget = ContextBudget { + max_tokens: 1_000, + top_k: 5, + graph_depth: 0, + }; + + let query = BenchQuery { + query_id: "q-exact-negated".to_string(), + text: format!("what does [{alias}] say? not touching klbr"), + reference_time: Some(3), + }; + let retrieval = pipeline.retrieve_evidence(query.clone(), budget).await?; + let context = pipeline.assemble_context(query, &retrieval, budget).await?; + + assert!(retrieval + .packets + .iter() + .any(|packet| packet.packet_kind == EvidencePacketKind::ExactRef)); + assert!(context.content.contains("borrowed the rust cli harness")); + Ok(()) + } + #[test] fn dense_embedding_text_is_bounded_for_local_embedder() { let long = "x".repeat(7_000); @@ -1525,8 +1995,14 @@ mod tests { let decision = apply_packet_rerank( packets, &[ - RerankResult { index: 0, score: 0.1 }, - RerankResult { index: 1, score: 0.9 }, + RerankResult { + index: 0, + score: 0.1, + }, + RerankResult { + index: 1, + score: 0.9, + }, ], &MemoryConfig::default(), ); @@ -1538,15 +2014,29 @@ mod tests { #[test] fn packet_rerank_does_not_demote_exact_ref_packets_below_score_only_hits() { let packets = vec![ - rerank_test_packet("pkt_exact", EvidencePacketKind::ExactRef, "exact cited body"), - rerank_test_packet("pkt_other", EvidencePacketKind::TurnWindow, "higher score body"), + rerank_test_packet( + "pkt_exact", + EvidencePacketKind::ExactRef, + "exact cited body", + ), + rerank_test_packet( + "pkt_other", + EvidencePacketKind::TurnWindow, + "higher score body", + ), ]; let decision = apply_packet_rerank( packets, &[ - RerankResult { index: 0, score: 0.1 }, - RerankResult { index: 1, score: 0.9 }, + RerankResult { + index: 0, + score: 0.1, + }, + RerankResult { + index: 1, + score: 0.9, + }, ], &MemoryConfig::default(), ); @@ -1567,8 +2057,14 @@ mod tests { let decision = apply_packet_rerank( packets, &[ - RerankResult { index: 0, score: 0.1 }, - RerankResult { index: 1, score: 0.9 }, + RerankResult { + index: 0, + score: 0.1, + }, + RerankResult { + index: 1, + score: 0.9, + }, ], &config, ); @@ -1596,7 +2092,9 @@ mod tests { "episode gist says the answer is Target", ); packet.refs.push("support_ref".to_string()); - packet.bodies.push("supporting turn repeats Target".to_string()); + packet + .bodies + .push("supporting turn repeats Target".to_string()); packet.signals.sources.push("dense".to_string()); packet.lanes.push(MemoryLane::Semantic); @@ -1621,9 +2119,10 @@ mod tests { ); for index in 0..24 { packet.refs.push(format!("support_ref_{index}")); - packet - .bodies - .push(format!("supporting neighbor {index} {}", "detail ".repeat(80))); + packet.bodies.push(format!( + "supporting neighbor {index} {}", + "detail ".repeat(80) + )); } let document = packet_rerank_document(&packet); @@ -1667,5 +2166,4 @@ mod tests { estimated_tokens: body.chars().count() / 4, } } - } -- 2.51.2