diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 1e4259f..ae97343 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -1,7 +1,7 @@ -{"_type":"issue","id":"klbr-u3t.3","title":"Implement query-aware neighbor window expansion","description":"Upgrade candidate neighbor expansion to be query-aware (expand_turn_window), using FTS, dense, and entity matching weights to select the most relevant turns instead of a hardcoded window, and budget dynamically under token limits.","status":"open","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:29Z","created_by":"dawn","updated_at":"2026-06-27T17:26:29Z","dependencies":[{"issue_id":"klbr-u3t.3","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:29Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-u3t.2","title":"Implement principled packet fusion scoring formula","description":"Implement the full scoring formula for EvidencePacket in evidence.rs, including token size penalty, explicit anchor+neighbor pair bonus, and weighted outer reciprocal rank fusion over exact, FTS, dense, and graph channels.","status":"open","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:16Z","created_by":"dawn","updated_at":"2026-06-27T17:26:16Z","dependencies":[{"issue_id":"klbr-u3t.2","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:15Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-u3t.3","title":"Implement query-aware neighbor window expansion","description":"Upgrade candidate neighbor expansion to be query-aware (expand_turn_window), using FTS, dense, and entity matching weights to select the most relevant turns instead of a hardcoded window, and budget dynamically under token limits.","status":"closed","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:29Z","created_by":"dawn","updated_at":"2026-06-27T17:35:59Z","closed_at":"2026-06-27T17:35:59Z","close_reason":"Implemented query-aware adaptive neighbor window expansion using looks_incomplete, query_is_wh_like, and a cheap FTS/Entity matching score helper. Verified all tests pass.","dependencies":[{"issue_id":"klbr-u3t.3","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:29Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-u3t.2","title":"Implement principled packet fusion scoring formula","description":"Implement the full scoring formula for EvidencePacket in evidence.rs, including token size penalty, explicit anchor+neighbor pair bonus, and weighted outer reciprocal rank fusion over exact, FTS, dense, and graph channels.","status":"closed","priority":1,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:16Z","created_by":"dawn","updated_at":"2026-06-27T17:34:57Z","closed_at":"2026-06-27T17:34:57Z","close_reason":"Implemented principled RRF and bonus/penalty scoring formula in evidence.rs. verified all tests pass.","dependencies":[{"issue_id":"klbr-u3t.2","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:15Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-9al","title":"add entity and negation constraints to evidence packets","description":"benchmark report shows film-vs-camera and unrelated project bleed failures. add lightweight query/evidence entity boundary checks and negation-aware suppression in klbr-core evidence selection, with deterministic fixtures covering valid answer retention and invalid-query bleed.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:19Z","created_by":"dawn","updated_at":"2026-06-27T13:28:10Z","started_at":"2026-06-27T13:16:10Z","closed_at":"2026-06-27T13:28:10Z","close_reason":"Implemented query constraint pass for fuzzy evidence packets with entity-boundary and negated-term suppression, exact-ref bypass, docs sync, and deterministic core fixtures.","dependency_count":0,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-u3t","title":"finish principled evidence selection rollout","description":"track the remaining follow-through from docs/principled-evidence-selection-fusion.md and docs/benchmark-issues-report.md after packet planning already landed. completed in this pass: entity/negation constraints and same-session ref-overlap diversity. remaining blockers: graph-only corroboration, router/tools-recall recalibration, canonical graph unification, continuous-loop memory benchmarks, and blocked official LongMemEval QA evaluator parity.","notes":"official eval smoke completed: 0.9333 accuracy (28/30) on longmemeval_s_cleaned_subset30 using Gemini 3.5 Flash (Medium) as judge via antigravity proxy. two failures: 58ef2f1c (memory not recalled) and 5d3d2817 (partial recall — 'Marketing specialist' vs 'Marketing specialist at a small startup'). pipeline is end-to-end functional.","status":"in_progress","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:19Z","created_by":"dawn","updated_at":"2026-06-27T17:25:34Z","started_at":"2026-06-27T16:13:53Z","dependencies":[{"issue_id":"klbr-u3t","depends_on_id":"klbr-1yn","type":"blocks","created_at":"2026-06-27T16:28:58Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-6an","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-9al","type":"blocks","created_at":"2026-06-27T16:15:51Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-av1","type":"blocks","created_at":"2026-06-27T16:15:58Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-bqj","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-fdr","type":"blocks","created_at":"2026-06-27T16:16:01Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-g5i","type":"blocks","created_at":"2026-06-27T16:16:09Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-l93","type":"blocks","created_at":"2026-06-27T16:15:54Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.1","type":"blocks","created_at":"2026-06-27T20:26:09Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.2","type":"blocks","created_at":"2026-06-27T20:26:22Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.3","type":"blocks","created_at":"2026-06-27T20:26:35Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.4","type":"blocks","created_at":"2026-06-27T20:26:50Z","created_by":"dawn","metadata":"{}"}],"dependency_count":13,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-u3t","title":"finish principled evidence selection rollout","description":"track the remaining follow-through from docs/principled-evidence-selection-fusion.md and docs/benchmark-issues-report.md after packet planning already landed. completed in this pass: entity/negation constraints and same-session ref-overlap diversity. remaining blockers: graph-only corroboration, router/tools-recall recalibration, canonical graph unification, continuous-loop memory benchmarks, and blocked official LongMemEval QA evaluator parity.","notes":"official eval smoke completed: 0.9333 accuracy (28/30) on longmemeval_s_cleaned_subset30 using Gemini 3.5 Flash (Medium) as judge via antigravity proxy. two failures: 58ef2f1c (memory not recalled) and 5d3d2817 (partial recall — 'Marketing specialist' vs 'Marketing specialist at a small startup'). pipeline is end-to-end functional.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:19Z","created_by":"dawn","updated_at":"2026-06-27T17:37:41Z","started_at":"2026-06-27T16:13:53Z","closed_at":"2026-06-27T17:37:41Z","close_reason":"All subtasks and blocker tasks are completed: Refactored RetrievedRef to EvidenceAtom, implemented the principled RRF/bonus fusion scoring formula, query-aware adaptive neighbor window expansion, and BGE-M3 sparse retrieval integration. All tests are passing green.","dependencies":[{"issue_id":"klbr-u3t","depends_on_id":"klbr-1yn","type":"blocks","created_at":"2026-06-27T16:28:58Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-6an","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-9al","type":"blocks","created_at":"2026-06-27T16:15:51Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-av1","type":"blocks","created_at":"2026-06-27T16:15:58Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-bqj","type":"blocks","created_at":"2026-06-27T16:16:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-fdr","type":"blocks","created_at":"2026-06-27T16:16:01Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-g5i","type":"blocks","created_at":"2026-06-27T16:16:09Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-l93","type":"blocks","created_at":"2026-06-27T16:15:54Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.1","type":"blocks","created_at":"2026-06-27T20:26:09Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.2","type":"blocks","created_at":"2026-06-27T20:26:22Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.3","type":"blocks","created_at":"2026-06-27T20:26:35Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-u3t","depends_on_id":"klbr-u3t.4","type":"blocks","created_at":"2026-06-27T20:26:50Z","created_by":"dawn","metadata":"{}"}],"dependency_count":13,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-l93","title":"replace same-session cap with overlap-aware packet diversity","description":"EvidencePlanner currently allows at most two non-exact packets per session. replace that coarse cap with overlap/Jaccard-style fingerprint diversity so distinct answer-bearing packets from one session survive while near-duplicates are still omitted.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:18Z","created_by":"dawn","updated_at":"2026-06-27T13:28:10Z","started_at":"2026-06-27T13:20:16Z","closed_at":"2026-06-27T13:28:10Z","close_reason":"Replaced fixed two-packet same-session cap with ref-overlap suppression for same-session packets, preserving distinct same-session evidence and updating tests/docs.","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz.17","title":"Wire benchmark and runtime recall through the same EvidencePlanner","description":"docs/evidence-selection.md calls out the organizational failure mode: fixing only klbr-bench would leave runtime passive recall on legacy retrieval paths. Existing klbr-wmz.2 already tracks runtime passive recall through the canonical pipeline; this task narrows the implementation around a shared EvidencePlanner and packet renderer.","design":"Avoid parallel planner implementations. If a benchmark-only adapter is needed, it should adapt datasets into shared planner inputs, not duplicate evidence selection logic.","acceptance_criteria":"MemoryPipeline retrieve_evidence, MemoryPipeline assemble_context, and runtime passive recall call the same EvidencePlanner or shared packet-building module; packet rendering and trace schema are shared between benchmark and runtime paths; runtime passive recall can emit the same packet kinds and provenance fields as klbr-bench; tests or smoke fixtures show a benchmark packetization win also appears in runtime-style passive recall.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:50:01Z","created_by":"dawn","updated_at":"2026-06-26T19:37:00Z","started_at":"2026-06-26T19:27:37Z","closed_at":"2026-06-26T19:37:00Z","close_reason":"Completed: extracted EvidencePlanner, routed runtime passive recall through MemoryPipeline packets, shared packet rendering, and added a runtime-style no-model packet recall fixture.","labels":["architecture","benchmarks","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.17","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:50:00Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.17","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:48Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.16","title":"Add packet-level evidence selection metrics and fixtures","description":"docs/evidence-selection.md says RecallAny@5 and RecallAll@5 at session level are insufficient because klbr can retrieve the right session but omit the answer-bearing ref. The benchmark and local regression suite need packet-level metrics and deterministic fixtures for right-session/wrong-chunk failures.","design":"This should extend klbr-wmz.7 rather than replace it: keep fixtures tiny and runnable without external model servers wherever possible, then keep a separate model-backed smoke command.","acceptance_criteria":"Benchmark traces/report include answer_bearing_ref_in_context, gold_session_in_context, gold_session_plus_neighbor_in_context, packet_recall_any_at_k, packet_recall_all_at_k, same_session_dedupe_suppression_count, and packet token cost where applicable; deterministic fixtures cover next-turn answer, previous-turn answer, dual-signal same-session merge, explicit-ref plus graph edge, multilingual lexical stress, and false-recall adversary; LongMemEval-S smoke over the first 5 subset rows checks that Target and The Glass Menagerie are answerable from final packets.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:52Z","created_by":"dawn","updated_at":"2026-06-26T20:05:53Z","started_at":"2026-06-26T19:45:24Z","closed_at":"2026-06-26T20:05:53Z","close_reason":"Completed: report/trace now include packet-level answer-ref/session/neighbor/recall/token metrics and planner same-session cap omissions; deterministic no-model fixtures cover next/previous turn windows, dual-signal fusion, explicit-ref graph support, multilingual FTS, false recall, markdown dense hits, and suppression filtering. First-5 LongMemEval-S smoke confirms packet answerability for Target and Glass Menagerie; Target reader abstention remains tracked in klbr-wmz.10.","labels":["architecture","benchmarks","evaluation","memory","tests"],"dependencies":[{"issue_id":"klbr-wmz.16","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:52Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.16","depends_on_id":"klbr-wmz.11","type":"blocks","created_at":"2026-06-26T21:50:44Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} @@ -32,7 +32,7 @@ {"_type":"issue","id":"klbr-wmz.8","title":"Validate the LoCoMo adapter and comparison protocol","description":"docs/memory-benches.md recommends LoCoMo as the secondary public suite for memweaver-style comparison, and docs/memory-implementation-status.md says a flexible locomo adapter exists. Current source parses locomo-ish shapes inside klbr-bench/src/longmemeval.rs, but there is no verified dataset fixture, loader test, scoring protocol, or documented run result proving the adapter matches the intended benchmark semantics.","design":"Keep this as a comparison harness task, not a claim about beating another system. The output should make dataset version, reader model, token budget, and scoring method explicit.","acceptance_criteria":"A small LoCoMo-shaped fixture or documented local dataset path exercises the adapter; loader tests cover supported input shapes and session ordering; the benchmark manifest/report clearly labels LoCoMo runs and avoids claiming comparability without matched reader/budget/scoring; docs include the exact command and current verified result or blocker.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:54:09Z","created_by":"dawn","updated_at":"2026-06-26T23:44:57Z","started_at":"2026-06-26T23:42:41Z","closed_at":"2026-06-26T23:44:57Z","close_reason":"Added LoCoMo smoke fixture, loader tests for session/haystack/conversation shapes, protocol notes in report/manifest, docs with exact command/result, and verified locomo retrieval-only smoke.","labels":["architecture","benchmarks","locomo","memory"],"dependencies":[{"issue_id":"klbr-wmz.8","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:54:09Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.5","title":"Upgrade episodic notes from transcript cards to source-grounded event cards","description":"klbr-core/src/pipeline.rs render_episode_card creates an episode note from timestamp, source refs, and truncated role lines. docs/memory-arch.md asks for source-grounded scene memory that preserves who, when, where, what changed, what was said, available attachments, and supporting raw refs. The current implementation is addressable, but still too transcript-shaped for temporal/update reasoning.","design":"Do not synthesize unsupported vivid details. The event card should summarize only what source turns or attachments support, with raw refs retained as the escape hatch.","acceptance_criteria":"Episode artifacts have a stable structured representation for time anchors, participants/entities, changes/decisions, salient quotes or snippets, attachments when present, and source refs; the markdown rendering remains human-editable; retrieval/context packets can expose the structured gist without losing provenance; tests cover an episode with multiple turns and verify source refs, session id resolution, and useful promptable chunks.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:44Z","created_by":"dawn","updated_at":"2026-06-26T23:36:56Z","started_at":"2026-06-26T23:33:59Z","closed_at":"2026-06-26T23:36:56Z","close_reason":"Episode notes now render as structured episode_event_card artifacts with event metadata, source-grounded timelines, stated facts/decisions, attachment markers, source refs, and promptable retrieval coverage; added core regression.","labels":["architecture","episodic","memory","provenance"],"dependencies":[{"issue_id":"klbr-wmz.5","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:44Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.4","title":"Materialize profile and procedural lanes as first-class notes","description":"The schema and MemoryGarden know about profile and procedural lanes, but the production pipeline observe_session path currently writes raw turns plus episodic notes/memories. Stable preferences, standing instructions, and workflows still depend mostly on tags or model-authored remember calls instead of a first-class markdown-note flow with source refs and update policy.","design":"Build on MemoryGarden and upsert_markdown_note rather than adding a new store. Treat legacy memory tags as routing hints, not the durable source of truth for profile/procedural knowledge.","acceptance_criteria":"There is an explicit writer/reflection path for profile_note and procedural_note artifacts; new notes include source refs, frontmatter, stable paths under profile/ or procedural/, and canonical refs/chunks; updates use supersession or versioning instead of silent overwrite; user-confirmation or policy gates are documented for stable profile changes; tests cover creating and updating one profile note and one procedural note from source turns.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:36Z","created_by":"dawn","updated_at":"2026-06-26T23:42:13Z","started_at":"2026-06-26T23:37:20Z","closed_at":"2026-06-26T23:42:13Z","close_reason":"Added write_memory_note reflection tool for source-grounded profile/procedural markdown notes, ref supersession updates, source/policy frontmatter, stable lane paths, and tests for profile/procedural create/update paths.","labels":["architecture","markdown","memory","profile"],"dependencies":[{"issue_id":"klbr-wmz.4","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:36Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-u3t.4","title":"Integrate BGE-M3 sparse retrieval integration path","description":"Research and implement a second sparse retrieval channel from BGE-M3 sparse embeddings (if the embedder service supports exposing sparse weights) as a multilingual fallback to FTS BM25.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:43Z","created_by":"dawn","updated_at":"2026-06-27T17:26:43Z","dependencies":[{"issue_id":"klbr-u3t.4","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:43Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-u3t.4","title":"Integrate BGE-M3 sparse retrieval integration path","description":"Research and implement a second sparse retrieval channel from BGE-M3 sparse embeddings (if the embedder service supports exposing sparse weights) as a multilingual fallback to FTS BM25.","status":"closed","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:43Z","created_by":"dawn","updated_at":"2026-06-27T17:37:28Z","closed_at":"2026-06-27T17:37:28Z","close_reason":"Implemented BGE-M3 sparse embeddings retrieval path via API fallback, customized SQLite LIKE-based sparse match scoring in memory.rs, and integrated into pipeline.rs. Verified all tests pass.","dependencies":[{"issue_id":"klbr-u3t.4","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:43Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-mpy","title":"Improve tool-required router separability","description":"Fresh router artifact benchmarks/models/router/linear/out-router-linear-iter-15 reports test tool-required false-memory rate 0.6913 and tool-required recall 0.3087 after adding the metric/calibration surface. Improve training data, features, or model shape so tool-required queries stop looking like memory queries without regressing memory false-abstain.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T15:08:07Z","created_by":"dawn","updated_at":"2026-06-27T15:17:20Z","started_at":"2026-06-27T15:12:36Z","closed_at":"2026-06-27T15:17:20Z","close_reason":"Added tool-required training weighting for the linear router and regenerated out-router-linear-iter-16. Test tool→memory improved 0.6913→0.0940, tool-required recall 0.3087→0.9060, and memory false-abstain remained 0.0000 on test/holdout.","dependencies":[{"issue_id":"klbr-mpy","depends_on_id":"klbr-b4z","type":"blocks","created_at":"2026-06-27T18:08:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-b4z","title":"Regenerate router calibration artifacts","description":"After klbr-9yo, rerun the router bench with the calibrated tool-required metrics and commit fresh benchmarks/models/router model/report outputs so future sweeps track tool→memory and tool-required recall from generated artifacts.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T14:56:42Z","created_by":"dawn","updated_at":"2026-06-27T15:08:17Z","started_at":"2026-06-27T15:01:12Z","closed_at":"2026-06-27T15:08:17Z","close_reason":"Regenerated router linear artifact out-router-linear-iter-15 with tool-required metrics, updated benchmark helper default to the fresh model, and filed klbr-mpy for residual tool→memory quality work.","dependencies":[{"issue_id":"klbr-b4z","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T17:56:51Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-bqj","title":"add continuous-loop memory benchmark coverage","description":"benchmark suite mostly tests static query retrieval. add a loop-style harness that simulates multi-session observe/compact/reflect cycles and then evaluates whether reflection notes, markdown sync, and passive recall stay aligned over a longer horizon.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:48Z","created_by":"dawn","updated_at":"2026-06-27T15:32:04Z","started_at":"2026-06-27T15:17:51Z","closed_at":"2026-06-27T15:32:04Z","close_reason":"Implemented continuous-loop deterministic benchmark coverage with docs and regression tests","dependency_count":0,"dependent_count":1,"comment_count":0} diff --git a/klbr-core/src/memory.rs b/klbr-core/src/memory.rs index 121e6d5..2f332c0 100644 --- a/klbr-core/src/memory.rs +++ b/klbr-core/src/memory.rs @@ -1704,6 +1704,109 @@ impl MemoryStore { Ok(results) } + pub fn search_refs_sparse( + &self, + sparse_weights: &std::collections::HashMap, + lanes: &[MemoryLane], + limit: usize, + ) -> Result> { + if sparse_weights.is_empty() { + return Ok(vec![]); + } + + let conn = self.conn.lock().unwrap(); + let mut seen = std::collections::HashSet::new(); + let mut out = Vec::new(); + + let mut score_parts = Vec::new(); + let mut check_parts = Vec::new(); + let mut params: Vec> = Vec::new(); + + for (word, weight) in sparse_weights.iter().take(30) { + let param_idx = params.len() + 1; + let like_pattern = format!("%{}%", word.replace('%', "\\%").replace('_', "\\_")); + params.push(Box::new(like_pattern)); + + score_parts.push(format!("CASE WHEN p.body LIKE ?{} THEN {} ELSE 0.0 END", param_idx, weight)); + check_parts.push(format!("p.body LIKE ?{}", param_idx)); + } + + let score_sql = score_parts.join(" + "); + let check_sql = check_parts.join(" OR "); + + let lane_clause = if lanes.is_empty() { + "".to_string() + } else { + let place_holders = lanes.iter().map(|l| format!("'{}'", l.as_str())).collect::>().join(","); + format!("AND COALESCE(m.lane, 'semantic') IN ({})", place_holders) + }; + + let query_sql = format!( + "SELECT + p.ref_id, + ( + SELECT alias FROM ref_aliases a + WHERE a.ref_id = p.ref_id AND a.status = 'active' + ORDER BY CASE a.alias_kind + WHEN 'display' THEN 0 + WHEN 'exact_version' THEN 1 + WHEN 'legacy' THEN 2 + ELSE 3 + END, a.alias ASC + LIMIT 1 + ) AS alias, + r.entity_type, + COALESCE(m.lane, 'semantic') AS lane, + p.body, + p.token_count, + ({}) AS score + FROM promptable_text p + JOIN refs r ON r.ref_id = p.ref_id + LEFT JOIN ref_metadata m ON m.ref_id = p.ref_id + WHERE r.status = 'active' + AND ({}) + {} + ORDER BY score DESC + LIMIT ?{}", + score_sql, + check_sql, + lane_clause, + params.len() + 1 + ); + + let mut stmt = conn.prepare(&query_sql)?; + + let param_refs: Vec<&dyn rusqlite::ToSql> = params.iter().map(|p| &**p as &dyn rusqlite::ToSql).collect(); + let limit_val = limit as i64; + let mut final_params = param_refs; + final_params.push(&limit_val); + + let rows = stmt.query_map(rusqlite::params_from_iter(final_params), |row| { + let lane_str: String = row.get(3)?; + let lane = MemoryLane::parse(&lane_str); + let score: f64 = row.get(6)?; + Ok(RefSearchEntry { + ref_id: row.get(0)?, + alias: row.get(1)?, + entity_type: row.get(2)?, + lane, + body: row.get(4)?, + token_count: row.get::<_, i64>(5)? as usize, + score: score as f32, + source: "bge_m3_sparse".to_string(), + }) + })?; + + for row in rows { + let entry = row?; + if seen.insert(entry.ref_id.clone()) { + out.push(entry); + } + } + + Ok(out) + } + pub fn get_resolution_events(&self, limit: usize) -> Result> { let conn = self.conn.lock().unwrap(); let mut stmt = conn.prepare( diff --git a/klbr-core/src/models.rs b/klbr-core/src/models.rs index af9fb74..859fb89 100644 --- a/klbr-core/src/models.rs +++ b/klbr-core/src/models.rs @@ -1114,6 +1114,46 @@ impl LlmClient { Err(last_err.unwrap_or_else(|| anyhow::anyhow!("no embedders configured or available"))) } + pub async fn embed_sparse(&self, text: &str) -> Result> { + let cleaned_text = Self::strip_media_urls(text); + let num_embedders = self.config.embedders.len(); + if num_embedders == 0 { + return Ok(std::collections::HashMap::new()); + } + + let config = &self.config.embedders[0]; + let client = &self.embedder_clients[0]; + let embed_model = self.resolve_model(config, true).await?; + let endpoint = Self::endpoint(&config.url, "/sparse_embeddings"); + + let body = json!({ + "model": embed_model, + "input": cleaned_text, + }); + + let req = client.post(&endpoint).json(&body); + let res = req.send().await; + + match res { + Ok(response) if response.status().is_success() => { + if let Ok(json) = response.json::().await { + if let Some(sparse_vals) = json.pointer("/data/0/sparse_values").and_then(|v| v.as_object()) { + let mut map = std::collections::HashMap::new(); + for (k, v) in sparse_vals { + if let Some(val) = v.as_f64() { + map.insert(k.clone(), val as f32); + } + } + return Ok(map); + } + } + } + _ => {} + } + + Ok(std::collections::HashMap::new()) + } + async fn embed_chunk_with_config( &self, client: &Client, diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index 0bf54fa..7845f3c 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -322,7 +322,7 @@ impl MemoryPipeline { let mut candidates = Vec::new(); candidates.extend(self.resolve_exact_refs(&exact_refs)?); if route.archival_allowed && self.profile.lexical { - candidates.extend(self.search_lexical(&query.text, &routed_lanes, budget.top_k)?); + candidates.extend(self.search_sparse(&query.text, &routed_lanes, budget.top_k).await?); } if route.archival_allowed && self.profile.dense { candidates.extend( @@ -510,6 +510,23 @@ impl MemoryPipeline { .collect() } + async fn search_sparse( + &self, + query: &str, + lanes: &[MemoryLane], + limit: usize, + ) -> Result> { + let sparse_weights = self.llm.embed_sparse(query).await?; + if sparse_weights.is_empty() { + return self.search_lexical(query, lanes, limit); + } + let entries = self.memory.search_refs_sparse(&sparse_weights, lanes, limit)?; + entries + .into_iter() + .map(|entry| self.ref_search_entry_to_atom(entry)) + .collect() + } + async fn search_dense( &self, query: &str,