diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index d930b0e..f4d2902 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -74,3 +74,4 @@ {"id":"int-c34adfde","kind":"field_change","created_at":"2026-06-30T12:55:18.472171415Z","actor":"dawn","issue_id":"klbr-54l.6","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately"}} {"id":"int-e9f364ec","kind":"field_change","created_at":"2026-06-30T12:55:18.89582463Z","actor":"dawn","issue_id":"klbr-54l","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"implemented trace-diff, planner hints/parser hardening, session/packet/render fixes, writer/whenloss diagnostics, docs, and final same-seed retrieval smoke; remaining passive misses are documented as follow-up opportunities"}} {"id":"int-788f0749","kind":"field_change","created_at":"2026-06-30T17:01:55.237649146Z","actor":"dawn","issue_id":"klbr-e28","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed: diagnosed renderer truncation of anchor/local answer bodies, landed prioritized packet rendering, added regression tests, and verified same-seed official QA improved from broken 0.32 and old 0.44 to 0.56."}} +{"id":"int-eba49ab6","kind":"field_change","created_at":"2026-06-30T17:48:55.880302146Z","actor":"dawn","issue_id":"klbr-d76","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Not planned for now: user wants folgezettel usefulness evaluated from real-world usage traces rather than a synthetic agentic eval. Revisit only after real usage exposes concrete failures or measurable trace-review needs."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 4b87802..f886e2b 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -41,6 +41,8 @@ {"_type":"issue","id":"klbr-wmz.2","title":"Route runtime passive recall through the canonical memory pipeline","description":"Bench retrieval now goes through MemoryPipeline, but runtime passive recall in klbr-core/src/agent.rs still embeds the prompt, calls MemoryStore::get_searchable, runs retrieval::retrieve_exact over legacy memories, then injects Context memory packets. That bypasses fts, exact refs, markdown notes, graph expansion, and the canonical lane/lifecycle path the docs describe.","design":"Avoid duplicating retrieval logic in agent.rs. Either make MemoryPipeline usable by AgentRuntime or extract a shared retrieval facade that both MemoryPipeline and runtime passive recall call.","acceptance_criteria":"Runtime passive recall uses the same lane-aware canonical retrieval and context packet assembly policy as the production pipeline; recalled packets can include fts/exact/dense/graph candidates from refs and markdown notes; archived/tombstoned/suppressed refs do not leak; klbr-core/src/instructions.md matches the actual memory packet format; tests or a focused integration fixture cover passive recall from a markdown note and from an explicit ref.","notes":"Runtime passive recall now calls MemoryPipeline::retrieve_evidence and injects shared EvidencePacket XML via Context::inject_evidence_packets; no-model runtime packet fixture covers turn-window expansion. Remaining acceptance is blocked on klbr-wmz.1 because dense search still starts from legacy memory rows before ref mapping.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:19Z","created_by":"dawn","updated_at":"2026-06-26T19:44:55Z","started_at":"2026-06-26T19:37:32Z","closed_at":"2026-06-26T19:44:55Z","close_reason":"Completed: runtime passive recall now uses MemoryPipeline/EvidencePlanner packets, instructions document current packet XML, and fixtures cover turn-window recall plus explicit markdown refs; dense canonical dependency completed in klbr-wmz.1.","labels":["architecture","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz.1","type":"blocks","created_at":"2026-06-26T22:37:53Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.1","title":"Move dense retrieval onto canonical refs and embedding_items","description":"Current source still has dense retrieval on the legacy memory-row surface: klbr-core/src/pipeline.rs search_dense calls MemoryStore::get_searchable and retrieval::retrieve_exact, while fts/exact retrieval uses canonical refs and promptable_text. The schema already has refs, promptable_text, markdown_note_chunks, and embedding_items, so dense retrieval should not be the odd path out.","design":"Prefer a ref-native embedding index backed by embedding_items. Backfill embeddings from promptable_text, keep memory-id aliases as compatibility aliases, and make klbr/full versus dense-only profiles exercise the same canonical identity layer as fts and exact retrieval.","acceptance_criteria":"Dense candidate generation works over canonical ref ids for memories, turn chunks, markdown note chunks, episode notes, profile notes, and procedural notes; lane and lifecycle filtering come from refs/ref_metadata instead of memory tags alone; benchmark traces return canonical refs for dense hits; regression tests cover a markdown-note-only hit and a tombstoned/suppressed ref not leaking through dense search.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:11Z","created_by":"dawn","updated_at":"2026-06-26T19:43:27Z","started_at":"2026-06-26T19:38:02Z","closed_at":"2026-06-26T19:43:27Z","close_reason":"Completed: dense candidate generation now lazily backfills embedding_items from active promptable refs, scores canonical ref embeddings directly, and tests markdown-note dense hits plus suppressed-ref filtering.","labels":["architecture","memory","refs","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.1","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:11Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz","title":"Finish memory architecture follow-through","description":"Tracks the remaining memory architecture work identified from docs/memory-arch.md, docs/memory-benches.md, docs/memory-implementation-status.md, and current klbr-core/klbr-bench source. Current status says pipeline, typed memory packets, markdown notes, edge mirroring, lifecycle projection, and benchmark runner integration exist; this epic is for gaps still present in source/docs.","acceptance_criteria":"Close when the child issues are complete, docs/memory-implementation-status.md is updated from current verification, and the architecture docs no longer point at missing or stale follow-up work.","status":"closed","priority":1,"issue_type":"epic","owner":"90008@klbr.net","created_at":"2026-06-26T17:52:54Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","closed_at":"2026-06-26T23:45:35Z","close_reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn.","labels":["architecture","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-4ls","title":"Design time-aware passive memory ranking","description":"LongMemEval temporal misses exposed that BenchQuery.reference_time is currently not used by dense/sparse candidate ranking, and temporal evidence is selected mostly by semantic/entity similarity. Design and implement a measured ranking policy that uses query reference_time and evidence timestamps to support relative-date, elapsed-time, and event-order questions without hard-coded dataset cues or broad context fanout.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T17:49:11Z","created_by":"dawn","updated_at":"2026-06-30T17:49:11Z","dependencies":[{"issue_id":"klbr-4ls","depends_on_id":"klbr-k7d","type":"blocks","created_at":"2026-06-30T20:49:19Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-k7d","title":"Run larger LongMemEval QA and inspect remaining misses","description":"Run a larger same-protocol LongMemEval-S QA sample after the anchor-window evidence rendering fix, compare against the sample-25 artifacts, inspect remaining miss categories while the run is in flight, and collect broader retrieval/memory-eval research perspectives before deciding next implementation work.","notes":"sample-25 anchor-window artifact has 11 official misses: 3 exact answer refs not selected, 1 selected but exact ref not rendered, 3 rendered refs with answer value not visible, and 5 value-visible reader failures. Temporal/query-time issue found: BenchQuery.reference_time is set from LongMemEval question_date, but assemble_context rendered only raw question text in \u003ctask\u003e; evidence facts/timeline rendered epoch seconds only. search_dense also currently ignores reference_time, but that needs a deliberate ranking design. Patched core presentation to include task reference_time/reference_date and packet date attrs; cargo fmt, klbr-core tests, and klbr-bench tests pass. Cancelled the foreground pre-patch sample-100 at 14/100; restarted patched sample-100 detached as pid 329401 with log benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample100_seed4937553249516211047_query_time_qa.log and output dir benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample100_seed4937553249516211047_query_time_qa.","status":"in_progress","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T17:31:26Z","created_by":"dawn","updated_at":"2026-06-30T17:48:45Z","started_at":"2026-06-30T17:31:31Z","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.6","title":"Probe passive writer retention for atomic facts and events","description":"Research on memory systems suggests passive QA can fail because write-time compression loses entity/slot/value/time facts before retrieval ever has a chance. Klbr now writes episode_event_card artifacts and generic fact_rows, but the next passive-recall improvement pass should add writer-side probes for counts, updates, preference constraints, and temporal facts so we can tell whether the passive writer retained the answer-bearing atom before tuning retrieval.","acceptance_criteria":"Fixtures or diagnostics check that answer-bearing entity/slot/value/time atoms exist in stored promptable artifacts before retrieval; failures are reported separately from packet/ranking misses; at least count, update_resolution, temporal order, and preference-constraint cases are covered.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:57Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately","labels":["bench","longmemeval","memory","writer"],"dependencies":[{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:56Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:18Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.5","title":"Make WhenLoss-style passive QA diagnostics first-class","description":"The passive QA bench should be treated as best-effort passive recall, not the whole memory-system score. The klbr-bench --diagnostic whenloss mode exists, but regression work needs a first-class joined report that separates write-side loss from retrieval/packet/rendering loss per question. Add a report path for tfc, oracle-evidence, complete-stored-memory, and retrieved-memory scores joined with packet metrics and local/official QA outcomes.","acceptance_criteria":"A normal regression run can emit per-question tfc/oe/csm/rm scores plus write_gap and retrieval_gap; report rows join those scores with op_plan, answer_bearing_ref_selected/rendered/in_context, answer_value_visible, and official/local QA labels; docs explicitly frame LongMemEval passive QA as best-effort passive recall rather than agentic folgezettel memory quality.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:41Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"whenloss diagnostic summaries are emitted into trace_summary rows and docs frame LongMemEval passive QA as best-effort passive recall","labels":["bench","diagnostics","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:41Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.4","title":"Add answer-session packet survival guardrail to context packing","description":"Current passive QA can spend the final read budget on unrelated high-scoring fts turn windows while a packet from a strongly supported answer session exists but is too late or too thin to render. Add a generic runtime-safe packing/ranking guardrail: when a session group has strong multi-signal support in stage_one, preserve at least one useful non-thin support packet from that session before unrelated single-signal packets, without using gold labels or answer ids outside benchmark metrics.","acceptance_criteria":"A fixture where stage_one finds the relevant session but context packing omits its useful packet fails before the change and passes after; traces expose first relevant/answer-session packet rank for diagnostics; answer_bearing_ref_rendered improves on old-pass/current-fail rows without packet_tokens_total_mean exploding back into unbounded context dumping.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:26Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"context packing/render guardrails now keep answer-session packets renderable under packet budgets; traces expose stage/packet/render ranks; targeted rows b5ef892d and gpt4_385a5000 pass answer-ref-in-context","labels":["bench","context-packing","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.4","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:25Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.4","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} @@ -66,7 +68,8 @@ {"_type":"issue","id":"klbr-wmz.8","title":"Validate the LoCoMo adapter and comparison protocol","description":"docs/memory-benches.md recommends LoCoMo as the secondary public suite for memweaver-style comparison, and docs/memory-implementation-status.md says a flexible locomo adapter exists. Current source parses locomo-ish shapes inside klbr-bench/src/longmemeval.rs, but there is no verified dataset fixture, loader test, scoring protocol, or documented run result proving the adapter matches the intended benchmark semantics.","design":"Keep this as a comparison harness task, not a claim about beating another system. The output should make dataset version, reader model, token budget, and scoring method explicit.","acceptance_criteria":"A small LoCoMo-shaped fixture or documented local dataset path exercises the adapter; loader tests cover supported input shapes and session ordering; the benchmark manifest/report clearly labels LoCoMo runs and avoids claiming comparability without matched reader/budget/scoring; docs include the exact command and current verified result or blocker.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:54:09Z","created_by":"dawn","updated_at":"2026-06-26T23:44:57Z","started_at":"2026-06-26T23:42:41Z","closed_at":"2026-06-26T23:44:57Z","close_reason":"Added LoCoMo smoke fixture, loader tests for session/haystack/conversation shapes, protocol notes in report/manifest, docs with exact command/result, and verified locomo retrieval-only smoke.","labels":["architecture","benchmarks","locomo","memory"],"dependencies":[{"issue_id":"klbr-wmz.8","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:54:09Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.5","title":"Upgrade episodic notes from transcript cards to source-grounded event cards","description":"klbr-core/src/pipeline.rs render_episode_card creates an episode note from timestamp, source refs, and truncated role lines. docs/memory-arch.md asks for source-grounded scene memory that preserves who, when, where, what changed, what was said, available attachments, and supporting raw refs. The current implementation is addressable, but still too transcript-shaped for temporal/update reasoning.","design":"Do not synthesize unsupported vivid details. The event card should summarize only what source turns or attachments support, with raw refs retained as the escape hatch.","acceptance_criteria":"Episode artifacts have a stable structured representation for time anchors, participants/entities, changes/decisions, salient quotes or snippets, attachments when present, and source refs; the markdown rendering remains human-editable; retrieval/context packets can expose the structured gist without losing provenance; tests cover an episode with multiple turns and verify source refs, session id resolution, and useful promptable chunks.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:44Z","created_by":"dawn","updated_at":"2026-06-26T23:36:56Z","started_at":"2026-06-26T23:33:59Z","closed_at":"2026-06-26T23:36:56Z","close_reason":"Episode notes now render as structured episode_event_card artifacts with event metadata, source-grounded timelines, stated facts/decisions, attachment markers, source refs, and promptable retrieval coverage; added core regression.","labels":["architecture","episodic","memory","provenance"],"dependencies":[{"issue_id":"klbr-wmz.5","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:44Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.4","title":"Materialize profile and procedural lanes as first-class notes","description":"The schema and MemoryGarden know about profile and procedural lanes, but the production pipeline observe_session path currently writes raw turns plus episodic notes/memories. Stable preferences, standing instructions, and workflows still depend mostly on tags or model-authored remember calls instead of a first-class markdown-note flow with source refs and update policy.","design":"Build on MemoryGarden and upsert_markdown_note rather than adding a new store. Treat legacy memory tags as routing hints, not the durable source of truth for profile/procedural knowledge.","acceptance_criteria":"There is an explicit writer/reflection path for profile_note and procedural_note artifacts; new notes include source refs, frontmatter, stable paths under profile/ or procedural/, and canonical refs/chunks; updates use supersession or versioning instead of silent overwrite; user-confirmation or policy gates are documented for stable profile changes; tests cover creating and updating one profile note and one procedural note from source turns.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:36Z","created_by":"dawn","updated_at":"2026-06-26T23:42:13Z","started_at":"2026-06-26T23:37:20Z","closed_at":"2026-06-26T23:42:13Z","close_reason":"Added write_memory_note reflection tool for source-grounded profile/procedural markdown notes, ref supersession updates, source/policy frontmatter, stable lane paths, and tests for profile/procedural create/update paths.","labels":["architecture","markdown","memory","profile"],"dependencies":[{"issue_id":"klbr-wmz.4","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:36Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-d76","title":"Design a separate folgezettel agentic memory eval","description":"LongMemEval passive QA is useful as best-effort passive recall, but it does not test the mk tool surface, titles-as-apis, sourced note writing, revisions, links, or trail-aware recall. Design a small agentic memory eval where the model must use mk_index, mk_search, mk_follow, mk_recall, mk_remember, mk_revise, mk_link, and mk_fleeting over a folgezettel tree, then measure sourced note quality, title usefulness, revision behavior, and later retrieval through trail_support.","acceptance_criteria":"A design doc or bench skeleton defines tasks, scoring, and traces for agentic folgezettel memory separately from LongMemEval passive QA; it includes at least note creation, revision/supersession, linking, search/follow/recall, and later answer-from-zettel scenarios; docs make clear this is not a replacement for passive LongMemEval recall.","status":"open","priority":3,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-30T11:14:04Z","created_by":"dawn","updated_at":"2026-06-30T11:14:04Z","labels":["agentic","bench","folgezettel","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-wj2","title":"Instrument real-world folgezettel usage review","description":"Replace the proposed synthetic folgezettel eval with real usage trace review. Capture or summarize mk_* tool use, whether later answers cite/follow notes, whether revisions and supersedes behavior prevent stale facts, and whether titles/links remain useful in later workflows. Use real failures before designing any benchmark.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T17:49:10Z","created_by":"dawn","updated_at":"2026-06-30T17:49:10Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-d76","title":"Design a separate folgezettel agentic memory eval","description":"LongMemEval passive QA is useful as best-effort passive recall, but it does not test the mk tool surface, titles-as-apis, sourced note writing, revisions, links, or trail-aware recall. Design a small agentic memory eval where the model must use mk_index, mk_search, mk_follow, mk_recall, mk_remember, mk_revise, mk_link, and mk_fleeting over a folgezettel tree, then measure sourced note quality, title usefulness, revision behavior, and later retrieval through trail_support.","acceptance_criteria":"A design doc or bench skeleton defines tasks, scoring, and traces for agentic folgezettel memory separately from LongMemEval passive QA; it includes at least note creation, revision/supersession, linking, search/follow/recall, and later answer-from-zettel scenarios; docs make clear this is not a replacement for passive LongMemEval recall.","status":"closed","priority":3,"issue_type":"feature","owner":"90008@klbr.net","created_at":"2026-06-30T11:14:04Z","created_by":"dawn","updated_at":"2026-06-30T17:48:56Z","closed_at":"2026-06-30T17:48:56Z","close_reason":"Not planned for now: user wants folgezettel usefulness evaluated from real-world usage traces rather than a synthetic agentic eval. Revisit only after real usage exposes concrete failures or measurable trace-review needs.","labels":["agentic","bench","folgezettel","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-u3t.4","title":"Integrate BGE-M3 sparse retrieval integration path","description":"Research and implement a second sparse retrieval channel from BGE-M3 sparse embeddings (if the embedder service supports exposing sparse weights) as a multilingual fallback to FTS BM25.","status":"closed","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T17:26:43Z","created_by":"dawn","updated_at":"2026-06-27T17:37:28Z","closed_at":"2026-06-27T17:37:28Z","close_reason":"Implemented BGE-M3 sparse embeddings retrieval path via API fallback, customized SQLite LIKE-based sparse match scoring in memory.rs, and integrated into pipeline.rs. Verified all tests pass.","dependencies":[{"issue_id":"klbr-u3t.4","depends_on_id":"klbr-u3t","type":"parent-child","created_at":"2026-06-27T20:26:43Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-mpy","title":"Improve tool-required router separability","description":"Fresh router artifact benchmarks/models/router/linear/out-router-linear-iter-15 reports test tool-required false-memory rate 0.6913 and tool-required recall 0.3087 after adding the metric/calibration surface. Improve training data, features, or model shape so tool-required queries stop looking like memory queries without regressing memory false-abstain.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T15:08:07Z","created_by":"dawn","updated_at":"2026-06-27T15:17:20Z","started_at":"2026-06-27T15:12:36Z","closed_at":"2026-06-27T15:17:20Z","close_reason":"Added tool-required training weighting for the linear router and regenerated out-router-linear-iter-16. Test tool→memory improved 0.6913→0.0940, tool-required recall 0.3087→0.9060, and memory false-abstain remained 0.0000 on test/holdout.","dependencies":[{"issue_id":"klbr-mpy","depends_on_id":"klbr-b4z","type":"blocks","created_at":"2026-06-27T18:08:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-b4z","title":"Regenerate router calibration artifacts","description":"After klbr-9yo, rerun the router bench with the calibrated tool-required metrics and commit fresh benchmarks/models/router model/report outputs so future sweeps track tool→memory and tool-required recall from generated artifacts.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T14:56:42Z","created_by":"dawn","updated_at":"2026-06-27T15:08:17Z","started_at":"2026-06-27T15:01:12Z","closed_at":"2026-06-27T15:08:17Z","close_reason":"Regenerated router linear artifact out-router-linear-iter-15 with tool-required metrics, updated benchmark helper default to the fresh model, and filed klbr-mpy for residual tool→memory quality work.","dependencies":[{"issue_id":"klbr-b4z","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T17:56:51Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs index 5058aac..4b2f740 100644 --- a/klbr-core/src/evidence.rs +++ b/klbr-core/src/evidence.rs @@ -1,6 +1,7 @@ use std::collections::{HashMap, HashSet}; use anyhow::Result; +use chrono::{SecondsFormat, TimeZone, Utc}; use crate::memory::{to_base36, MemoryLane, MemoryStore, ResolvedRef}; @@ -544,6 +545,9 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str ); if let Some(timestamp) = fact.timestamp { attrs.push_str(&format!(" t=\"{}\"", timestamp)); + if let Some(date) = format_timestamp_utc(timestamp) { + attrs.push_str(&format!(" date=\"{}\"", xml_escape(&date))); + } } if let Some(value) = &fact.value_text { attrs.push_str(&format!(" value=\"{}\"", xml_escape(value))); @@ -583,9 +587,12 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str .filter_map(|index| packet.timeline.get(index)) { let line = format!( - " {}\n", + " {}\n", xml_escape(&event.ref_id), event.timestamp, + format_timestamp_utc(event.timestamp) + .map(|date| format!(" date=\"{}\"", xml_escape(&date))) + .unwrap_or_default(), event.ordinal, xml_escape(event.session_id.as_deref().unwrap_or("")), xml_escape(&truncate_chars(&event.text, 96)), @@ -612,6 +619,15 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str out } +fn format_timestamp_utc(timestamp: i64) -> Option { + if timestamp <= 0 { + return None; + } + Utc.timestamp_opt(timestamp, 0) + .single() + .map(|dt| dt.to_rfc3339_opts(SecondsFormat::Secs, true)) +} + fn push_packet_body( out: &mut String, packet: &EvidencePacket, @@ -1176,6 +1192,7 @@ mod tests { assert!(rendered.contains("")); assert!(rendered.contains("")); assert!(rendered.contains("ref=\"ref_fact\"")); + assert!(rendered.contains("date=\"1970-01-01T00:00:42Z\"")); assert!(rendered.contains("value=\"27:45\"")); assert!(rendered.contains(""); let content = format!( - "\n{}\n\n\n\n{}\n", + "\n{}\n\n\n{}", packets, - xml_escape(&query.text) + render_task(&query) ); let estimated_tokens = content.chars().count() / 4; Ok(AssembledContext { @@ -1537,6 +1538,28 @@ fn xml_escape(value: &str) -> String { .replace('\'', "'") } +fn render_task(query: &BenchQuery) -> String { + let task = xml_escape(&query.text); + match query.reference_time.and_then(format_reference_time) { + Some(reference_date) => format!( + "\n{}\n", + query.reference_time.unwrap_or_default(), + xml_escape(&reference_date), + task + ), + None => format!("\n{}\n", task), + } +} + +fn format_reference_time(timestamp: i64) -> Option { + if timestamp <= 0 { + return None; + } + Utc.timestamp_opt(timestamp, 0) + .single() + .map(|dt| dt.to_rfc3339_opts(SecondsFormat::Secs, true)) +} + fn normalized_turn_role(role: &str) -> &'static str { match role { "user" => "user", @@ -2023,6 +2046,10 @@ mod tests { let context = pipeline.assemble_context(query, &retrieval, budget).await?; assert!(context.content.contains("The Glass Menagerie")); + assert!(context.content.contains("reference_time=\"3\"")); + assert!(context + .content + .contains("reference_date=\"1970-01-01T00:00:03Z\"")); Ok(()) }