diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index 5fbef87..9174a46 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -78,3 +78,4 @@ {"id":"int-ce8a7dc1","kind":"field_change","created_at":"2026-06-30T20:47:27.910556331Z","actor":"dawn","issue_id":"klbr-k7d","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed larger query-time LongMemEval QA run, inspected miss buckets, and recorded follow-up work."}} {"id":"int-01ad9c52","kind":"field_change","created_at":"2026-06-30T21:01:48.359954498Z","actor":"dawn","issue_id":"klbr-2jh","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed structural inspection and diagnostic fix; filed aggregate value representation follow-up."}} {"id":"int-6ca0e2c5","kind":"field_change","created_at":"2026-06-30T21:11:58.591402412Z","actor":"dawn","issue_id":"klbr-jae","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented typed evidence value mentions for fact rows, rendered them in memory packets, and made aggregate synthesis use only finite same-kind/unit aggregateable values. Ordinals, dates, and mixed-unit rows now abstain instead of falling back to first-number arithmetic; no prompt fixes were added."}} +{"id":"int-5d3aa77d","kind":"field_change","created_at":"2026-06-30T21:57:30.338732657Z","actor":"dawn","issue_id":"klbr-4ls","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented time-aware session-first ranking by threading ref event timestamps into EvidenceAtom and applying a bounded reference_time temporal score for episodic/live candidates, with canonical ref_timestamp preferred over promptable computed_at and sentinel/future safeguards. Stage-one traces now expose timestamp and temporal_score. Verified with klbr-core tests (166 passed, 1 ignored), klbr-bench tests (19 passed), fmt/diff checks, and retrieval-only LongMemEval-S smokes: sample20 at /tmp/klbr-time-aware-sample20 and corrected sample5 at /tmp/klbr-time-aware-sample5-v2."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index bcf42bd..8fd4c78 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -60,7 +60,7 @@ {"_type":"issue","id":"klbr-ds9","title":"refactor and simplify core agent loop without losing behaviour","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T21:20:30Z","created_by":"dawn","updated_at":"2026-06-30T21:20:30Z","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-jae","title":"Design aggregate evidence value representation","description":"LongMemEval aggregate misses show that a prompt fix would be benchmark-shaped and brittle. The structural gap is evidence representation: EvidenceFactRow currently stores only the first numeric value from a chunk, often capturing list ordinals, dates, or distractors instead of the values needed for counts/sums. Design and implement a general value representation for memory evidence that can keep multiple numeric mentions with nearby unit/currency/date/entity context, then only enable deterministic aggregate synthesis when the rendered facts are grounded and unambiguous.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T21:01:47Z","created_by":"dawn","updated_at":"2026-06-30T21:11:59Z","started_at":"2026-06-30T21:02:38Z","closed_at":"2026-06-30T21:11:59Z","close_reason":"Implemented typed evidence value mentions for fact rows, rendered them in memory packets, and made aggregate synthesis use only finite same-kind/unit aggregateable values. Ordinals, dates, and mixed-unit rows now abstain instead of falling back to first-number arithmetic; no prompt fixes were added.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-2jh","title":"Inspect aggregate and value-synthesis LongMemEval misses","description":"The sample-100 query-time LongMemEval run showed 12 official failures where the answer ref was rendered but the answer value was not visible, plus 10 value-visible reader or synthesis failures. Inspect aggregate_count, aggregate_sum, lookup, and update_resolution misses to decide whether packet rendering, fact extraction, or final reader prompting needs the next fix. Use measured miss buckets rather than broad context fanout.","notes":"Inspected sample-100 aggregate/value misses without prompt changes. User explicitly rejected prompt-fix/hardcoding route. Findings: many aggregate misses are not simple retrieval failures; literal answer_value_visible is misleading for derived answers like 5+3=8 or component expenses summing to 85, and short numeric answers can produce false positives from unrelated dates/list ordinals. Also found a real diagnostic bug: assemble_context marked every selected packet ref as used/rendered even when render_evidence_packet omitted that ref under packet budget. Fixed core ref accounting so render_evidence_packet_with_refs reports emitted body/fact/timeline refs and assemble_context.used_refs tracks only those. Tightened bench diagnostics so answer_value_visible requires an answer-bearing ref to have been emitted, and critical_fact_row_visible requires a fact row for an answer-bearing ref. Targeted retrieval-only rerun for 46a3abf7 now reports answer_bearing_ref_selected=1.0 but rendered/in_context/value_visible/critical_fact_row_visible=0.0, correctly rebucketing it as selected_not_rendered instead of value-visible reader failure. Tests: cargo fmt --check, cargo test -p klbr-core, cargo test -p klbr-bench, git diff --check.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T20:47:37Z","created_by":"dawn","updated_at":"2026-06-30T21:01:48Z","started_at":"2026-06-30T20:49:47Z","closed_at":"2026-06-30T21:01:48Z","close_reason":"Completed structural inspection and diagnostic fix; filed aggregate value representation follow-up.","dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-4ls","title":"Design time-aware passive memory ranking","description":"LongMemEval temporal misses exposed that BenchQuery.reference_time is currently not used by dense/sparse candidate ranking, and temporal evidence is selected mostly by semantic/entity similarity. Design and implement a measured ranking policy that uses query reference_time and evidence timestamps to support relative-date, elapsed-time, and event-order questions without hard-coded dataset cues or broad context fanout.","notes":"sample-100 query-time run finished with temporal-reasoning accuracy 0.5556 over 27 examples. On the 25 overlapping sample-25 ids, query-date rendering recovered b46e15ee and gpt4_b5700ca9, both relative-date examples. Remaining temporal failures are still mostly order_or_rank: 12 temporal misses total, including 5 rendered-value-not-visible, 3 selected-not-rendered, 2 not-selected, and 2 value-visible reader/synthesis. Design should use BenchQuery.reference_time plus evidence timestamps during candidate or packet ranking, not dataset cue hacks.","status":"in_progress","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T17:49:11Z","created_by":"dawn","updated_at":"2026-06-30T21:18:34Z","started_at":"2026-06-30T21:18:34Z","dependencies":[{"issue_id":"klbr-4ls","depends_on_id":"klbr-k7d","type":"blocks","created_at":"2026-06-30T20:49:19Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-4ls","title":"Design time-aware passive memory ranking","description":"LongMemEval temporal misses exposed that BenchQuery.reference_time is currently not used by dense/sparse candidate ranking, and temporal evidence is selected mostly by semantic/entity similarity. Design and implement a measured ranking policy that uses query reference_time and evidence timestamps to support relative-date, elapsed-time, and event-order questions without hard-coded dataset cues or broad context fanout.","notes":"sample-100 query-time run finished with temporal-reasoning accuracy 0.5556 over 27 examples. On the 25 overlapping sample-25 ids, query-date rendering recovered b46e15ee and gpt4_b5700ca9, both relative-date examples. Remaining temporal failures are still mostly order_or_rank: 12 temporal misses total, including 5 rendered-value-not-visible, 3 selected-not-rendered, 2 not-selected, and 2 value-visible reader/synthesis. Design should use BenchQuery.reference_time plus evidence timestamps during candidate or packet ranking, not dataset cue hacks.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T17:49:11Z","created_by":"dawn","updated_at":"2026-06-30T21:57:30Z","started_at":"2026-06-30T21:18:34Z","closed_at":"2026-06-30T21:57:30Z","close_reason":"Implemented time-aware session-first ranking by threading ref event timestamps into EvidenceAtom and applying a bounded reference_time temporal score for episodic/live candidates, with canonical ref_timestamp preferred over promptable computed_at and sentinel/future safeguards. Stage-one traces now expose timestamp and temporal_score. Verified with klbr-core tests (166 passed, 1 ignored), klbr-bench tests (19 passed), fmt/diff checks, and retrieval-only LongMemEval-S smokes: sample20 at /tmp/klbr-time-aware-sample20 and corrected sample5 at /tmp/klbr-time-aware-sample5-v2.","dependencies":[{"issue_id":"klbr-4ls","depends_on_id":"klbr-k7d","type":"blocks","created_at":"2026-06-30T20:49:19Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-k7d","title":"Run larger LongMemEval QA and inspect remaining misses","description":"Run a larger same-protocol LongMemEval-S QA sample after the anchor-window evidence rendering fix, compare against the sample-25 artifacts, inspect remaining miss categories while the run is in flight, and collect broader retrieval/memory-eval research perspectives before deciding next implementation work.","notes":"sample-100 query-time run finished in benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample100_seed4937553249516211047_query_time_qa. Official LongMemEval-S QA accuracy: 0.61 over 100 evaluated, 0 skipped. Retrieval metrics: RecallAny@5 0.98, RecallAll@5 0.85, PacketRecallAny@5 0.98, PacketRecallAll@5 0.78, answer_bearing_ref_selected 0.81, answer_bearing_ref_in_context 0.74, answer_value_visible 0.47. Per type accuracy: single-session-assistant 0.9091, knowledge-update 0.8667, single-session-user 0.7143, temporal-reasoning 0.5556, multi-session 0.4074, single-session-preference 0.3333. Miss buckets among 39 official failures: 12 answer ref not selected, 5 selected not rendered, 12 rendered but answer value not visible, 10 value-visible reader or synthesis failures. Misses by type: 16 multi-session, 12 temporal-reasoning, 4 single-session-user, 4 single-session-preference, 2 knowledge-update, 1 single-session-assistant. Misses by op: 17 order_or_rank, 6 aggregate_count, 5 aggregate_sum, 4 preference_recommendation, 4 lookup, 3 update_resolution. Compared on the 25 overlapping sample-25 ids, official labels moved 14/25 to 15/25: recovered 6b168ec8, b46e15ee, gpt4_b5700ca9; regressed 51a45a95 and b5ef892d. The query reference date rendering specifically helped relative-date examples, but sample100 still supports separate follow-up for time-aware ranking and aggregate/value synthesis.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T17:31:26Z","created_by":"dawn","updated_at":"2026-06-30T20:47:28Z","started_at":"2026-06-30T17:31:31Z","closed_at":"2026-06-30T20:47:28Z","close_reason":"Completed larger query-time LongMemEval QA run, inspected miss buckets, and recorded follow-up work.","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.6","title":"Probe passive writer retention for atomic facts and events","description":"Research on memory systems suggests passive QA can fail because write-time compression loses entity/slot/value/time facts before retrieval ever has a chance. Klbr now writes episode_event_card artifacts and generic fact_rows, but the next passive-recall improvement pass should add writer-side probes for counts, updates, preference constraints, and temporal facts so we can tell whether the passive writer retained the answer-bearing atom before tuning retrieval.","acceptance_criteria":"Fixtures or diagnostics check that answer-bearing entity/slot/value/time atoms exist in stored promptable artifacts before retrieval; failures are reported separately from packet/ranking misses; at least count, update_resolution, temporal order, and preference-constraint cases are covered.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:57Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately","labels":["bench","longmemeval","memory","writer"],"dependencies":[{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:56Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.6","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:18Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.5","title":"Make WhenLoss-style passive QA diagnostics first-class","description":"The passive QA bench should be treated as best-effort passive recall, not the whole memory-system score. The klbr-bench --diagnostic whenloss mode exists, but regression work needs a first-class joined report that separates write-side loss from retrieval/packet/rendering loss per question. Add a report path for tfc, oracle-evidence, complete-stored-memory, and retrieved-memory scores joined with packet metrics and local/official QA outcomes.","acceptance_criteria":"A normal regression run can emit per-question tfc/oe/csm/rm scores plus write_gap and retrieval_gap; report rows join those scores with op_plan, answer_bearing_ref_selected/rendered/in_context, answer_value_visible, and official/local QA labels; docs explicitly frame LongMemEval passive QA as best-effort passive recall rather than agentic folgezettel memory quality.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:41Z","created_by":"dawn","updated_at":"2026-06-30T12:55:18Z","closed_at":"2026-06-30T12:55:18Z","close_reason":"whenloss diagnostic summaries are emitted into trace_summary rows and docs frame LongMemEval passive QA as best-effort passive recall","labels":["bench","diagnostics","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:41Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.5","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:17Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} diff --git a/docs/memory-benches.md b/docs/memory-benches.md index ff25bf4..a327842 100644 --- a/docs/memory-benches.md +++ b/docs/memory-benches.md @@ -1,7 +1,7 @@ # memory benchmark protocol status: current -updated: 2026-06-29 +updated: 2026-07-01 this is the active benchmark protocol doc. if benchmark status conflicts with [`memory-implementation-status.md`](./memory-implementation-status.md), verify @@ -64,8 +64,9 @@ rtk cargo run -p klbr-bench -- run \ --out /tmp/klbr-fact-row-trace-smoke ``` -the trace should include `stage_one.session_candidates`, `coverage`, packet -`fact_rows`, and rendered packet metrics. the report should include +the trace should include `stage_one.session_candidates`, including per-session +`timestamp` and `temporal_score` when a query reference time is available, +`coverage`, packet `fact_rows`, and rendered packet metrics. the report should include `critical_fact_row_visible` and `collect_mode_fact_group_recall`; these metrics separate selected refs, rendered refs, value visibility, and structured fact-row visibility instead of folding everything into one recall number. diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index cf54809..af60663 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -1,7 +1,7 @@ # memory architecture implementation status status: current -updated: 2026-06-30 +updated: 2026-07-01 read this with [`long-term-memory-arch.md`](./long-term-memory-arch.md). this file says what exists today; the architecture doc says where the memory stack is @@ -73,8 +73,11 @@ going next. - session/event-first retrieval has an initial production profile: `klbr-full`, `dense-only`, and explicit `session-first` profiles group archival candidates by `session_id`, prefer episodic/event-card anchors when present, then add - chunk fts/sparse/dense hits as packet enrichment. retrieval traces now include - `stage_one.session_candidates`. + chunk fts/sparse/dense hits as packet enrichment. when `BenchQuery` provides + `reference_time`, session-first ranking applies a bounded temporal score from + candidate timestamps, favoring recent episodic/live evidence at or before the + query time without query cue lists or extra fanout. retrieval traces now + include `stage_one.session_candidates` with timestamp and `temporal_score`. - `fts-only` keeps episode notes for fts but skips embedded episode-memory rows, so lexical ablations no longer require the dense embedder during ingest. - collect-mode retrieval is wired through `OpPlan::collect_all` for aggregate, diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs index ab2ce27..e27d58a 100644 --- a/klbr-core/src/evidence.rs +++ b/klbr-core/src/evidence.rs @@ -134,6 +134,7 @@ pub struct EvidenceAtom { pub rank_in_source: usize, pub score: f32, pub token_count: usize, + pub timestamp: Option, pub body: String, pub anchor_kind: AnchorKind, } @@ -1162,6 +1163,7 @@ fn resolved_refs_to_ids(refs: Vec) -> Vec { .collect() } +#[cfg(test)] fn first_numeric_value(text: &str) -> (Option, Option) { primary_numeric_value(&numeric_value_mentions(text)) } @@ -1951,6 +1953,7 @@ mod tests { rank_in_source: 1, score: root_entry.score, token_count: root_entry.token_count, + timestamp: root_entry.timestamp, body: root_entry.body, anchor_kind: AnchorKind::QueryMatch, }; diff --git a/klbr-core/src/memory.rs b/klbr-core/src/memory.rs index 2f9d31d..aa77651 100644 --- a/klbr-core/src/memory.rs +++ b/klbr-core/src/memory.rs @@ -147,6 +147,7 @@ pub struct RefSearchEntry { pub lane: MemoryLane, pub body: String, pub token_count: usize, + pub timestamp: Option, pub score: f32, pub source: String, } @@ -1309,7 +1310,8 @@ impl MemoryStore { r.entity_type, COALESCE(m.lane, 'semantic') AS lane, p.body, - p.token_count + p.token_count, + p.computed_at FROM promptable_text p JOIN refs r ON r.ref_id = p.ref_id LEFT JOIN ref_metadata m ON m.ref_id = p.ref_id @@ -1341,6 +1343,7 @@ impl MemoryStore { lane: MemoryLane::parse(&lane_raw), body: row.get(4)?, token_count: row.get::<_, i64>(5)? as usize, + timestamp: row.get(6)?, score: 0.0, source: "complete_stored".to_string(), }); @@ -1376,7 +1379,8 @@ impl MemoryStore { r.entity_type, COALESCE(m.lane, 'semantic') AS lane, p.body, - p.token_count + p.token_count, + p.computed_at FROM promptable_text p JOIN refs r ON r.ref_id = p.ref_id LEFT JOIN ref_metadata m ON m.ref_id = p.ref_id @@ -1398,6 +1402,7 @@ impl MemoryStore { lane: MemoryLane::parse(&lane_raw), body: row.get(4)?, token_count: row.get::<_, i64>(5)? as usize, + timestamp: row.get(6)?, score: 0.0, source: source.to_string(), }) @@ -1517,6 +1522,7 @@ impl MemoryStore { COALESCE(m.lane, 'semantic') AS lane, p.body, p.token_count, + p.computed_at, e.embedding_blob, e.embedding_dim FROM embedding_items e @@ -1533,11 +1539,11 @@ impl MemoryStore { if !lane_filter.is_empty() && !lane_filter.contains(&lane_raw.as_str()) { continue; } - let embedding_dim: i64 = row.get(7)?; + let embedding_dim: i64 = row.get(8)?; if embedding_dim as usize != query_embedding.len() { continue; } - let embedding_blob: Vec = row.get(6)?; + let embedding_blob: Vec = row.get(7)?; let embedding = bytes_to_f32s(&embedding_blob); if embedding.len() != query_embedding.len() { continue; @@ -1549,6 +1555,7 @@ impl MemoryStore { lane: MemoryLane::parse(&lane_raw), body: row.get(4)?, token_count: row.get::<_, i64>(5)? as usize, + timestamp: row.get(6)?, score: cosine_distance(query_embedding, &embedding), source: "dense".to_string(), }); @@ -1699,6 +1706,7 @@ impl MemoryStore { COALESCE(m.lane, 'semantic') AS lane, f.body, COALESCE(p.token_count, length(f.body) / 4) AS token_count, + p.computed_at, bm25(promptable_text_fts) AS score FROM promptable_text_fts f JOIN refs r ON r.ref_id = f.ref_id @@ -1723,7 +1731,8 @@ impl MemoryStore { lane: MemoryLane::parse(&lane_raw), body: row.get(4)?, token_count: row.get::<_, i64>(5)? as usize, - score: row.get::<_, f64>(6)? as f32, + timestamp: row.get(6)?, + score: row.get::<_, f64>(7)? as f32, source: "fts".to_string(), }); if results.len() >= limit { @@ -1763,6 +1772,7 @@ impl MemoryStore { COALESCE(m.lane, 'semantic') AS lane, f.body, COALESCE(p.token_count, length(f.body) / 4) AS token_count, + p.computed_at, bm25(promptable_text_trigram) AS score FROM promptable_text_trigram f JOIN refs r ON r.ref_id = f.ref_id @@ -1787,7 +1797,8 @@ impl MemoryStore { lane: MemoryLane::parse(&lane_raw), body: row.get(4)?, token_count: row.get::<_, i64>(5)? as usize, - score: row.get::<_, f64>(6)? as f32, + timestamp: row.get(6)?, + score: row.get::<_, f64>(7)? as f32, source: "fts_trigram".to_string(), }); if results.len() >= limit { @@ -1859,6 +1870,7 @@ impl MemoryStore { COALESCE(m.lane, 'semantic') AS lane, p.body, p.token_count, + p.computed_at, ({}) AS score FROM promptable_text p JOIN refs r ON r.ref_id = p.ref_id @@ -1887,7 +1899,7 @@ impl MemoryStore { let rows = stmt.query_map(rusqlite::params_from_iter(final_params), |row| { let lane_str: String = row.get(3)?; let lane = MemoryLane::parse(&lane_str); - let score: f64 = row.get(6)?; + let score: f64 = row.get(7)?; Ok(RefSearchEntry { ref_id: row.get(0)?, alias: row.get(1)?, @@ -1895,6 +1907,7 @@ impl MemoryStore { lane, body: row.get(4)?, token_count: row.get::<_, i64>(5)? as usize, + timestamp: row.get(6)?, score: score as f32, source: "sparse".to_string(), }) diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index 6fa01bf..a36c1dc 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -10,8 +10,7 @@ use crate::{ evidence::{ render_evidence_packet_with_refs, stable_base36, AnchorKind, EntityType, EvidenceCoverageTrace, EvidenceFactRow, EvidenceOmittedRef, EvidencePacket, - EvidencePacketKind, EvidencePlanner, EvidenceValueKind, EvidenceValueMention, - RetrievalSource, + EvidencePacketKind, EvidencePlanner, RetrievalSource, }, memory::{ to_base26_suffix, to_base36, MarkdownNoteInput, MemoryLane, MemoryStore, RefSearchEntry, @@ -142,6 +141,10 @@ pub struct StageOneSessionCandidate { pub session_id: Option, pub anchor_ref: String, pub anchor_entity_type: EntityType, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub timestamp: Option, + #[serde(default)] + pub temporal_score: f32, pub sources: Vec, pub seed_refs: Vec, pub score: f32, @@ -430,7 +433,7 @@ impl MemoryPipeline { let ranked_limit = packet_limit.saturating_mul(4).max(candidate_limit); let (candidates, stage_one) = if self.profile.session_first { - rank_session_first_candidates(candidates, ranked_limit) + rank_session_first_candidates(candidates, ranked_limit, query.reference_time) } else { let candidates = rank_seed_candidates(candidates, ranked_limit); let stage_one = trace_from_ranked_candidates("source_interleave", &candidates); @@ -727,6 +730,7 @@ impl MemoryPipeline { rank_in_source: 1, score, token_count: data.token_count.unwrap_or_else(|| body.chars().count() / 4), + timestamp: self.memory.ref_timestamp(ref_id).ok().flatten(), body, anchor_kind: if source == "exact" { AnchorKind::ExplicitRef @@ -737,6 +741,12 @@ impl MemoryPipeline { } fn ref_search_entry_to_atom(&self, entry: RefSearchEntry) -> Result { + let timestamp = self + .memory + .ref_timestamp(&entry.ref_id) + .ok() + .flatten() + .or(entry.timestamp); Ok(EvidenceAtom { session_id: self.memory.ref_session_id(&entry.ref_id).ok().flatten(), ref_id: entry.ref_id, @@ -747,6 +757,7 @@ impl MemoryPipeline { rank_in_source: 1, score: entry.score, token_count: entry.token_count, + timestamp, body: entry.body, anchor_kind: AnchorKind::QueryMatch, }) @@ -858,6 +869,7 @@ fn resolved_to_entry(resolved: ResolvedRef, source: &str) -> Option, limit: usize) -> Vec, limit: usize, + reference_time: Option, ) -> (Vec, StageOneTrace) { assign_source_ranks(&mut candidates); @@ -938,8 +951,8 @@ fn rank_session_first_candidates( let mut groups = groups.into_values().collect::>(); groups.sort_by(|left, right| { right - .score() - .partial_cmp(&left.score()) + .score(reference_time) + .partial_cmp(&left.score(reference_time)) .unwrap_or(std::cmp::Ordering::Equal) .then_with(|| left.group_key.cmp(&right.group_key)) }); @@ -967,7 +980,7 @@ fn rank_session_first_candidates( if out.len() >= limit { break; } - let trace = group.trace(); + let trace = group.trace(reference_time); if traced_groups.insert(trace.group_key.clone()) { session_candidates.push(trace); } @@ -1011,7 +1024,7 @@ impl CandidateSessionGroup { self.candidates.push(candidate); } - fn score(&self) -> f32 { + fn score(&self, reference_time: Option) -> f32 { let source_score = [ RetrievalSource::Dense, RetrievalSource::Sparse, @@ -1034,10 +1047,26 @@ impl CandidateSessionGroup { 0.0 }; let source_count_bonus = 0.15 * (1.0 + self.source_names().len() as f32).ln(); - source_score + session_anchor_bonus + source_count_bonus + source_score + + session_anchor_bonus + + source_count_bonus + + self.temporal_score(reference_time) + } + + fn temporal_score(&self, reference_time: Option) -> f32 { + reference_time + .and_then(|reference_time| { + self.candidates + .iter() + .map(|candidate| candidate_temporal_score(candidate, reference_time)) + .max_by(|left, right| { + left.partial_cmp(right).unwrap_or(std::cmp::Ordering::Equal) + }) + }) + .unwrap_or(0.0) } - fn trace(&self) -> StageOneSessionCandidate { + fn trace(&self, reference_time: Option) -> StageOneSessionCandidate { let anchor = self .candidates .iter() @@ -1048,13 +1077,19 @@ impl CandidateSessionGroup { session_id: self.session_id.clone(), anchor_ref: anchor.ref_id.clone(), anchor_entity_type: anchor.entity_type, + timestamp: self + .candidates + .iter() + .filter_map(|candidate| candidate.timestamp) + .max(), + temporal_score: self.temporal_score(reference_time), sources: self.source_names(), seed_refs: self .candidates .iter() .map(|candidate| candidate.ref_id.clone()) .collect(), - score: self.score(), + score: self.score(reference_time), } } @@ -1134,6 +1169,33 @@ fn candidate_rank_score(candidate: &EvidenceAtom) -> f32 { source_weight(candidate.source) / (20.0 + candidate.rank_in_source.max(1) as f32) } +fn candidate_temporal_score(candidate: &EvidenceAtom, reference_time: i64) -> f32 { + if !is_plausible_temporal_timestamp(reference_time) { + return 0.0; + } + if !matches!(candidate.lane, MemoryLane::Live | MemoryLane::Episodic) { + return 0.0; + } + let Some(timestamp) = candidate.timestamp else { + return 0.0; + }; + if !is_plausible_temporal_timestamp(timestamp) { + return 0.0; + } + let age_seconds = reference_time.saturating_sub(timestamp); + if age_seconds < 0 { + return -0.15; + } + let age_days = age_seconds as f32 / 86_400.0; + 0.45 / (1.0 + (1.0 + age_days).log10()) +} + +fn is_plausible_temporal_timestamp(timestamp: i64) -> bool { + const YEAR_2000_UTC: i64 = 946_684_800; + const YEAR_2100_UTC: i64 = 4_102_444_800; + (YEAR_2000_UTC..=YEAR_2100_UTC).contains(×tamp) +} + fn source_weight(source: RetrievalSource) -> f32 { match source { RetrievalSource::Exact => 3.0, @@ -1172,6 +1234,8 @@ fn trace_from_ranked_candidates(strategy: &str, candidates: &[EvidenceAtom]) -> session_id: candidate.session_id.clone(), anchor_ref: candidate.ref_id.clone(), anchor_entity_type: candidate.entity_type, + timestamp: candidate.timestamp, + temporal_score: 0.0, sources: vec![candidate.source.as_str().to_string()], seed_refs: vec![candidate.ref_id.clone()], score: candidate_rank_score(candidate), @@ -1835,7 +1899,10 @@ fn dense_embedding_text(body: &str) -> String { #[cfg(test)] mod tests { use super::*; - use crate::evidence::{EvidencePacketKind, EvidencePacketSignals, EvidenceSourceSignal}; + use crate::evidence::{ + EvidencePacketKind, EvidencePacketSignals, EvidenceSourceSignal, EvidenceValueKind, + EvidenceValueMention, + }; use crate::planner::QueryOp; use tempfile::NamedTempFile; @@ -1906,6 +1973,7 @@ mod tests { ), ], 8, + None, ); assert_eq!(trace.strategy, "session_first"); @@ -1977,6 +2045,7 @@ mod tests { ), ], 3, + None, ); assert_eq!(trace.session_candidates[0].anchor_ref, "s1_episode_a"); @@ -1989,6 +2058,61 @@ mod tests { ); } + #[test] + fn session_first_uses_reference_time_for_temporal_ranking() { + let reference_time = 1_700_000_000; + let mut old = seed_atom( + "old_episode", + Some("old"), + EntityType::Episode, + RetrievalSource::Dense, + 1, + MemoryLane::Episodic, + ); + old.timestamp = Some(reference_time - 90 * 86_400); + let mut recent = seed_atom( + "recent_episode", + Some("recent"), + EntityType::Episode, + RetrievalSource::Dense, + 1, + MemoryLane::Episodic, + ); + recent.timestamp = Some(reference_time - 2 * 86_400); + let recent_timestamp = recent.timestamp; + + let (ranked, trace) = + rank_session_first_candidates(vec![old, recent], 4, Some(reference_time)); + + assert_eq!( + trace + .session_candidates + .first() + .and_then(|candidate| candidate.session_id.as_deref()), + Some("recent") + ); + assert_eq!(trace.session_candidates[0].anchor_ref, "recent_episode"); + assert_eq!(trace.session_candidates[0].timestamp, recent_timestamp); + assert!(trace.session_candidates[0].temporal_score > 0.0); + assert_eq!(ranked[0].ref_id, "recent_episode"); + } + + #[test] + fn temporal_score_ignores_unbounded_sentinel_timestamps() { + let mut sentinel = seed_atom( + "sentinel_episode", + Some("sentinel"), + EntityType::Episode, + RetrievalSource::Dense, + 1, + MemoryLane::Episodic, + ); + sentinel.timestamp = Some(253_370_764_800); + + assert_eq!(candidate_temporal_score(&sentinel, 253_370_764_800), 0.0); + assert_eq!(candidate_temporal_score(&sentinel, 1_700_000_000), 0.0); + } + #[tokio::test] async fn turn_window_keeps_anchor_when_previous_turn_chunks_fill_cap() -> Result<()> { let tmp = NamedTempFile::new()?; @@ -3034,6 +3158,7 @@ mod tests { rank_in_source, score: 0.0, token_count: 8, + timestamp: None, body: format!("body for {ref_id}"), anchor_kind: AnchorKind::QueryMatch, }