From 466e5d3e80bc9bcd80e9b421241f7235f8f23523 Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Wed, 1 Jul 2026 00:12:15 +0300 Subject: [PATCH] add typed evidence value mentions --- .beads/interactions.jsonl | 1 + .beads/issues.jsonl | 2 +- docs/memory-implementation-status.md | 15 +- klbr-core/src/evidence.rs | 436 +++++++++++++++++++++++++-- klbr-core/src/pipeline.rs | 163 +++++++++- 5 files changed, 580 insertions(+), 37 deletions(-) diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index efb0352..5fbef87 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -77,3 +77,4 @@ {"id":"int-eba49ab6","kind":"field_change","created_at":"2026-06-30T17:48:55.880302146Z","actor":"dawn","issue_id":"klbr-d76","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Not planned for now: user wants folgezettel usefulness evaluated from real-world usage traces rather than a synthetic agentic eval. Revisit only after real usage exposes concrete failures or measurable trace-review needs."}} {"id":"int-ce8a7dc1","kind":"field_change","created_at":"2026-06-30T20:47:27.910556331Z","actor":"dawn","issue_id":"klbr-k7d","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed larger query-time LongMemEval QA run, inspected miss buckets, and recorded follow-up work."}} {"id":"int-01ad9c52","kind":"field_change","created_at":"2026-06-30T21:01:48.359954498Z","actor":"dawn","issue_id":"klbr-2jh","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed structural inspection and diagnostic fix; filed aggregate value representation follow-up."}} +{"id":"int-6ca0e2c5","kind":"field_change","created_at":"2026-06-30T21:11:58.591402412Z","actor":"dawn","issue_id":"klbr-jae","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented typed evidence value mentions for fact rows, rendered them in memory packets, and made aggregate synthesis use only finite same-kind/unit aggregateable values. Ordinals, dates, and mixed-unit rows now abstain instead of falling back to first-number arithmetic; no prompt fixes were added."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 3422884..ec5d653 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -41,7 +41,7 @@ {"_type":"issue","id":"klbr-wmz.2","title":"Route runtime passive recall through the canonical memory pipeline","description":"Bench retrieval now goes through MemoryPipeline, but runtime passive recall in klbr-core/src/agent.rs still embeds the prompt, calls MemoryStore::get_searchable, runs retrieval::retrieve_exact over legacy memories, then injects Context memory packets. That bypasses fts, exact refs, markdown notes, graph expansion, and the canonical lane/lifecycle path the docs describe.","design":"Avoid duplicating retrieval logic in agent.rs. Either make MemoryPipeline usable by AgentRuntime or extract a shared retrieval facade that both MemoryPipeline and runtime passive recall call.","acceptance_criteria":"Runtime passive recall uses the same lane-aware canonical retrieval and context packet assembly policy as the production pipeline; recalled packets can include fts/exact/dense/graph candidates from refs and markdown notes; archived/tombstoned/suppressed refs do not leak; klbr-core/src/instructions.md matches the actual memory packet format; tests or a focused integration fixture cover passive recall from a markdown note and from an explicit ref.","notes":"Runtime passive recall now calls MemoryPipeline::retrieve_evidence and injects shared EvidencePacket XML via Context::inject_evidence_packets; no-model runtime packet fixture covers turn-window expansion. Remaining acceptance is blocked on klbr-wmz.1 because dense search still starts from legacy memory rows before ref mapping.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:19Z","created_by":"dawn","updated_at":"2026-06-26T19:44:55Z","started_at":"2026-06-26T19:37:32Z","closed_at":"2026-06-26T19:44:55Z","close_reason":"Completed: runtime passive recall now uses MemoryPipeline/EvidencePlanner packets, instructions document current packet XML, and fixtures cover turn-window recall plus explicit markdown refs; dense canonical dependency completed in klbr-wmz.1.","labels":["architecture","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz.1","type":"blocks","created_at":"2026-06-26T22:37:53Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.1","title":"Move dense retrieval onto canonical refs and embedding_items","description":"Current source still has dense retrieval on the legacy memory-row surface: klbr-core/src/pipeline.rs search_dense calls MemoryStore::get_searchable and retrieval::retrieve_exact, while fts/exact retrieval uses canonical refs and promptable_text. The schema already has refs, promptable_text, markdown_note_chunks, and embedding_items, so dense retrieval should not be the odd path out.","design":"Prefer a ref-native embedding index backed by embedding_items. Backfill embeddings from promptable_text, keep memory-id aliases as compatibility aliases, and make klbr/full versus dense-only profiles exercise the same canonical identity layer as fts and exact retrieval.","acceptance_criteria":"Dense candidate generation works over canonical ref ids for memories, turn chunks, markdown note chunks, episode notes, profile notes, and procedural notes; lane and lifecycle filtering come from refs/ref_metadata instead of memory tags alone; benchmark traces return canonical refs for dense hits; regression tests cover a markdown-note-only hit and a tombstoned/suppressed ref not leaking through dense search.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:11Z","created_by":"dawn","updated_at":"2026-06-26T19:43:27Z","started_at":"2026-06-26T19:38:02Z","closed_at":"2026-06-26T19:43:27Z","close_reason":"Completed: dense candidate generation now lazily backfills embedding_items from active promptable refs, scores canonical ref embeddings directly, and tests markdown-note dense hits plus suppressed-ref filtering.","labels":["architecture","memory","refs","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.1","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:11Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz","title":"Finish memory architecture follow-through","description":"Tracks the remaining memory architecture work identified from docs/memory-arch.md, docs/memory-benches.md, docs/memory-implementation-status.md, and current klbr-core/klbr-bench source. Current status says pipeline, typed memory packets, markdown notes, edge mirroring, lifecycle projection, and benchmark runner integration exist; this epic is for gaps still present in source/docs.","acceptance_criteria":"Close when the child issues are complete, docs/memory-implementation-status.md is updated from current verification, and the architecture docs no longer point at missing or stale follow-up work.","status":"closed","priority":1,"issue_type":"epic","owner":"90008@klbr.net","created_at":"2026-06-26T17:52:54Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","closed_at":"2026-06-26T23:45:35Z","close_reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn.","labels":["architecture","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-jae","title":"Design aggregate evidence value representation","description":"LongMemEval aggregate misses show that a prompt fix would be benchmark-shaped and brittle. The structural gap is evidence representation: EvidenceFactRow currently stores only the first numeric value from a chunk, often capturing list ordinals, dates, or distractors instead of the values needed for counts/sums. Design and implement a general value representation for memory evidence that can keep multiple numeric mentions with nearby unit/currency/date/entity context, then only enable deterministic aggregate synthesis when the rendered facts are grounded and unambiguous.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T21:01:47Z","created_by":"dawn","updated_at":"2026-06-30T21:01:47Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-jae","title":"Design aggregate evidence value representation","description":"LongMemEval aggregate misses show that a prompt fix would be benchmark-shaped and brittle. The structural gap is evidence representation: EvidenceFactRow currently stores only the first numeric value from a chunk, often capturing list ordinals, dates, or distractors instead of the values needed for counts/sums. Design and implement a general value representation for memory evidence that can keep multiple numeric mentions with nearby unit/currency/date/entity context, then only enable deterministic aggregate synthesis when the rendered facts are grounded and unambiguous.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T21:01:47Z","created_by":"dawn","updated_at":"2026-06-30T21:11:59Z","started_at":"2026-06-30T21:02:38Z","closed_at":"2026-06-30T21:11:59Z","close_reason":"Implemented typed evidence value mentions for fact rows, rendered them in memory packets, and made aggregate synthesis use only finite same-kind/unit aggregateable values. Ordinals, dates, and mixed-unit rows now abstain instead of falling back to first-number arithmetic; no prompt fixes were added.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-2jh","title":"Inspect aggregate and value-synthesis LongMemEval misses","description":"The sample-100 query-time LongMemEval run showed 12 official failures where the answer ref was rendered but the answer value was not visible, plus 10 value-visible reader or synthesis failures. Inspect aggregate_count, aggregate_sum, lookup, and update_resolution misses to decide whether packet rendering, fact extraction, or final reader prompting needs the next fix. Use measured miss buckets rather than broad context fanout.","notes":"Inspected sample-100 aggregate/value misses without prompt changes. User explicitly rejected prompt-fix/hardcoding route. Findings: many aggregate misses are not simple retrieval failures; literal answer_value_visible is misleading for derived answers like 5+3=8 or component expenses summing to 85, and short numeric answers can produce false positives from unrelated dates/list ordinals. Also found a real diagnostic bug: assemble_context marked every selected packet ref as used/rendered even when render_evidence_packet omitted that ref under packet budget. Fixed core ref accounting so render_evidence_packet_with_refs reports emitted body/fact/timeline refs and assemble_context.used_refs tracks only those. Tightened bench diagnostics so answer_value_visible requires an answer-bearing ref to have been emitted, and critical_fact_row_visible requires a fact row for an answer-bearing ref. Targeted retrieval-only rerun for 46a3abf7 now reports answer_bearing_ref_selected=1.0 but rendered/in_context/value_visible/critical_fact_row_visible=0.0, correctly rebucketing it as selected_not_rendered instead of value-visible reader failure. Tests: cargo fmt --check, cargo test -p klbr-core, cargo test -p klbr-bench, git diff --check.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T20:47:37Z","created_by":"dawn","updated_at":"2026-06-30T21:01:48Z","started_at":"2026-06-30T20:49:47Z","closed_at":"2026-06-30T21:01:48Z","close_reason":"Completed structural inspection and diagnostic fix; filed aggregate value representation follow-up.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-4ls","title":"Design time-aware passive memory ranking","description":"LongMemEval temporal misses exposed that BenchQuery.reference_time is currently not used by dense/sparse candidate ranking, and temporal evidence is selected mostly by semantic/entity similarity. Design and implement a measured ranking policy that uses query reference_time and evidence timestamps to support relative-date, elapsed-time, and event-order questions without hard-coded dataset cues or broad context fanout.","notes":"sample-100 query-time run finished with temporal-reasoning accuracy 0.5556 over 27 examples. On the 25 overlapping sample-25 ids, query-date rendering recovered b46e15ee and gpt4_b5700ca9, both relative-date examples. Remaining temporal failures are still mostly order_or_rank: 12 temporal misses total, including 5 rendered-value-not-visible, 3 selected-not-rendered, 2 not-selected, and 2 value-visible reader/synthesis. Design should use BenchQuery.reference_time plus evidence timestamps during candidate or packet ranking, not dataset cue hacks.","status":"open","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-30T17:49:11Z","created_by":"dawn","updated_at":"2026-06-30T20:47:38Z","dependencies":[{"issue_id":"klbr-4ls","depends_on_id":"klbr-k7d","type":"blocks","created_at":"2026-06-30T20:49:19Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-k7d","title":"Run larger LongMemEval QA and inspect remaining misses","description":"Run a larger same-protocol LongMemEval-S QA sample after the anchor-window evidence rendering fix, compare against the sample-25 artifacts, inspect remaining miss categories while the run is in flight, and collect broader retrieval/memory-eval research perspectives before deciding next implementation work.","notes":"sample-100 query-time run finished in benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample100_seed4937553249516211047_query_time_qa. Official LongMemEval-S QA accuracy: 0.61 over 100 evaluated, 0 skipped. Retrieval metrics: RecallAny@5 0.98, RecallAll@5 0.85, PacketRecallAny@5 0.98, PacketRecallAll@5 0.78, answer_bearing_ref_selected 0.81, answer_bearing_ref_in_context 0.74, answer_value_visible 0.47. Per type accuracy: single-session-assistant 0.9091, knowledge-update 0.8667, single-session-user 0.7143, temporal-reasoning 0.5556, multi-session 0.4074, single-session-preference 0.3333. Miss buckets among 39 official failures: 12 answer ref not selected, 5 selected not rendered, 12 rendered but answer value not visible, 10 value-visible reader or synthesis failures. Misses by type: 16 multi-session, 12 temporal-reasoning, 4 single-session-user, 4 single-session-preference, 2 knowledge-update, 1 single-session-assistant. Misses by op: 17 order_or_rank, 6 aggregate_count, 5 aggregate_sum, 4 preference_recommendation, 4 lookup, 3 update_resolution. Compared on the 25 overlapping sample-25 ids, official labels moved 14/25 to 15/25: recovered 6b168ec8, b46e15ee, gpt4_b5700ca9; regressed 51a45a95 and b5ef892d. The query reference date rendering specifically helped relative-date examples, but sample100 still supports separate follow-up for time-aware ranking and aggregate/value synthesis.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T17:31:26Z","created_by":"dawn","updated_at":"2026-06-30T20:47:28Z","started_at":"2026-06-30T17:31:31Z","closed_at":"2026-06-30T20:47:28Z","close_reason":"Completed larger query-time LongMemEval QA run, inspected miss buckets, and recorded follow-up work.","dependency_count":0,"dependent_count":1,"comment_count":0} diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index 84a38c0..cf54809 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -92,8 +92,11 @@ going next. previous turns cannot produce packets whose `anchor_ref` is missing from `refs`. - evidence packets now carry compact `fact_rows` and `timeline` rows with - supporting refs while preserving raw packet bodies as provenance. rendered - `` include `` and `` blocks before bodies. + supporting refs while preserving raw packet bodies as provenance. fact rows + keep typed numeric value mentions (`number`, `currency`, `percent`, + `duration`, `date`, `time`, `ordinal`) so aggregate synthesis does not have + to trust the first number in a body. rendered `` include + `` and `` blocks before bodies. - `MemoryPipeline::answer` has an initial planned synthesis path for aggregate count/sum/avg, chronological ordering, previous/latest update resolution, and preference recommendations with personal-support guards. @@ -116,8 +119,8 @@ going next. ## open direction - harden the structured planner path and add model-backed planner fixtures. -- replace the current generic fact-row extraction with richer entity/slot/unit - extraction when the writer/planner path can provide it. +- harden fact-row entity/slot extraction around the typed value mentions when + the writer/planner path can provide stronger structure. - add deeper multilingual retrieval fixtures so english lexical shortcuts cannot become hidden policy. current coverage includes no-space cjk trigram fts, but cross-lingual paraphrase, turkish suffix variation, diacritics, and @@ -138,8 +141,8 @@ current result: cargo check -p klbr-core: 0 errors, 1 pre-existing warning cargo check -p klbr-bench: 0 errors, 1 pre-existing warning focused folgezettel tests: mk surface 2 passed; trail-support packet test 1 passed -klbr-core tests: 156 passed, 1 ignored -klbr-bench tests: 18 passed +klbr-core tests: 164 passed, 1 ignored +klbr-bench tests: 19 passed continuous-loop smoke: passed with the pre-existing agent.rs dead-code warning session-first/fts-only trace smoke: evaluated 1, CandidateSessionRecallAny@5 1.0000 ``` diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs index 2b8bc73..ab2ce27 100644 --- a/klbr-core/src/evidence.rs +++ b/klbr-core/src/evidence.rs @@ -1,7 +1,9 @@ use std::collections::{HashMap, HashSet}; +use std::sync::OnceLock; use anyhow::Result; use chrono::{SecondsFormat, TimeZone, Utc}; +use regex::Regex; use crate::memory::{to_base36, MemoryLane, MemoryStore, ResolvedRef}; @@ -190,6 +192,63 @@ pub struct EvidencePacket { pub estimated_tokens: usize, } +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum EvidenceValueKind { + Number, + Currency, + Percent, + Duration, + Date, + Time, + Ordinal, +} + +impl EvidenceValueKind { + pub fn as_str(&self) -> &'static str { + match self { + Self::Number => "number", + Self::Currency => "currency", + Self::Percent => "percent", + Self::Duration => "duration", + Self::Date => "date", + Self::Time => "time", + Self::Ordinal => "ordinal", + } + } +} + +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct EvidenceValueMention { + pub text: String, + pub number: Option, + pub kind: EvidenceValueKind, + pub unit: Option, +} + +impl EvidenceValueMention { + pub(crate) fn aggregate_signature(&self) -> Option { + let number = self.number?; + if !number.is_finite() { + return None; + } + if !matches!( + self.kind, + EvidenceValueKind::Number + | EvidenceValueKind::Currency + | EvidenceValueKind::Percent + | EvidenceValueKind::Duration + ) { + return None; + } + Some(format!( + "{}:{}", + self.kind.as_str(), + self.unit.as_deref().unwrap_or("") + )) + } +} + #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] pub struct EvidenceFactRow { pub row_id: String, @@ -200,6 +259,8 @@ pub struct EvidenceFactRow { pub ordinal: usize, pub value_text: Option, pub value_number: Option, + #[serde(default)] + pub values: Vec, pub text: String, } @@ -439,7 +500,8 @@ impl EvidencePlanner { .ref_session_id(ref_id)? .or_else(|| fallback_session_id.map(str::to_string)); let timestamp = self.memory.ref_timestamp(ref_id)?; - let (value_text, value_number) = first_numeric_value(body); + let values = numeric_value_mentions(body); + let (value_text, value_number) = primary_numeric_value(&values); let text = truncate_chars(body, 180); let row_seed = format!("{ref_id}:{ordinal}:{}", text); Ok(EvidenceFactRow { @@ -451,6 +513,7 @@ impl EvidencePlanner { ordinal, value_text, value_number, + values, text, }) }) @@ -572,6 +635,12 @@ pub fn render_evidence_packet_with_refs( if let Some(value) = fact.value_number { attrs.push_str(&format!(" number=\"{}\"", value)); } + if !fact.values.is_empty() { + attrs.push_str(&format!( + " values=\"{}\"", + xml_escape(&render_value_mentions(&fact.values)) + )); + } let line = format!( " {}\n", attrs, @@ -1094,35 +1163,320 @@ fn resolved_refs_to_ids(refs: Vec) -> Vec { } fn first_numeric_value(text: &str) -> (Option, Option) { - let mut current = String::new(); - let mut has_digit = false; - for ch in text.chars() { - if ch.is_ascii_digit() - || (has_digit && matches!(ch, '.' | ',' | ':' | '/' | '-')) - || (!has_digit && matches!(ch, '+' | '-')) - { - if ch.is_ascii_digit() { - has_digit = true; + primary_numeric_value(&numeric_value_mentions(text)) +} + +fn primary_numeric_value(values: &[EvidenceValueMention]) -> (Option, Option) { + values + .iter() + .find(|value| value.aggregate_signature().is_some()) + .or_else(|| values.first()) + .map(|value| (Some(value.text.clone()), value.number)) + .unwrap_or((None, None)) +} + +fn numeric_value_mentions(text: &str) -> Vec { + numeric_token_regex() + .find_iter(text) + .filter(|mat| !token_embedded_in_word(text, mat.start(), mat.end())) + .map(|mat| { + let token = mat.as_str(); + let token_lower = token.to_ascii_lowercase(); + let next_word = next_word_after(text, mat.end()); + let previous_word = previous_word_before(text, mat.start()); + let (kind, unit, mention_end) = classify_numeric_mention( + text, + mat.start(), + mat.end(), + token, + &token_lower, + previous_word.as_deref(), + next_word.as_ref(), + ); + let text = text[mat.start()..mention_end].trim().to_string(); + let number = parse_mention_number(token, &kind); + EvidenceValueMention { + text, + number, + kind, + unit, } - current.push(ch); - } else if has_digit { + }) + .collect() +} + +fn numeric_token_regex() -> &'static Regex { + static NUMERIC_TOKEN_RE: OnceLock = OnceLock::new(); + NUMERIC_TOKEN_RE.get_or_init(|| { + Regex::new( + r"\$?[+-]?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?(?:[:/-]\d{1,4})*(?:%|st|nd|rd|th)?", + ) + .expect("numeric value regex compiles") + }) +} + +fn classify_numeric_mention( + text: &str, + start: usize, + end: usize, + token: &str, + token_lower: &str, + previous_word: Option<&str>, + next_word: Option<&(String, usize)>, +) -> (EvidenceValueKind, Option, usize) { + let next_word_text = next_word.map(|(word, _)| word.as_str()); + let next_word_end = next_word.map(|(_, end)| *end).unwrap_or(end); + + if is_list_marker(text, start, end) { + return (EvidenceValueKind::Ordinal, None, end); + } + if is_currency_mention(token, previous_word, next_word_text) { + let mention_end = next_word_text + .filter(|word| is_currency_word(word)) + .map(|_| next_word_end) + .unwrap_or(end); + return ( + EvidenceValueKind::Currency, + Some("$".to_string()), + mention_end, + ); + } + if is_percent_mention(token_lower, next_word_text) { + let mention_end = next_word_text + .filter(|word| matches!(*word, "percent" | "percentage")) + .map(|_| next_word_end) + .unwrap_or(end); + return ( + EvidenceValueKind::Percent, + Some("%".to_string()), + mention_end, + ); + } + if token.contains(':') { + return (EvidenceValueKind::Time, None, end); + } + if is_date_like(token, token_lower, previous_word, next_word_text) { + return (EvidenceValueKind::Date, None, end); + } + if has_ordinal_suffix(token_lower) { + return (EvidenceValueKind::Ordinal, None, end); + } + if let Some(unit) = next_word_text.and_then(duration_unit) { + return (EvidenceValueKind::Duration, Some(unit), next_word_end); + } + (EvidenceValueKind::Number, None, end) +} + +fn token_embedded_in_word(text: &str, start: usize, end: usize) -> bool { + let before = text[..start].chars().next_back(); + let after = text[end..].chars().next(); + let before_word = before.is_some_and(|ch| ch.is_ascii_alphabetic() || ch == '_'); + let after_word = after.is_some_and(|ch| ch.is_ascii_alphabetic() || ch == '_'); + before_word || after_word +} + +fn next_word_after(text: &str, index: usize) -> Option<(String, usize)> { + let mut word_start = None; + for (offset, ch) in text[index..].char_indices() { + if ch.is_whitespace() || ch == '-' { + continue; + } + if ch.is_ascii_alphabetic() { + word_start = Some(index + offset); + } + break; + } + let start = word_start?; + let mut end = start; + for (offset, ch) in text[start..].char_indices() { + if ch.is_ascii_alphabetic() { + end = start + offset + ch.len_utf8(); + } else { + break; + } + } + Some((text[start..end].to_ascii_lowercase(), end)) +} + +fn previous_word_before(text: &str, index: usize) -> Option { + let prefix = &text[..index]; + let mut end = prefix.len(); + while let Some((pos, ch)) = prefix[..end].char_indices().next_back() { + if ch.is_whitespace() || ch == '-' { + end = pos; + } else { break; + } + } + let mut start = end; + while let Some((pos, ch)) = prefix[..start].char_indices().next_back() { + if ch.is_ascii_alphabetic() { + start = pos; } else { - current.clear(); + break; } } - if !has_digit { - return (None, None); + (start < end).then(|| prefix[start..end].to_ascii_lowercase()) +} + +fn is_list_marker(text: &str, start: usize, end: usize) -> bool { + let Some(punct) = text[end..].chars().next() else { + return false; + }; + if !matches!(punct, '.' | ')') { + return false; + } + let after_punct = end + punct.len_utf8(); + if text[after_punct..] + .chars() + .next() + .is_some_and(|ch| !ch.is_whitespace()) + { + return false; } - let value_text = current - .trim_matches(|ch: char| !ch.is_ascii_digit()) - .to_string(); - if value_text.is_empty() { - return (None, None); + let line_start = text[..start] + .rfind('\n') + .map(|index| index + 1) + .unwrap_or(0); + let prefix = text[line_start..start].trim(); + prefix.is_empty() || matches!(prefix, "-" | "*" | "+") +} + +fn is_currency_mention(token: &str, previous_word: Option<&str>, next_word: Option<&str>) -> bool { + token.starts_with('$') + || previous_word.is_some_and(is_currency_word) + || next_word.is_some_and(is_currency_word) +} + +fn is_currency_word(word: &str) -> bool { + matches!(word, "dollar" | "dollars" | "usd" | "buck" | "bucks") +} + +fn is_percent_mention(token_lower: &str, next_word: Option<&str>) -> bool { + token_lower.ends_with('%') + || next_word.is_some_and(|word| matches!(word, "percent" | "percentage")) +} + +fn is_date_like( + token: &str, + token_lower: &str, + previous_word: Option<&str>, + next_word: Option<&str>, +) -> bool { + if token.contains('/') || token[1.min(token.len())..].contains('-') { + return true; + } + if previous_word.is_some_and(is_month_word) || next_word.is_some_and(is_month_word) { + return true; + } + let normalized = strip_numeric_affixes(token_lower); + let is_year = normalized + .parse::() + .is_ok_and(|year| (1900..=2100).contains(&year)); + is_year + && (previous_word.is_some_and(is_date_context_word) + || next_word.is_some_and(is_date_context_word)) +} + +fn is_month_word(word: &str) -> bool { + matches!( + word, + "january" + | "jan" + | "february" + | "feb" + | "march" + | "mar" + | "april" + | "apr" + | "may" + | "june" + | "jun" + | "july" + | "jul" + | "august" + | "aug" + | "september" + | "sep" + | "sept" + | "october" + | "oct" + | "november" + | "nov" + | "december" + | "dec" + ) +} + +fn is_date_context_word(word: &str) -> bool { + is_month_word(word) || matches!(word, "date" | "year" | "birthday" | "anniversary") +} + +fn has_ordinal_suffix(token_lower: &str) -> bool { + ["st", "nd", "rd", "th"].iter().any(|suffix| { + token_lower + .strip_suffix(suffix) + .is_some_and(ends_with_digit) + }) +} + +fn duration_unit(word: &str) -> Option { + let unit = match word { + "day" | "days" => "day", + "week" | "weeks" => "week", + "month" | "months" => "month", + "year" | "years" => "year", + "hour" | "hours" | "hr" | "hrs" => "hour", + "minute" | "minutes" | "min" | "mins" => "minute", + "second" | "seconds" | "sec" | "secs" => "second", + "night" | "nights" => "night", + _ => return None, + }; + Some(unit.to_string()) +} + +fn parse_mention_number(token: &str, kind: &EvidenceValueKind) -> Option { + if matches!( + kind, + EvidenceValueKind::Date | EvidenceValueKind::Time | EvidenceValueKind::Ordinal + ) { + return None; } - let normalized = value_text.replace(',', ""); - let value_number = normalized.parse::().ok(); - (Some(value_text), value_number) + let normalized = strip_numeric_affixes(&token.to_ascii_lowercase()).replace(',', ""); + normalized.parse::().ok() +} + +fn strip_numeric_affixes(token: &str) -> String { + let token = token.trim_start_matches('$').trim_end_matches('%'); + ["st", "nd", "rd", "th"] + .iter() + .find_map(|suffix| token.strip_suffix(suffix)) + .unwrap_or(token) + .to_string() +} + +fn ends_with_digit(value: &str) -> bool { + value + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_digit()) +} + +fn render_value_mentions(values: &[EvidenceValueMention]) -> String { + values + .iter() + .take(8) + .map(|value| { + let mut rendered = format!("{}:{}", value.kind.as_str(), value.text); + if let Some(unit) = &value.unit { + rendered.push('['); + rendered.push_str(unit); + rendered.push(']'); + } + rendered + }) + .collect::>() + .join(";") } pub(crate) fn stable_base36(value: &str) -> String { @@ -1208,6 +1562,12 @@ mod tests { ordinal: 0, value_text: Some("27:45".to_string()), value_number: None, + values: vec![EvidenceValueMention { + text: "27:45".to_string(), + number: None, + kind: EvidenceValueKind::Time, + unit: None, + }], text: "personal best was 27:45".to_string(), }]; packet.timeline = vec![EvidenceTimelineEvent { @@ -1225,9 +1585,36 @@ mod tests { assert!(rendered.contains("ref=\"ref_fact\"")); assert!(rendered.contains("date=\"1970-01-01T00:00:42Z\"")); assert!(rendered.contains("value=\"27:45\"")); + assert!(rendered.contains("values=\"time:27:45\"")); assert!(rendered.contains(">(), + vec![ + EvidenceValueKind::Ordinal, + EvidenceValueKind::Currency, + EvidenceValueKind::Duration, + EvidenceValueKind::Date, + ] + ); + assert_eq!(values[1].text, "$40"); + assert_eq!(values[1].number, Some(40.0)); + assert_eq!(values[1].unit.as_deref(), Some("$")); + assert_eq!(values[2].text, "3 days"); + assert_eq!(values[2].number, Some(3.0)); + assert_eq!(values[2].unit.as_deref(), Some("day")); + assert_eq!( + first_numeric_value(text), + (Some("$40".to_string()), Some(40.0)) + ); + } + #[test] fn rendered_packet_with_refs_tracks_only_emitted_evidence_refs() { let mut packet = test_packet_with_refs( @@ -1281,6 +1668,7 @@ mod tests { ordinal, value_text: Some(format!("{ordinal}")), value_number: Some(ordinal as f64), + values: vec![], text: "fact row with enough text to consume visible packet budget".repeat(3), }) .collect(); @@ -1347,6 +1735,7 @@ mod tests { ordinal, value_text: None, value_number: None, + values: vec![], text: packet.bodies[ordinal].clone(), }) .collect(); @@ -1399,6 +1788,7 @@ mod tests { ordinal, value_text: None, value_number: None, + values: vec![], text: packet.bodies[ordinal].clone(), }) .collect(); diff --git a/klbr-core/src/pipeline.rs b/klbr-core/src/pipeline.rs index f6ce5e8..6fa01bf 100644 --- a/klbr-core/src/pipeline.rs +++ b/klbr-core/src/pipeline.rs @@ -10,7 +10,8 @@ use crate::{ evidence::{ render_evidence_packet_with_refs, stable_base36, AnchorKind, EntityType, EvidenceCoverageTrace, EvidenceFactRow, EvidenceOmittedRef, EvidencePacket, - EvidencePacketKind, EvidencePlanner, RetrievalSource, + EvidencePacketKind, EvidencePlanner, EvidenceValueKind, EvidenceValueMention, + RetrievalSource, }, memory::{ to_base26_suffix, to_base36, MarkdownNoteInput, MemoryLane, MemoryStore, RefSearchEntry, @@ -1274,16 +1275,54 @@ fn numeric_fact_values(facts: &[EvidenceFactRow]) -> Vec { let turn_chunk_values = facts .iter() .filter(|fact| fact.entity_type == "turn_chunk") - .filter_map(|fact| fact.value_number) - .filter(|value| value.is_finite()) + .flat_map(aggregate_values_for_fact) .collect::>(); - if !turn_chunk_values.is_empty() { - return turn_chunk_values; + let values = if !turn_chunk_values.is_empty() { + turn_chunk_values + } else { + facts + .iter() + .flat_map(aggregate_values_for_fact) + .collect::>() + }; + if values.is_empty() { + return Vec::new(); } - facts + let signatures = values .iter() - .filter_map(|fact| fact.value_number) + .map(|value| value.signature.as_str()) + .collect::>(); + if signatures.len() != 1 { + return Vec::new(); + } + values.into_iter().map(|value| value.number).collect() +} + +#[derive(Debug, Clone)] +struct AggregateValue { + number: f64, + signature: String, +} + +fn aggregate_values_for_fact(fact: &EvidenceFactRow) -> Vec { + if !fact.values.is_empty() { + return fact + .values + .iter() + .filter_map(|value| { + let number = value.number?; + let signature = value.aggregate_signature()?; + Some(AggregateValue { number, signature }) + }) + .collect(); + } + fact.value_number .filter(|value| value.is_finite()) + .map(|number| AggregateValue { + number, + signature: "legacy:number".to_string(), + }) + .into_iter() .collect() } @@ -2641,6 +2680,105 @@ mod tests { assert_eq!(avg.as_deref(), Some("17.17")); } + #[test] + fn planned_synthesis_ignores_ordinals_and_dates_when_summing() { + let mut packet = fact_packet( + "pkt_date", + "ref_date", + Some("s1"), + Some(10), + Some("1"), + Some(1.0), + "1. appointment was on 2026-06-30", + MemoryLane::Episodic, + ); + packet.fact_rows[0].values = vec![ + EvidenceValueMention { + text: "1".to_string(), + number: None, + kind: EvidenceValueKind::Ordinal, + unit: None, + }, + EvidenceValueMention { + text: "2026-06-30".to_string(), + number: None, + kind: EvidenceValueKind::Date, + unit: None, + }, + ]; + + let answer = + synthesize_planned_answer(&OpPlan::for_op(QueryOp::AggregateSum, "test"), &[packet]); + + assert_eq!(answer, None); + } + + #[test] + fn planned_synthesis_uses_multiple_grounded_values_with_one_unit() { + let mut packet = fact_packet( + "pkt_duration", + "ref_duration", + Some("s1"), + Some(10), + Some("5 days"), + Some(5.0), + "stayed 5 days, then extended by 3 days", + MemoryLane::Episodic, + ); + packet.fact_rows[0].values = vec![ + EvidenceValueMention { + text: "5 days".to_string(), + number: Some(5.0), + kind: EvidenceValueKind::Duration, + unit: Some("day".to_string()), + }, + EvidenceValueMention { + text: "3 days".to_string(), + number: Some(3.0), + kind: EvidenceValueKind::Duration, + unit: Some("day".to_string()), + }, + ]; + + let answer = + synthesize_planned_answer(&OpPlan::for_op(QueryOp::AggregateSum, "test"), &[packet]); + + assert_eq!(answer.as_deref(), Some("8")); + } + + #[test] + fn planned_synthesis_abstains_on_mixed_value_units() { + let mut packet = fact_packet( + "pkt_mixed", + "ref_mixed", + Some("s1"), + Some(10), + Some("$40"), + Some(40.0), + "spent $40 and waited 3 days", + MemoryLane::Episodic, + ); + packet.fact_rows[0].values = vec![ + EvidenceValueMention { + text: "$40".to_string(), + number: Some(40.0), + kind: EvidenceValueKind::Currency, + unit: Some("$".to_string()), + }, + EvidenceValueMention { + text: "3 days".to_string(), + number: Some(3.0), + kind: EvidenceValueKind::Duration, + unit: Some("day".to_string()), + }, + ]; + + let answer = + synthesize_planned_answer(&OpPlan::for_op(QueryOp::AggregateSum, "test"), &[packet]); + + assert_eq!(answer, None); + } + #[test] fn planned_synthesis_orders_fact_rows_by_timestamp() { let packets = vec![ @@ -2927,6 +3065,17 @@ mod tests { ordinal: 0, value_text: value_text.map(str::to_string), value_number, + values: value_text + .zip(value_number) + .map(|(text, number)| { + vec![EvidenceValueMention { + text: text.to_string(), + number: Some(number), + kind: EvidenceValueKind::Number, + unit: None, + }] + }) + .unwrap_or_default(), text: text.to_string(), }], timeline: timestamp -- 2.51.2