From e65a301d7c64c40565ce5c5c570eecd77aa455b7 Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Tue, 30 Jun 2026 20:02:45 +0300 Subject: [PATCH] prioritize anchor bodies in evidence packets --- .beads/interactions.jsonl | 1 + .beads/issues.jsonl | 1 + klbr-core/src/evidence.rs | 210 +++++++++++++++++++++++++++++++++++--- 3 files changed, 198 insertions(+), 14 deletions(-) diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index f7e4697..d930b0e 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -73,3 +73,4 @@ {"id":"int-aa7c2ef1","kind":"field_change","created_at":"2026-06-30T12:55:18.069202784Z","actor":"dawn","issue_id":"klbr-54l.5","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"whenloss diagnostic summaries are emitted into trace_summary rows and docs frame LongMemEval passive QA as best-effort passive recall"}} {"id":"int-c34adfde","kind":"field_change","created_at":"2026-06-30T12:55:18.472171415Z","actor":"dawn","issue_id":"klbr-54l.6","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately"}} {"id":"int-e9f364ec","kind":"field_change","created_at":"2026-06-30T12:55:18.89582463Z","actor":"dawn","issue_id":"klbr-54l","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"implemented trace-diff, planner hints/parser hardening, session/packet/render fixes, writer/whenloss diagnostics, docs, and final same-seed retrieval smoke; remaining passive misses are documented as follow-up opportunities"}} +{"id":"int-788f0749","kind":"field_change","created_at":"2026-06-30T17:01:55.237649146Z","actor":"dawn","issue_id":"klbr-e28","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed: diagnosed renderer truncation of anchor/local answer bodies, landed prioritized packet rendering, added regression tests, and verified same-seed official QA improved from broken 0.32 and old 0.44 to 0.56."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 1a4a438..4b87802 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -1,3 +1,4 @@ +{"_type":"issue","id":"klbr-e28","title":"Investigate LongMemEval same-seed QA regression after context-survival fixes","description":"Same 25-question LongMemEval-S stratified sample improved retrieval/context survival but official QA accuracy fell from 44% to 32%. Need isolate whether reader prompt, deterministic synthesis, op planning, packet formatting, or context packing causes answerable rows to produce I-don't-know / bad arithmetic/date answers despite answer-bearing refs in context.","acceptance_criteria":"Root cause is documented from trace evidence for flipped rows; either a narrow fix lands with regression coverage/bench evidence, or follow-up beads capture remaining work precisely.","notes":"Root cause: same-seed QA dropped because packet selection/retrieval found the right sessions, but render_evidence_packet emitted facts/timeline and earlier chronological bodies before the anchor/local answer bodies. With per-packet context caps, literal answer-bearing text was truncated even when selected refs were correct. Narrow fix: render anchor body + adjacent local window first, then prioritized facts/timeline, then remaining bodies. Evidence: old baseline 2026-06-30_qa_sample25_seed4937553249516211047_current official accuracy 0.44; broken postfix 2026-06-30_qa_sample25_seed4937553249516211047_postfix_qa official accuracy 0.32; fixed anchor_window_qa official accuracy 0.56 with RecallAny@5 0.96, RecallAll@5 0.84, answer_bearing_ref_selected 0.80, rendered/in_context 0.72. Regression coverage added for late anchor and adjacent anchor bodies; cargo fmt --check, cargo test -p klbr-core, cargo test -p klbr-bench passed.","status":"closed","priority":1,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T16:19:09Z","created_by":"dawn","updated_at":"2026-06-30T17:01:55Z","started_at":"2026-06-30T16:19:13Z","closed_at":"2026-06-30T17:01:55Z","close_reason":"Completed: diagnosed renderer truncation of anchor/local answer bodies, landed prioritized packet rendering, added regression tests, and verified same-seed official QA improved from broken 0.32 and old 0.44 to 0.56.","labels":["bench","longmemeval","memory","qa"],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-54l.3","title":"Stabilize op planner labels for LongMemEval update and count cases","description":"The same-seed regression changed planner behavior heavily: old op_plan_counts had update_resolution=10 and lookup=3, while current has update_resolution=0 and lookup=15. That can disable collect/update behavior and shrink packet budgets even when retrieval finds the right area. Harden the structured planner path and fallback behavior for aggregate_count, aggregate_sum, aggregate_avg, order_or_rank, update_resolution, and preference_recommendation without reintroducing hidden English cue lists as the main policy.","acceptance_criteria":"Known klbr-54l rows get stable op labels across reruns or model-parser fallback; update_resolution no longer collapses to lookup on the same 25-row sample; tests cover malformed structured planner output and paraphrased update/count questions; implementation remains language-agnostic rather than relying on hardcoded English cue lists.","status":"closed","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:17Z","created_by":"dawn","updated_at":"2026-06-30T12:55:17Z","closed_at":"2026-06-30T12:55:17Z","close_reason":"planner parser accepts tolerant op labels and LongMemEval category hints cover update/temporal/preference cases; tests and same-seed sample show update/count labels restored","labels":["bench","longmemeval","memory","planner"],"dependencies":[{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:17Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.2","title":"Expand session-first episode anchors into supporting turn-window packets","description":"The klbr-54l trace research found a concrete failure class: stage_one finds the right session, but session-first emits thin episode_bridge packets from episode/card candidates and the final context drops the actual answer-bearing turn span. For 6b168ec8, the right session is found and a late episode bridge contains the three-bikes evidence, but the rendered context spends budget on irrelevant fts turn windows. In session-first mode, strong episode/session anchors should enrich or produce supporting turn_window packets from the same session's best seed refs instead of relying only on expand_edges from the episode card.","acceptance_criteria":"A production-pipeline fixture covers the candidate-session-hit / thin-episode-bridge-dropped failure; when the answer session is a strong stage_one group, packet planning includes a non-thin supporting turn_window from that session; 6b168ec8 no longer loses the three-bikes evidence before reader time without increasing global read budget.","status":"closed","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:10Z","created_by":"dawn","updated_at":"2026-06-30T12:55:17Z","closed_at":"2026-06-30T12:55:17Z","close_reason":"session-first representatives now preserve real turn chunks and turn-window expansion keeps the anchor ref; 6b168ec8 retrieval smoke has answer ref rendered/in context","labels":["bench","evidence-packets","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-54l.1","title":"Trace-diff old-pass/current-fail LongMemEval passive QA rows","description":"Build a reusable trace-diff for the same-seed LongMemEval-S 25-row regression in klbr-54l before changing ranking or planner code. Compare baseline benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z against current benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current for old-pass/current-fail ids: 6b168ec8, c14c00dd, b5ef892d, 46a3abf7, 720133ac, gpt4_385a5000, dad224aa. Report per question: op_plan, stage_one answer-session rank, candidate vs packet recall, packet order/kind/session, answer-bearing selected/rendered/in-context, answer_value_visible, critical_fact_row_visible, packet/context tokens, omissions, and reader hypothesis.","acceptance_criteria":"A command, test, or bench report emits a per-question diff for the listed ids; 6b168ec8 clearly shows the candidate-session-hit but answer-packet-not-rendered class; output is durable enough to rerun during klbr-54l without hand-inspecting trace.jsonl.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:02Z","created_by":"dawn","updated_at":"2026-06-30T12:55:16Z","started_at":"2026-06-30T11:33:29Z","closed_at":"2026-06-30T12:55:16Z","close_reason":"implemented trace_summary.jsonl/md and trace-diff command; verified against old/current/fixed LongMemEval-S regression rows","labels":["bench","longmemeval","memory","passive-recall"],"dependencies":[{"issue_id":"klbr-54l.1","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:02Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":6,"comment_count":0} diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs index 5e02318..5058aac 100644 --- a/klbr-core/src/evidence.rs +++ b/klbr-core/src/evidence.rs @@ -505,12 +505,36 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str xml_escape(packet.session_id.as_deref().unwrap_or("")), packet.estimated_tokens.min(max_tokens), ); + let closing_chars = " \n".len(); + let anchor_body_index = packet + .refs + .iter() + .position(|ref_id| ref_id == &packet.anchor_ref); + let mut priority_body_indices = Vec::new(); + for index in prioritized_indices(packet.bodies.len(), anchor_body_index) + .into_iter() + .take(3) + { + if push_packet_body(&mut out, packet, index, max_chars, closing_chars) { + priority_body_indices.push(index); + } + } if !packet.fact_rows.is_empty() { let open = " \n"; let close = " \n"; if out.chars().count() + open.len() + close.len() + " \n".len() <= max_chars { out.push_str(open); - for fact in packet.fact_rows.iter().take(8) { + for fact in prioritized_indices( + packet.fact_rows.len(), + packet + .fact_rows + .iter() + .position(|fact| fact.ref_id == packet.anchor_ref), + ) + .into_iter() + .take(8) + .filter_map(|index| packet.fact_rows.get(index)) + { let mut attrs = format!( "ref=\"{}\" entity_type=\"{}\" ord=\"{}\" session=\"{}\"", xml_escape(&fact.ref_id), @@ -547,7 +571,17 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str let close = " \n"; if out.chars().count() + open.len() + close.len() + " \n".len() <= max_chars { out.push_str(open); - for event in packet.timeline.iter().take(8) { + for event in prioritized_indices( + packet.timeline.len(), + packet + .timeline + .iter() + .position(|event| event.ref_id == packet.anchor_ref), + ) + .into_iter() + .take(8) + .filter_map(|index| packet.timeline.get(index)) + { let line = format!( " {}\n", xml_escape(&event.ref_id), @@ -566,24 +600,61 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str out.push_str(close); } } - let closing_chars = " \n".len(); - for (index, body) in packet.bodies.iter().enumerate() { - let ref_id = packet.refs.get(index).unwrap_or(&packet.anchor_ref); - let prefix = format!(" ", xml_escape(ref_id)); - let suffix = "\n"; - let remaining_chars = max_chars.saturating_sub(out.chars().count() + closing_chars); - if remaining_chars <= prefix.chars().count() + suffix.len() { + for index in prioritized_indices(packet.bodies.len(), anchor_body_index) { + if priority_body_indices.contains(&index) { + continue; + } + if !push_packet_body(&mut out, packet, index, max_chars, closing_chars) { break; } - let body_chars = remaining_chars - prefix.chars().count() - suffix.len(); - out.push_str(&prefix); - out.push_str(&xml_escape(&truncate_chars(body, body_chars))); - out.push_str(suffix); } out.push_str(" \n"); out } +fn push_packet_body( + out: &mut String, + packet: &EvidencePacket, + index: usize, + max_chars: usize, + closing_chars: usize, +) -> bool { + let Some(body) = packet.bodies.get(index) else { + return false; + }; + let ref_id = packet.refs.get(index).unwrap_or(&packet.anchor_ref); + let prefix = format!(" ", xml_escape(ref_id)); + let suffix = "\n"; + let remaining_chars = max_chars.saturating_sub(out.chars().count() + closing_chars); + if remaining_chars <= prefix.chars().count() + suffix.len() { + return false; + } + let body_chars = remaining_chars - prefix.chars().count() - suffix.len(); + out.push_str(&prefix); + out.push_str(&xml_escape(&truncate_chars(body, body_chars))); + out.push_str(suffix); + true +} + +fn prioritized_indices(len: usize, priority: Option) -> Vec { + let mut indices = Vec::with_capacity(len); + if let Some(priority) = priority.filter(|priority| *priority < len) { + indices.push(priority); + if priority > 0 { + indices.push(priority - 1); + } + if priority + 1 < len { + indices.push(priority + 1); + } + } + for index in 0..len { + if !indices.contains(&index) { + indices.push(index); + } + } + indices +} + fn evidence_coverage_trace( collect_mode: bool, available_packets: &[EvidencePacket], @@ -1124,7 +1195,7 @@ mod tests { packet.bodies = packet .refs .iter() - .map(|ref_id| format!("{ref_id} {}", "long body text ".repeat(80))) + .map(|ref_id| format!("{ref_id} body")) .collect(); packet.fact_rows = packet .refs @@ -1161,6 +1232,117 @@ mod tests { assert!(rendered.contains("answer-bearing anchor says Sunday mass was on March 19th" + )); + assert!(rendered.chars().count() <= 220 * 4); + } + + #[test] + fn rendered_packet_prioritizes_adjacent_anchor_bodies() { + let mut packet = test_packet_with_refs( + "pkt_adjacent_anchor", + &["ref_before", "ref_anchor", "ref_after", "ref_later"], + "fts", + 1, + Some("s1"), + ); + packet.anchor_ref = "ref_anchor".to_string(); + packet.bodies = vec![ + "previous user turn says Coastal Cleanup was on March 7th".to_string(), + "assistant acknowledges the Coastal Cleanup event".to_string(), + "next assistant detail".to_string(), + "low priority filler ".repeat(160), + ]; + packet.fact_rows = packet + .refs + .iter() + .enumerate() + .map(|(ordinal, ref_id)| EvidenceFactRow { + row_id: format!("fact_{ordinal}"), + ref_id: ref_id.clone(), + entity_type: "turn_chunk".to_string(), + session_id: Some("s1".to_string()), + timestamp: Some(42 + ordinal as i64), + ordinal, + value_text: None, + value_number: None, + text: packet.bodies[ordinal].clone(), + }) + .collect(); + + let rendered = render_evidence_packet(&packet, 180); + + assert!(rendered.contains( + "assistant acknowledges the Coastal Cleanup event" + )); + assert!(rendered.contains( + "previous user turn says Coastal Cleanup was on March 7th" + )); + assert!(rendered.chars().count() <= 180 * 4); + } + #[test] fn packet_fusion_merges_channel_signals() { let mut packet = test_packet("pkt_same", "ref_same", "fts", 1, Some("s1")); -- 2.51.2