From e65a301d7c64c40565ce5c5c570eecd77aa455b7 Mon Sep 17 00:00:00 2001
From: dawn <90008@klbr.net>
Date: Tue, 30 Jun 2026 20:02:45 +0300
Subject: [PATCH] prioritize anchor bodies in evidence packets
---
.beads/interactions.jsonl | 1 +
.beads/issues.jsonl | 1 +
klbr-core/src/evidence.rs | 210 +++++++++++++++++++++++++++++++++++---
3 files changed, 198 insertions(+), 14 deletions(-)
diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl
index f7e4697..d930b0e 100644
--- a/.beads/interactions.jsonl
+++ b/.beads/interactions.jsonl
@@ -73,3 +73,4 @@
{"id":"int-aa7c2ef1","kind":"field_change","created_at":"2026-06-30T12:55:18.069202784Z","actor":"dawn","issue_id":"klbr-54l.5","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"whenloss diagnostic summaries are emitted into trace_summary rows and docs frame LongMemEval passive QA as best-effort passive recall"}}
{"id":"int-c34adfde","kind":"field_change","created_at":"2026-06-30T12:55:18.472171415Z","actor":"dawn","issue_id":"klbr-54l.6","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"trace rows include writer_retention probes for answer refs and episode cards; same-seed sample reports writer vs retrieval misses separately"}}
{"id":"int-e9f364ec","kind":"field_change","created_at":"2026-06-30T12:55:18.89582463Z","actor":"dawn","issue_id":"klbr-54l","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"implemented trace-diff, planner hints/parser hardening, session/packet/render fixes, writer/whenloss diagnostics, docs, and final same-seed retrieval smoke; remaining passive misses are documented as follow-up opportunities"}}
+{"id":"int-788f0749","kind":"field_change","created_at":"2026-06-30T17:01:55.237649146Z","actor":"dawn","issue_id":"klbr-e28","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed: diagnosed renderer truncation of anchor/local answer bodies, landed prioritized packet rendering, added regression tests, and verified same-seed official QA improved from broken 0.32 and old 0.44 to 0.56."}}
diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl
index 1a4a438..4b87802 100644
--- a/.beads/issues.jsonl
+++ b/.beads/issues.jsonl
@@ -1,3 +1,4 @@
+{"_type":"issue","id":"klbr-e28","title":"Investigate LongMemEval same-seed QA regression after context-survival fixes","description":"Same 25-question LongMemEval-S stratified sample improved retrieval/context survival but official QA accuracy fell from 44% to 32%. Need isolate whether reader prompt, deterministic synthesis, op planning, packet formatting, or context packing causes answerable rows to produce I-don't-know / bad arithmetic/date answers despite answer-bearing refs in context.","acceptance_criteria":"Root cause is documented from trace evidence for flipped rows; either a narrow fix lands with regression coverage/bench evidence, or follow-up beads capture remaining work precisely.","notes":"Root cause: same-seed QA dropped because packet selection/retrieval found the right sessions, but render_evidence_packet emitted facts/timeline and earlier chronological bodies before the anchor/local answer bodies. With per-packet context caps, literal answer-bearing text was truncated even when selected refs were correct. Narrow fix: render anchor body + adjacent local window first, then prioritized facts/timeline, then remaining bodies. Evidence: old baseline 2026-06-30_qa_sample25_seed4937553249516211047_current official accuracy 0.44; broken postfix 2026-06-30_qa_sample25_seed4937553249516211047_postfix_qa official accuracy 0.32; fixed anchor_window_qa official accuracy 0.56 with RecallAny@5 0.96, RecallAll@5 0.84, answer_bearing_ref_selected 0.80, rendered/in_context 0.72. Regression coverage added for late anchor and adjacent anchor bodies; cargo fmt --check, cargo test -p klbr-core, cargo test -p klbr-bench passed.","status":"closed","priority":1,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T16:19:09Z","created_by":"dawn","updated_at":"2026-06-30T17:01:55Z","started_at":"2026-06-30T16:19:13Z","closed_at":"2026-06-30T17:01:55Z","close_reason":"Completed: diagnosed renderer truncation of anchor/local answer bodies, landed prioritized packet rendering, added regression tests, and verified same-seed official QA improved from broken 0.32 and old 0.44 to 0.56.","labels":["bench","longmemeval","memory","qa"],"dependency_count":0,"dependent_count":0,"comment_count":0}
{"_type":"issue","id":"klbr-54l.3","title":"Stabilize op planner labels for LongMemEval update and count cases","description":"The same-seed regression changed planner behavior heavily: old op_plan_counts had update_resolution=10 and lookup=3, while current has update_resolution=0 and lookup=15. That can disable collect/update behavior and shrink packet budgets even when retrieval finds the right area. Harden the structured planner path and fallback behavior for aggregate_count, aggregate_sum, aggregate_avg, order_or_rank, update_resolution, and preference_recommendation without reintroducing hidden English cue lists as the main policy.","acceptance_criteria":"Known klbr-54l rows get stable op labels across reruns or model-parser fallback; update_resolution no longer collapses to lookup on the same 25-row sample; tests cover malformed structured planner output and paraphrased update/count questions; implementation remains language-agnostic rather than relying on hardcoded English cue lists.","status":"closed","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:17Z","created_by":"dawn","updated_at":"2026-06-30T12:55:17Z","closed_at":"2026-06-30T12:55:17Z","close_reason":"planner parser accepts tolerant op labels and LongMemEval category hints cover update/temporal/preference cases; tests and same-seed sample show update/count labels restored","labels":["bench","longmemeval","memory","planner"],"dependencies":[{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:17Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.3","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0}
{"_type":"issue","id":"klbr-54l.2","title":"Expand session-first episode anchors into supporting turn-window packets","description":"The klbr-54l trace research found a concrete failure class: stage_one finds the right session, but session-first emits thin episode_bridge packets from episode/card candidates and the final context drops the actual answer-bearing turn span. For 6b168ec8, the right session is found and a late episode bridge contains the three-bikes evidence, but the rendered context spends budget on irrelevant fts turn windows. In session-first mode, strong episode/session anchors should enrich or produce supporting turn_window packets from the same session's best seed refs instead of relying only on expand_edges from the episode card.","acceptance_criteria":"A production-pipeline fixture covers the candidate-session-hit / thin-episode-bridge-dropped failure; when the answer session is a strong stage_one group, packet planning includes a non-thin supporting turn_window from that session; 6b168ec8 no longer loses the three-bikes evidence before reader time without increasing global read budget.","status":"closed","priority":1,"issue_type":"bug","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:10Z","created_by":"dawn","updated_at":"2026-06-30T12:55:17Z","closed_at":"2026-06-30T12:55:17Z","close_reason":"session-first representatives now preserve real turn chunks and turn-window expansion keeps the anchor ref; 6b168ec8 retrieval smoke has answer ref rendered/in context","labels":["bench","evidence-packets","longmemeval","memory"],"dependencies":[{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:10Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-54l.2","depends_on_id":"klbr-54l.1","type":"blocks","created_at":"2026-06-30T14:14:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0}
{"_type":"issue","id":"klbr-54l.1","title":"Trace-diff old-pass/current-fail LongMemEval passive QA rows","description":"Build a reusable trace-diff for the same-seed LongMemEval-S 25-row regression in klbr-54l before changing ranking or planner code. Compare baseline benchmarks/runs/longmemeval-s/klbr-full/2026-06-28_210327.693783Z against current benchmarks/runs/longmemeval-s/klbr-full/2026-06-30_qa_sample25_seed4937553249516211047_current for old-pass/current-fail ids: 6b168ec8, c14c00dd, b5ef892d, 46a3abf7, 720133ac, gpt4_385a5000, dad224aa. Report per question: op_plan, stage_one answer-session rank, candidate vs packet recall, packet order/kind/session, answer-bearing selected/rendered/in-context, answer_value_visible, critical_fact_row_visible, packet/context tokens, omissions, and reader hypothesis.","acceptance_criteria":"A command, test, or bench report emits a per-question diff for the listed ids; 6b168ec8 clearly shows the candidate-session-hit but answer-packet-not-rendered class; output is durable enough to rerun during klbr-54l without hand-inspecting trace.jsonl.","status":"closed","priority":1,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-30T11:13:02Z","created_by":"dawn","updated_at":"2026-06-30T12:55:16Z","started_at":"2026-06-30T11:33:29Z","closed_at":"2026-06-30T12:55:16Z","close_reason":"implemented trace_summary.jsonl/md and trace-diff command; verified against old/current/fixed LongMemEval-S regression rows","labels":["bench","longmemeval","memory","passive-recall"],"dependencies":[{"issue_id":"klbr-54l.1","depends_on_id":"klbr-54l","type":"parent-child","created_at":"2026-06-30T14:13:02Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":6,"comment_count":0}
diff --git a/klbr-core/src/evidence.rs b/klbr-core/src/evidence.rs
index 5e02318..5058aac 100644
--- a/klbr-core/src/evidence.rs
+++ b/klbr-core/src/evidence.rs
@@ -505,12 +505,36 @@ pub fn render_evidence_packet(packet: &EvidencePacket, max_tokens: usize) -> Str
xml_escape(packet.session_id.as_deref().unwrap_or("")),
packet.estimated_tokens.min(max_tokens),
);
+ let closing_chars = " \n".len();
+ let anchor_body_index = packet
+ .refs
+ .iter()
+ .position(|ref_id| ref_id == &packet.anchor_ref);
+ let mut priority_body_indices = Vec::new();
+ for index in prioritized_indices(packet.bodies.len(), anchor_body_index)
+ .into_iter()
+ .take(3)
+ {
+ if push_packet_body(&mut out, packet, index, max_chars, closing_chars) {
+ priority_body_indices.push(index);
+ }
+ }
if !packet.fact_rows.is_empty() {
let open = "