diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 8f96621..85f5284 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -21,7 +21,7 @@ {"_type":"issue","id":"klbr-f0p","title":"Remove tools lane and refactor router to 2-class setup","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T01:06:06Z","created_by":"dawn","updated_at":"2026-06-27T01:06:10Z","closed_at":"2026-06-27T01:06:10Z","close_reason":"Closed","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.18","title":"Keep packet rerank documents under local reranker batch limits","description":"Current full-profile run for LongMemEval-S question 51a45a95 with --rerank-packets produced packet_rerank.status=error from localhost:8003: input 1184 tokens is too large for physical batch size 1024. Packet rerank documents need a hard budget that samples every body without exceeding the local bge-reranker/llama.cpp batch/context limits, and traces should keep the service error visible.","status":"closed","priority":2,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T23:25:26Z","created_by":"dawn","updated_at":"2026-06-26T23:27:53Z","started_at":"2026-06-26T23:25:36Z","closed_at":"2026-06-26T23:27:53Z","close_reason":"Bounded packet rerank documents, added many-body regression, verified klbr-core tests and 51a45a95 packet rerank smoke now returns scores and Target.","labels":["architecture","benchmarks","memory","rerank"],"dependencies":[{"issue_id":"klbr-wmz.18","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-27T02:25:26Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.19","title":"Stabilize reader answers when evidence packets contain the gold answer","description":"Current source retrieves the answer-bearing ref for LongMemEval-S question 51a45a95: answer_bearing_ref_in_context=true, gold_session_plus_neighbor_in_context=true, but a full-profile non-rerank QA run answered 'I don't know' for gold answer Target. A rerank attempt still failed to run due packet size, and that run answered 'Your email inbox'. Treat this as reader/evidence interpretation work, not retrieval recall: add a focused regression or reader/answer-verifier change so where/action questions prefer supported merchant/store/venue evidence over discovery-channel text when the gold answer is in packet context.","status":"closed","priority":2,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T23:25:26Z","created_by":"dawn","updated_at":"2026-06-26T23:33:01Z","started_at":"2026-06-26T23:30:28Z","closed_at":"2026-06-26T23:33:01Z","close_reason":"Centralized reader prompt, made packet-local adjacent bodies explicit evidence windows, added prompt contract test, and verified 51a45a95 non-rerank now answers Target with answer-bearing refs in context.","labels":["architecture","benchmarks","memory","reader"],"dependencies":[{"issue_id":"klbr-wmz.19","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-27T02:25:26Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-1yn","title":"Run real official LongMemEval QA evaluator","description":"The benchmark harness now captures official evaluator stdout/stderr/status and parses official_eval.json when --official-eval-cmd writes it, but this repo does not include the upstream LongMemEval evaluator environment/command. Provide or install the official evaluator, run a small checked-in/local LongMemEval-S QA smoke with answer generation, and record the resulting official metrics in the run artifacts/docs.","notes":"Blocked on external upstream LongMemEval evaluator command/environment. The repo harness supports --official-eval-cmd and parses official_eval.json, but cannot run a real official QA score until that evaluator is provided or installed.\nunblocked locally: scripts/evaluate_qa_local.py added — reimplementation of official evaluate_qa.py using the antigravity proxy at http://127.0.0.1:8000/v1 as judge. no OPENAI_API_KEY needed. nix-shell -p python3Packages.openai provides the dep. see benchmarks/README.md 'Official QA Evaluation (local judge)' section for full --official-eval-cmd invocation.","status":"blocked","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T22:53:31Z","created_by":"dawn","updated_at":"2026-06-27T15:53:42Z","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-1yn","title":"Run real official LongMemEval QA evaluator","description":"The benchmark harness now captures official evaluator stdout/stderr/status and parses official_eval.json when --official-eval-cmd writes it, but this repo does not include the upstream LongMemEval evaluator environment/command. Provide or install the official evaluator, run a small checked-in/local LongMemEval-S QA smoke with answer generation, and record the resulting official metrics in the run artifacts/docs.","notes":"Blocked on external upstream LongMemEval evaluator command/environment. The repo harness supports --official-eval-cmd and parses official_eval.json, but cannot run a real official QA score until that evaluator is provided or installed.\nunblocked locally: scripts/evaluate_qa_local.py added — reimplementation of official evaluate_qa.py using the antigravity proxy at http://127.0.0.1:8000/v1 as judge. no OPENAI_API_KEY needed. nix-shell -p python3Packages.openai provides the dep. see benchmarks/README.md 'Official QA Evaluation (local judge)' section for full --official-eval-cmd invocation.","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T22:53:31Z","created_by":"dawn","updated_at":"2026-06-27T15:55:30Z","closed_at":"2026-06-27T15:55:30Z","close_reason":"local judge available via scripts/evaluate_qa_local.py — uses antigravity proxy at 127.0.0.1:8000/v1, no OPENAI_API_KEY needed. smoke tested. see benchmarks/README.md for --official-eval-cmd usage.","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-d67","title":"document klbr.kdl and bench.kdl secrets and usage","description":"add documentation to AGENTS.md specifying that klbr.kdl and bench.kdl are config files containing potential secrets, used by klbr and the bench suite respectively","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:40:37Z","created_by":"dawn","updated_at":"2026-06-26T22:41:08Z","started_at":"2026-06-26T22:40:43Z","closed_at":"2026-06-26T22:41:08Z","close_reason":"documented klbr.kdl and bench.kdl config usage and secrets under the config.rs section in AGENTS.md","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-djb","title":"update agents.md to reflect latest code state, get rid of anything stale - dont touch rtk / beads instructions","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:27:02Z","created_by":"dawn","updated_at":"2026-06-26T22:31:09Z","started_at":"2026-06-26T22:28:05Z","closed_at":"2026-06-26T22:31:09Z","close_reason":"updated AGENTS.md to match the latest codebase architecture (KDL configs, models.rs, memory lanes, router, evidence, garden, twilight discord)","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.15","title":"Rerank answer-bearing evidence packets instead of raw chunks","description":"docs/evidence-selection.md recommends reranking EvidencePacket objects, not raw chunks or whole sessions. Chunk-level reranking can prefer the query-adjacent setup chunk and still miss the answer-bearing neighbor; whole-session reranking is too broad. Packet reranking should judge whether a budgeted evidence packet contains enough support to answer.","design":"Treat reranking as a tie-breaker until packet-level fixtures show lift. The reranker prompt/input should ask for answer-bearing evidence, not generic relevance.","acceptance_criteria":"The reranker path can score packet bodies that include anchor, local window, episode gist, refs, and source signals; exact-ref packets cannot be suppressed by reranker score alone; traces log pre-rerank and post-rerank packet order; reranker thresholds operate on packets and cannot drop all strong evidence for a gold session without a traceable reason; tests cover a packet that beats its raw anchor chunk because the neighbor contains the answer.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:44Z","created_by":"dawn","updated_at":"2026-06-26T23:19:52Z","started_at":"2026-06-26T23:10:05Z","closed_at":"2026-06-26T23:19:52Z","close_reason":"Implemented packet-level reranking with bounded packet documents, exact-ref protection, traceable threshold fallback, bench flag wiring, tests, and live smoke.","labels":["architecture","context","memory","rerank","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:44Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:39Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} diff --git a/klbr-bench/src/main.rs b/klbr-bench/src/main.rs index d0d57e5..aafbaaf 100644 --- a/klbr-bench/src/main.rs +++ b/klbr-bench/src/main.rs @@ -276,13 +276,32 @@ async fn main() -> Result<()> { longmemeval::run_command(&args).await } "longmem" => { - if args.len() != 5 && args.len() != 6 { + // positional: dataset config outdir [llm_url] + // optional flags: --official-eval-cmd + let official_eval_cmd = args + .windows(2) + .find(|w| w[0] == "--official-eval-cmd") + .map(|w| w[1].clone()); + let mut positional: Vec<&str> = Vec::new(); + let mut skip_next = false; + for a in args.iter().skip(2) { + if skip_next { + skip_next = false; + continue; + } + if a.starts_with("--") { + skip_next = true; + continue; + } + positional.push(a.as_str()); + } + if positional.len() < 3 || positional.len() > 4 { bail!( - "usage: cargo run -p klbr-bench -- longmem [llm_url]" + "usage: cargo run -p klbr-bench -- longmem [llm_url] [--official-eval-cmd ]" ); } - let llm_url = args.get(5).cloned(); - run_longmem_command(&args[2], &args[3], &args[4], llm_url).await + let llm_url = positional.get(3).map(|s| s.to_string()); + run_longmem_command(positional[0], positional[1], positional[2], llm_url, official_eval_cmd).await } "longmem-retrieval" => { if args.len() != 5 && args.len() != 7 { @@ -5339,6 +5358,7 @@ async fn run_longmem_command( config_path: &str, output_dir: &str, llm_url_override: Option, + official_eval_cmd: Option, ) -> Result<()> { let dataset_path = Path::new(dataset_path); let config_path = Path::new(config_path); @@ -5872,6 +5892,33 @@ async fn run_longmem_command( println!("Average Hot Query Latency: {:.2} ms", avg_hot_query); println!("--------------------------------------------------------------------------------"); + if let Some(cmd) = official_eval_cmd { + let hypothesis = output_dir.join("longmem_hypothesis.jsonl"); + let out_json = output_dir.join("official_eval.json"); + let cmd_expanded = cmd + .replace("{hypothesis}", &hypothesis.to_string_lossy()) + .replace("{out}", &out_json.to_string_lossy()) + .replace("{run_dir}", &output_dir.to_string_lossy()); + eprintln!("[official-eval] running: {}", cmd_expanded); + let status = std::process::Command::new("sh") + .arg("-c") + .arg(&cmd_expanded) + .status() + .with_context(|| format!("failed to run official eval command: {cmd_expanded}"))?; + if !status.success() { + eprintln!("[official-eval] command exited with status: {:?}", status.code()); + } else { + eprintln!("[official-eval] done — results at {}", out_json.display()); + if let Ok(bytes) = fs::read(&out_json) { + if let Ok(val) = serde_json::from_slice::(&bytes) { + if let Some(acc) = val.get("accuracy") { + println!("Official QA accuracy: {}", acc); + } + } + } + } + } + Ok(()) }