diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index 93446e2..a3acd8c 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -49,3 +49,5 @@ {"id":"int-401d0a5b","kind":"field_change","created_at":"2026-06-28T11:18:16.246397759Z","actor":"dawn","issue_id":"klbr-a9b.3","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented replace_symbol_body, insert_before_symbol, and insert_after_symbol using generic tree-sitter symbol ranges."}} {"id":"int-21f83166","kind":"field_change","created_at":"2026-06-28T11:20:24.873425877Z","actor":"dawn","issue_id":"klbr-a9b.7","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Added code-intel docs, registration coverage, and recorded focused verification commands."}} {"id":"int-c8b11ed4","kind":"field_change","created_at":"2026-06-28T11:20:38.81144388Z","actor":"dawn","issue_id":"klbr-a9b","extra":{"field":"status","new_value":"closed","old_value":"open","reason":"Completed first-pass generic tree-sitter file-local code retrieval, navigation, symbolic edits, tests, and docs."}} +{"id":"int-ca7d767e","kind":"field_change","created_at":"2026-06-28T16:16:07.053939933Z","actor":"dawn","issue_id":"klbr-8dq","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed: LongMemEval pipeline run now defaults --out under benchmarks/runs."}} +{"id":"int-fb2d145f","kind":"field_change","created_at":"2026-06-28T16:16:07.11528074Z","actor":"dawn","issue_id":"klbr-oba","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Completed: added reproducible random LongMemEval sampling with question_type stratification."}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 6689084..1789c2c 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -22,6 +22,8 @@ {"_type":"issue","id":"klbr-wmz.2","title":"Route runtime passive recall through the canonical memory pipeline","description":"Bench retrieval now goes through MemoryPipeline, but runtime passive recall in klbr-core/src/agent.rs still embeds the prompt, calls MemoryStore::get_searchable, runs retrieval::retrieve_exact over legacy memories, then injects Context memory packets. That bypasses fts, exact refs, markdown notes, graph expansion, and the canonical lane/lifecycle path the docs describe.","design":"Avoid duplicating retrieval logic in agent.rs. Either make MemoryPipeline usable by AgentRuntime or extract a shared retrieval facade that both MemoryPipeline and runtime passive recall call.","acceptance_criteria":"Runtime passive recall uses the same lane-aware canonical retrieval and context packet assembly policy as the production pipeline; recalled packets can include fts/exact/dense/graph candidates from refs and markdown notes; archived/tombstoned/suppressed refs do not leak; klbr-core/src/instructions.md matches the actual memory packet format; tests or a focused integration fixture cover passive recall from a markdown note and from an explicit ref.","notes":"Runtime passive recall now calls MemoryPipeline::retrieve_evidence and injects shared EvidencePacket XML via Context::inject_evidence_packets; no-model runtime packet fixture covers turn-window expansion. Remaining acceptance is blocked on klbr-wmz.1 because dense search still starts from legacy memory rows before ref mapping.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:19Z","created_by":"dawn","updated_at":"2026-06-26T19:44:55Z","started_at":"2026-06-26T19:37:32Z","closed_at":"2026-06-26T19:44:55Z","close_reason":"Completed: runtime passive recall now uses MemoryPipeline/EvidencePlanner packets, instructions document current packet XML, and fixtures cover turn-window recall plus explicit markdown refs; dense canonical dependency completed in klbr-wmz.1.","labels":["architecture","memory","retrieval","runtime"],"dependencies":[{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:19Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.2","depends_on_id":"klbr-wmz.1","type":"blocks","created_at":"2026-06-26T22:37:53Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.1","title":"Move dense retrieval onto canonical refs and embedding_items","description":"Current source still has dense retrieval on the legacy memory-row surface: klbr-core/src/pipeline.rs search_dense calls MemoryStore::get_searchable and retrieval::retrieve_exact, while fts/exact retrieval uses canonical refs and promptable_text. The schema already has refs, promptable_text, markdown_note_chunks, and embedding_items, so dense retrieval should not be the odd path out.","design":"Prefer a ref-native embedding index backed by embedding_items. Backfill embeddings from promptable_text, keep memory-id aliases as compatibility aliases, and make klbr/full versus dense-only profiles exercise the same canonical identity layer as fts and exact retrieval.","acceptance_criteria":"Dense candidate generation works over canonical ref ids for memories, turn chunks, markdown note chunks, episode notes, profile notes, and procedural notes; lane and lifecycle filtering come from refs/ref_metadata instead of memory tags alone; benchmark traces return canonical refs for dense hits; regression tests cover a markdown-note-only hit and a tombstoned/suppressed ref not leaking through dense search.","status":"closed","priority":1,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T17:53:11Z","created_by":"dawn","updated_at":"2026-06-26T19:43:27Z","started_at":"2026-06-26T19:38:02Z","closed_at":"2026-06-26T19:43:27Z","close_reason":"Completed: dense candidate generation now lazily backfills embedding_items from active promptable refs, scores canonical ref embeddings directly, and tests markdown-note dense hits plus suppressed-ref filtering.","labels":["architecture","memory","refs","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.1","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T20:53:11Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-wmz","title":"Finish memory architecture follow-through","description":"Tracks the remaining memory architecture work identified from docs/memory-arch.md, docs/memory-benches.md, docs/memory-implementation-status.md, and current klbr-core/klbr-bench source. Current status says pipeline, typed memory packets, markdown notes, edge mirroring, lifecycle projection, and benchmark runner integration exist; this epic is for gaps still present in source/docs.","acceptance_criteria":"Close when the child issues are complete, docs/memory-implementation-status.md is updated from current verification, and the architecture docs no longer point at missing or stale follow-up work.","status":"closed","priority":1,"issue_type":"epic","owner":"90008@klbr.net","created_at":"2026-06-26T17:52:54Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","closed_at":"2026-06-26T23:45:35Z","close_reason":"All 19 memory architecture child issues are closed; docs/status were updated from current verification; remaining official evaluator run is tracked separately as external blocked klbr-1yn.","labels":["architecture","memory"],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-oba","title":"add stratified random longmemeval sampling","description":"support reproducible random subset runs for the LongMemEval pipeline, stratified by question_type so small incremental qa runs still cover categories.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T16:09:37Z","created_by":"dawn","updated_at":"2026-06-28T16:16:07Z","started_at":"2026-06-28T16:10:01Z","closed_at":"2026-06-28T16:16:07Z","close_reason":"Completed: added reproducible random LongMemEval sampling with question_type stratification.","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"klbr-8dq","title":"default longmemeval qa output under benchmarks/runs","description":"make the qa bench less annoying to run by defaulting its output directory under benchmarks/runs when --out is not supplied.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T15:55:48Z","created_by":"dawn","updated_at":"2026-06-28T16:16:07Z","started_at":"2026-06-28T15:56:03Z","closed_at":"2026-06-28T16:16:07Z","close_reason":"Completed: LongMemEval pipeline run now defaults --out under benchmarks/runs.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-a9b.7","title":"Document and test tree-sitter code tools","description":"Add focused fixtures, regression tests, and docs for the tree-sitter-first code tools. Tests should cover multiple representative languages through the generic registry, symbol extraction, overview/find/read flows, symbolic edits, terse OK/error behavior, and the explicitly skipped diagnostics/project-wide/LSP scope.","acceptance_criteria":"Core tests cover supported generic file-local behavior and failure modes; docs explain parser/query-bundle availability, the first-pass boundary, and future expansion path; the epic's verification commands are recorded after implementation lands.","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:32Z","created_by":"dawn","updated_at":"2026-06-28T11:20:25Z","started_at":"2026-06-28T11:18:27Z","closed_at":"2026-06-28T11:20:25Z","close_reason":"Added code-intel docs, registration coverage, and recorded focused verification commands.","labels":["code-intel","docs","tests","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:32Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b.2","type":"blocks","created_at":"2026-06-28T13:46:51Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b.3","type":"blocks","created_at":"2026-06-28T13:46:50Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.7","depends_on_id":"klbr-a9b.5","type":"blocks","created_at":"2026-06-28T13:46:58Z","created_by":"dawn","metadata":"{}"}],"dependency_count":3,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-a9b.4","title":"Implement read_symbol and range extraction","description":"Add a tool/path for retrieving the exact source body for a single file-local symbol selected by name path. This is the retrieval step agents should use before symbolic replacement, mirroring Serena's expectation that replace_symbol_body follows a prior body read.","acceptance_criteria":"read_symbol returns the selected symbol body and location for fixture languages covered by the generic extractor; ambiguous or missing symbols return concise errors; returned body range is the same range used by replace_symbol_body.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:31Z","created_by":"dawn","updated_at":"2026-06-28T11:11:25Z","started_at":"2026-06-28T11:09:41Z","closed_at":"2026-06-28T11:11:25Z","close_reason":"Implemented read_symbol using the shared generic tree-sitter symbol matcher and body range extraction.","labels":["code-intel","retrieval","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.4","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:31Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.4","depends_on_id":"klbr-a9b.8","type":"blocks","created_at":"2026-06-28T13:46:50Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-a9b.5","title":"Add best-effort file-local references and declaration lookup","description":"Add file-local references/declaration helpers using tree-sitter tags/locals/name matching. This is explicitly best-effort syntax-derived navigation, not typechecker-accurate LSP behavior. Project-wide reference/declaration resolution remains out of scope for the first implementation.","acceptance_criteria":"For fixture files, the tool can find same-file call/reference captures for a selected symbol and jump from a captured reference to a same-file definition when names are unambiguous; ambiguous or unsupported cases return concise best-effort errors.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-28T10:46:30Z","created_by":"dawn","updated_at":"2026-06-28T11:14:47Z","started_at":"2026-06-28T11:11:41Z","closed_at":"2026-06-28T11:14:47Z","close_reason":"Implemented best-effort same-file find_references and find_definition tools over generic tree-sitter tags.","labels":["code-intel","retrieval","tools","tree-sitter"],"dependencies":[{"issue_id":"klbr-a9b.5","depends_on_id":"klbr-a9b","type":"parent-child","created_at":"2026-06-28T13:46:30Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-a9b.5","depends_on_id":"klbr-a9b.8","type":"blocks","created_at":"2026-06-28T13:46:50Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} diff --git a/benchmarks/README.md b/benchmarks/README.md index eb1ddf1..ad608ec 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -134,16 +134,20 @@ For that comparison, use the report's all-row `RecallAny@5` because MemWeave ave `evaluate_qa.py` that points at the local antigravity proxy instead of OpenAI. Same prompt templates, same positional interface, writes `official_eval.json` which the bench harness reads. -Requires `python3Packages.openai` — run under nix-shell: +Requires `python3Packages.openai`. without `--out`, the pipeline runner writes to +`benchmarks/runs////`. ```bash # Smoke run against the 30-question subset (fast, no OPENAI_API_KEY needed): -cargo run -p klbr-bench -- longmem \ - benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ - benchmarks/inputs/configs/mvp_rerank_support_calibrated.json \ - benchmarks/runs/longmem-eval-smoke \ +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --profile klbr-full \ + --top-k 5 \ + --budget-read 5000 \ + --graph-depth 1 \ --official-eval-cmd \ - "nix-shell -p python3Packages.openai --run \ + "rtk nix-shell -p python3Packages.openai --run \ 'python3 scripts/evaluate_qa_local.py \ \"Gemini 3.5 Flash (Medium)\" \ {hypothesis} \ @@ -152,18 +156,41 @@ cargo run -p klbr-bench -- longmem \ --out {out}'" ``` -For the full longmemeval-S run, swap out `subset30` for `longmemeval_s_cleaned.json` in both -the data path and the `--official-eval-cmd` ref file argument. +For a stratified random incremental QA run over the full dataset, sample by +`question_type` and pin the seed: + +```bash +rtk cargo run -p klbr-bench -- run \ + --suite longmemeval-s \ + --data benchmarks/inputs/datasets/longmemeval_s_cleaned.json \ + --profile klbr-full \ + --top-k 5 \ + --budget-read 5000 \ + --graph-depth 1 \ + --sample 20 \ + --sample-seed 42 \ + --official-eval-cmd \ + "rtk nix-shell -p python3Packages.openai --run \ + 'python3 scripts/evaluate_qa_local.py \ + \"Gemini 3.5 Flash (Medium)\" \ + {hypothesis} \ + benchmarks/inputs/datasets/longmemeval_s_cleaned.json \ + --base-url http://127.0.0.1:8000/v1 \ + --out {out}'" +``` + +For the full longmemeval-S run, omit `--sample` and use `longmemeval_s_cleaned.json` +in both the data path and the `--official-eval-cmd` ref file argument. The script also works standalone against a pre-generated hypothesis file: ```bash -nix-shell -p python3Packages.openai --run "python3 scripts/evaluate_qa_local.py \ +rtk nix-shell -p python3Packages.openai --run "python3 scripts/evaluate_qa_local.py \ 'Gemini 3.5 Flash (Medium)' \ - benchmarks/runs/longmem-eval-smoke/hypothesis.jsonl \ + /hypothesis.jsonl \ benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ --base-url http://127.0.0.1:8000/v1 \ - --out benchmarks/runs/longmem-eval-smoke/official_eval.json" + --out /official_eval.json" ``` To use a different model from the proxy (`curl http://127.0.0.1:8000/v1/models` to list), just diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index a976f89..e81286c 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -224,14 +224,16 @@ rtk cargo run -p klbr-bench -- run \ --budget-read 5000 \ --graph-depth 1 \ --limit 5 \ - --official-eval-cmd '' \ - --out /tmp/klbr-lme-s-official + --official-eval-cmd '' ``` the hook stores status/stdout/stderr in `official_eval.txt`, parses `official_eval.json` when the command writes it, and surfaces the result in -`report.json` / `report.md`. the upstream official LongMemEval evaluator command -is not checked into this repo, so a real official qa score is still blocked on +`report.json` / `report.md`. without `--out`, the run command writes under +`benchmarks/runs////`. for incremental qa runs, +`--sample 20 --sample-seed 42` picks a reproducible random subset stratified by +`question_type`. the upstream official LongMemEval evaluator command is not +checked into this repo, so a real official qa score is still blocked on providing that command/environment. ## open gaps diff --git a/klbr-bench/src/longmemeval.rs b/klbr-bench/src/longmemeval.rs index 231c98a..f634ba5 100644 --- a/klbr-bench/src/longmemeval.rs +++ b/klbr-bench/src/longmemeval.rs @@ -1,4 +1,5 @@ -use std::collections::{HashMap, HashSet}; +use std::cmp::Ordering; +use std::collections::{BTreeMap, HashMap, HashSet}; use std::fs::{self, File}; use std::io::{BufWriter, Write}; use std::path::{Path, PathBuf}; @@ -6,6 +7,7 @@ use std::process::Command; use std::time::{Duration, Instant}; use anyhow::{bail, Context as _, Result}; +use rand::{rngs::StdRng, seq::SliceRandom, Rng as _, SeedableRng}; use serde::{Deserialize, Serialize}; use serde_json::json; @@ -68,6 +70,40 @@ struct OfficialEvalResult { parsed_metrics: Option, } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum SampleMode { + Stratified, + Random, +} + +impl SampleMode { + fn parse(value: &str) -> Result { + match value { + "stratified" | "stratified-random" => Ok(Self::Stratified), + "random" => Ok(Self::Random), + other => bail!("unknown --sample-mode {other}; expected stratified or random"), + } + } + + fn as_str(self) -> &'static str { + match self { + Self::Stratified => "stratified", + Self::Random => "random", + } + } +} + +#[derive(Debug, Clone, Serialize)] +struct SampleReport { + requested: usize, + selected: usize, + available: usize, + seed: u64, + mode: &'static str, + question_type_counts: BTreeMap, + question_ids: Vec, +} + fn load_questions_for_suite(suite: &str, data_path: &str) -> Result> { let bytes = fs::read(data_path).with_context(|| format!("failed to read {data_path}"))?; if suite.eq_ignore_ascii_case("locomo") { @@ -261,10 +297,165 @@ fn suite_protocol_note(suite: &str) -> Option<&'static str> { } } +fn default_pipeline_run_dir(suite: &str, profile: &str) -> PathBuf { + let timestamp = chrono::Utc::now() + .format("%Y-%m-%d_%H%M%S%.6fZ") + .to_string(); + PathBuf::from("benchmarks") + .join("runs") + .join(path_component(suite)) + .join(path_component(profile)) + .join(timestamp) +} + +fn path_component(value: &str) -> String { + let component = value + .chars() + .map(|ch| { + if ch.is_ascii_alphanumeric() || matches!(ch, '-' | '_' | '.') { + ch + } else { + '-' + } + }) + .collect::(); + let trimmed = component.trim_matches('-'); + if trimmed.is_empty() { + "unnamed".to_string() + } else { + trimmed.to_string() + } +} + +fn sample_questions( + data: Vec, + requested: usize, + seed: u64, + mode: SampleMode, +) -> Vec { + if requested >= data.len() { + return data; + } + + let mut rng = StdRng::seed_from_u64(seed); + match mode { + SampleMode::Random => { + let mut indexed = data.into_iter().enumerate().collect::>(); + indexed.shuffle(&mut rng); + indexed.truncate(requested); + indexed.sort_by_key(|(idx, _)| *idx); + indexed.into_iter().map(|(_, item)| item).collect() + } + SampleMode::Stratified => stratified_sample_questions(data, requested, &mut rng), + } +} + +fn stratified_sample_questions( + data: Vec, + requested: usize, + rng: &mut StdRng, +) -> Vec { + let total = data.len(); + let mut buckets = BTreeMap::>::new(); + for (idx, item) in data.into_iter().enumerate() { + buckets + .entry(item.question_type.clone()) + .or_default() + .push((idx, item)); + } + for bucket in buckets.values_mut() { + bucket.shuffle(rng); + } + + let mut allocations = BTreeMap::::new(); + if requested < buckets.len() { + let mut keys = buckets.keys().cloned().collect::>(); + keys.shuffle(rng); + for key in keys.into_iter().take(requested) { + allocations.insert(key, 1); + } + } else { + let mut assigned = 0usize; + let mut remainders = Vec::<(f64, u64, String)>::new(); + for (key, bucket) in &buckets { + let exact = requested as f64 * bucket.len() as f64 / total as f64; + let allocation = (exact.floor() as usize).max(1).min(bucket.len()); + assigned += allocation; + allocations.insert(key.clone(), allocation); + remainders.push((exact - allocation as f64, rng.gen(), key.clone())); + } + remainders.sort_by(|left, right| { + right + .0 + .partial_cmp(&left.0) + .unwrap_or(Ordering::Equal) + .then_with(|| left.1.cmp(&right.1)) + }); + + let mut remaining = requested.saturating_sub(assigned); + while remaining > 0 { + let mut progressed = false; + for (_, _, key) in &remainders { + let Some(allocation) = allocations.get_mut(key) else { + continue; + }; + if *allocation < buckets[key].len() { + *allocation += 1; + remaining -= 1; + progressed = true; + if remaining == 0 { + break; + } + } + } + if !progressed { + break; + } + } + } + + let mut selected = Vec::with_capacity(requested.min(total)); + for (key, bucket) in buckets { + selected.extend( + bucket + .into_iter() + .take(*allocations.get(&key).unwrap_or(&0)), + ); + } + selected.sort_by_key(|(idx, _)| *idx); + selected.into_iter().map(|(_, item)| item).collect() +} + +fn question_type_counts(data: &[LongMemEvalQuestion]) -> BTreeMap { + let mut counts = BTreeMap::new(); + for item in data { + *counts.entry(item.question_type.clone()).or_insert(0) += 1; + } + counts +} + #[cfg(test)] mod tests { use super::*; + fn test_question(question_id: &str, question_type: &str) -> LongMemEvalQuestion { + LongMemEvalQuestion { + question_id: question_id.to_string(), + question_type: question_type.to_string(), + question: format!("question {question_id}"), + question_date: default_question_date(), + answer: serde_json::Value::String("answer".to_string()), + answer_session_ids: vec!["session".to_string()], + haystack_dates: vec![default_question_date()], + haystack_session_ids: vec!["session".to_string()], + haystack_sessions: vec![vec![LongMemEvalMessage { + role: "user".to_string(), + content: "answer".to_string(), + has_answer: Some(true), + }]], + } + } + #[test] fn locomo_loader_parses_session_objects_and_preserves_order() -> Result<()> { let questions = load_locomo_questions( @@ -350,6 +541,56 @@ mod tests { assert!(note.contains("matched dataset version")); assert!(suite_protocol_note("longmemeval-s").is_none()); } + + #[test] + fn default_pipeline_run_dir_lives_under_benchmarks_runs() { + let out_dir = default_pipeline_run_dir("longmemeval-s", "raw-turns/fts-only"); + let parts = out_dir + .components() + .map(|component| component.as_os_str().to_string_lossy().to_string()) + .collect::>(); + assert_eq!( + &parts[..4], + ["benchmarks", "runs", "longmemeval-s", "raw-turns-fts-only"] + ); + assert!(parts[4].ends_with('Z')); + } + + #[test] + fn stratified_sample_questions_balances_question_types() { + let data = (0..12) + .map(|idx| { + let question_type = match idx % 3 { + 0 => "single-session-user", + 1 => "multi-session", + _ => "temporal-reasoning", + }; + test_question(&format!("q{idx:02}"), question_type) + }) + .collect::>(); + + let sampled = sample_questions(data.clone(), 6, 42, SampleMode::Stratified); + let repeated = sample_questions(data, 6, 42, SampleMode::Stratified); + let sampled_ids = sampled + .iter() + .map(|item| item.question_id.as_str()) + .collect::>(); + let repeated_ids = repeated + .iter() + .map(|item| item.question_id.as_str()) + .collect::>(); + + assert_eq!(sampled.len(), 6); + assert_eq!(sampled_ids, repeated_ids); + assert_eq!( + question_type_counts(&sampled), + BTreeMap::from([ + ("multi-session".to_string(), 2), + ("single-session-user".to_string(), 2), + ("temporal-reasoning".to_string(), 2), + ]) + ); + } } pub(crate) fn parse_date_to_timestamp(date_str: &str) -> i64 { @@ -1008,8 +1249,10 @@ pub async fn run_command(args: &[String]) -> Result<()> { pub async fn run_pipeline_command(args: &[String]) -> Result<()> { let suite = optional_arg(args, "--suite").unwrap_or_else(|| "longmemeval-s".to_string()); let data_path = required_arg(args, "--data")?; - let out_dir = PathBuf::from(required_arg(args, "--out")?); let profile = optional_arg(args, "--profile").unwrap_or_else(|| "klbr-full".to_string()); + let out_dir = optional_arg(args, "--out") + .map(PathBuf::from) + .unwrap_or_else(|| default_pipeline_run_dir(&suite, &profile)); let top_k = optional_arg(args, "--top-k") .and_then(|value| value.parse::().ok()) .unwrap_or(8); @@ -1020,6 +1263,27 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { .and_then(|value| value.parse::().ok()) .unwrap_or(1); let limit = optional_arg(args, "--limit").and_then(|value| value.parse::().ok()); + let sample = optional_arg(args, "--sample") + .map(|value| { + value + .parse::() + .with_context(|| format!("invalid --sample value {value}")) + }) + .transpose()?; + if sample == Some(0) { + bail!("--sample must be greater than 0"); + } + let sample_mode = optional_arg(args, "--sample-mode") + .map(|value| SampleMode::parse(&value)) + .transpose()? + .unwrap_or(SampleMode::Stratified); + let sample_seed = optional_arg(args, "--sample-seed") + .map(|value| { + value + .parse::() + .with_context(|| format!("invalid --sample-seed value {value}")) + }) + .transpose()?; let question_id_filter = optional_arg(args, "--question-id"); let retrieval_only = has_flag(args, "--retrieval-only"); let diagnostic = optional_arg(args, "--diagnostic"); @@ -1036,6 +1300,22 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { anyhow::bail!("no question_id {question_id} found in {data_path}"); } } + let sample_report = if let Some(requested) = sample { + let available = data.len(); + let seed = sample_seed.unwrap_or_else(|| rand::thread_rng().gen()); + data = sample_questions(data, requested, seed, sample_mode); + Some(SampleReport { + requested, + selected: data.len(), + available, + seed, + mode: sample_mode.as_str(), + question_type_counts: question_type_counts(&data), + question_ids: data.iter().map(|item| item.question_id.clone()).collect(), + }) + } else { + None + }; let mut config = Config::load_bench()?; if let Some(llm_url) = optional_arg(args, "--llm-url") { @@ -1292,6 +1572,15 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { let answerable_denominator = answerable.max(1) as f64; let evaluated_denominator = evaluated.max(1) as f64; + let sample_summary = sample_report + .as_ref() + .map(|sample| { + format!( + "{} requested={} selected={} available={} seed={}", + sample.mode, sample.requested, sample.selected, sample.available, sample.seed + ) + }) + .unwrap_or_else(|| "none".to_string()); let report = json!({ "suite": suite, "profile": profile, @@ -1307,6 +1596,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { "top_k": top_k, "budget_read": budget_read, "graph_depth": graph_depth, + "sample": sample_report.clone(), "recall_any_at_5": if answerable == 0 { 0.0 } else { recall_any_at_5 as f64 / answerable as f64 }, "recall_all_at_5": if answerable == 0 { 0.0 } else { recall_all_at_5 as f64 / answerable as f64 }, "packet_metrics": { @@ -1329,11 +1619,12 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { fs::write( out_dir.join("report.md"), format!( - "# klbr memory pipeline benchmark\n\n- suite: `{}`\n- profile: `{}`\n- evaluated: `{}`\n- answerable: `{}`\n- RecallAny@5: `{:.4}`\n- RecallAll@5: `{:.4}`\n- PacketRecallAny@{}: `{:.4}`\n- PacketRecallAll@{}: `{:.4}`\n- answer_bearing_ref_in_context: `{:.4}`\n- gold_session_plus_neighbor_in_context: `{:.4}`\n- same_session_dedupe_suppression_count: `{}`\n- packet_tokens_mean: `{:.2}`\n- retrieval_only: `{}`\n- official_eval: `{}`\n", + "# klbr memory pipeline benchmark\n\n- suite: `{}`\n- profile: `{}`\n- evaluated: `{}`\n- answerable: `{}`\n- sample: `{}`\n- RecallAny@5: `{:.4}`\n- RecallAll@5: `{:.4}`\n- PacketRecallAny@{}: `{:.4}`\n- PacketRecallAll@{}: `{:.4}`\n- answer_bearing_ref_in_context: `{:.4}`\n- gold_session_plus_neighbor_in_context: `{:.4}`\n- same_session_dedupe_suppression_count: `{}`\n- packet_tokens_mean: `{:.2}`\n- retrieval_only: `{}`\n- official_eval: `{}`\n", suite, profile, evaluated, answerable, + sample_summary, if answerable == 0 { 0.0 } else { recall_any_at_5 as f64 / answerable as f64 }, if answerable == 0 { 0.0 } else { recall_all_at_5 as f64 / answerable as f64 }, top_k, @@ -1374,6 +1665,7 @@ pub async fn run_pipeline_command(args: &[String]) -> Result<()> { "packet_rerank": config.memory.rerank, "protocol_note": protocol_note, "question_id": question_id_filter.clone(), + "sample": sample_report, }))?, )?; println!("wrote pipeline benchmark outputs to {}", out_dir.display()); diff --git a/klbr-bench/src/main.rs b/klbr-bench/src/main.rs index 8bd274b..a90b574 100644 --- a/klbr-bench/src/main.rs +++ b/klbr-bench/src/main.rs @@ -214,7 +214,7 @@ async fn main() -> Result<()> { let args: Vec = env::args().collect(); if args.len() < 2 { bail!( - "usage:\n cargo run -p klbr-bench -- run --suite longmemeval-s --data --out [--retrieval-only]\n cargo run -p klbr-bench -- retrieval \n cargo run -p klbr-bench -- passive-recall \n cargo run -p klbr-bench -- router \n cargo run -p klbr-bench -- router-multi [dataset2.json ...]\n cargo run -p klbr-bench -- router-multi-linear [dataset2.json ...]\n cargo run -p klbr-bench -- longmem [llm_url]\n cargo run -p klbr-bench -- longmem-retrieval [split.json subset]\n cargo run -p klbr-bench -- continuous-loop \n cargo run -p klbr-bench -- dump-tools \n cargo run -p klbr-bench -- sweep [score_start score_end score_step margin_start margin_end margin_step [support_start support_end support_step]]" + "usage:\n cargo run -p klbr-bench -- run --suite longmemeval-s --data [--sample ] [--sample-seed ] [--sample-mode stratified|random] [--out ] [--retrieval-only]\n cargo run -p klbr-bench -- retrieval \n cargo run -p klbr-bench -- passive-recall \n cargo run -p klbr-bench -- router \n cargo run -p klbr-bench -- router-multi [dataset2.json ...]\n cargo run -p klbr-bench -- router-multi-linear [dataset2.json ...]\n cargo run -p klbr-bench -- longmem [llm_url]\n cargo run -p klbr-bench -- longmem-retrieval [split.json subset]\n cargo run -p klbr-bench -- continuous-loop \n cargo run -p klbr-bench -- dump-tools \n cargo run -p klbr-bench -- sweep [score_start score_end score_step margin_start margin_end margin_step [support_start support_end support_step]]" ); }