diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 7af08ba..8f96621 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -21,7 +21,7 @@ {"_type":"issue","id":"klbr-f0p","title":"Remove tools lane and refactor router to 2-class setup","status":"closed","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T01:06:06Z","created_by":"dawn","updated_at":"2026-06-27T01:06:10Z","closed_at":"2026-06-27T01:06:10Z","close_reason":"Closed","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.18","title":"Keep packet rerank documents under local reranker batch limits","description":"Current full-profile run for LongMemEval-S question 51a45a95 with --rerank-packets produced packet_rerank.status=error from localhost:8003: input 1184 tokens is too large for physical batch size 1024. Packet rerank documents need a hard budget that samples every body without exceeding the local bge-reranker/llama.cpp batch/context limits, and traces should keep the service error visible.","status":"closed","priority":2,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T23:25:26Z","created_by":"dawn","updated_at":"2026-06-26T23:27:53Z","started_at":"2026-06-26T23:25:36Z","closed_at":"2026-06-26T23:27:53Z","close_reason":"Bounded packet rerank documents, added many-body regression, verified klbr-core tests and 51a45a95 packet rerank smoke now returns scores and Target.","labels":["architecture","benchmarks","memory","rerank"],"dependencies":[{"issue_id":"klbr-wmz.18","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-27T02:25:26Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.19","title":"Stabilize reader answers when evidence packets contain the gold answer","description":"Current source retrieves the answer-bearing ref for LongMemEval-S question 51a45a95: answer_bearing_ref_in_context=true, gold_session_plus_neighbor_in_context=true, but a full-profile non-rerank QA run answered 'I don't know' for gold answer Target. A rerank attempt still failed to run due packet size, and that run answered 'Your email inbox'. Treat this as reader/evidence interpretation work, not retrieval recall: add a focused regression or reader/answer-verifier change so where/action questions prefer supported merchant/store/venue evidence over discovery-channel text when the gold answer is in packet context.","status":"closed","priority":2,"issue_type":"bug","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T23:25:26Z","created_by":"dawn","updated_at":"2026-06-26T23:33:01Z","started_at":"2026-06-26T23:30:28Z","closed_at":"2026-06-26T23:33:01Z","close_reason":"Centralized reader prompt, made packet-local adjacent bodies explicit evidence windows, added prompt contract test, and verified 51a45a95 non-rerank now answers Target with answer-bearing refs in context.","labels":["architecture","benchmarks","memory","reader"],"dependencies":[{"issue_id":"klbr-wmz.19","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-27T02:25:26Z","created_by":"dawn","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"klbr-1yn","title":"Run real official LongMemEval QA evaluator","description":"The benchmark harness now captures official evaluator stdout/stderr/status and parses official_eval.json when --official-eval-cmd writes it, but this repo does not include the upstream LongMemEval evaluator environment/command. Provide or install the official evaluator, run a small checked-in/local LongMemEval-S QA smoke with answer generation, and record the resulting official metrics in the run artifacts/docs.","notes":"Blocked on external upstream LongMemEval evaluator command/environment. The repo harness supports --official-eval-cmd and parses official_eval.json, but cannot run a real official QA score until that evaluator is provided or installed.","status":"blocked","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T22:53:31Z","created_by":"dawn","updated_at":"2026-06-26T23:45:35Z","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-1yn","title":"Run real official LongMemEval QA evaluator","description":"The benchmark harness now captures official evaluator stdout/stderr/status and parses official_eval.json when --official-eval-cmd writes it, but this repo does not include the upstream LongMemEval evaluator environment/command. Provide or install the official evaluator, run a small checked-in/local LongMemEval-S QA smoke with answer generation, and record the resulting official metrics in the run artifacts/docs.","notes":"Blocked on external upstream LongMemEval evaluator command/environment. The repo harness supports --official-eval-cmd and parses official_eval.json, but cannot run a real official QA score until that evaluator is provided or installed.\nunblocked locally: scripts/evaluate_qa_local.py added — reimplementation of official evaluate_qa.py using the antigravity proxy at http://127.0.0.1:8000/v1 as judge. no OPENAI_API_KEY needed. nix-shell -p python3Packages.openai provides the dep. see benchmarks/README.md 'Official QA Evaluation (local judge)' section for full --official-eval-cmd invocation.","status":"blocked","priority":2,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-26T22:53:31Z","created_by":"dawn","updated_at":"2026-06-27T15:53:42Z","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-d67","title":"document klbr.kdl and bench.kdl secrets and usage","description":"add documentation to AGENTS.md specifying that klbr.kdl and bench.kdl are config files containing potential secrets, used by klbr and the bench suite respectively","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:40:37Z","created_by":"dawn","updated_at":"2026-06-26T22:41:08Z","started_at":"2026-06-26T22:40:43Z","closed_at":"2026-06-26T22:41:08Z","close_reason":"documented klbr.kdl and bench.kdl config usage and secrets under the config.rs section in AGENTS.md","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-djb","title":"update agents.md to reflect latest code state, get rid of anything stale - dont touch rtk / beads instructions","status":"closed","priority":2,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T22:27:02Z","created_by":"dawn","updated_at":"2026-06-26T22:31:09Z","started_at":"2026-06-26T22:28:05Z","closed_at":"2026-06-26T22:31:09Z","close_reason":"updated AGENTS.md to match the latest codebase architecture (KDL configs, models.rs, memory lanes, router, evidence, garden, twilight discord)","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-wmz.15","title":"Rerank answer-bearing evidence packets instead of raw chunks","description":"docs/evidence-selection.md recommends reranking EvidencePacket objects, not raw chunks or whole sessions. Chunk-level reranking can prefer the query-adjacent setup chunk and still miss the answer-bearing neighbor; whole-session reranking is too broad. Packet reranking should judge whether a budgeted evidence packet contains enough support to answer.","design":"Treat reranking as a tie-breaker until packet-level fixtures show lift. The reranker prompt/input should ask for answer-bearing evidence, not generic relevance.","acceptance_criteria":"The reranker path can score packet bodies that include anchor, local window, episode gist, refs, and source signals; exact-ref packets cannot be suppressed by reranker score alone; traces log pre-rerank and post-rerank packet order; reranker thresholds operate on packets and cannot drop all strong evidence for a gold session without a traceable reason; tests cover a packet that beats its raw anchor chunk because the neighbor contains the answer.","status":"closed","priority":2,"issue_type":"feature","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-26T18:49:44Z","created_by":"dawn","updated_at":"2026-06-26T23:19:52Z","started_at":"2026-06-26T23:10:05Z","closed_at":"2026-06-26T23:19:52Z","close_reason":"Implemented packet-level reranking with bounded packet documents, exact-ref protection, traceable threshold fallback, bench flag wiring, tests, and live smoke.","labels":["architecture","context","memory","rerank","retrieval"],"dependencies":[{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz","type":"parent-child","created_at":"2026-06-26T21:49:44Z","created_by":"dawn","metadata":"{}"},{"issue_id":"klbr-wmz.15","depends_on_id":"klbr-wmz.13","type":"blocks","created_at":"2026-06-26T21:50:39Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} diff --git a/benchmarks/README.md b/benchmarks/README.md index 2fe7af8..eb1ddf1 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -127,3 +127,44 @@ cargo run -p klbr-bench -- longmem-retrieval \ ``` For that comparison, use the report's all-row `RecallAny@5` because MemWeave averages all 450 held-out questions, including `_abs` rows. + +## Official QA Evaluation (local judge) + +`scripts/evaluate_qa_local.py` is a self-contained reimplementation of the official LongMemEval +`evaluate_qa.py` that points at the local antigravity proxy instead of OpenAI. Same prompt +templates, same positional interface, writes `official_eval.json` which the bench harness reads. + +Requires `python3Packages.openai` — run under nix-shell: + +```bash +# Smoke run against the 30-question subset (fast, no OPENAI_API_KEY needed): +cargo run -p klbr-bench -- longmem \ + benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + benchmarks/inputs/configs/mvp_rerank_support_calibrated.json \ + benchmarks/runs/longmem-eval-smoke \ + --official-eval-cmd \ + "nix-shell -p python3Packages.openai --run \ + 'python3 scripts/evaluate_qa_local.py \ + \"Gemini 3.5 Flash (Medium)\" \ + {hypothesis} \ + benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --base-url http://127.0.0.1:8000/v1 \ + --out {out}'" +``` + +For the full longmemeval-S run, swap out `subset30` for `longmemeval_s_cleaned.json` in both +the data path and the `--official-eval-cmd` ref file argument. + +The script also works standalone against a pre-generated hypothesis file: + +```bash +nix-shell -p python3Packages.openai --run "python3 scripts/evaluate_qa_local.py \ + 'Gemini 3.5 Flash (Medium)' \ + benchmarks/runs/longmem-eval-smoke/hypothesis.jsonl \ + benchmarks/inputs/datasets/longmemeval_s_cleaned_subset30.json \ + --base-url http://127.0.0.1:8000/v1 \ + --out benchmarks/runs/longmem-eval-smoke/official_eval.json" +``` + +To use a different model from the proxy (`curl http://127.0.0.1:8000/v1/models` to list), just +replace the first positional argument with any model id the proxy advertises. diff --git a/scripts/evaluate_qa_local.py b/scripts/evaluate_qa_local.py new file mode 100755 index 0000000..2e77f13 --- /dev/null +++ b/scripts/evaluate_qa_local.py @@ -0,0 +1,216 @@ +#!/usr/bin/env python3 +""" +Local-judge reimplementation of the official LongMemEval evaluate_qa.py. + +Compatible with the official script's positional argument interface: + evaluate_qa_local.py + +Additional flags: + --base-url URL OpenAI-compatible base URL [default: http://127.0.0.1:8000/v1] + --api-key KEY API key [default: EMPTY] + --out PATH Where to write official_eval.json [default: .official_eval.json] + +The --out path is intended to be filled in by the klbr bench harness via +the {out} placeholder in --official-eval-cmd. + +Exit code 0 on success, 1 on error. +""" + +import json +import sys +import os +import argparse +from pathlib import Path + +try: + from openai import OpenAI +except ImportError: + print("ERROR: openai package not available. Run with:", file=sys.stderr) + print(" nix-shell -p python3Packages.openai --run 'python3 scripts/evaluate_qa_local.py ...'", file=sys.stderr) + sys.exit(1) + + +TASK_TEMPLATES = { + "default": ( + "I will give you a question, a correct answer, and a response from a model. " + "Please answer yes if the response contains the correct answer. Otherwise, answer no. " + "If the response is equivalent to the correct answer or contains all the intermediate " + "steps to get the correct answer, you should also answer yes. If the response only " + "contains a subset of the information required by the answer, answer no. " + "\n\nQuestion: {question}\n\nCorrect Answer: {answer}\n\nModel Response: {response}" + "\n\nIs the model response correct? Answer yes or no only." + ), + "temporal-reasoning": ( + "I will give you a question, a correct answer, and a response from a model. " + "Please answer yes if the response contains the correct answer. Otherwise, answer no. " + "If the response is equivalent to the correct answer or contains all the intermediate " + "steps to get the correct answer, you should also answer yes. If the response only " + "contains a subset of the information required by the answer, answer no. In addition, " + "do not penalize off-by-one errors for the number of days. If the question asks for " + "the number of days/weeks/months, etc., and the model makes off-by-one errors (e.g., " + "predicting 19 days when the answer is 18), the model's response is still correct. " + "\n\nQuestion: {question}\n\nCorrect Answer: {answer}\n\nModel Response: {response}" + "\n\nIs the model response correct? Answer yes or no only." + ), + "knowledge-update": ( + "I will give you a question, a correct answer, and a response from a model. " + "Please answer yes if the response contains the correct answer. Otherwise, answer no. " + "If the response contains some previous information along with an updated answer, the " + "response should be considered as correct as long as the updated answer is the required answer." + "\n\nQuestion: {question}\n\nCorrect Answer: {answer}\n\nModel Response: {response}" + "\n\nIs the model response correct? Answer yes or no only." + ), + "single-session-preference": ( + "I will give you a question, a rubric for desired personalized response, and a response " + "from a model. Please answer yes if the response satisfies the desired response. Otherwise, " + "answer no. The model does not need to reflect all the points in the rubric. The response " + "is correct as long as it recalls and utilizes the user's personal information correctly." + "\n\nQuestion: {question}\n\nRubric: {answer}\n\nModel Response: {response}" + "\n\nIs the model response correct? Answer yes or no only." + ), + "abstention": ( + "I will give you an unanswerable question, an explanation, and a response from a model. " + "Please answer yes if the model correctly identifies the question as unanswerable. The " + "model could say that the information is incomplete, or some other information is given " + "but the asked information is not." + "\n\nQuestion: {question}\n\nExplanation: {answer}\n\nModel Response: {response}" + "\n\nDoes the model correctly identify the question as unanswerable? Answer yes or no only." + ), +} + +ABSTENTION_TASK_TYPES = { + "single-session-user", "single-session-assistant", "multi-session", + "temporal-reasoning", "knowledge-update", "single-session-preference", +} + + +def get_prompt(task: str, question: str, answer: str, response: str, is_abstention: bool) -> str: + if is_abstention: + tmpl = TASK_TEMPLATES["abstention"] + elif task in TASK_TEMPLATES: + tmpl = TASK_TEMPLATES[task] + else: + tmpl = TASK_TEMPLATES["default"] + return tmpl.format(question=question, answer=answer, response=response) + + +def load_jsonl_or_json(path: str): + text = Path(path).read_text() + stripped = text.strip() + # JSONL: multiple lines, each a json object + lines = [l for l in stripped.splitlines() if l.strip()] + if len(lines) > 1: + try: + return [json.loads(line) for line in lines] + except json.JSONDecodeError: + pass + # single JSON value + parsed = json.loads(stripped) + if isinstance(parsed, list): + return parsed + return [parsed] + + +def main(): + parser = argparse.ArgumentParser(description="LongMemEval local judge evaluator") + parser.add_argument("metric_model", help="Model name to use as judge") + parser.add_argument("hyp_file", help="Path to hypothesis JSONL") + parser.add_argument("ref_file", help="Path to reference JSON (original dataset)") + parser.add_argument("--base-url", default="http://127.0.0.1:8000/v1", + help="OpenAI-compatible base URL") + parser.add_argument("--api-key", default="EMPTY", help="API key (use EMPTY for local)") + parser.add_argument("--out", default=None, + help="Path to write official_eval.json (default: .official_eval.json)") + parser.add_argument("--verbose", action="store_true", default=True) + + args = parser.parse_args() + + out_path = Path(args.out) if args.out else Path(args.hyp_file + ".official_eval.json") + + client = OpenAI(api_key=args.api_key, base_url=args.base_url) + + hypotheses = load_jsonl_or_json(args.hyp_file) + references = load_jsonl_or_json(args.ref_file) + qid2data = {e["question_id"]: e for e in references} + qid2type = {e["question_id"]: e["question_type"] for e in references} + qtypes = set(qid2type.values()) + qtype2acc: dict[str, list[int]] = {t: [] for t in qtypes} + + logs = [] + skipped = 0 + + for entry in hypotheses: + qid = entry["question_id"] + if qid not in qid2type: + print(f"WARNING: skipping {qid} — not in reference data", file=sys.stderr) + skipped += 1 + continue + + qtype = qid2type[qid] + q = qid2data[qid]["question"] + ans = qid2data[qid]["answer"] + hyp = entry["hypothesis"] + is_abs = "_abs" in qid + + prompt = get_prompt(qtype, q, ans, hyp, is_abs) + + try: + stream = client.chat.completions.create( + model=args.metric_model, + messages=[{"role": "user", "content": prompt}], + n=1, + temperature=0, + max_tokens=10, + stream=True, + ) + chunks = [] + for chunk in stream: + if chunk.choices and chunk.choices[0].delta.content: + chunks.append(chunk.choices[0].delta.content) + eval_response = "".join(chunks).strip() + except Exception as e: + print(f"ERROR on {qid}: {e}", file=sys.stderr) + eval_response = "no" + + label = "yes" in eval_response.lower() + entry["autoeval_label"] = {"model": args.metric_model, "label": label} + logs.append(entry) + qtype2acc.setdefault(qtype, []).append(1 if label else 0) + + if args.verbose: + print(json.dumps({ + "question_id": qid, + "question": q, + "answer": ans, + "hypothesis": hyp, + "autoeval_label": label, + "eval_response": eval_response, + }), flush=True) + + # compute final metrics + all_labels = [1 if e["autoeval_label"]["label"] else 0 for e in logs] + overall_acc = sum(all_labels) / len(all_labels) if all_labels else 0.0 + + per_type = {} + for t, vals in qtype2acc.items(): + per_type[t] = {"accuracy": sum(vals) / len(vals) if vals else 0.0, "count": len(vals)} + + result = { + "model": args.metric_model, + "base_url": args.base_url, + "evaluated": len(logs), + "skipped": skipped, + "accuracy": overall_acc, + "per_type": per_type, + "logs": logs, + } + + out_path.write_text(json.dumps(result, indent=2)) + print(f"\nAccuracy: {overall_acc:.4f} ({len(logs)} evaluated, {skipped} skipped)") + for t, d in sorted(per_type.items()): + print(f"\t{t}: {d['accuracy']:.4f} ({d['count']})") + print(f"Saved to {out_path}") + + +if __name__ == "__main__": + main()