diff --git a/.beads/interactions.jsonl b/.beads/interactions.jsonl index 53fcc55..f84e71c 100644 --- a/.beads/interactions.jsonl +++ b/.beads/interactions.jsonl @@ -31,3 +31,4 @@ {"id":"int-61df751e","kind":"field_change","created_at":"2026-06-27T15:12:00.165278848Z","actor":"dawn","issue_id":"klbr-av1","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Graph-only packets are now omitted unless exact/fts/dense corroboration is present; omission traces use graph_only_uncorroborated and regression tests cover omitted and corroborated graph packets."}} {"id":"int-d9f9e698","kind":"field_change","created_at":"2026-06-27T15:17:19.888479202Z","actor":"dawn","issue_id":"klbr-mpy","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Added tool-required training weighting for the linear router and regenerated out-router-linear-iter-16. Test tool→memory improved 0.6913→0.0940, tool-required recall 0.3087→0.9060, and memory false-abstain remained 0.0000 on test/holdout."}} {"id":"int-0d70493b","kind":"field_change","created_at":"2026-06-27T15:32:03.759629206Z","actor":"dawn","issue_id":"klbr-bqj","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Implemented continuous-loop deterministic benchmark coverage with docs and regression tests"}} +{"id":"int-4dcf4d28","kind":"field_change","created_at":"2026-06-27T15:41:01.505996492Z","actor":"dawn","issue_id":"klbr-6an","extra":{"field":"status","new_value":"closed","old_value":"in_progress","reason":"Unified memory provenance reads and tombstone propagation behind canonical ref edges; legacy memory_edges remains compatibility/backfill storage"}} diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index d3e3b87..7af08ba 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -32,4 +32,4 @@ {"_type":"issue","id":"klbr-mpy","title":"Improve tool-required router separability","description":"Fresh router artifact benchmarks/models/router/linear/out-router-linear-iter-15 reports test tool-required false-memory rate 0.6913 and tool-required recall 0.3087 after adding the metric/calibration surface. Improve training data, features, or model shape so tool-required queries stop looking like memory queries without regressing memory false-abstain.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T15:08:07Z","created_by":"dawn","updated_at":"2026-06-27T15:17:20Z","started_at":"2026-06-27T15:12:36Z","closed_at":"2026-06-27T15:17:20Z","close_reason":"Added tool-required training weighting for the linear router and regenerated out-router-linear-iter-16. Test tool→memory improved 0.6913→0.0940, tool-required recall 0.3087→0.9060, and memory false-abstain remained 0.0000 on test/holdout.","dependencies":[{"issue_id":"klbr-mpy","depends_on_id":"klbr-b4z","type":"blocks","created_at":"2026-06-27T18:08:16Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"klbr-b4z","title":"Regenerate router calibration artifacts","description":"After klbr-9yo, rerun the router bench with the calibrated tool-required metrics and commit fresh benchmarks/models/router model/report outputs so future sweeps track tool→memory and tool-required recall from generated artifacts.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T14:56:42Z","created_by":"dawn","updated_at":"2026-06-27T15:08:17Z","started_at":"2026-06-27T15:01:12Z","closed_at":"2026-06-27T15:08:17Z","close_reason":"Regenerated router linear artifact out-router-linear-iter-15 with tool-required metrics, updated benchmark helper default to the fresh model, and filed klbr-mpy for residual tool→memory quality work.","dependencies":[{"issue_id":"klbr-b4z","depends_on_id":"klbr-9yo","type":"blocks","created_at":"2026-06-27T17:56:51Z","created_by":"dawn","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"klbr-bqj","title":"add continuous-loop memory benchmark coverage","description":"benchmark suite mostly tests static query retrieval. add a loop-style harness that simulates multi-session observe/compact/reflect cycles and then evaluates whether reflection notes, markdown sync, and passive recall stay aligned over a longer horizon.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:48Z","created_by":"dawn","updated_at":"2026-06-27T15:32:04Z","started_at":"2026-06-27T15:17:51Z","closed_at":"2026-06-27T15:32:04Z","close_reason":"Implemented continuous-loop deterministic benchmark coverage with docs and regression tests","dependency_count":0,"dependent_count":1,"comment_count":0} -{"_type":"issue","id":"klbr-6an","title":"unify memory edge tables behind canonical refs","description":"memory.rs still has memory_edges and ref-level edges as separate provenance graphs. design and migrate toward one transactional canonical-ref links graph so packet planning, provenance, supersession, and markdown-note evidence traverse the same substrate.","status":"open","priority":3,"issue_type":"task","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:44Z","created_by":"dawn","updated_at":"2026-06-27T13:15:44Z","dependency_count":0,"dependent_count":1,"comment_count":0} +{"_type":"issue","id":"klbr-6an","title":"unify memory edge tables behind canonical refs","description":"memory.rs still has memory_edges and ref-level edges as separate provenance graphs. design and migrate toward one transactional canonical-ref links graph so packet planning, provenance, supersession, and markdown-note evidence traverse the same substrate.","status":"closed","priority":3,"issue_type":"task","assignee":"dawn","owner":"90008@klbr.net","created_at":"2026-06-27T13:15:44Z","created_by":"dawn","updated_at":"2026-06-27T15:41:01Z","started_at":"2026-06-27T15:32:41Z","closed_at":"2026-06-27T15:41:01Z","close_reason":"Unified memory provenance reads and tombstone propagation behind canonical ref edges; legacy memory_edges remains compatibility/backfill storage","dependency_count":0,"dependent_count":1,"comment_count":0} diff --git a/docs/benchmark-issues-report.md b/docs/benchmark-issues-report.md index bb18e92..e4fabdd 100644 --- a/docs/benchmark-issues-report.md +++ b/docs/benchmark-issues-report.md @@ -58,7 +58,7 @@ the current numbers show a clear mismatch between coarse retrieval success and a ### A. fragmented graph system - **symptom**: two separate, disjoint graph implementations. - **cause**: the codebase implements `memory_edges` (for memory-id database provenance) and `edges` (for ref-id reflink relations) as independent tables in [memory.rs](file:///home/mayer/proj/klbr/klbr-core/src/memory.rs). they are only bridged ad hoc, which prevents a unified traversal of the memory-alias space. -- **status/fix**: graph-only evidence packets are now omitted unless the packet also has exact, lexical, or dense corroboration. this prevents graph expansion from surfacing unsupported top evidence, but does not replace the remaining graph-table unification work. +- **status/fix**: memory-id provenance APIs now read canonical ref `edges`, tombstone descendant propagation follows canonical incoming edges, and `add_edge` requires the canonical mirror to succeed. legacy `memory_edges` remains as compatibility/backfill storage. graph-only evidence packets are also omitted unless exact, lexical, or dense corroboration exists. ### B. unutilized l2/l3 memory layers - **symptom**: the schema defines `MemoryLayer::L1` but no compaction logic actually backfills or creates L2/L3 layers. @@ -73,7 +73,6 @@ the current numbers show a clear mismatch between coarse retrieval success and a ## 4. actionable remediation roadmap -1. **unify the graphs**: merge `memory_edges` and `edges` into one transactional links graph in [memory.rs](file:///home/mayer/proj/klbr/klbr-core/src/memory.rs). -2. **strict entity mapping**: add entity extraction to the [evidence.rs](file:///home/mayer/proj/klbr/klbr-core/src/evidence.rs) planning phase to block cross-entity hallucinations (e.g. film vs camera). -3. **improve router separability**: use the fresh tool-required metrics to improve router training data/features if runtime memory-vs-tool routing remains too permissive. -4. **unblock `klbr-1yn`**: import the official LongMemEval Python evaluation script to establish baseline QA accuracy parity. +1. **strict entity mapping**: add entity extraction to the [evidence.rs](file:///home/mayer/proj/klbr/klbr-core/src/evidence.rs) planning phase to block cross-entity hallucinations (e.g. film vs camera). +2. **implement l2/l3 compaction automation**: synthesize durable distilled summaries and hierarchical entity nodes in [agent.rs](file:///home/mayer/proj/klbr/klbr-core/src/agent.rs). +3. **unblock `klbr-1yn`**: import the official LongMemEval Python evaluation script to establish baseline QA accuracy parity. diff --git a/docs/memory-arch.md b/docs/memory-arch.md index e7bfeae..03f3d32 100644 --- a/docs/memory-arch.md +++ b/docs/memory-arch.md @@ -47,7 +47,7 @@ the zettelkasten part matters for a different reason. ahrens describes the appea the cleanest target is a **markdown-first, sqlite-indexed memory garden**. markdown files are the human-readable source of truth for durable semantic notes and project memory; sqlite is the machine-optimized mirror for indexing, retrieval, versions, refs, backlinks, embeddings, fts, entity aliases, and prompt assembly. ahrens explicitly frames zettelkasten principles as tool-agnostic and says they can be implemented analog or digital across tools like zettlr, the archive, roam, and obsidian; sqlite fts5 already gives you efficient lexical retrieval, ranking, snippets, phrase search, and proximity search without leaving your current stack. that combination is enough for the architecture you want. citeturn9view0turn14view0 -the architectural spine should be a **single universal `refs` layer**. right now your uploaded design already gestures at that with reflinks, aliases, promptable text, and resolution events, but it still leaves too much behavior split across old and new tables. i would make `ref` the only canonical identity everywhere. a raw chat line gets a ref. a chunk within that line gets a child ref. an episodic event card gets a ref. a semantic markdown note and each paragraph inside it get refs. a compaction summary gets a ref. an attachment caption gets a ref. links are always `ref -> ref`, versions are always `ref -> ref`, and lifecycle state is always attached to canonical refs first, with indexes and projections derived from that. the current `memory_edges` versus `edges` split should disappear into one transactional `links` table over canonical refs. fileciteturn0file0 +the architectural spine should be a **single universal `refs` layer**. right now your uploaded design already gestures at that with reflinks, aliases, promptable text, and resolution events, but it still leaves too much behavior split across old and new tables. i would make `ref` the only canonical identity everywhere. a raw chat line gets a ref. a chunk within that line gets a child ref. an episodic event card gets a ref. a semantic markdown note and each paragraph inside it get refs. a compaction summary gets a ref. an attachment caption gets a ref. links are always `ref -> ref`, versions are always `ref -> ref`, and lifecycle state is always attached to canonical refs first, with indexes and projections derived from that. the runtime provenance path now reads canonical `edges`, with `memory_edges` kept as a compatibility/backfill projection rather than a second traversal source. fileciteturn0file0 on top of that spine, i would define four lanes. the **live context lane** is session-scoped and authoritative for short-range discourse, unresolved pronouns, recent tool outputs, and “do that” style callbacks. the **episodic lane** stores append-only event cards that summarize concrete interactions with temporal anchors and supporting raw refs. the **semantic lane** stores durable, human-facing markdown notes with folgezettel-style ids, backlinks, and citations to episodes or documents. the **profile/procedural lane** stores stable preferences, standing instructions, identity facts, and operation patterns that change slowly and should often require explicit user confirmation to update. this is similar in spirit to the multi-layer or multi-memory direction explored by memgpt, memorybank, timem, and remem, but grounded in your existing sqlite/ref architecture rather than a fresh abstraction stack. citeturn2academia1turn4academia0turn3academia3turn13academia5 diff --git a/docs/memory-implementation-status.md b/docs/memory-implementation-status.md index 746ac17..a976f89 100644 --- a/docs/memory-implementation-status.md +++ b/docs/memory-implementation-status.md @@ -20,7 +20,7 @@ updated: 2026-06-27 - stable chunk refs - fts rows through `promptable_text` - `MemoryGarden` can write markdown note files and sync them back into sqlite. -- legacy `memory_edges` now mirror into canonical `edges`, and `sync_reference_indexes` backfills old rows. +- legacy `memory_edges` now mirror into canonical `edges`, `sync_reference_indexes` backfills old rows, and memory-id provenance APIs read canonical ref edges. - memory status changes now project onto canonical refs and rebuild fts visibility. - note, episode, and chunk refs can resolve benchmark session ids from ref metadata/frontmatter. - dense retrieval now searches canonical refs through `embedding_items`, with diff --git a/klbr-core/src/memory.rs b/klbr-core/src/memory.rs index e88d9e1..121e6d5 100644 --- a/klbr-core/src/memory.rs +++ b/klbr-core/src/memory.rs @@ -1166,11 +1166,11 @@ impl MemoryStore { ], |row| row.get(0), )?; - let _ = mirror_memory_edge(&conn, input); + mirror_memory_edge(&conn, input)?; return Ok(id); } let id = conn.last_insert_rowid(); - let _ = mirror_memory_edge(&conn, input); + mirror_memory_edge(&conn, input)?; Ok(id) } @@ -1189,38 +1189,29 @@ impl MemoryStore { pub fn provenance_counts(&self, memory_id: i64) -> Result> { let conn = self.conn.lock().unwrap(); - let mut stmt = conn.prepare( - "SELECT edge_type, COUNT(*) - FROM memory_edges - WHERE from_memory_id = ?1 - GROUP BY edge_type", - )?; - let mut counts = stmt - .query_map(params![memory_id], |row| { - Ok(( - parse_edge_type(&row.get::<_, String>(0)?), - row.get::<_, i64>(1)? as usize, - )) - })? - .filter_map(|row| row.ok()) - .collect::>(); + let mut counts = Vec::<(MemoryEdgeType, usize)>::new(); + for edge in canonical_memory_edges(&conn, memory_id, MemoryEdgeDirection::Outgoing)? { + if let Some((_, count)) = counts + .iter_mut() + .find(|(edge_type, _)| *edge_type == edge.edge_type) + { + *count += 1; + } else { + counts.push((edge.edge_type, 1)); + } + } counts.sort_by_key(|(edge_type, _)| edge_type_order(edge_type)); Ok(counts) } fn edges(&self, column: &str, memory_id: i64) -> Result> { let conn = self.conn.lock().unwrap(); - let mut stmt = conn.prepare(&format!( - "SELECT id, from_memory_id, to_memory_id, edge_type, metadata, ts - FROM memory_edges - WHERE {column} = ?1 - ORDER BY ts ASC, id ASC" - ))?; - let results = stmt - .query_map(params![memory_id], row_to_memory_edge)? - .filter_map(|row| row.ok()) - .collect(); - Ok(results) + let direction = match column { + "from_memory_id" => MemoryEdgeDirection::Outgoing, + "to_memory_id" => MemoryEdgeDirection::Incoming, + _ => anyhow::bail!("unsupported memory edge column {column}"), + }; + canonical_memory_edges(&conn, memory_id, direction) } pub fn provenance_sources(&self, memory_id: i64, max_depth: usize) -> Result> { @@ -1237,14 +1228,8 @@ impl MemoryStore { if depth >= max_depth { continue; } - let mut stmt = conn.prepare( - "SELECT to_memory_id FROM memory_edges - WHERE from_memory_id = ?1 - ORDER BY ts ASC, id ASC", - )?; - let rows = stmt.query_map(params![current_id], |row| row.get::<_, i64>(0))?; - for row in rows { - let source_id = row?; + for edge in canonical_memory_edges(&conn, current_id, MemoryEdgeDirection::Outgoing)? { + let source_id = edge.to_memory_id; if seen.insert(source_id) { source_ids.push(source_id); frontier.push((source_id, depth + 1)); @@ -3725,20 +3710,66 @@ fn memory_exists(conn: &Connection, id: i64) -> Result { Ok(rows.next()?.is_some()) } +#[derive(Debug, Clone, Copy)] +enum MemoryEdgeDirection { + Outgoing, + Incoming, +} + +fn canonical_memory_edges( + conn: &Connection, + memory_id: i64, + direction: MemoryEdgeDirection, +) -> Result> { + let filter = match direction { + MemoryEdgeDirection::Outgoing => "src.memory_id", + MemoryEdgeDirection::Incoming => "dst.memory_id", + }; + let mut stmt = conn.prepare(&format!( + "WITH memory_ref AS ( + SELECT + r.ref_id, + CASE + WHEN r.entity_type = 'memory' THEN r.entity_id + WHEN r.entity_type = 'memory_version' THEN v.memory_id + END AS memory_id + FROM refs r + LEFT JOIN memory_versions v + ON r.entity_type = 'memory_version' + AND v.version_id = r.entity_id + WHERE r.entity_type IN ('memory', 'memory_version') + ) + SELECT + e.edge_id, + src.memory_id AS from_memory_id, + dst.memory_id AS to_memory_id, + e.rel_type, + e.metadata, + e.created_at + FROM edges e + JOIN memory_ref src ON src.ref_id = e.src_ref_id + JOIN memory_ref dst ON dst.ref_id = e.dst_ref_id + WHERE e.edge_state != 'deleted' + AND e.rel_type IN ('derived_from', 'supersedes', 'supports') + AND src.memory_id IS NOT NULL + AND dst.memory_id IS NOT NULL + AND {filter} = ?1 + ORDER BY e.created_at ASC, e.edge_id ASC" + ))?; + let edges = stmt + .query_map(params![memory_id], row_to_canonical_memory_edge)? + .collect::>>()?; + Ok(edges) +} + fn provenance_descendants(conn: &Connection, memory_id: i64) -> Result> { let mut seen = std::collections::HashSet::new(); let mut frontier = vec![memory_id]; let mut descendants = Vec::new(); while let Some(current_id) = frontier.pop() { - let mut stmt = conn.prepare( - "SELECT from_memory_id FROM memory_edges - WHERE to_memory_id = ?1 - ORDER BY ts ASC, id ASC", - )?; - let rows = stmt.query_map(params![current_id], |row| row.get::<_, i64>(0))?; - for row in rows { - let descendant_id = row?; + for edge in canonical_memory_edges(conn, current_id, MemoryEdgeDirection::Incoming)? { + let descendant_id = edge.from_memory_id; if seen.insert(descendant_id) { descendants.push(descendant_id); frontier.push(descendant_id); @@ -3839,7 +3870,7 @@ fn memory_by_id(conn: &Connection, id: i64) -> Result> { })) } -fn row_to_memory_edge(row: &rusqlite::Row<'_>) -> rusqlite::Result { +fn row_to_canonical_memory_edge(row: &rusqlite::Row<'_>) -> rusqlite::Result { let metadata: String = row.get(4)?; Ok(MemoryEdge { id: row.get(0)?, @@ -4394,6 +4425,56 @@ mod tests { Ok(()) } + #[test] + fn test_memory_provenance_reads_canonical_ref_edges_without_legacy_rows() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + + let source_id = store.store_with_metadata(&test_input( + "canonical source fact", + MemoryStatus::Archived, + vec!["project:klbr".to_string()], + vec![1.0, 0.0, 0.0, 0.0], + ))?; + let summary_id = store.store_with_metadata(&test_input( + "canonical summary", + MemoryStatus::Active, + vec!["summary".to_string()], + vec![1.0, 0.0, 0.0, 0.0], + ))?; + let source_ref = format!("m{}", to_base36(source_id as u64)); + let summary_ref = format!("m{}", to_base36(summary_id as u64)); + + store.add_reflink_edge(&summary_ref, &source_ref, "derived_from")?; + + let legacy_edges: i64 = store.conn().lock().unwrap().query_row( + "SELECT COUNT(*) FROM memory_edges + WHERE from_memory_id = ?1 OR to_memory_id = ?1", + params![summary_id], + |row| row.get(0), + )?; + assert_eq!(legacy_edges, 0); + + assert_eq!( + store.provenance_counts(summary_id)?, + vec![(MemoryEdgeType::DerivedFrom, 1)] + ); + let edges = store.edges_from(summary_id)?; + assert_eq!(edges.len(), 1); + assert_eq!(edges[0].from_memory_id, summary_id); + assert_eq!(edges[0].to_memory_id, source_id); + assert_eq!(edges[0].edge_type, MemoryEdgeType::DerivedFrom); + let incoming = store.edges_to(source_id)?; + assert_eq!(incoming.len(), 1); + assert_eq!(incoming[0].from_memory_id, summary_id); + + let sources = store.provenance_sources(summary_id, 1)?; + assert_eq!(sources.len(), 1); + assert_eq!(sources[0].memory_id, source_id); + assert_eq!(sources[0].status, MemoryStatus::Archived); + Ok(()) + } + #[test] fn test_provenance_sources_respects_depth_limit() -> Result<()> { let tmp = NamedTempFile::new()?; @@ -4525,6 +4606,48 @@ mod tests { Ok(()) } + #[test] + fn test_tombstone_suppresses_descendants_from_canonical_ref_edges() -> Result<()> { + let tmp = NamedTempFile::new()?; + let store = MemoryStore::open(tmp.path().to_str().unwrap(), 4)?; + + let source_id = store.store_with_metadata(&test_input( + "canonical private source", + MemoryStatus::Active, + vec!["private".to_string()], + vec![1.0, 0.0, 0.0, 0.0], + ))?; + let derived_id = store.store_with_metadata(&test_input( + "canonical derived summary", + MemoryStatus::Active, + vec!["summary".to_string()], + vec![1.0, 0.0, 0.0, 0.0], + ))?; + let source_ref = format!("m{}", to_base36(source_id as u64)); + let derived_ref = format!("m{}", to_base36(derived_id as u64)); + + store.add_reflink_edge(&derived_ref, &source_ref, "derived_from")?; + let legacy_edges: i64 = store.conn().lock().unwrap().query_row( + "SELECT COUNT(*) FROM memory_edges + WHERE from_memory_id = ?1 OR to_memory_id = ?1", + params![derived_id], + |row| row.get(0), + )?; + assert_eq!(legacy_edges, 0); + + store.tombstone_memory(source_id, Some("canonical privacy delete"))?; + + assert_eq!( + store.get_memory(source_id)?.unwrap().status, + MemoryStatus::Tombstoned + ); + assert_eq!( + store.get_memory(derived_id)?.unwrap().status, + MemoryStatus::Suppressed + ); + Ok(()) + } + #[test] fn dense_ref_search_returns_markdown_note_chunks() -> Result<()> { let tmp = NamedTempFile::new()?;