From 40beb75f52c1bdb55c245fa12bc2aa7df22dd760 Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Sun, 15 Mar 2026 22:17:25 -0600 Subject: [PATCH] =?UTF-8?q?KG-8:=20graph=20API=20improvements=20=E2=80=94?= =?UTF-8?q?=20edge=20type=20stats,=20fuzzy=20name=20bridging?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add explicit_edge_count and co_occurrence_edge_count to graph stats response so frontend can display signal quality indicators. Add fuzzy name bridging to _build_name_to_node_id() using the 5-tier matching from think.entities.matching. KG entity names that differ from canonical names (e.g., "Jer Miller" vs "Jeremie Miller") now resolve correctly via exact, case-insensitive, slug, first-word, and rapidfuzz matching — same strategy used everywhere else in entity resolution. Facet consistency verified: all three signal query paths (explicit edges, co-occurrence edges, strength scoring) already apply facet filtering consistently. With KG-6's facet assignments, facet switching now produces meaningfully scoped graphs. Co-Authored-By: Claude Opus 4.6 (1M context) --- apps/home/routes.py | 47 ++++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 44 insertions(+), 3 deletions(-) diff --git a/apps/home/routes.py b/apps/home/routes.py index 378f18497..b19af7d2d 100644 --- a/apps/home/routes.py +++ b/apps/home/routes.py @@ -132,6 +132,12 @@ def api_graph(): finally: conn.close() + # Count edges by type + explicit_edge_count = sum(1 for e in edges if e.get("edge_type") == "explicit") + co_occurrence_edge_count = sum( + 1 for e in edges if e.get("edge_type") == "co_occurrence" + ) + return jsonify( { "nodes": nodes, @@ -139,6 +145,8 @@ def api_graph(): "stats": { "total_entities": total_entities, "total_signals": total_signals, + "explicit_edge_count": explicit_edge_count, + "co_occurrence_edge_count": co_occurrence_edge_count, }, } ) @@ -292,15 +300,23 @@ def _build_name_to_node_id( conn: sqlite3.Connection, node_ids: set[str], ) -> dict[str, str]: - """Map signal entity_names to node IDs (entity_ids) for edge matching.""" + """Map signal entity_names to node IDs (entity_ids) for edge matching. + + Uses slug matching first, then falls back to fuzzy matching via + find_matching_entity() for KG entity names that differ from canonical + names (e.g., "Jer Miller" -> "jeremie_miller"). + """ + import json as _json + from think.entities.core import entity_slug + from think.entities.matching import find_matching_entity rows = conn.execute( - "SELECT entity_id, name FROM entities WHERE source='identity'" + "SELECT entity_id, name, aka FROM entities WHERE source='identity'" ).fetchall() result: dict[str, str] = {} - for entity_id, name in rows: + for entity_id, name, _aka in rows: if entity_id in node_ids: result[name] = entity_id result[entity_id] = entity_id @@ -319,4 +335,29 @@ def _build_name_to_node_id( if slug in node_ids: result[sname] = slug + # Fuzzy matching for remaining unmapped signal names using the 5-tier + # matching from think.entities.matching (handles "Jer Miller" vs + # "Jeremie Miller" and similar KG name variants). + unmapped = [sname for (sname,) in signal_names if sname not in result] + if unmapped: + entity_dicts = [] + for entity_id, name, aka_str in rows: + if entity_id not in node_ids: + continue + d: dict[str, Any] = {"id": entity_id, "name": name, "aka": []} + if aka_str: + try: + aka_list = _json.loads(aka_str) + if isinstance(aka_list, list): + d["aka"] = aka_list + except (ValueError, TypeError): + pass + entity_dicts.append(d) + + if entity_dicts: + for sname in unmapped: + match = find_matching_entity(sname, entity_dicts) + if match: + result[sname] = match["id"] + return result -- 2.51.2