From 3f953e48e959c3ee7e354ccfef2b9284c8b97337 Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Wed, 5 Aug 2026 12:56:19 -0600 Subject: [PATCH] unify entity matcher comparisons on normalize_resolution_query Replace case-only comparison keys with the shared Unicode normalization path and record the intentional fixture divergences from that behavior change. --- .../solstone-core-entity/src/fixture_tests.rs | 79 ++++- .../solstone-core-entity/src/matcher.rs | 84 +++-- .../solstone-core-entity/src/test_support.rs | 33 +- .../entity_matching_native_divergences.json | 309 ++++++++++++++++++ 4 files changed, 472 insertions(+), 33 deletions(-) create mode 100644 core/fixtures/entity_matching_native_divergences.json diff --git a/core/crates/solstone-core-entity/src/fixture_tests.rs b/core/crates/solstone-core-entity/src/fixture_tests.rs index 575b0a68b..85bc4cf76 100644 --- a/core/crates/solstone-core-entity/src/fixture_tests.rs +++ b/core/crates/solstone-core-entity/src/fixture_tests.rs @@ -14,8 +14,8 @@ use crate::{ use super::test_support::{ SweepVerificationError, entity_identity_fixture, entity_matching_fixture, - normalization_divergences, normalization_divergences_fixture, slug_divergences, - slug_divergences_fixture, verify_raw_sweep_digest, verify_sweep_digest, + matching_divergences_fixture, normalization_divergences, normalization_divergences_fixture, + slug_divergences, slug_divergences_fixture, verify_raw_sweep_digest, verify_sweep_digest, }; const JOURNAL_IO_ALLOWED: &[&str] = &[ @@ -133,6 +133,7 @@ fn collect_production_sources(directory: &Path, sources: &mut Vec) { path.file_name().and_then(|name| name.to_str()), Some( "fixture_tests.rs" + | "resolution_tests.rs" | "store_tests.rs" | "test_support.rs" | "trust_lock_tests.rs" @@ -237,6 +238,7 @@ fn ambiguity_id_vectors_match_fixture() { #[test] fn matching_vectors_match_fixture() { let fixture = entity_matching_fixture(); + let divergences_fixture = matching_divergences_fixture(); // A loader that parses the file, misses the array and yields nothing would // satisfy every assertion below perfectly. The declared count is what makes // these tests rather than formalities. @@ -255,7 +257,61 @@ fn matching_vectors_match_fixture() { .count(), fixture.refusal_count ); - for vector in fixture.vectors { + assert_eq!( + divergences_fixture.entries.len(), + divergences_fixture.counts.total + ); + assert_eq!( + divergences_fixture.counts.tier_changes + divergences_fixture.counts.refusal_to_match, + divergences_fixture.counts.total + ); + let mut divergences = HashMap::with_capacity(divergences_fixture.entries.len()); + for divergence in divergences_fixture.entries { + assert!( + divergence.fixture_index < fixture.vectors.len(), + "divergence fixture index is out of range: {}", + divergence.fixture_index + ); + assert_ne!( + divergence.reference_outcome, divergence.native_outcome, + "divergence fixture entry is a no-op: {}", + divergence.fixture_index + ); + assert!( + divergences + .insert(divergence.fixture_index, divergence) + .is_none(), + "duplicate divergence fixture index" + ); + } + for (index, vector) in fixture.vectors.into_iter().enumerate() { + let expected = if let Some(divergence) = divergences.remove(&index) { + assert_eq!( + divergence.query, vector.query, + "divergence query does not match matching fixture at index {index}" + ); + let candidate_ids = vector + .candidates + .iter() + .map(|candidate| { + candidate + .id + .as_deref() + .expect("divergence candidates include ids") + }) + .collect::>(); + assert_eq!( + divergence.candidate_ids, candidate_ids, + "divergence candidates do not match matching fixture at index {index}" + ); + assert_eq!( + divergence.reference_outcome, vector.outcome, + "divergence reference does not match matching fixture at index {index}" + ); + divergence.native_outcome + } else { + vector.outcome.clone() + }; let candidates: Vec = vector .candidates .into_iter() @@ -268,7 +324,7 @@ fn matching_vectors_match_fixture() { .collect(); let result = find_matching_entity(&vector.query, &candidates, fixture.fuzzy_threshold); - if !vector.outcome.matched { + if !expected.matched { assert_eq!(result, None, "{:?}", vector.query); continue; } @@ -276,8 +332,7 @@ fn matching_vectors_match_fixture() { let result = result.expect("matched fixture vector resolves a candidate"); assert_eq!( result.candidate_index, - vector - .outcome + expected .candidate_index .expect("matched fixture vector has candidate index"), "{:?}", @@ -285,17 +340,13 @@ fn matching_vectors_match_fixture() { ); assert_eq!( result.tier as u8, - vector - .outcome - .tier - .expect("matched fixture vector has tier"), + expected.tier.expect("matched fixture vector has tier"), "{:?}", vector.query ); assert_eq!( result.tier.is_high_confidence(), - vector - .outcome + expected .high_confidence .expect("matched fixture vector has confidence"), "{:?}", @@ -313,6 +364,10 @@ fn matching_vectors_match_fixture() { result.tier as u8 <= fixture.high_confidence_max_tier ); } + assert!( + divergences.is_empty(), + "divergence fixture index was not visited" + ); } #[test] diff --git a/core/crates/solstone-core-entity/src/matcher.rs b/core/crates/solstone-core-entity/src/matcher.rs index de620d6ae..6ac6792fc 100644 --- a/core/crates/solstone-core-entity/src/matcher.rs +++ b/core/crates/solstone-core-entity/src/matcher.rs @@ -22,6 +22,7 @@ use std::cmp::Ordering; use std::collections::BTreeMap; use std::collections::btree_map::Entry; +use crate::normalize_resolution_query; use crate::slug::entity_slug; #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] @@ -66,7 +67,7 @@ pub fn find_matching_entity( return None; } - let detected_lower = detected_name.to_lowercase(); + let detected_lower = normalize_resolution_query(detected_name); let detected_slug = entity_slug(detected_name); let mut exact_case_map: BTreeMap = BTreeMap::new(); @@ -84,7 +85,7 @@ pub fn find_matching_entity( } exact_case_map.insert(name.to_string(), candidate_index); - lower_map.insert(name.to_lowercase(), candidate_index); + lower_map.insert(normalize_resolution_query(name), candidate_index); if entity_id.is_empty() { let name_slug = entity_slug(name); @@ -93,20 +94,20 @@ pub fn find_matching_entity( } } else { exact_case_map.insert(entity_id.to_string(), candidate_index); - lower_map.insert(entity_id.to_lowercase(), candidate_index); + lower_map.insert(normalize_resolution_query(entity_id), candidate_index); id_map.insert(entity_id.to_string(), candidate_index); } for aka in &candidate.aka { if !aka.is_empty() { exact_case_map.insert(aka.clone(), candidate_index); - lower_map.insert(aka.to_lowercase(), candidate_index); + lower_map.insert(normalize_resolution_query(aka), candidate_index); } } for email in &candidate.emails { if !email.is_empty() { - email_map.insert(email.to_lowercase(), candidate_index); + email_map.insert(normalize_resolution_query(email), candidate_index); } } @@ -153,7 +154,7 @@ pub fn find_matching_entity( } if let Some(detected_first) = detected_name.split_whitespace().next() { - let detected_first = detected_first.to_lowercase(); + let detected_first = normalize_resolution_query(detected_first); if detected_first != detected_lower && char_len(&detected_first) >= 3 && let Some(matches) = first_word_map.get(&detected_first) @@ -174,8 +175,11 @@ pub fn find_matching_entity( if candidate.name.is_empty() { return None; } - token_subset_match(&detected_lower, &candidate.name.to_lowercase()) - .then_some(candidate_index) + token_subset_match( + &detected_lower, + &normalize_resolution_query(&candidate.name), + ) + .then_some(candidate_index) }) .collect(); if subset_matches.len() == 1 { @@ -189,8 +193,11 @@ pub fn find_matching_entity( if candidate.name.is_empty() { return None; } - prefix_token_match(&detected_lower, &candidate.name.to_lowercase()) - .then_some(candidate_index) + prefix_token_match( + &detected_lower, + &normalize_resolution_query(&candidate.name), + ) + .then_some(candidate_index) }) .collect(); if prefix_matches.len() == 1 { @@ -214,7 +221,7 @@ fn entity_match(candidate_index: usize, tier: MatchTier) -> EntityNameMatch { } } -fn token_subset_match(name_a_lower: &str, name_b_lower: &str) -> bool { +pub(crate) fn token_subset_match(name_a_lower: &str, name_b_lower: &str) -> bool { let tokens_a: Vec<&str> = unique_sorted_tokens(name_a_lower); let tokens_b: Vec<&str> = unique_sorted_tokens(name_b_lower); let (shorter, longer) = match tokens_a.len().cmp(&tokens_b.len()) { @@ -224,7 +231,7 @@ fn token_subset_match(name_a_lower: &str, name_b_lower: &str) -> bool { shorter.len() >= 2 && shorter.iter().all(|token| longer.contains(token)) } -fn prefix_token_match(name_a_lower: &str, name_b_lower: &str) -> bool { +pub(crate) fn prefix_token_match(name_a_lower: &str, name_b_lower: &str) -> bool { let mut sorted_a: Vec<&str> = name_a_lower.split_whitespace().collect(); let mut sorted_b: Vec<&str> = name_b_lower.split_whitespace().collect(); sorted_a.sort_unstable(); @@ -240,15 +247,18 @@ fn prefix_token_match(name_a_lower: &str, name_b_lower: &str) -> bool { } fn first_word_key(name: &str) -> Option { - let first_word = name.split_whitespace().next()?.to_lowercase(); + let first_word = normalize_resolution_query(name) + .split_whitespace() + .next()? + .to_owned(); (char_len(&first_word) >= 3).then_some(first_word) } -fn first_word_match(query_lower: &str, entity_name: &str) -> bool { +pub(crate) fn first_word_match(query_lower: &str, entity_name: &str) -> bool { char_len(query_lower) >= 3 && first_word_key(entity_name).as_deref() == Some(query_lower) } -fn single_token_first_word_match(query_first: &str, entity_name: &str) -> bool { +pub(crate) fn single_token_first_word_match(query_first: &str, entity_name: &str) -> bool { !entity_name.is_empty() && entity_name.split_whitespace().count() == 1 && first_word_match(query_first, entity_name) @@ -261,7 +271,7 @@ fn unique_sorted_tokens(text: &str) -> Vec<&str> { tokens } -fn token_sort(text: &str) -> String { +pub(crate) fn token_sort(text: &str) -> String { let mut tokens: Vec<&str> = text.split_whitespace().collect(); tokens.sort_unstable(); tokens.join(" ") @@ -287,7 +297,7 @@ fn extract_one_fuzzy( best.map(|(_score, candidate_index)| candidate_index) } -fn char_len(text: &str) -> usize { +pub(crate) fn char_len(text: &str) -> usize { text.chars().count() } @@ -450,6 +460,26 @@ mod tests { ); } + #[test] + fn unified_normalization_resolves_opaque_unicode_pairs_at_tier_2() { + let cases = [ + ("Straße Handel", "STRASSE HANDEL"), + ("ΟΔΥΣΣΕΥΣ", "οδυσσευσ"), + ("firefly labs", "firefly labs"), + ]; + for (query, name) in cases { + let candidates = [candidate(Some("xx_opaque_identity"), name, &[], &[])]; + assert_match( + query, + &candidates, + 90.0, + 0, + Some("xx_opaque_identity"), + MatchTier::CaseInsensitive, + ); + } + } + #[test] fn email_matches_tier_3_when_query_contains_at() { let candidates = [candidate( @@ -816,10 +846,24 @@ mod tests { } #[test] - fn leading_and_trailing_space_queries_miss_first_word() { + fn leading_and_trailing_space_queries_match_first_word_after_normalization() { let candidates = [candidate(Some("jg"), "Javier Garcia", &[], &[])]; - assert_no_match(" Javier", &candidates, 100.0); - assert_no_match("Javier ", &candidates, 100.0); + assert_match( + " Javier", + &candidates, + 100.0, + 0, + Some("jg"), + MatchTier::FirstWord, + ); + assert_match( + "Javier ", + &candidates, + 100.0, + 0, + Some("jg"), + MatchTier::FirstWord, + ); } #[test] diff --git a/core/crates/solstone-core-entity/src/test_support.rs b/core/crates/solstone-core-entity/src/test_support.rs index 160d8f64d..af1dd067e 100644 --- a/core/crates/solstone-core-entity/src/test_support.rs +++ b/core/crates/solstone-core-entity/src/test_support.rs @@ -22,6 +22,10 @@ const ENTITY_NORMALIZATION_DIVERGENCES_FIXTURE: &str = include_str!(concat!( env!("CARGO_MANIFEST_DIR"), "/../../fixtures/entity_normalization_native_divergences.json" )); +const ENTITY_MATCHING_DIVERGENCES_FIXTURE: &str = include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../fixtures/entity_matching_native_divergences.json" +)); #[derive(Debug, Deserialize)] pub(crate) struct EntityIdentityFixture { @@ -111,7 +115,7 @@ pub(crate) struct MatchingCandidate { pub(crate) emails: Vec, } -#[derive(Debug, Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, Deserialize)] pub(crate) struct MatchingOutcome { pub(crate) matched: bool, pub(crate) candidate_index: Option, @@ -119,6 +123,28 @@ pub(crate) struct MatchingOutcome { pub(crate) high_confidence: Option, } +#[derive(Debug, Deserialize)] +pub(crate) struct MatchingDivergencesFixture { + pub(crate) counts: MatchingDivergenceCounts, + pub(crate) entries: Vec, +} + +#[derive(Debug, Deserialize)] +pub(crate) struct MatchingDivergenceCounts { + pub(crate) total: usize, + pub(crate) tier_changes: usize, + pub(crate) refusal_to_match: usize, +} + +#[derive(Debug, Deserialize)] +pub(crate) struct MatchingDivergenceRecord { + pub(crate) candidate_ids: Vec, + pub(crate) fixture_index: usize, + pub(crate) reference_outcome: MatchingOutcome, + pub(crate) native_outcome: MatchingOutcome, + pub(crate) query: String, +} + #[derive(Debug, Deserialize)] pub(crate) struct SlugDivergencesFixture { pub(crate) counts: DivergenceCounts, @@ -175,6 +201,11 @@ pub(crate) fn entity_matching_fixture() -> EntityMatchingFixture { serde_json::from_str(ENTITY_MATCHING_FIXTURE).expect("parse entity matching fixture") } +pub(crate) fn matching_divergences_fixture() -> MatchingDivergencesFixture { + serde_json::from_str(ENTITY_MATCHING_DIVERGENCES_FIXTURE) + .expect("parse entity matching divergences fixture") +} + pub(crate) fn slug_divergences_fixture() -> SlugDivergencesFixture { serde_json::from_str(ENTITY_SLUG_DIVERGENCES_FIXTURE) .expect("parse entity slug divergences fixture") diff --git a/core/fixtures/entity_matching_native_divergences.json b/core/fixtures/entity_matching_native_divergences.json new file mode 100644 index 000000000..6fc0f5c96 --- /dev/null +++ b/core/fixtures/entity_matching_native_divergences.json @@ -0,0 +1,309 @@ +{ + "cause": "The native matcher now uses NFKC, Python-compatible whitespace collapse, and full Unicode case folding for its tier-2, tier-3, tier-5, tier-6, and tier-7 comparisons. The recorded matching fixture remains the pre-unification reference, so entries here state the deliberate native answers where that reference differs.", + "counts": { + "refusal_to_match": 3, + "tier_changes": 12, + "total": 15 + }, + "entries": [ + { + "candidate_ids": [ + "strasse_handels_gmbh" + ], + "fixture_index": 19, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "Straße Handels GmbH", + "reason": "U+00DF sharp s full-case-folds to ss, so tier 2 matches the candidate instead of falling through to slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "muller_werke" + ], + "fixture_index": 20, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "Müller Werke", + "reason": "NFKC composes the decomposed umlaut before the tier-2 comparison, so the candidate matches before slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "strasse_handel" + ], + "fixture_index": 21, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "Straße Handel", + "reason": "U+00DF sharp s full-case-folds to ss, so tier 2 matches the uppercase expansion instead of slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "strasse_handel", + "zeta_holdings" + ], + "fixture_index": 22, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "Straße Handel", + "reason": "U+00DF sharp s full-case-folds to ss, so tier 2 selects the same first candidate before slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "strasse_handel" + ], + "fixture_index": 23, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "STRASSE HANDEL", + "reason": "U+00DF sharp s full-case-folds to ss, so tier 2 matches the reverse pair instead of slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "strasse_handel", + "zeta_holdings" + ], + "fixture_index": 24, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "STRASSE HANDEL", + "reason": "U+00DF sharp s full-case-folds to ss, so tier 2 selects the same first candidate before slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "firefly_labs" + ], + "fixture_index": 25, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "firefly labs", + "reason": "NFKC expands the U+FB01 fi ligature before tier 2, so the match no longer waits for slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "firefly_labs", + "zeta_holdings" + ], + "fixture_index": 26, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "firefly labs", + "reason": "NFKC expands the U+FB01 fi ligature before tier 2, preserving the selected candidate while bypassing slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "ffluent_works" + ], + "fixture_index": 27, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "ffluent works", + "reason": "NFKC expands the U+FB04 ffl ligature before tier 2, so the match no longer waits for slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "ffluent_works", + "zeta_holdings" + ], + "fixture_index": 28, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "ffluent works", + "reason": "NFKC expands the U+FB04 ffl ligature before tier 2, preserving the selected candidate while bypassing slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "m_lab_research" + ], + "fixture_index": 29, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "µ-lab research", + "reason": "NFKC maps U+00B5 micro sign to Greek mu before tier 2, so the match no longer waits for slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "m_lab_research", + "zeta_holdings" + ], + "fixture_index": 30, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "µ-lab research", + "reason": "NFKC maps U+00B5 micro sign to Greek mu before tier 2, preserving the selected candidate while bypassing slug tier 4.", + "reference_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 4 + } + }, + { + "candidate_ids": [ + "xx_opaque_identity" + ], + "fixture_index": 31, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "Straße Handel", + "reason": "U+00DF sharp s full-case-folds to ss, turning the former non-slug refusal into a tier-2 match.", + "reference_outcome": { + "matched": false + } + }, + { + "candidate_ids": [ + "xx_opaque_identity" + ], + "fixture_index": 32, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "ΟΔΥΣΣΕΥΣ", + "reason": "Full case folding maps Greek final sigma and medial sigma to the same comparison form, turning the former non-slug refusal into a tier-2 match.", + "reference_outcome": { + "matched": false + } + }, + { + "candidate_ids": [ + "xx_opaque_identity" + ], + "fixture_index": 33, + "native_outcome": { + "candidate_index": 0, + "high_confidence": true, + "matched": true, + "tier": 2 + }, + "query": "firefly labs", + "reason": "NFKC expands the U+FB01 fi ligature, turning the former non-slug refusal into a tier-2 match.", + "reference_outcome": { + "matched": false + } + } + ], + "note": "Deliberate native matcher divergences from entity_matching.json after matcher comparison keys unified on normalize_resolution_query. The existing matching fixture remains the pre-unification reference corpus.", + "when_this_file_reddens": "A later failure is a finding that matcher tier behavior changed again. Re-evaluate every matching vector and make a deliberate decision; do not silently regenerate this file to absorb the change.", + "why_the_per_entry_assertion_is_required": "The declared total can still be correct when a divergence points at the wrong vector, records the wrong reference outcome, or describes a native answer that no longer occurs. Each entry must therefore be checked against both the frozen reference vector and the native matcher result, so the narrow confidence-boundary changes cannot be hidden by an aggregate count." +} -- 2.51.2