diff --git a/core/fixtures/entity_matching.json b/core/fixtures/entity_matching.json index 4400bf6f1..a76a20b3f 100644 --- a/core/fixtures/entity_matching.json +++ b/core/fixtures/entity_matching.json @@ -414,7 +414,7 @@ "type": "Person" } ], - "note": "case-fold divergent \u2014 German sharp s against its uppercase expansion", + "note": "case-operation \u2014 German sharp s against its uppercase expansion", "outcome": { "candidate_index": 0, "high_confidence": true, @@ -436,7 +436,7 @@ "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 German sharp s against its uppercase expansion", + "note": "case-operation with competition \u2014 German sharp s against its uppercase expansion", "outcome": { "candidate_index": 0, "high_confidence": true, @@ -453,7 +453,7 @@ "type": "Person" } ], - "note": "case-fold divergent \u2014 the same pair, reversed", + "note": "case-operation \u2014 the same pair, reversed", "outcome": { "candidate_index": 0, "high_confidence": true, @@ -475,7 +475,7 @@ "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 the same pair, reversed", + "note": "case-operation with competition \u2014 the same pair, reversed", "outcome": { "candidate_index": 0, "high_confidence": true, @@ -487,25 +487,25 @@ { "candidates": [ { - "id": "odusseus_shipping", - "name": "\u03bf\u03b4\u03c5\u03c3\u03c3\u03b5\u03c5\u03c2 Shipping", + "id": "firefly_labs", + "name": "firefly labs", "type": "Person" } ], - "note": "case-fold divergent \u2014 Greek final sigma", + "note": "case-operation \u2014 the fi ligature against its expansion", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, - "tier": 2 + "tier": 4 }, - "query": "\u039f\u0394\u03a5\u03a3\u03a3\u0395\u03a5\u03a3 Shipping" + "query": "\ufb01refly labs" }, { "candidates": [ { - "id": "odusseus_shipping", - "name": "\u03bf\u03b4\u03c5\u03c3\u03c3\u03b5\u03c5\u03c2 Shipping", + "id": "firefly_labs", + "name": "firefly labs", "type": "Person" }, { @@ -514,37 +514,37 @@ "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 Greek final sigma", + "note": "case-operation with competition \u2014 the fi ligature against its expansion", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, - "tier": 2 + "tier": 4 }, - "query": "\u039f\u0394\u03a5\u03a3\u03a3\u0395\u03a5\u03a3 Shipping" + "query": "\ufb01refly labs" }, { "candidates": [ { - "id": "odusseus_shipping", - "name": "\u039f\u0394\u03a5\u03a3\u03a3\u0395\u03a5\u03a3 Shipping", + "id": "ffluent_works", + "name": "ffluent works", "type": "Person" } ], - "note": "case-fold divergent \u2014 Greek final sigma, reversed", + "note": "case-operation \u2014 the ffl ligature against its expansion", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, - "tier": 2 + "tier": 4 }, - "query": "\u03bf\u03b4\u03c5\u03c3\u03c3\u03b5\u03c5\u03c2 Shipping" + "query": "\ufb04uent works" }, { "candidates": [ { - "id": "odusseus_shipping", - "name": "\u039f\u0394\u03a5\u03a3\u03a3\u0395\u03a5\u03a3 Shipping", + "id": "ffluent_works", + "name": "ffluent works", "type": "Person" }, { @@ -553,37 +553,37 @@ "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 Greek final sigma, reversed", + "note": "case-operation with competition \u2014 the ffl ligature against its expansion", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, - "tier": 2 + "tier": 4 }, - "query": "\u03bf\u03b4\u03c5\u03c3\u03c3\u03b5\u03c5\u03c2 Shipping" + "query": "\ufb04uent works" }, { "candidates": [ { - "id": "firefly_labs", - "name": "firefly labs", + "id": "m_lab_research", + "name": "\u03bc-lab research", "type": "Person" } ], - "note": "case-fold divergent \u2014 the fi ligature against its expansion", + "note": "case-operation \u2014 micro sign against Greek mu", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, "tier": 4 }, - "query": "\ufb01refly labs" + "query": "\u00b5-lab research" }, { "candidates": [ { - "id": "firefly_labs", - "name": "firefly labs", + "id": "m_lab_research", + "name": "\u03bc-lab research", "type": "Person" }, { @@ -592,164 +592,174 @@ "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 the fi ligature against its expansion", + "note": "case-operation with competition \u2014 micro sign against Greek mu", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, "tier": 4 }, - "query": "\ufb01refly labs" + "query": "\u00b5-lab research" }, { "candidates": [ { - "id": "ffluent_works", - "name": "ffluent works", + "id": "xx_opaque_identity", + "name": "STRASSE HANDEL", "type": "Person" } ], - "note": "case-fold divergent \u2014 the ffl ligature against its expansion", + "note": "case-operation across the confidence boundary \u2014 German sharp s, slug tier disabled", "outcome": { - "candidate_index": 0, - "high_confidence": true, - "matched": true, - "tier": 4 + "matched": false }, - "query": "\ufb04uent works" + "query": "Stra\u00dfe Handel" }, { "candidates": [ { - "id": "ffluent_works", - "name": "ffluent works", + "id": "xx_opaque_identity", + "name": "\u03bf\u03b4\u03c5\u03c3\u03c3\u03b5\u03c5\u03c3", "type": "Person" - }, + } + ], + "note": "case-operation across the confidence boundary \u2014 Greek medial sigma, slug tier disabled", + "outcome": { + "matched": false + }, + "query": "\u039f\u0394\u03a5\u03a3\u03a3\u0395\u03a5\u03a3" + }, + { + "candidates": [ { - "id": "zeta_holdings", - "name": "Zeta Holdings", + "id": "xx_opaque_identity", + "name": "firefly labs", "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 the ffl ligature against its expansion", + "note": "case-operation across the confidence boundary \u2014 the fi ligature, slug tier disabled", + "outcome": { + "matched": false + }, + "query": "\ufb01refly labs" + }, + { + "candidates": [ + { + "id": "trading", + "name": "\uabb3\uab83\uab79 Trading", + "type": "Person" + } + ], + "note": "case-neutral control \u2014 Cherokee \u2014 case-neutral under both operations", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, - "tier": 4 + "tier": 2 }, - "query": "\ufb04uent works" + "query": "\u13e3\u13b3\u13a9 Trading" }, { "candidates": [ { - "id": "m_lab_research", - "name": "\u03bc-lab research", + "id": "istanbul_works", + "name": "istanbul works", "type": "Person" } ], - "note": "case-fold divergent \u2014 micro sign against Greek mu", + "note": "case-neutral control \u2014 Turkish dotted capital I \u2014 neutral outside the Turkic profile", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, "tier": 4 }, - "query": "\u00b5-lab research" + "query": "\u0130stanbul Works" }, { "candidates": [ { - "id": "m_lab_research", - "name": "\u03bc-lab research", - "type": "Person" - }, - { - "id": "zeta_holdings", - "name": "Zeta Holdings", + "id": "odusseus_shipping", + "name": "\u03bf\u03b4\u03c5\u03c3\u03c3\u03b5\u03c5\u03c2 Shipping", "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 micro sign against Greek mu", + "note": "case-neutral control \u2014 Greek FINAL sigma \u2014 neutral; the medial form is the divergent one", "outcome": { "candidate_index": 0, "high_confidence": true, "matched": true, - "tier": 4 + "tier": 2 }, - "query": "\u00b5-lab research" + "query": "\u039f\u0394\u03a5\u03a3\u03a3\u0395\u03a5\u03a3 Shipping" }, { "candidates": [ { - "id": "trading", - "name": "\uabb3\uab85\uab79 Trading", + "id": "bartholomew_kensington", + "name": "Bartholomew Kensington", "type": "Person" } ], - "note": "case-fold divergent \u2014 Cherokee, which folds toward uppercase", + "note": "tier 7 \u2014 prefix token, same token count, unambiguous", "outcome": { - "matched": false + "candidate_index": 0, + "high_confidence": false, + "matched": true, + "tier": 7 }, - "query": "\u13e3\u13b3\u13a9 Trading" + "query": "Barth Kensington" }, { "candidates": [ { - "id": "trading", - "name": "\uabb3\uab85\uab79 Trading", + "id": "bartholomew_kensington", + "name": "Bartholomew Kensington", "type": "Person" }, { - "id": "zeta_holdings", - "name": "Zeta Holdings", + "id": "barthelemy_kensington", + "name": "Barthelemy Kensington", "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 Cherokee, which folds toward uppercase", + "note": "REFUSAL \u2014 prefix token is ambiguous across two candidates", "outcome": { "matched": false }, - "query": "\u13e3\u13b3\u13a9 Trading" + "query": "Barth Kensington" }, { "candidates": [ { - "id": "istanbul_works", - "name": "istanbul works", + "id": "jonathan_smythe", + "name": "Jonathan Smythe", "type": "Person" } ], - "note": "case-fold divergent \u2014 Turkish dotted capital I", + "note": "tier 8 \u2014 fuzzy, above the threshold", "outcome": { "candidate_index": 0, - "high_confidence": true, + "high_confidence": false, "matched": true, - "tier": 4 + "tier": 8 }, - "query": "\u0130stanbul Works" + "query": "Jonathon Smythe" }, { "candidates": [ { - "id": "istanbul_works", - "name": "istanbul works", - "type": "Person" - }, - { - "id": "zeta_holdings", - "name": "Zeta Holdings", + "id": "jonathan_smythe", + "name": "Jonathan Smythe", "type": "Person" } ], - "note": "case-fold divergent with competition \u2014 Turkish dotted capital I", + "note": "tier 8 \u2014 fuzzy, far below the threshold, no match", "outcome": { - "candidate_index": 0, - "high_confidence": true, - "matched": true, - "tier": 4 + "matched": false }, - "query": "\u0130stanbul Works" + "query": "Zxqvwm Ptkkkl" }, { "candidates": [ diff --git a/core/fixtures/entity_normalization_native_divergences.json b/core/fixtures/entity_normalization_native_divergences.json index bdaf365e2..df7fd2d2e 100644 --- a/core/fixtures/entity_normalization_native_divergences.json +++ b/core/fixtures/entity_normalization_native_divergences.json @@ -457,5 +457,6 @@ "note": "Codepoints where the native normalize_resolution_query intentionally differs from the reference implementation. NOT produced by `make core-fixtures` -- it needs both implementations, and the fixture generator only has one. The native test recomputes the whole sweep, substitutes reference_normalized for each codepoint listed here, and asserts the digest equals the reference digest in entity_identity.json. It ALSO asserts, per entry and before any substitution, that the native result equals native_normalized -- otherwise the substitution would discard the implementation's output on exactly the codepoints most likely to be wrong.", "reachability": "EVERY entry is unassigned in the reference's Unicode revision, so no such character can appear in a name typed today, and none can already be stored. The set is expected to SHRINK toward empty as the reference revision advances -- an entry disappearing is convergence, not regression, but it must still be a deliberate edit rather than a silent one, which is what the per-entry assertion enforces.", "reference_unicode_version": "15.1.0", + "when_this_file_reddens": "TWO dependencies jointly pin this set -- the compatibility normalization and the case fold carry separate Unicode table revisions, and 37 of these entries involve a normalization change while 54 involve a fold change. A bump to either is the expected trigger, and reddening is the pin working. The repair is to rebuild both implementations, re-diff the full sweep, and rewrite this file wholesale. Hand-editing a single entry to make a digest agree is never the repair -- it edits the oracle to match the thing it exists to check.", "why_not_pinned_to_the_reference": "Matching the older tables would mean hand-special-casing these codepoints inside the function that derives ambiguity identity. That function's output is hashed into an id an owner's recorded answers are found by, so it is the last place to accumulate special cases for characters nobody can type." } diff --git a/core/fixtures/entity_slug_native_divergences.json b/core/fixtures/entity_slug_native_divergences.json index 51143c261..c24cf9b53 100644 --- a/core/fixtures/entity_slug_native_divergences.json +++ b/core/fixtures/entity_slug_native_divergences.json @@ -1,4 +1,5 @@ { + "backfill_scope": "The backfill obligation is the ASSIGNED count, not the letter count. Letters are the vivid cases but they are a minority: the assigned set also includes the 25 C1 controls -- the mojibake path, exactly how a Windows-1252 round trip arrives in imported data -- nine modifier tone bars used in linguistic transcription, a currency sign, and an Arabic ligature. Scoping a backfill from the letter count alone misses most of it, in a store where the failure mode is permanent.", "cause": "The two implementations transliterate through different data tables, and the native one carries the newer set. This is a data difference, not a logic difference, which is why it is invisible to review and has to be pinned by a sweep.", "counts": { "assigned_in_reference_unicode": 62, @@ -701,8 +702,9 @@ } ], "note": "Codepoints where the native entity_slug intentionally differs from the reference implementation. NOT produced by `make core-fixtures` -- it needs both implementations, and the fixture generator only has one. The native test recomputes the whole sweep, substitutes reference_slug for each codepoint listed here, and asserts the resulting digest equals the reference digest in entity_identity.json. That catches a divergence appearing at any codepoint NOT listed here.", - "reachability": "A slug is a directory label, and identity is written beside the data rather than derived from it, so a differing slug does not fork an identity. The reachable surface is the tier-4 lookup, where a query typed as an id may fall through to the fuzzy tier for one of these names; a query typed as the NAME still matches at the earlier tiers.", + "reachability": "A journal-level slug is a directory label, and identity is written beside the data rather than derived from it, so a differing slug does not fork a journal-level identity. Two surfaces are genuinely reachable. (1) The tier-4 lookup: a query typed as an id may fall through to the fuzzy tier for one of these names, though a query typed as the NAME still matches at the earlier tiers. (2) Facet-scoped entity memory, which keys its directory by a slug RE-DERIVED from the name on every access -- for these codepoints that directory stops being derivable once the native implementation is authoritative, and it is a first-run backfill's job to fix.", "reference_unicode_version": "15.1.0", + "when_this_file_reddens": "Bumping either the transliteration or the normalization dependency is the expected trigger, and reddening is the pin working. The repair is to rebuild BOTH implementations, re-diff the full sweep, and rewrite this file wholesale. Hand-editing a single entry to make a digest agree is never the repair -- it edits the oracle to match the thing the oracle exists to check.", "why_the_newer_table_wins": "The older table maps several living-script letters to nothing, so two distinct names collapse to the same slug -- and a facet-scoped save then raises `names may slugify to same value` rather than storing them. Cherokee is the clearest case: seven of its letters are in this list. Preserving the older behaviour would preserve that collision.", "why_the_per_entry_assertion_is_required": "The digest alone cannot see these 99 at all: their output is substituted away before hashing, so an implementation that regressed to the older table -- or returned an empty string -- would still produce the reference digest. The 99 listed here are precisely the codepoints most likely to be wrong, so the test MUST also assert, per entry and before any substitution, that the implementation's own result equals native_slug, and that native_slug and reference_slug differ. That is what makes a listed divergence disappearing go red." } diff --git a/scripts/entity_corpus.py b/scripts/entity_corpus.py index 583024775..dfc66acbd 100644 --- a/scripts/entity_corpus.py +++ b/scripts/entity_corpus.py @@ -351,26 +351,70 @@ def _match_cases() -> list[dict[str, Any]]: ): case(a, [_candidate(b)], f"unicode pair — {a!r} against {b!r}") - # 🔴 Case-fold-divergent pairs. These exist to pin which case operation the - # matcher uses. Simple lowercasing and full case folding disagree on every - # pair below, and the disagreement lands on the tier — which is to say, on - # whether the store resolves silently or stops and asks the owner. Without - # them nothing in the corpus can tell the two operations apart, because - # every other vector here is case-fold-neutral. + # 🔴 Case-operation vectors. These exist to pin *which* case operation the + # matcher applies, because simple lowercasing and full case folding are + # different functions and the difference lands on the tier — which is to + # say, on whether the store resolves silently or stops and asks the owner. + # Without them the corpus cannot tell the two apart: every other vector + # here is case-neutral. for a, b, note in ( ("Straße Handel", "STRASSE HANDEL", "German sharp s against its uppercase expansion"), ("STRASSE HANDEL", "Straße Handel", "the same pair, reversed"), - ("ΟΔΥΣΣΕΥΣ Shipping", "οδυσσευς Shipping", "Greek final sigma"), - ("οδυσσευς Shipping", "ΟΔΥΣΣΕΥΣ Shipping", "Greek final sigma, reversed"), ("firefly labs", "firefly labs", "the fi ligature against its expansion"), ("ffluent works", "ffluent works", "the ffl ligature against its expansion"), ("µ-lab research", "μ-lab research", "micro sign against Greek mu"), - ("ᏣᎳᎩ Trading", "ꮳꮅꭹ Trading", "Cherokee, which folds toward uppercase"), - ("İstanbul Works", "istanbul works", "Turkish dotted capital I"), ): - case(a, [_candidate(b)], f"case-fold divergent — {note}") + case(a, [_candidate(b)], f"case-operation — {note}") case(a, [_candidate(b), _candidate("Zeta Holdings")], - f"case-fold divergent with competition — {note}") + f"case-operation with competition — {note}") + + # 🔴 The same phenomena, with the slug tier DISABLED. The vectors above all + # resolve through the slug tier, because a candidate's id is the slug of its + # own name and transliteration already erases the distinction — so swapping + # the case operation only moves them between two high-confidence tiers. + # Giving the candidate an id that is *not* the slug of its name removes that + # path, and the case operation alone then decides between "no match at all" + # and a high-confidence match. That is the tier-4 boundary — the line + # between asking the owner and deciding for them — and nothing else in this + # corpus crosses it. + for a, b, note in ( + ("Straße Handel", "STRASSE HANDEL", "German sharp s, slug tier disabled"), + ("ΟΔΥΣΣΕΥΣ", "οδυσσευσ", "Greek medial sigma, slug tier disabled"), + ("firefly labs", "firefly labs", "the fi ligature, slug tier disabled"), + ): + cases.append( + { + "query": a, + "candidates": [ + {"id": "xx_opaque_identity", "name": b, "type": "Person"} + ], + "note": f"case-operation across the confidence boundary — {note}", + } + ) + + # Controls: real case pairs that are case-neutral, kept so the corpus does + # not imply coverage it lacks. Cherokee is the interesting one — its cases + # agree under BOTH operations, so it cannot produce a divergence here at + # all, and a reader who assumes otherwise will misread what is covered. + for a, b, note in ( + ("ᏣᎳᎩ Trading", "ꮳꮃꭹ Trading", "Cherokee — case-neutral under both operations"), + ("İstanbul Works", "istanbul works", "Turkish dotted capital I — neutral outside the Turkic profile"), + ("ΟΔΥΣΣΕΥΣ Shipping", "οδυσσευς Shipping", "Greek FINAL sigma — neutral; the medial form is the divergent one"), + ): + case(a, [_candidate(b)], f"case-neutral control — {note}") + + # Tiers 7 and 8 are otherwise unrepresented, so a move that hollowed them + # out would keep every count and drop two tiers. + # Tier 7 requires the SAME token count, pairwise equal or a >=4-char prefix. + case("Barth Kensington", [_candidate("Bartholomew Kensington")], + "tier 7 — prefix token, same token count, unambiguous") + case("Barth Kensington", + [_candidate("Bartholomew Kensington"), _candidate("Barthelemy Kensington")], + "REFUSAL — prefix token is ambiguous across two candidates") + case("Jonathon Smythe", [_candidate("Jonathan Smythe")], + "tier 8 — fuzzy, above the threshold") + case("Zxqvwm Ptkkkl", [_candidate("Jonathan Smythe")], + "tier 8 — fuzzy, far below the threshold, no match") # Larger populations, so ambiguity detection sees real competition. for name in _MATCH_NAMES: case(name, [_candidate(n) for n in _MATCH_NAMES], "full population")