From d63beb49ccf11bdc7e5c537cccb1445985efcc6b Mon Sep 17 00:00:00 2001 From: zzstoatzz Date: Tue, 26 May 2026 23:38:50 -0500 Subject: [PATCH] index: v2 CJK/Japanese key derivation (normalize + shared bigram tokenizer) normalize.zig: preserve Han + Hiragana, fold katakana->hiragana + full-width ASCII->ASCII (NFKC-lite); Latin fold table unchanged; VERSION -> v2-cjk-bigram (forces re-index via the snapshot loader's normalize_version gate). prefixes.zig: one shared UTF-8-codepoint-safe tokenizer (emitTokenKeys) used by BOTH builder (displayPrefixKeys) and serving (queryKeys). Segments Latin vs CJK runs; CJK runs emit adjacent bigrams + a leading unigram (no full unigram indexing); Latin runs unchanged. 100/100 index tests pass. Refs docs/i18n-cjk-search.md. Not yet: scoring boosts, manifest cjk_key_count + promote floor gate, full re-index (separate go/no-go). Co-Authored-By: Claude Opus 4.7 (1M context) --- services/src/index/normalize.zig | 180 ++++++++++++++++++++------- services/src/index/prefixes.zig | 203 ++++++++++++++++++++++++++++--- 2 files changed, 325 insertions(+), 58 deletions(-) diff --git a/services/src/index/normalize.zig b/services/src/index/normalize.zig index 80c9525..855987b 100644 --- a/services/src/index/normalize.zig +++ b/services/src/index/normalize.zig @@ -25,18 +25,21 @@ //! `_` because display names commonly have it ("alice_b") and we //! tokenize on whitespace later anyway. //! -//! Scope (v1): ASCII letters/digits + a hand-rolled fold table for +//! Scope (v2): ASCII letters/digits + a hand-rolled fold table for //! Latin-1 Supplement (U+0080..U+00FF) and Latin Extended-A -//! (U+0100..U+017F). Covers the vast majority of accented Latin -//! characters in handles + display names. Everything outside that range -//! (CJK, Arabic, Hebrew, emoji, symbols, combining marks) is dropped. +//! (U+0100..U+017F), PLUS Japanese/CJK: Han, Hiragana, and Katakana +//! (folded to Hiragana), with NFKC-lite width folding (full-width ASCII → +//! ASCII, ideographic/middle-dot → space). CJK codepoints are PRESERVED +//! inline in the output; the prefix tokenizer (prefixes.zig) segments +//! script runs and bigrams CJK runs. See docs/i18n-cjk-search.md. //! -//! KNOWN LIMITATION: non-Latin scripts are unsearchable by handle or -//! display name. A Japanese display "佐藤" produces the empty string and -//! emits no `n:` prefix keys. This is acceptable for v1 (most atproto -//! handles are ASCII; display names searched by users are dominantly -//! Latin) but should be revisited as the network internationalizes. -//! Tracked in docs/typeahead-index-design.md. +//! STILL DROPPED: other non-Latin scripts (Arabic, Hebrew, Hangul, Thai…), +//! emoji, symbols, combining marks. Known gap: half-width katakana is kept +//! as-is (not cross-form folded to full-width). These are later versions. +//! +//! LOAD-BEARING: bumping the supported set or any fold changes `VERSION`, +//! which forces a full re-index (the snapshot loader rejects a snapshot +//! whose recorded normalize_version disagrees with the running binary). //! //! Output character sets: //! query(): a-z 0-9 . - _ plus single ' ' between tokens @@ -46,7 +49,7 @@ const std = @import("std"); const Allocator = std.mem.Allocator; -pub const VERSION: []const u8 = "v1-ascii-latin1"; +pub const VERSION: []const u8 = "v2-cjk-bigram"; /// Hard cap on normalized output length. Atproto handles are ≤253 chars; /// display names are ≤64 codepoints (≤256 bytes UTF-8). Queries arrive @@ -179,9 +182,79 @@ pub fn handle(arena: Allocator, raw: []const u8) ![]const u8 { return normalizeWith(arena, raw, .drop); } +// ─── v2: CJK (Japanese) support ──────────────────────────────────────────── +// Checked BEFORE the Latin fold path in normalizeWith. The Latin fold table +// above is unchanged; CJK/width handling lives here so the two stay decoupled. + +/// CJK codepoints we index. Han is preserved; katakana is folded to hiragana +/// (so サトウ and さとう collide); hiragana preserved. Returns the canonical +/// codepoint to emit, or null if `cp` is not CJK. NOTE: half-width katakana is +/// preserved as-is (not cross-form folded) — a known v2 gap, rare in names. +fn cjkFold(cp: u21) ?u21 { + if (cp >= 0x3041 and cp <= 0x3096) return cp; // hiragana + if (cp >= 0x30A1 and cp <= 0x30F6) return cp - 0x60; // katakana → hiragana + if (cp == 0x30FC) return cp; // ー prolonged sound mark (shared by kana) + if (cp >= 0xFF66 and cp <= 0xFF9D) return cp; // half-width katakana (as-is) + if (cp >= 0x3400 and cp <= 0x4DBF) return cp; // CJK Ext A + if (cp >= 0x4E00 and cp <= 0x9FFF) return cp; // CJK Unified Ideographs + if (cp >= 0xF900 and cp <= 0xFAFF) return cp; // CJK Compatibility Ideographs + return null; +} + +/// Full-width / ideographic forms that fold to a single ASCII byte (handled +/// before CJK + Latin). Full-width `A-Z0-9!…` → ASCII; ideographic and +/// katakana middle-dot spaces → ' ' (a separator). Returns the raw ASCII byte +/// (not yet lowercased) or null. +fn widthFoldAscii(cp: u21) ?u8 { + if (cp >= 0xFF01 and cp <= 0xFF5E) return @intCast(cp - 0xFEE0); // full-width ASCII + if (cp == 0x3000) return ' '; // ideographic space + if (cp == 0x30FB or cp == 0xFF65) return ' '; // ・ middle dot → token separator + return null; +} + +/// Classify + append one already-lowercased ASCII byte through the shared +/// separator/letter logic. Returns false when MAX_OUTPUT is reached (stop). +/// Used by both the width-fold path and the Latin fold loop — one code path, +/// no "almost-same" duplication. +fn emitAscii( + out: *std.ArrayListUnmanaged(u8), + c: u8, + underscore: UnderscorePolicy, + pending_space: *bool, + has_non_space: *bool, +) bool { + if (out.items.len >= MAX_OUTPUT) return false; + const cls = classifyAscii(c); + const effective: CharClass = switch (cls) { + .underscore => switch (underscore) { + .keep => .letter_or_digit, + .drop => .drop, + }, + else => cls, + }; + switch (effective) { + .drop => {}, + .separator => { + if (has_non_space.*) pending_space.* = true; + }, + .letter_or_digit, .handle_internal => { + if (pending_space.*) { + out.appendAssumeCapacity(' '); + pending_space.* = false; + } + out.appendAssumeCapacity(c); + has_non_space.* = true; + }, + .underscore => unreachable, // resolved above + } + return true; +} + fn normalizeWith(arena: Allocator, raw: []const u8, underscore: UnderscorePolicy) ![]const u8 { var out: std.ArrayListUnmanaged(u8) = .empty; - try out.ensureTotalCapacityPrecise(arena, @min(raw.len + 4, MAX_OUTPUT)); + // +8 slack so a multi-byte CJK codepoint appended at the MAX_OUTPUT + // boundary still fits before the next-iteration break. + try out.ensureTotalCapacityPrecise(arena, @min(raw.len + 4, MAX_OUTPUT + 8)); var pending_space = false; // becomes true when a separator runs; emitted lazily var has_non_space = false; // leading separators get dropped @@ -201,34 +274,31 @@ fn normalizeWith(arena: Allocator, raw: []const u8, underscore: UnderscorePolicy }; i += cp_len; - const folded = foldCodepoint(cp); - if (folded.len == 0) continue; + // 1. full-width / ideographic → ASCII, then shared classify path. + if (widthFoldAscii(cp)) |raw_c| { + const c = if (raw_c >= 'A' and raw_c <= 'Z') raw_c + 32 else raw_c; + if (!emitAscii(&out, c, underscore, &pending_space, &has_non_space)) break; + continue; + } - for (folded.bytes[0..folded.len]) |c| { - if (out.items.len >= MAX_OUTPUT) break; - const cls = classifyAscii(c); - const effective: CharClass = switch (cls) { - .underscore => switch (underscore) { - .keep => .letter_or_digit, - .drop => .drop, - }, - else => cls, - }; - switch (effective) { - .drop => {}, - .separator => { - if (has_non_space) pending_space = true; - }, - .letter_or_digit, .handle_internal => { - if (pending_space) { - out.appendAssumeCapacity(' '); - pending_space = false; - } - out.appendAssumeCapacity(c); - has_non_space = true; - }, - .underscore => unreachable, // resolved above + // 2. CJK: preserve (katakana already folded to hiragana). Emitted as + // its UTF-8 bytes — letter-class, lazy leading space like ASCII. + if (cjkFold(cp)) |kept| { + var buf: [4]u8 = undefined; + const n = std.unicode.utf8Encode(kept, &buf) catch continue; + if (pending_space) { + out.appendAssumeCapacity(' '); + pending_space = false; } + for (buf[0..n]) |b| out.appendAssumeCapacity(b); + has_non_space = true; + continue; + } + + // 3. Latin / ASCII fold path (table above, unchanged). + const folded = foldCodepoint(cp); + for (folded.bytes[0..folded.len]) |c| { + if (!emitAscii(&out, c, underscore, &pending_space, &has_non_space)) break; } } @@ -298,10 +368,36 @@ test "internal runs of separators collapse to single space" { try expectNorm("foo , , , bar", "foo bar"); } -test "emoji and CJK dropped" { - try expectNorm("nate 👋", "nate"); - try expectNorm("hello 你好", "hello"); +test "emoji dropped, CJK preserved (v2)" { + try expectNorm("nate 👋", "nate"); // emoji still dropped try expectNorm("🚀rocket🚀", "rocket"); + try expectNorm("hello 你好", "hello 你好"); // CJK now preserved +} + +test "v2: japanese kana + han preserved" { + try expectNorm("日本", "日本"); // Han preserved + try expectNorm("さとう", "さとう"); // hiragana preserved + try expectNorm("佐藤太郎", "佐藤太郎"); // kanji run preserved +} + +test "v2: katakana folds to hiragana" { + try expectNorm("サトウ", "さとう"); // katakana → hiragana (collides with さとう) + try expectNorm("トーキョー", "とーきょー"); // ー prolonged mark preserved +} + +test "v2: full-width ascii folds to ascii" { + try expectNorm("Nate", "nate"); // full-width latin → ascii lowercase + try expectNorm("hello123", "hello123"); +} + +test "v2: ideographic + middle-dot spaces separate tokens" { + try expectNorm("ジョン・スミス", "じょん すみす"); // ・ → space; katakana → hiragana + try expectNorm("東京 タワー", "東京 たわー"); // ideographic space → space +} + +test "v2: mixed latin + cjk preserved inline (segmented later)" { + try expectNorm("Tokyo東京", "tokyo東京"); // no space inserted; tokenizer segments runs + try expectNorm("東京2020", "東京2020"); } test "combining marks dropped (pre-composed already folded above)" { @@ -377,5 +473,5 @@ test "query() keeps underscore (user may type a username with _)" { test "version constant is stable" { // Don't change this without bumping VERSION and rebuilding the index. - try testing.expectEqualStrings("v1-ascii-latin1", VERSION); + try testing.expectEqualStrings("v2-cjk-bigram", VERSION); } diff --git a/services/src/index/prefixes.zig b/services/src/index/prefixes.zig index 6a5e9cb..780bb13 100644 --- a/services/src/index/prefixes.zig +++ b/services/src/index/prefixes.zig @@ -79,26 +79,138 @@ pub fn displayTokens(arena: Allocator, normalized_display: []const u8) ![][]cons return tokens.toOwnedSlice(arena); } -/// Derive every `n:` prefix key for a normalized display name. Tokenizes -/// then emits prefixes of each token (lengths 1..PREFIX_MAX per token). +/// Max codepoints bigrammed from a single CJK run (defense against a +/// pathological all-CJK display name). Phase 0 avg is ~4 codepoints. +pub const CJK_RUN_MAX: usize = 24; + +/// CJK codepoints we index: Han, Hiragana (katakana is folded to Hiragana +/// upstream in normalize.zig), and half-width katakana. Must agree with the +/// set normalize.zig preserves — anything preserved-and-non-ASCII is CJK. +fn isCjkCodepoint(cp: u21) bool { + return (cp >= 0x3040 and cp <= 0x30FF) // hiragana + (folded-away) katakana + ー + or (cp >= 0x3400 and cp <= 0x4DBF) // CJK Ext A + or (cp >= 0x4E00 and cp <= 0x9FFF) // CJK Unified Ideographs + or (cp >= 0xF900 and cp <= 0xFAFF) // CJK Compatibility Ideographs + or (cp >= 0xFF66 and cp <= 0xFF9D); // half-width katakana +} + +/// Emit `n:` keys for one contiguous CJK run (codepoint-safe — all slices +/// land on UTF-8 boundaries). Adjacent-codepoint BIGRAMS (`n:日本`, `n:本太`) +/// approximate word boundaries for a script with no spaces; the query side +/// intersects them (the existing multi-token `n:` AND). Plus the LEADING +/// UNIGRAM (`n:日`) so a 1-char in-progress query matches. +/// for_query=false (builder): bigrams + leading unigram. +/// for_query=true (query): bigrams; or the single unigram when the run +/// is one codepoint (no every-char unigram). +fn appendCjkKeys( + arena: Allocator, + keys: *std.ArrayListUnmanaged([]const u8), + run: []const u8, + for_query: bool, +) !void { + // Byte offsets of each codepoint boundary, capped at CJK_RUN_MAX cps. + var bound: [CJK_RUN_MAX + 1]usize = undefined; + var n: usize = 0; + bound[0] = 0; + const view = std.unicode.Utf8View.init(run) catch return; + var it = view.iterator(); + while (it.nextCodepointSlice()) |cps| { + if (n >= CJK_RUN_MAX) break; + bound[n + 1] = bound[n] + cps.len; + n += 1; + } + if (n == 0) return; + if (n == 1) { + try keys.append(arena, try formatKey(arena, .display_token, run[0..bound[1]])); + return; + } + var i: usize = 0; + while (i + 1 < n) : (i += 1) { + try keys.append(arena, try formatKey(arena, .display_token, run[bound[i]..bound[i + 2]])); + } + if (!for_query) { + try keys.append(arena, try formatKey(arena, .display_token, run[0..bound[1]])); // leading unigram + } +} + +/// Emit `n:` keys for one Latin (ASCII, post-normalize) run. +/// for_query=false (builder): every prefix, length 1..PREFIX_MAX. +/// for_query=true (query): the full run as a single key. +fn appendLatinKeys( + arena: Allocator, + keys: *std.ArrayListUnmanaged([]const u8), + run: []const u8, + for_query: bool, +) !void { + if (run.len == 0) return; + if (for_query) { + try keys.append(arena, try formatKey(arena, .display_token, run)); + return; + } + const max = @min(run.len, PREFIX_MAX); // Latin run is ASCII → byte == codepoint + var i: usize = 0; + while (i < max) : (i += 1) { + try keys.append(arena, try formatKey(arena, .display_token, run[0 .. i + 1])); + } +} + +/// The shared, script-aware `n:`-key tokenizer. Segments one whitespace-token +/// into contiguous Latin vs CJK runs (codepoint-safe) and emits keys per run. +/// BOTH the builder (`displayPrefixKeys`, for_query=false) and the serving +/// path (`queryKeys`, for_query=true) go through here — one code path, so the +/// index and the query can never disagree on what a CJK key looks like. +fn emitTokenKeys( + arena: Allocator, + keys: *std.ArrayListUnmanaged([]const u8), + token: []const u8, + for_query: bool, +) !void { + const view = std.unicode.Utf8View.init(token) catch { + try appendLatinKeys(arena, keys, token, for_query); // invalid utf8: treat as ascii + return; + }; + var it = view.iterator(); + var run_start: usize = 0; + var pos: usize = 0; + var cur_cjk: ?bool = null; + while (it.nextCodepointSlice()) |cps| { + const cp = std.unicode.utf8Decode(cps) catch { + pos += cps.len; + continue; + }; + const this_cjk = isCjkCodepoint(cp); + if (cur_cjk) |ck| { + if (ck != this_cjk) { + try emitRun(arena, keys, token[run_start..pos], ck, for_query); + run_start = pos; + } + } + cur_cjk = this_cjk; + pos += cps.len; + } + if (cur_cjk) |ck| try emitRun(arena, keys, token[run_start..pos], ck, for_query); +} + +fn emitRun( + arena: Allocator, + keys: *std.ArrayListUnmanaged([]const u8), + run: []const u8, + is_cjk: bool, + for_query: bool, +) !void { + if (is_cjk) try appendCjkKeys(arena, keys, run, for_query) else try appendLatinKeys(arena, keys, run, for_query); +} + +/// Derive every `n:` key for a normalized display name. Latin runs emit +/// prefixes; CJK runs emit bigrams + a leading unigram (see emitTokenKeys). /// Duplicate keys across tokens are NOT deduplicated here — the builder's /// group-by pass naturally dedupes. pub fn displayPrefixKeys(arena: Allocator, normalized_display: []const u8) ![][]const u8 { const tokens = try displayTokens(arena, normalized_display); if (tokens.len == 0) return &.{}; - // Worst case: every token at PREFIX_MAX length → up to - // MAX_DISPLAY_TOKENS * PREFIX_MAX keys. var keys: std.ArrayListUnmanaged([]const u8) = .empty; - try keys.ensureTotalCapacity(arena, tokens.len * PREFIX_MAX); - - for (tokens) |tok| { - const max = @min(tok.len, PREFIX_MAX); - var i: usize = 0; - while (i < max) : (i += 1) { - keys.appendAssumeCapacity(try formatKey(arena, .display_token, tok[0 .. i + 1])); - } - } + for (tokens) |tok| try emitTokenKeys(arena, &keys, tok, false); return keys.toOwnedSlice(arena); } @@ -227,10 +339,11 @@ pub fn queryKeys(arena: Allocator, normalized_query: []const u8) ![][]const u8 { try keys.append(arena, try formatKey(arena, .domain_suffix, normalized_query)); } + // CJK-aware: a CJK query token yields its bigram set (intersected by the + // serving layer); a Latin token yields its single full-run key. Same + // tokenizer the builder uses, so index + query agree. const tokens = try displayTokens(arena, normalized_query); - for (tokens) |tok| { - try keys.append(arena, try formatKey(arena, .display_token, tok)); - } + for (tokens) |tok| try emitTokenKeys(arena, &keys, tok, true); return keys.toOwnedSlice(arena); } @@ -537,3 +650,61 @@ test "deriveAllUnique enables clean set-diff for overlay rename" { try testing.expect(has_h_alice); try testing.expect(has_e_alicefm); } + +// ─── v2: CJK / Japanese ──────────────────────────────────────────────────── + +fn keySet(arena: Allocator, keys: [][]const u8) !std.StringHashMap(void) { + var s = std.StringHashMap(void).init(arena); + for (keys) |k| try s.put(k, {}); + return s; +} + +test "v2: displayPrefixKeys bigrams a CJK run + leading unigram" { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + const keys = try displayPrefixKeys(arena.allocator(), "日本太郎"); + // bigrams 日本, 本太, 太郎 + leading unigram 日 → 4 keys + try testing.expectEqual(@as(usize, 4), keys.len); + var s = try keySet(arena.allocator(), keys); + try testing.expect(s.contains("n:日本")); + try testing.expect(s.contains("n:本太")); + try testing.expect(s.contains("n:太郎")); + try testing.expect(s.contains("n:日")); +} + +test "v2: queryKeys bigrams a CJK query" { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + var s = try keySet(arena.allocator(), try queryKeys(arena.allocator(), "佐藤太郎")); + try testing.expect(s.contains("n:佐藤")); + try testing.expect(s.contains("n:藤太")); + try testing.expect(s.contains("n:太郎")); +} + +test "v2: single CJK char query → leading unigram only" { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + var s = try keySet(arena.allocator(), try queryKeys(arena.allocator(), "日")); + try testing.expect(s.contains("n:日")); +} + +test "v2: mixed latin+cjk token segments into runs" { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + var s = try keySet(arena.allocator(), try displayPrefixKeys(arena.allocator(), "tokyo東京")); + try testing.expect(s.contains("n:t")); // latin prefix + try testing.expect(s.contains("n:tokyo")); // latin full + try testing.expect(s.contains("n:東京")); // cjk bigram + try testing.expect(s.contains("n:東")); // cjk leading unigram +} + +test "v2: query bigrams are a subset of indexed keys (builder/query agree)" { + var arena = std.heap.ArenaAllocator.init(testing.allocator); + defer arena.deinit(); + var idx = try keySet(arena.allocator(), try displayPrefixKeys(arena.allocator(), "佐藤太郎")); + // a partial query "佐藤太" must produce only n: keys present in the index + const q = try queryKeys(arena.allocator(), "佐藤太"); + for (q) |k| { + if (std.mem.startsWith(u8, k, "n:")) try testing.expect(idx.contains(k)); + } +} -- 2.51.2