diff --git a/scripts/build-fonts.ts b/scripts/build-fonts.ts index 408a7cb..95bfe3d 100644 --- a/scripts/build-fonts.ts +++ b/scripts/build-fonts.ts @@ -77,7 +77,7 @@ for (const name of await readdir(FONT_DIR)) await mkdir(NOTICES_DIR, { recursive: true }) await writeFile(join(NOTICES_DIR, 'noto-ofl.txt'), await raw('font/OFL.txt')) -/** Collect the characters each region needs, in row order (commonness order). */ +/** Collect the characters each region needs, in display-priority order. */ const needed: Record = Object.fromEntries( REGIONS.map((r) => [r, [] as string[]]), ) @@ -101,9 +101,26 @@ const collect = (row: (typeof data.rows)[number]) => { for (const form of formsOf(row)) need(form.font, form.char) } +/** + * The table can sort by any of the four regional frequency corpora. Raw row + * order cannot stand in for that priority: merged rows such as 箇 and 幷 have + * very common mainland forms but deliberately live much later in the dataset. + * Putting a row at its best observed rank keeps every region's first frequency + * page together at the front of each font instead of pulling distant chunks. + * Stable sorting preserves dataset order for equal and unranked rows. + */ +const bestFrequencyRank = (row: (typeof data.rows)[number]): number => + Math.min( + ...(row.freq?.filter((rank): rank is number => rank !== null) ?? []), + Number.MAX_SAFE_INTEGER, + ) + const hero = data.rows.find((row) => row.key === HERO_KEY) if (hero) collect(hero) -for (const row of data.rows) collect(row) +for (const row of data.rows.toSorted( + (a, b) => bestFrequencyRank(a) - bestFrequencyRank(b), +)) + collect(row) /** Sort codepoints and merge consecutive runs to keep unicode-range short. */ function unicodeRange(chars: string[]): string { diff --git a/scripts/tests/fonts.test.ts b/scripts/tests/fonts.test.ts index b79b298..1baa412 100644 --- a/scripts/tests/fonts.test.ts +++ b/scripts/tests/fonts.test.ts @@ -11,8 +11,19 @@ import { readdirSync, readFileSync } from 'node:fs' import { join } from 'node:path' import * as fontkit from 'fontkit' import { describe, expect, it } from 'vitest' -import { fontIndexOf } from '../../shared/row.ts' -import { REGIONS, type CharsData } from '../../shared/types.ts' +import { frequencyRankOf } from '../../shared/frequency.ts' +import { + fontIndexOf, + fontRegionOf, + projectSignature, + varietyOf, +} from '../../shared/row.ts' +import { + FREQUENCY_REGIONS, + REGIONS, + type CharsData, + type FrequencyRegion, +} from '../../shared/types.ts' import { partitionSignature } from '../cmap.ts' import { DATA_DIR, FONT_DIR } from '../sources.ts' @@ -27,13 +38,17 @@ const fontsOf = (region: string) => .filter( (f) => f.startsWith(`hanji-sans-${region}-`) && f.endsWith('.woff2'), ) - .map((f) => fontkit.create(readFileSync(join(FONT_DIR, f))) as fontkit.Font) + .toSorted() + .map((name) => ({ + name, + font: fontkit.create(readFileSync(join(FONT_DIR, name))) as fontkit.Font, + })) const fonts = Object.fromEntries(REGIONS.map((r) => [r, fontsOf(r)])) /** Glyph IDs are not comparable across subsets, so compare outlines. */ function outline(region: string, char: string): string { - for (const font of fonts[region]) { + for (const { font } of fonts[region]) { const [glyph] = font.layout(char).glyphs // A missing character falls back to .notdef, which is glyph 0 if (glyph && glyph.id !== 0) return glyph.path.toSVG() @@ -41,6 +56,16 @@ function outline(region: string, char: string): string { throw new Error(`no chunk of hanji-sans-${region} carries ${char}`) } +/** Which generated chunk carries this region's character. */ +function shardOf(region: string, char: string): number { + const codePoint = char.codePointAt(0)! + for (const { name, font } of fonts[region]) { + if (font.glyphForCodePoint(codePoint).id === 0) continue + return Number(/-(\d+)\.woff2$/.exec(name)![1]!) + } + throw new Error(`no chunk of hanji-sans-${region} carries ${char}`) +} + const outlinesOf = (key: string) => { const row = rows.get(key)! return row.chars.map((char, i) => outline(REGIONS[fontIndexOf(row, i)], char)) @@ -93,6 +118,49 @@ describe('subset coverage', () => { }) }) +describe('subset loading', () => { + // Stored-region indices in the default presentation order. Korea and the + // optional old-form cell start hidden. + const visible = [0, 3, 1, 2] as const + const pageSize = 100 + + function firstFrequencyPage(region: FrequencyRegion) { + return data.rows + .filter((row) => varietyOf(projectSignature(row.glyph, visible)) > 1) + .toSorted((a, b) => { + const left = frequencyRankOf(a, region) + const right = frequencyRankOf(b, region) + if (left !== right && (left === null || right === null)) + return left === null ? 1 : -1 + return ( + (left ?? Number.MAX_SAFE_INTEGER) - + (right ?? Number.MAX_SAFE_INTEGER) || + a.key.codePointAt(0)! - b.key.codePointAt(0)! + ) + }) + .slice(0, pageSize) + } + + it.each(FREQUENCY_REGIONS)( + 'keeps the first %s frequency page in one shard per visible font', + (frequencyRegion) => { + const shards = new Set() + for (const row of firstFrequencyPage(frequencyRegion)) { + for (const index of visible) { + const region = fontRegionOf(row, index) + shards.add(`${region}-${shardOf(region, row.chars[index]!)}`) + } + } + + // The hero is visible above the list and follows the same four faces. + for (const region of ['cn', 'jp', 'hk', 'tw'] as const) + shards.add(`${region}-${shardOf(region, '返')}`) + + expect([...shards].toSorted()).toEqual(['cn-0', 'hk-0', 'jp-0', 'tw-0']) + }, + ) +}) + describe('interface subset coverage', () => { it('carries every Korean reading in every locale font', () => { const hangul = [