diff --git a/.prettierignore b/.prettierignore index efdc50f..732d7d7 100644 --- a/.prettierignore +++ b/.prettierignore @@ -5,3 +5,7 @@ app/assets/strokes/ # Generated by scripts/build-fonts.ts alongside the woff2 files app/assets/fonts/fonts-*.css app/generated/ + +# Pinned upstream licence notices copied verbatim by the font build. +public/notices/plangothic-ofl.txt +public/notices/wenjin-mincho-ofl.md diff --git a/README.ja-JP.md b/README.ja-JP.md index 83cb007..8834ce7 100644 --- a/README.ja-JP.md +++ b/README.ja-JP.md @@ -16,7 +16,7 @@ ## 収録範囲 -中国大陸の『通用規範漢字表』(2013年)、台湾の『常用國字標準字體表』(1982年)、香港の『常用字字形表』、日本の『常用漢字表』(2010年)、韓国の『漢文教育用基礎漢字』(2000年)の和集合を収録しています。韓国の字表には常用漢字1,800字が含まれます。簡体字と繁体字、日本の新字体と旧字体、韓国式異体字など、コードポイントをまたぐ対応を1行に統合した結果、全体で **8,449行** です。このうち1,695行は5地域の字形が完全に一致し、129行は5地域すべてで異なります。 +中国大陸の『通用規範漢字表』(2013年)、台湾の『常用國字標準字體表』(1982年)、香港の『常用字字形表』、日本の『常用漢字表』(2010年)、韓国の『漢文教育用基礎漢字』(2000年)の和集合を収録しています。韓国の字表には常用漢字1,800字が含まれます。簡体字と繁体字、日本の新字体と旧字体、韓国式異体字など、コードポイントをまたぐ対応を1行に統合した結果、全体で **8,449行** です。このうち1,692行は5地域の字形が完全に一致し、129行は5地域すべてで異なります。 台湾の『次常用國字表』は、既存行の二級収録状態と候補を示すためだけに使い、新しい行の生成には使いません。6,343件の主要項目のうち、他にない3,599件は明示的に本プロダクトの対象外としています。 @@ -40,6 +40,8 @@ ページ上ではこの判定に基づいてセルを再グループ化します。同形と判定されたセルは、グループ内の一地域のNotoフォントを共通して使い、画面上でも実際に同じ輪郭を表示します。そのため、上記ルールで除外された地域版間の細かな差異は表示されません。`scripts/tests/fonts.test.ts` はfontkitで生成フォントの実際のアウトラインを取り出し、判定と画面表示が一致することを1字ずつ検証します。 +明確なコードポイント間の対応は、Noto/Source Hanが対応先の文字を収録していなくても元のコードポイントへ戻しません。たとえば `𬒗 → 𥗽` は二つのコードポイントとして表示します。ビルドは、現在のデータで表示が必要ながらNotoにないコードポイントを自動収集し、同梱用の補充WOFF2サブセットを生成します。ゴシック体はPlangothic P1、明朝体はWenJin Mincho P2を使い、選択したフォントに一つでも未収録コードポイントがあればビルドを明示的に失敗させます。ページには常に実際のUnicodeテキストを残すため、端末フォントには依存しません。これらの補充字形は表示専用で、地域差の判定には使いません。 + ## 制限事項 - 本アプリが比較するのは、一般的なゴシック体と明朝体における印刷字形だけです。手書きの慣習は対象外で、教科書体の例示字形も基準にしません。日本語の教科書体は主に日本語教育向けに設計され、中国大陸・香港・台湾・韓国と同じグリフプールを共有する正式な地域版がありません。見た目の近い別々のフォントを組み合わせると、地域差とフォント固有のデザイン差を分離できません。条件を揃えるため、5地域版を同時に提供する同系統のフォントファミリーだけを使います。 @@ -63,7 +65,7 @@ ```bash pnpm install -pnpm build:data # 字表とフォントサブセットを生成。初回は約261 MiBをダウンロードし、以後はキャッシュを使用 +pnpm build:data # 字表とフォントサブセットを生成。初回は約302 MiBをダウンロードし、以後はキャッシュを使用 pnpm update:sources # サードパーティデータの更新を確認して固定。変更があればダウンロードして再生成 pnpm dev pnpm test @@ -94,7 +96,7 @@ Cloudflareの **Settings → Domains & Routes** でproductionドメインを接 各字群の詳細ページはそれぞれ独立したHTMLとして生成されます。ページデータはローカルbundleに含まれるため、ルートごとの追加 `_payload.json` を生成するpayload extractionは無効にしています。地域異体字の別名については、リダイレクト専用ページを生成しません。Static Assetsがまず `404.html` とHTTP 404を返し、その後Nuxtのクライアントミドルウェアが対応する行へ移動します。これにより、検索エンジンが別名を成功ページとして重複登録することを避けます。実際に存在しないURLはHTTP 404のままです。`@nuxtjs/sitemap` は静的生成時にすべてのcanonicalページを `/sitemap.xml` へ書き出し、`@nuxtjs/robots` は `/robots.txt` を生成してsitemapの場所を通知します。両方の絶対URLには `NUXT_SITE_URL` を使用します。GitHub Actionsは同名のリポジトリ変数を優先し、未設定の場合はリポジトリのhomepageを使用します。他の環境では `pnpm generate` の実行時に設定する必要があります。PRプレビューのビルドでは `NUXT_SITE_ENV=preview` によりインデックス登録を禁止します。`public/_headers` では、内容ハッシュ付きの `_nuxt/*` に長期immutableキャッシュを設定し、安定した `/notices/*`、sitemap、robotsのURLには `no-cache`、`/data/chars.json` には1時間の `max-age` と1日の `stale-while-revalidate` を指定します。 -サードパーティ資産の具体的なcommit、公式添付ファイル識別子、SHA-256は `data/sources.lock.json` に記録しています。更新時には `pnpm update:sources` を実行します。バージョンのある上流データについてはバージョン番号を解決し、バージョンのない公式直リンクについては改めて検証します。内容が変わっていればlockfileを更新してデータを再生成し、まったく変わっていなければ生成をスキップします。ビルド時の `pnpm build:data` はlockfileに従い、約 **261 MiB** の元データをダウンロードして検証します。このうち195 MiBは10個のNoto CJKフォントです。明示的な更新を行っていない直リンクの内容が変わった場合は、チェックサム不一致として失敗し、黙ってデータに取り込むことはありません。Actionsでは元データのダウンロードと生成フォントを別々にキャッシュします。前者はlockfileだけで決まり、後者はlockfile、実際の生成スクリプト、関連依存関係、locale、字表から決まります。フォント入力が完全に同じ場合は、データ生成を省略します。 +サードパーティ資産の具体的なcommit、GitHub release tag、公式添付ファイル識別子、SHA-256は `data/sources.lock.json` に記録しています。更新時には `pnpm update:sources` を実行します。GitHubブランチ、最新release、Unicodeバージョンを解決し、バージョンのない公式直リンクについては改めて検証します。内容が変わっていればlockfileを更新してデータを再生成し、まったく変わっていなければ生成をスキップします。ビルド時の `pnpm build:data` はlockfileに従い、約 **302 MiB** の元データをダウンロードして検証します。このうち195 MiBは10個のNoto CJKフォントで、約40 MiBは2個の補充フォントです。明示的な更新を行っていない直リンクの内容が変わった場合は、チェックサム不一致として失敗し、黙ってデータに取り込むことはありません。Actionsでは元データのダウンロードと生成フォントを別々にキャッシュします。前者はlockfileだけで決まり、後者はlockfile、実際の生成スクリプト、関連依存関係、locale、字表から決まります。フォント入力が完全に同じ場合は、データ生成を省略します。 筆順シャードは `pnpm build:dataset` が `app/assets/strokes/` に生成し、リポジトリにはコミットしません。デプロイ処理はテストと静的生成の前に毎回再生成し、Viteが内容ハッシュ付きのファイル名で出力します。付属ライセンスは `public/notices/` の安定URLに置き、再検証を必須にします。同じ字グループ内で筆画順に並べた輪郭が完全に一致する場合は、最初のバリアントとその中心線だけを保存し、画面上でも対応する地域を1つの選択肢にまとめます。ページ読み込み時に所属シャードを一度だけ取得し、その後の地域切替ではメモリ上の字グループデータを再利用します。サイト全体で、ここから解析した輪郭数を第一候補の画数として使います。 @@ -115,6 +117,8 @@ pnpm deploy | ------------------------------------------------ | --------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------- | | 5地域の字形差異の判定 | [Adobe Source Han Sans / Serif(CMapリソース)](https://github.com/adobe-fonts/source-han-sans) | [SIL OFL 1.1](https://openfontlicense.org/) | | ページ表示用フォント | [Noto Sans / Noto Serif(CJKを含む)](https://github.com/notofonts/noto-cjk) | [SIL OFL 1.1](https://openfontlicense.org/) | +| Noto未収録字のゴシック体補完 | [Plangothic P1](https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project) | [SIL OFL 1.1](https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project/blob/main/LICENSE-OFL.txt) | +| Noto未収録字の明朝体補完 | [WenJin Mincho P2](https://github.com/takushun-wu/WenJinMincho) | [SIL OFL 1.1](https://github.com/takushun-wu/WenJinMincho/blob/main/LICENSE.md) | | 簡体・繁体、香港・台湾異体字、日本新旧字体の対応 | [OpenCC(中国語文字変換)](https://github.com/BYVoid/OpenCC) | [Apache-2.0](https://www.apache.org/licenses/LICENSE-2.0) | | 中国・台湾・日本・韓国の筆順と画数 | [AnimCJK](https://github.com/parsimonhi/animCJK) | [Arphic Public License](https://github.com/parsimonhi/animCJK/blob/master/licenses/APL/english/ARPHICPL.TXT) | | 5地域の標準字表 | [zispace/hanzi-chars](https://github.com/zispace/hanzi-chars) | [リポジトリに記載なし](https://github.com/zispace/hanzi-chars) | @@ -129,9 +133,9 @@ pnpm deploy 原典となる標準は、中国大陸の『通用規範漢字表』(2013年)、台湾の『常用國字標準字體表』(1982年)、香港の『常用字字形表』、日本の『常用漢字表』(2010年)と『学年別漢字配当表』(2017年)、韓国の『漢文教育用基礎漢字』(2000年)です。 -1字ずつ比較できるツール [tofu.tools](https://tofu.tools/) は本プロジェクトの先行例で、同じくNotoファミリーを使って地域字形を区別しています。 +1字ずつ比較できるツール [tofu.tools](https://tofu.tools/) は本プロジェクトの先行例で、同じくNotoファミリーを使って地域字形を区別しています。希少字の補充字形を提供してくださったPlangothicとWenJin Minchoのメンテナーにも感謝します。 -フォントはNoto Sans CJKとNoto Serif CJK(SIL OFL 1.1)を本アプリで使う文字にサブセット化したものです。ライセンス文は [`/notices/noto-ofl.txt`](public/notices/noto-ofl.txt) に同梱しています。生成データファイルは上記の出典から派生しているため、それぞれのライセンスに従ってください。項目ごとの変換方法と帰属表示は、公開されている [`/notices/data-sources.md`](public/notices/data-sources.md) にも記載しています。 +フォントはNoto Sans CJK、Noto Serif CJK、Plangothic P1、WenJin Mincho P2(すべてSIL OFL 1.1)を本アプリで使う文字にサブセット化したものです。ライセンス文は [`/notices/noto-ofl.txt`](public/notices/noto-ofl.txt)、[`/notices/plangothic-ofl.txt`](public/notices/plangothic-ofl.txt)、[`/notices/wenjin-mincho-ofl.md`](public/notices/wenjin-mincho-ofl.md) に同梱しています。生成データファイルは上記の出典から派生しているため、それぞれのライセンスに従ってください。項目ごとの変換方法と帰属表示は、公開されている [`/notices/data-sources.md`](public/notices/data-sources.md) にも記載しています。 ## License diff --git a/README.md b/README.md index dc0fd57..1047d55 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ ## 收录范围 -收录《通用规范汉字表》(2013)、臺灣《常用國字標準字體表》(1982)、香港《常用字字形表》、日本《常用漢字表》(2010)、韩国《漢文教育用基礎漢字》(2000)五份字表的并集。韩国字表含 1,800 个常用汉字。把简繁、日本新旧字体和韩式异体这类跨码点的对应合并成一行后,共 **8,449 行**——其中 1,695 行五地字形完全一致,129 行五地各不相同。 +收录《通用规范汉字表》(2013)、臺灣《常用國字標準字體表》(1982)、香港《常用字字形表》、日本《常用漢字表》(2010)、韩国《漢文教育用基礎漢字》(2000)五份字表的并集。韩国字表含 1,800 个常用汉字。把简繁、日本新旧字体和韩式异体这类跨码点的对应合并成一行后,共 **8,449 行**——其中 1,692 行五地字形完全一致,129 行五地各不相同。 台湾《次常用國字表》只为已有行提供二级收录状态和候选,不参与生成新行;其 6,343 个主条目中有 3,599 个独有条目明确在产品范围之外。 @@ -40,6 +40,8 @@ 页面会按这份判定重新分组:被判为同形的格子统一借用组内一个地区的 Noto 字体,使屏幕上也真正呈现同一轮廓。相应地,被上述规则过滤的地区版本细小差异不会显示。`scripts/tests/fonts.test.ts` 会用 fontkit 取出生成字体的真实轮廓,逐字验证判定与画面显示一致。 +明确的跨码点映射不会因为 Noto/Source Han 未收录目标字而退回原码点。例如 `𬒗 → 𥗽` 仍会显示为两个码点。构建会从当前数据自动收集 Noto 缺少但页面需要显示的码点,并生成项目内置的补充 WOFF2 子集:黑体取 Plangothic P1,宋体取 WenJin Mincho P2;所选字体缺少任一码点时,构建会明确失败。页面始终保留真正的 Unicode 文本,不依赖用户设备的本机字体。这两套补充字形只负责显示,不参与地区差异判定。 + ## 局限 - 本应用只比较通用黑体与宋体中的印刷字形,不涵盖手写习惯,也不以教科书体的示范字形为准。日语教科书体主要为日语教学设计,并没有与中、港、台、韩共享同一字形池的正式地区版本;若拼接风格相近但来源不同的字体,地区差异与字体自身的设计差异就无法分开。为了控制变量,本应用只能选用同时提供五地版本的同源字体系列。 @@ -63,7 +65,7 @@ ```bash pnpm install -pnpm build:data # 生成字表与字体子集,首次会下载约 261 MiB 原始数据,之后走缓存 +pnpm build:data # 生成字表与字体子集,首次会下载约 302 MiB 原始数据,之后走缓存 pnpm update:sources # 检查并锁定新版第三方数据;有变化时下载并重新生成 pnpm dev pnpm test @@ -94,7 +96,7 @@ Cloudflare Worker 名称须为 `hanji`,与 `wrangler.json` 中的 `name` 一 每个字组详情页都会生成独立 HTML;页面数据在本地 bundle 中,因此关闭了每路由额外生成 `_payload.json` 的 payload extraction。地区异体别名不另外生成跳转页:它先由 Static Assets 返回 `404.html` 和 HTTP 404,再由 Nuxt 客户端中间件跳到所属行;搜索引擎不会把 alias 当作成功页面重复收录。真正未知的地址保持 HTTP 404。`@nuxtjs/sitemap` 会在静态生成时把全部 canonical 页面写入 `/sitemap.xml`,`@nuxtjs/robots` 生成 `/robots.txt` 并公布 sitemap 地址。两者的绝对 URL 来自 `NUXT_SITE_URL`;GitHub Actions 优先读取同名仓库变量,未设置时使用仓库 homepage,其他环境需在运行 `pnpm generate` 时设置。PR 预览构建通过 `NUXT_SITE_ENV=preview` 禁止索引。`public/_headers` 给带内容哈希的 `_nuxt/*` 设长期 immutable 缓存,让稳定的 `/notices/*`、sitemap 和 robots URL 使用 `no-cache`,并为 `/data/chars.json` 设置 1 小时的 `max-age` 与 1 天的 `stale-while-revalidate`。 -第三方资产的具体 commit、官方附件标识与 SHA-256 记录在 `data/sources.lock.json`;需要升级时运行 `pnpm update:sources`。它会解析有版本上游的版本号,并重新校验没有版本号的官方直链;内容有变化时更新 lockfile 并直接重新生成数据,完全未变则跳过生成。构建时 `pnpm build:data` 会按 lockfile 下载并校验约 **261 MiB** 原始数据(其中 195 MiB 是十份 Noto CJK 字体);任何未显式更新的直链内容变化都会因校验和不符而失败,不会静默进入数据。Actions 分开缓存原始下载与生成字体:前者只由 lockfile 决定,后者由 lockfile、实际生成脚本、相关依赖、locale 与字表决定;字体输入完全不变时跳过数据生成。 +第三方资产的具体 commit、GitHub release tag、官方附件标识与 SHA-256 记录在 `data/sources.lock.json`;需要升级时运行 `pnpm update:sources`。它会解析 GitHub 分支、最新 release 与 Unicode 版本,并重新校验没有版本号的官方直链;内容有变化时更新 lockfile 并直接重新生成数据,完全未变则跳过生成。构建时 `pnpm build:data` 会按 lockfile 下载并校验约 **302 MiB** 原始数据(其中 195 MiB 是十份 Noto CJK 字体,另有约 40 MiB 的两份补充字体);任何未显式更新的直链内容变化都会因校验和不符而失败,不会静默进入数据。Actions 分开缓存原始下载与生成字体:前者只由 lockfile 决定,后者由 lockfile、实际生成脚本、相关依赖、locale 与字表决定;字体输入完全不变时跳过数据生成。 笔顺分片由 `pnpm build:dataset` 生成到 `app/assets/strokes/`,不提交到仓库;部署流程会在测试和静态生成前重建,再由 Vite 输出带内容哈希的文件名。随附授权保留在 `public/notices/` 的稳定 URL 下并要求重新验证。同一字组内,按笔画顺序排列的轮廓完全一致时只保存第一份变体及其中线,界面也把对应地区合并为一个选择项;页面加载时只获取一次所属分片,之后切换地区直接复用内存中的字组数据。整站使用这里解析出的轮廓数量作为首选笔画数。 @@ -115,6 +117,8 @@ pnpm deploy | -------------------------------- | --------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------- | | 判定五地字形差异 | [Adobe Source Han Sans / Serif(CMap 资源)](https://github.com/adobe-fonts/source-han-sans) | [SIL OFL 1.1](https://openfontlicense.org/) | | 页面展示用字体 | [Noto Sans / Noto Serif(含 CJK)](https://github.com/notofonts/noto-cjk) | [SIL OFL 1.1](https://openfontlicense.org/) | +| 补充Noto未收录的黑体字形 | [Plangothic P1](https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project) | [SIL OFL 1.1](https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project/blob/main/LICENSE-OFL.txt) | +| 补充Noto未收录的宋体字形 | [WenJin Mincho P2](https://github.com/takushun-wu/WenJinMincho) | [SIL OFL 1.1](https://github.com/takushun-wu/WenJinMincho/blob/main/LICENSE.md) | | 简繁、港台异体、日本新旧字体对应 | [OpenCC 开放中文转换](https://github.com/BYVoid/OpenCC) | [Apache-2.0](https://www.apache.org/licenses/LICENSE-2.0) | | 中、台、日、韩笔顺动画与笔画数 | [AnimCJK](https://github.com/parsimonhi/animCJK) | [Arphic Public License](https://github.com/parsimonhi/animCJK/blob/master/licenses/APL/english/ARPHICPL.TXT) | | 五地标准字表 | [zispace/hanzi-chars](https://github.com/zispace/hanzi-chars) | [仓库未声明](https://github.com/zispace/hanzi-chars) | @@ -129,9 +133,9 @@ pnpm deploy 原始规范出处:《通用规范汉字表》(2013)、臺灣《常用國字標準字體表》(1982)、香港《常用字字形表》、日本《常用漢字表》(2010)与《学年別漢字配当表》(2017)、韩国《漢文教育用基礎漢字》(2000)。 -逐字对照工具 [tofu.tools](https://tofu.tools/) 是本项目的先行者,同样用 Noto 系列区分地区字形。 +逐字对照工具 [tofu.tools](https://tofu.tools/) 是本项目的先行者,同样用 Noto 系列区分地区字形。感谢 Plangothic 与 WenJin Mincho 的维护者提供生僻字补充字形。 -字体为 Noto Sans CJK 与 Noto Serif CJK(SIL OFL 1.1)按本应用用字子集化后的产物,声明随附于 [`/notices/noto-ofl.txt`](public/notices/noto-ofl.txt)。生成的数据文件派生自上述来源,请遵守各自许可;逐项转换方式与署名也写入公开的 [`/notices/data-sources.md`](public/notices/data-sources.md)。 +字体为 Noto Sans CJK、Noto Serif CJK、Plangothic P1 与 WenJin Mincho P2(均为 SIL OFL 1.1)按本应用用字子集化后的产物。声明分别随附于 [`/notices/noto-ofl.txt`](public/notices/noto-ofl.txt)、[`/notices/plangothic-ofl.txt`](public/notices/plangothic-ofl.txt) 与 [`/notices/wenjin-mincho-ofl.md`](public/notices/wenjin-mincho-ofl.md)。生成的数据文件派生自上述来源,请遵守各自许可;逐项转换方式与署名也写入公开的 [`/notices/data-sources.md`](public/notices/data-sources.md)。 ## License diff --git a/app/components/CharCells.vue b/app/components/CharCells.vue index 0ad3d5a..fe7ca6f 100644 --- a/app/components/CharCells.vue +++ b/app/components/CharCells.vue @@ -65,7 +65,7 @@ const cells = computed(() => > /** * BCP-47 tag per column. This is accessibility semantics, unrelated to the - * interface language; it also drives the system's own CJK fallback when a - * subset has not loaded. The kyujitai is Japan's own, so it is tagged as + * interface language. The kyujitai is Japan's own, so it is tagged as * Japanese too. */ export const COLUMN_LANG: Record = { diff --git a/data/sources.lock.json b/data/sources.lock.json index 5633a25..aba1d53 100644 --- a/data/sources.lock.json +++ b/data/sources.lock.json @@ -2,14 +2,17 @@ "version": 1, "revisions": { "github:zispace/hanzi-chars@main": "105544909cab18db1fedfd41a42c87408fbaac59", - "github:BYVoid/OpenCC@master": "4f90418b9ed73a91023897095c762e5fdaadc016", + "github:BYVoid/OpenCC@master": "f5de4dd6eb96186e5ab9973043e19873ed678288", "github:parsimonhi/animCJK@master": "ec5e17cca76c87587790bcbce5ea0b4d4fb753d6", "github:adobe-fonts/source-han-sans@master": "0b993716f6910f0c8e00f957c767ab3cf5cb7602", "github:adobe-fonts/source-han-serif@master": "4356704ff7c68e9a84bf2bd489c9ebe186a9e2ef", "github:ruddfawcett/hanziDB.csv@master": "3e9f908d25f70674862343b201f6b641a36cd226", "github:scriptin/kanji-frequency@master": "62df93626e51a61c3dec58b51bfa20bef79491d7", "github:notofonts/noto-cjk@main": "f8d157532fbfaeda587e826d4cd5b21a49186f7c", + "github:Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project@main": "cfcb7cc869562443d6c6f85c935d0dd8dbdd69b6", + "github:takushun-wu/WenJinMincho@main": "a9b9653cf12857e10bdf1c22455afaa4e21dd9a3", "github:notofonts/notofonts.github.io@main": "3c16704cb6f6e7c02268f7bc0cf86aaee598d16f", + "github-release:Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project": "V2.9.5795", "unicode:ucd": "17.0.0" }, "assets": { @@ -59,27 +62,27 @@ "size": 7466 }, "opencc/STCharacters.txt": { - "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/4f90418b9ed73a91023897095c762e5fdaadc016/data/dictionary/STCharacters.txt", + "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/f5de4dd6eb96186e5ab9973043e19873ed678288/data/dictionary/STCharacters.txt", "sha256": "a0ca1601c70648cf48b33c3c6210ccbecc5c7eead4b4c3daf76587ba2c03582b", "size": 36026 }, "opencc/TSCharacters.txt": { - "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/4f90418b9ed73a91023897095c762e5fdaadc016/data/dictionary/TSCharacters.txt", + "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/f5de4dd6eb96186e5ab9973043e19873ed678288/data/dictionary/TSCharacters.txt", "sha256": "737c21c66f55a419dd6956cb3089476cdefc5a36877452631617696df1e5d925", "size": 104516 }, "opencc/TWVariants.txt": { - "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/4f90418b9ed73a91023897095c762e5fdaadc016/data/dictionary/TWVariants.txt", + "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/f5de4dd6eb96186e5ab9973043e19873ed678288/data/dictionary/TWVariants.txt", "sha256": "e187278e119c427ca561180ac5da5b20e9f8681190458f35c327ce499e95a6a5", "size": 986 }, "opencc/HKVariants.txt": { - "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/4f90418b9ed73a91023897095c762e5fdaadc016/data/dictionary/HKVariants.txt", + "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/f5de4dd6eb96186e5ab9973043e19873ed678288/data/dictionary/HKVariants.txt", "sha256": "e5cd4345303224587102f2c9e4d2b67d2b7e349c6ce9152e4a118f4656cf7302", "size": 1001 }, "opencc/JPShinjitaiCharacters.txt": { - "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/4f90418b9ed73a91023897095c762e5fdaadc016/data/dictionary/JPShinjitaiCharacters.txt", + "url": "https://raw.githubusercontent.com/BYVoid/OpenCC/f5de4dd6eb96186e5ab9973043e19873ed678288/data/dictionary/JPShinjitaiCharacters.txt", "sha256": "12cec7250b873ef52b36d8f92218d4f92c0aaf5d8cd7c58fe42d9785bdcdc43a", "size": 12442 }, @@ -238,6 +241,26 @@ "sha256": "77b4b741f864d27f15e90f275b17106dde90b2ad28f82bab72dc95805db5fb42", "size": 24539640 }, + "font/PlangothicP1-Regular.ttf": { + "url": "https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project/releases/download/V2.9.5795/PlangothicP1-Regular.ttf", + "sha256": "550b5d0775b15405946b18f4843df439a51e69508d7e6778d94c1f7a53dc5ad6", + "size": 20410664 + }, + "font/Plangothic-OFL.txt": { + "url": "https://raw.githubusercontent.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project/cfcb7cc869562443d6c6f85c935d0dd8dbdd69b6/LICENSE-OFL.txt", + "sha256": "e564f06d018e7b95bc3594c96a17f1d41865af4038c375e7aa974dd69df38602", + "size": 4302 + }, + "font/WenJinMinchoP2-Regular.otf": { + "url": "https://raw.githubusercontent.com/takushun-wu/WenJinMincho/a9b9653cf12857e10bdf1c22455afaa4e21dd9a3/otf/WenJinMinchoP2-Regular.otf", + "sha256": "8ed6e637aafd659e517b2af230b66ae587f23412223c054a29c1e2e64dbf1642", + "size": 22041624 + }, + "font/WenJinMincho-OFL.md": { + "url": "https://raw.githubusercontent.com/takushun-wu/WenJinMincho/a9b9653cf12857e10bdf1c22455afaa4e21dd9a3/LICENSE.md", + "sha256": "3f47d592920dbb67d90e2b248b26295703b3892cfeec465f1814fd71b2e902ef", + "size": 5356 + }, "font/NotoSans-VF.ttf": { "url": "https://raw.githubusercontent.com/notofonts/notofonts.github.io/3c16704cb6f6e7c02268f7bc0cf86aaee598d16f/fonts/NotoSans/unhinted/variable-ttf/NotoSans%5Bwdth,wght%5D.ttf", "sha256": "511cc04b442ffcc3b24d65c2cc780bdffaf482b20ba0aebdc8f27b37017ff0f5", diff --git a/docs/known-issues.md b/docs/known-issues.md index b69aa4c..f9ad1cd 100644 --- a/docs/known-issues.md +++ b/docs/known-issues.md @@ -27,14 +27,15 @@ ## 公共数据语义 -| 字段 | 含义 | -| -------------- | ----------------------------------------------------------------------------------- | -| `chars` | 五地展示形;未收录时仍保留可绘制的参考字形 | -| `listing` | 每格在该地区字表中的身份:`primary`、`glossed` 或 `unlisted` | -| `alternatives` | 该地区已收录、未展示且明确属于当前字组的形式;可搜索、作网址别名并显示字典引用 | -| `aka` | 同一字组的其他名称,不表示地区候选 | -| `uncertain` | 双向未确认关系,记录相关行、触发字形和地区;只用于详情页“另见” | -| `old` | 仅由 `JPShinjitaiCharacters` 的明确关系产生的日本旧字体,不由行名与日本列的差异反推 | +| 字段 | 含义 | +| ------------------ | ----------------------------------------------------------------------------------- | +| `chars` | 五地展示形;未收录时仍保留参考字形,明确映射不会因项目字体缺字而改回原码点 | +| `listing` | 每格在该地区字表中的身份:`primary`、`glossed` 或 `unlisted` | +| `alternatives` | 该地区已收录、未展示且明确属于当前字组的形式;可搜索、作网址别名并显示字典引用 | +| `aka` | 同一字组的其他名称,不表示地区候选 | +| `uncertain` | 双向未确认关系,记录相关行、触发字形和地区;只用于详情页“另见” | +| `old` | 仅由 `JPShinjitaiCharacters` 的明确关系产生的日本旧字体,不由行名与日本列的差异反推 | +| `supplementalFont` | 黑体、宋体各自需要跳过Noto、改用项目内置补充字体的列 | `listing` 的三个状态直接来自原始字表: @@ -46,7 +47,7 @@ ## 字形与笔画 -字形比较使用 Source Han Sans 与 Source Han Serif 的地区 CMap。同一套字体内 CID 相同即视为同形;黑体或宋体任一方认为两地同形,最终就按同形处理。这样会过滤只由一套字体作出的设计区分,但也意味着本应用衡量的是 Source Han 的地区设计,而不是规范文件本身。 +字形比较使用 Source Han Sans 与 Source Han Serif 的地区 CMap。同一套字体内 CID 相同即视为同形;黑体或宋体任一方认为两地同形,最终就按同形处理。这样会过滤只由一套字体作出的设计区分,但也意味着本应用衡量的是 Source Han 的地区设计,而不是规范文件本身。若明确的跨码点目标不在两套 CMap 中,同一码点的未测量列按一组保留,并通过 `supplementalFont` 改用项目内置的 Plangothic P1 或 WenJin Mincho P2 子集;不同码点仍保持不同。补充字形只用于渲染,不作为地区差异证据。 字表筛选、排序与详情页共用同一套地区笔画数,优先级为笔顺数据的轮廓数 → `kAlternateTotalStrokes` → 日本列的 `kRSAdobe_Japan1_6` → `kTotalStrokes`。香港复用笔顺的同形回退,按台湾、大陆、日本、韩国的顺序取首个可用数据。例如 `以` 五列为 `4 / 5 / 5 / 5 / 5`。 @@ -58,7 +59,7 @@ - 每格 `listing` 与原始主条目、括注和未收录状态一致; - `alternatives` 不属于其他最终字组; - 五份建行主表的每个主条目都被展示或记录为候选,否则构建失败; -- 所有展示形和日本旧字体都能由生成的字体子集绘制。 +- 每个展示形和日本旧字体都能由最终生成的Noto或补充字体子集绘制;标记为 `supplementalFont` 的码点还必须由对应的补充字体源覆盖。 ## 已知限制 diff --git a/public/notices/data-sources.md b/public/notices/data-sources.md index 574a741..6b4e9fd 100644 --- a/public/notices/data-sources.md +++ b/public/notices/data-sources.md @@ -7,14 +7,28 @@ Hanji 的程序代码采用 MIT 许可;来源数据及其派生字段仍须遵 - 用途:判定五地字形差异 - 来源:https://github.com/adobe-fonts/source-han-sans - 许可:SIL OFL 1.1 (https://openfontlicense.org/) -- 备注:每套字体的五份地区 CMap 给出「码点 → CID」映射,同一字形池内 CID 相同即同一字形。判定取黑体与宋体的并集。 +- 备注:每套字体的五份地区CMap给出「码点 → CID」映射,同一字形池内CID相同即同一字形。判定取黑体与宋体的并集。 ## Noto Sans / Noto Serif(含 CJK) - 用途:页面展示用字体 - 来源:https://github.com/notofonts/noto-cjk - 许可:SIL OFL 1.1 (https://openfontlicense.org/) -- 备注:汉字取自 CJK 版本,拉丁字母与数字取自拉丁版本,都按应用内用字子集化后自托管,OFL 声明随附于 /notices/noto-ofl.txt。 +- 备注:汉字取自CJK版本,拉丁字母与数字取自拉丁版本,都按应用内用字子集化后自托管,OFL声明随附于 /notices/noto-ofl.txt。 + +## Plangothic P1 + +- 用途:补充Noto未收录的黑体字形 +- 来源:https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project +- 许可:SIL OFL 1.1 (https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project/blob/main/LICENSE-OFL.txt) +- 备注:按当前数据中Noto未收录但需要显示的码点,动态生成重命名后的网页字体子集;这些字形不参与地区差异判定。OFL声明随附于/notices/plangothic-ofl.txt。 + +## WenJin Mincho P2 + +- 用途:补充Noto未收录的宋体字形 +- 来源:https://github.com/takushun-wu/WenJinMincho +- 许可:SIL OFL 1.1 (https://github.com/takushun-wu/WenJinMincho/blob/main/LICENSE.md) +- 备注:按当前数据中Noto未收录但需要显示的码点,动态生成重命名后的网页字体子集;这些字形不参与地区差异判定。OFL声明随附于/notices/wenjin-mincho-ofl.md。 ## OpenCC 开放中文转换 diff --git a/public/notices/plangothic-ofl.txt b/public/notices/plangothic-ofl.txt new file mode 100644 index 0000000..77b1731 --- /dev/null +++ b/public/notices/plangothic-ofl.txt @@ -0,0 +1,91 @@ +This Font Software is licensed under the SIL Open Font License, Version 1.1. +This license is copied below, and is also available with a FAQ at: +http://scripts.sil.org/OFL + + +----------------------------------------------------------- +SIL OPEN FONT LICENSE Version 1.1 - 26 February 2007 +----------------------------------------------------------- + +PREAMBLE +The goals of the Open Font License (OFL) are to stimulate worldwide +development of collaborative font projects, to support the font creation +efforts of academic and linguistic communities, and to provide a free and +open framework in which fonts may be shared and improved in partnership +with others. + +The OFL allows the licensed fonts to be used, studied, modified and +redistributed freely as long as they are not sold by themselves. The +fonts, including any derivative works, can be bundled, embedded, +redistributed and/or sold with any software provided that any reserved +names are not used by derivative works. The fonts and derivatives, +however, cannot be released under any other type of license. The +requirement for fonts to remain under this license does not apply +to any document created using the fonts or their derivatives. + +DEFINITIONS +"Font Software" refers to the set of files released by the Copyright +Holder(s) under this license and clearly marked as such. This may +include source files, build scripts and documentation. + +"Reserved Font Name" refers to any names specified as such after the +copyright statement(s). + +"Original Version" refers to the collection of Font Software components as +distributed by the Copyright Holder(s). + +"Modified Version" refers to any derivative made by adding to, deleting, +or substituting -- in part or in whole -- any of the components of the +Original Version, by changing formats or by porting the Font Software to a +new environment. + +"Author" refers to any designer, engineer, programmer, technical +writer or other person who contributed to the Font Software. + +PERMISSION & CONDITIONS +Permission is hereby granted, free of charge, to any person obtaining +a copy of the Font Software, to use, study, copy, merge, embed, modify, +redistribute, and sell modified and unmodified copies of the Font +Software, subject to the following conditions: + +1) Neither the Font Software nor any of its individual components, +in Original or Modified Versions, may be sold by itself. + +2) Original or Modified Versions of the Font Software may be bundled, +redistributed and/or sold with any software, provided that each copy +contains the above copyright notice and this license. These can be +included either as stand-alone text files, human-readable headers or +in the appropriate machine-readable metadata fields within text or +binary files as long as those fields can be easily viewed by the user. + +3) No Modified Version of the Font Software may use the Reserved Font +Name(s) unless explicit written permission is granted by the corresponding +Copyright Holder. This restriction only applies to the primary font name as +presented to the users. + +4) The name(s) of the Copyright Holder(s) or the Author(s) of the Font +Software shall not be used to promote, endorse or advertise any +Modified Version, except to acknowledge the contribution(s) of the +Copyright Holder(s) and the Author(s) or with their explicit written +permission. + +5) The Font Software, modified or unmodified, in part or in whole, +must be distributed entirely under this license, and must not be +distributed under any other license. The requirement for fonts to +remain under this license does not apply to any document created +using the Font Software. + +TERMINATION +This license becomes null and void if any of the above conditions are +not met. + +DISCLAIMER +THE FONT SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTIES OF +MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT +OF COPYRIGHT, PATENT, TRADEMARK, OR OTHER RIGHT. IN NO EVENT SHALL THE +COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +INCLUDING ANY GENERAL, SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL +DAMAGES, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +FROM, OUT OF THE USE OR INABILITY TO USE THE FONT SOFTWARE OR FROM +OTHER DEALINGS IN THE FONT SOFTWARE. diff --git a/public/notices/wenjin-mincho-ofl.md b/public/notices/wenjin-mincho-ofl.md new file mode 100644 index 0000000..75c26d3 --- /dev/null +++ b/public/notices/wenjin-mincho-ofl.md @@ -0,0 +1,126 @@ +Copyright (c) 2024-2026, Takushun Wu (https://github.com/takushun-wu/), +with Reserved Font Name 'WenJin Mincho', '文津宋体', '文津宋體', '文津明朝', '문진(文津) 명조'. + +Copyright 2010-2025 Adobe (http://www.adobe.com/), with Reserved Font +Name 'Source'. All Rights Reserved. Source is a trademark of Adobe in the United States +and/or other countries. + +Copyright 2021-2025 Tamcy (https://github.com/chiron-fonts/chiron-sung-hk). + +Copyright 2022-2025 Shanggu Fonts (https://github.com/GuiWonder/Shanggu). + +© 2007–2024 Adobe, But Ko, CMEX, Creative Commons Corporation, GlyphWiki & Night Koo. + +Copyright 2022 The Noto Project Authors (https://github.com/notofonts/). Noto is a trademark of Google Inc. + +Copyright © 2015 Google Inc. + +Saudi Riyal Font © Emran Alhaddad - Used under SIL Open Font License 1.1 +  + + +This Font Software is licensed under the SIL Open Font License, Version 1.1. +This license is copied below, and is also available with a FAQ at: +https\://openfontlicense.org +  + +\---------------------------------------------------------------------- + +#### SIL OPEN FONT LICENSE Version 1.1 - 26 February 2007 + +\---------------------------------------------------------------------- + +  + +PREAMBLE +----------- + +The goals of the Open Font License (OFL) are to stimulate worldwide +development of collaborative font projects, to support the font creation +efforts of academic and linguistic communities, and to provide a free and +open framework in which fonts may be shared and improved in partnership +with others. + +The OFL allows the licensed fonts to be used, studied, modified and +redistributed freely as long as they are not sold by themselves. The +fonts, including any derivative works, can be bundled, embedded, +redistributed and/or sold with any software provided that any reserved +names are not used by derivative works. The fonts and derivatives, +however, cannot be released under any other type of license. The +requirement for fonts to remain under this license does not apply +to any document created using the fonts or their derivatives. + +DEFINITIONS +----------- + +"Font Software" refers to the set of files released by the Copyright +Holder(s) under this license and clearly marked as such. This may +include source files, build scripts and documentation. + +"Reserved Font Name" refers to any names specified as such after the +copyright statement(s). + +"Original Version" refers to the collection of Font Software components as +distributed by the Copyright Holder(s). + +"Modified Version" refers to any derivative made by adding to, deleting, +or substituting -- in part or in whole -- any of the components of the +Original Version, by changing formats or by porting the Font Software to a +new environment. + +"Author" refers to any designer, engineer, programmer, technical +writer or other person who contributed to the Font Software. + +PERMISSION & CONDITIONS +----------- + +Permission is hereby granted, free of charge, to any person obtaining +a copy of the Font Software, to use, study, copy, merge, embed, modify, +redistribute, and sell modified and unmodified copies of the Font +Software, subject to the following conditions: + +1) Neither the Font Software nor any of its individual components, +in Original or Modified Versions, may be sold by itself. + +2) Original or Modified Versions of the Font Software may be bundled, +redistributed and/or sold with any software, provided that each copy +contains the above copyright notice and this license. These can be +included either as stand-alone text files, human-readable headers or +in the appropriate machine-readable metadata fields within text or +binary files as long as those fields can be easily viewed by the user. + +3) No Modified Version of the Font Software may use the Reserved Font +Name(s) unless explicit written permission is granted by the corresponding +Copyright Holder. This restriction only applies to the primary font name as +presented to the users. + +4) The name(s) of the Copyright Holder(s) or the Author(s) of the Font +Software shall not be used to promote, endorse or advertise any +Modified Version, except to acknowledge the contribution(s) of the +Copyright Holder(s) and the Author(s) or with their explicit written +permission. + +5) The Font Software, modified or unmodified, in part or in whole, +must be distributed entirely under this license, and must not be +distributed under any other license. The requirement for fonts to +remain under this license does not apply to any document created +using the Font Software. + +TERMINATION +----------- + +This license becomes null and void if any of the above conditions are +not met. + +DISCLAIMER +----------- + +THE FONT SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTIES OF +MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT +OF COPYRIGHT, PATENT, TRADEMARK, OR OTHER RIGHT. IN NO EVENT SHALL THE +COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +INCLUDING ANY GENERAL, SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL +DAMAGES, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +FROM, OUT OF THE USE OR INABILITY TO USE THE FONT SOFTWARE OR FROM +OTHER DEALINGS IN THE FONT SOFTWARE. diff --git a/scripts/build-data.ts b/scripts/build-data.ts index 1292757..6e38da4 100644 --- a/scripts/build-data.ts +++ b/scripts/build-data.ts @@ -13,7 +13,19 @@ import { mkdir, readFile, rm, writeFile } from 'node:fs/promises' import { join } from 'node:path' import { format, resolveConfig } from 'prettier' +import { fontIndexOf } from '../shared/row.ts' import { SOURCES } from '../shared/sources.ts' +import { + STYLES, + type CharRow, + type Column, + type ListedAlternative, + type Region, + type RegionalTuple, + type Stats, + type Style, + type UncertainRelation, +} from '../shared/types.ts' import { buildStrokeData } from './build-strokes.ts' import { parseCMap, partitionSignature } from './cmap.ts' import { @@ -31,15 +43,6 @@ import { reverseDict, ROOT, } from './sources.ts' -import type { - CharRow, - ListedAlternative, - Region, - RegionalTuple, - Stats, - UncertainRelation, -} from '../shared/types.ts' - const dict = async (name: string) => parseDict(await rawText(`opencc/${name}.txt`)) @@ -360,7 +363,10 @@ function pick( * glyph. Source Han Sans hands Japan its own glyph for a fifth of common * characters; for a couple of hundred of those the serif faces do not, which * marks the difference as that typeface's design decision rather than a - * regional one. Agreement in either face is therefore taken as agreement. + * regional one. Agreement in either face is therefore taken as agreement. If + * neither face covers a codepoint, its same-codepoint columns remain one + * unmeasured group; different codepoints still remain distinct and are drawn + * by the project's supplemental fonts. * * The union of two equivalence relations need not be transitive, so the groups * are the connected components rather than the pairs themselves. @@ -368,6 +374,7 @@ function pick( function unifiedSignature( sans: (number | undefined)[], serif: (number | undefined)[], + codePoints?: (number | undefined)[], ): string { const parent = sans.map((_, i) => i) const find = (x: number): number => @@ -378,7 +385,15 @@ function unifiedSignature( for (let i = 0; i < parent.length; i++) for (let j = i + 1; j < parent.length; j++) { - if (!agree(sans, i, j) && !agree(serif, i, j)) continue + const sameUnmeasuredCodePoint = + codePoints?.[i] !== undefined && + codePoints[i] === codePoints[j] && + sans[i] === undefined && + sans[j] === undefined && + serif[i] === undefined && + serif[j] === undefined + if (!agree(sans, i, j) && !agree(serif, i, j) && !sameUnmeasuredCodePoint) + continue const [ri, rj] = [find(i), find(j)] if (ri !== rj) parent[rj] = ri } @@ -437,12 +452,9 @@ function buildRow( char, ]) as RegionalTuple, alternatives?: CharRow['alternatives'], -): CharRow | undefined { +): CharRow { const codePoints = chars.map((c) => c.codePointAt(0)) as RegionalTuple const sansCids = codePoints.map((cp, i) => sansCMaps[i]!.get(cp)) - // A character Noto does not cover can be neither drawn nor judged - if (sansCids.includes(undefined)) return undefined - const serifCids = codePoints.map((cp, i) => serifCMaps[i]!.get(cp)) /** @@ -456,13 +468,14 @@ function buildRow( const oldPoint = oldChar?.codePointAt(0) const oldSans = oldPoint === undefined ? undefined : sansCMaps[JP_INDEX]!.get(oldPoint) - const hasOld = oldChar !== undefined && oldSans !== undefined + const hasOld = oldChar !== undefined const oldSerif = oldPoint === undefined ? undefined : serifCMaps[JP_INDEX]!.get(oldPoint) const signature = unifiedSignature( hasOld ? [...sansCids, oldSans] : sansCids, hasOld ? [...serifCids, oldSerif] : serifCids, + hasOld ? [...codePoints, oldPoint] : codePoints, ) const glyph = signature.slice(0, REGION_IDS.length) const cp = partitionSignature(codePoints) @@ -512,6 +525,26 @@ function buildRow( common: tier.filter(Boolean).length, ...(Object.keys(readings).length > 0 ? { readings } : {}), } + + const mapsByStyle: Record = { + sans: sansCMaps, + serif: serifCMaps, + } + const supplementalFont: Partial> = {} + for (const style of STYLES) { + const maps = mapsByStyle[style] + const columns: Column[] = [] + for (const [region, id] of REGION_IDS.entries()) { + const font = fontIndexOf(row, region) + if (!maps[font]!.has(codePoints[region]!)) columns.push(id) + } + if (hasOld && oldPoint !== undefined && !maps[JP_INDEX]!.has(oldPoint)) + columns.push('old') + if (columns.length > 0) supplementalFont[style] = columns + } + if (Object.keys(supplementalFont).length > 0) + row.supplementalFont = supplementalFont + candidatesByRow.set(row, candidates) return row } @@ -522,7 +555,7 @@ for (const key of keys) { (forms) => forms[0] ?? key, ) as RegionalTuple const row = buildRow(key, chars, undefined, candidates) - if (row) rows.push(row) + rows.push(row) if (++done % 1000 === 0) console.error(` ${done}/${keys.size}`) } @@ -815,10 +848,6 @@ function fold(all: CharRow[]): CharRow[] { const aka = names.filter((name) => name !== key) const folded = buildRow(key!, chars, aka, regionalCandidates, alternatives) - if (!folded) - throw new Error( - `cannot render folded row ${key}: ${REGION_IDS.map((id, region) => `${id}:${chars[region]}`).join(' ')}; names=${names.join(' ')} candidates=${JSON.stringify(regionalCandidates)}`, - ) rootsByRow.set(folded, [groupRoot]) out.push(folded) } @@ -832,10 +861,10 @@ rows = fold(rows).filter((row) => row.common > 0) /** * Every primary source entry must be represented as either the displayed form * or a listed alternative. A few mainland entries have an ST mapping whose - * orthodox target is absent from the reverse TS table (昵 -> 暱, 稆 -> 穭), or - * whose target cannot be drawn in every regional font. In that case the source - * entry itself is still drawable and gets a conservative, unconverted row - * rather than disappearing during normalization. + * orthodox target is absent from the reverse TS table (昵 -> 暱, 稆 -> 穭). + * Anything normalization still failed to account for gets its own row rather + * than disappearing; font coverage is recorded separately by buildRow and + * never changes the codepoint selected here. */ const rowEntries: Set[] = [ new Set([...entries['cn-1'], ...entries['cn-2'], ...entries['cn-3']]), @@ -852,19 +881,14 @@ const accountsFor = (region: number, char: string) => (entry) => entry.char === char, ), ) -const unrenderable: string[] = [] for (const [region, source] of rowEntries.entries()) for (const char of source) { if (accountsFor(region, char)) continue const chars = REGION_IDS.map(() => char) as RegionalTuple const fallback = buildRow(char, chars) - if (fallback) { - rootsByRow.set(fallback, [rootName(char)]) - rows.push(fallback) - } else unrenderable.push(`${REGION_IDS[region]}:${char}`) + rootsByRow.set(fallback, [rootName(char)]) + rows.push(fallback) } -if (unrenderable.length > 0) - throw new Error(`unrenderable source entries: ${unrenderable.join(' ')}`) /** * Substituting a form can leave two rows with the same five columns -- one of diff --git a/scripts/build-fonts.ts b/scripts/build-fonts.ts index 95bfe3d..3af9a7d 100644 --- a/scripts/build-fonts.ts +++ b/scripts/build-fonts.ts @@ -36,7 +36,7 @@ import { type Locale, } from '../app/locales/index.ts' import { dictLinks, formsOf } from '../shared/links.ts' -import { fontIndexOf } from '../shared/row.ts' +import { fontIndexOf, usesSupplementalFont } from '../shared/row.ts' import { SOURCES } from '../shared/sources.ts' import { REGIONS, STYLES, type CharsData, type Style } from '../shared/types.ts' import { DATA_DIR, FONT_DIR, NOTICES_DIR, raw, ROOT } from './sources.ts' @@ -54,6 +54,16 @@ const NOTO: Record = { const otf = (style: Style, region: string) => `font/Noto${style === 'sans' ? 'Sans' : 'Serif'}CJK${NOTO[region]}-Regular.otf` +const SUPPLEMENTAL_FONT: Record = { + sans: 'font/PlangothicP1-Regular.ttf', + serif: 'font/WenJinMinchoP2-Regular.otf', +} + +const SUPPLEMENTAL_FAMILY: Record = { + sans: 'Hanji Rare Sans', + serif: 'Hanji Rare Serif', +} + /** Characters per chunk. Smaller means a lighter first paint but more * @font-face rules and more requests. */ const CHUNK_SIZE = 400 @@ -75,30 +85,75 @@ for (const name of await readdir(FONT_DIR)) if (name.endsWith('.woff2')) await unlink(join(FONT_DIR, name)) await mkdir(NOTICES_DIR, { recursive: true }) -await writeFile(join(NOTICES_DIR, 'noto-ofl.txt'), await raw('font/OFL.txt')) +await Promise.all([ + writeFile(join(NOTICES_DIR, 'noto-ofl.txt'), await raw('font/OFL.txt')), + writeFile( + join(NOTICES_DIR, 'plangothic-ofl.txt'), + await raw('font/Plangothic-OFL.txt'), + ), + writeFile( + join(NOTICES_DIR, 'wenjin-mincho-ofl.md'), + await raw('font/WenJinMincho-OFL.md'), + ), +]) /** Collect the characters each region needs, in display-priority order. */ -const needed: Record = Object.fromEntries( - REGIONS.map((r) => [r, [] as string[]]), -) -const seen: Record> = Object.fromEntries( - REGIONS.map((r) => [r, new Set()]), -) +const needed = Object.fromEntries( + STYLES.map((style) => [ + style, + Object.fromEntries(REGIONS.map((region) => [region, [] as string[]])), + ]), +) as Record> +const seen = Object.fromEntries( + STYLES.map((style) => [ + style, + Object.fromEntries(REGIONS.map((region) => [region, new Set()])), + ]), +) as Record>> +const supplementalNeeded = Object.fromEntries( + STYLES.map((style) => [style, [] as string[]]), +) as Record +const supplementalSeen = Object.fromEntries( + STYLES.map((style) => [style, new Set()]), +) as Record> + +const need = (style: Style, region: (typeof REGIONS)[number], char: string) => { + if (seen[style][region].has(char)) return + seen[style][region].add(char) + needed[style][region].push(char) +} -const need = (region: string, char: string) => { - if (seen[region]!.has(char)) return - seen[region]!.add(char) - needed[region]!.push(char) +const supplement = (style: Style, char: string) => { + if (supplementalSeen[style].has(char)) return + supplementalSeen[style].add(char) + supplementalNeeded[style].push(char) } const collect = (row: (typeof data.rows)[number]) => { - for (let i = 0; i < REGIONS.length; i++) - need(REGIONS[fontIndexOf(row, i)]!, row.chars[i]!) - // The Japanese column also shows kyujitai, which has no group to share with - if (row.old) need('jp', row.old.char) - // A key or merged-in name the columns never show still appears on the - // character page, next to the references that look it up - for (const form of formsOf(row)) need(form.font, form.char) + for (const style of STYLES) { + for (let i = 0; i < REGIONS.length; i++) { + const column = REGIONS[i]! + if (usesSupplementalFont(row, column, style)) { + supplement(style, row.chars[i]!) + continue + } + need(style, REGIONS[fontIndexOf(row, i)]!, row.chars[i]!) + } + // The Japanese column also shows kyujitai, which has no group to share + // with. + if (row.old) { + if (usesSupplementalFont(row, 'old', style)) + supplement(style, row.old.char) + else need(style, 'jp', row.old.char) + } + // A key or merged-in name the columns never show still appears on the + // character page, next to the references that look it up. + for (const form of formsOf(row)) { + if (form.column && usesSupplementalFont(row, form.column, style)) + supplement(style, form.char) + else need(style, form.font, form.char) + } + } } /** @@ -211,7 +266,7 @@ const styleChunks: Record = { sans: [], serif: [] } for (const style of STYLES) { for (const region of REGIONS) { - const chars = needed[region]! + const chars = needed[style][region] for (let start = 0, index = 0; start < chars.length; start += CHUNK_SIZE) { const chunk = chars.slice(start, start + CHUNK_SIZE) const file = `hanji-${style}-${region}-${index}.woff2` @@ -231,6 +286,57 @@ for (const style of STYLES) { } } +/** + * Source Han/Noto can omit explicit mapping targets in the current data. Keep + * them as real Unicode text and provide one small face per style instead of + * depending on an unknown system-font fallback chain. The source font must + * cover every dynamically collected target or the build fails above output. + */ +for (const style of STYLES) { + const chars = supplementalNeeded[style] + if (chars.length === 0) continue + + const source = fontkit.create( + Buffer.from(await raw(SUPPLEMENTAL_FONT[style])), + ) as fontkit.Font + const missing = chars.filter( + (char) => source.glyphForCodePoint(char.codePointAt(0)!).id === 0, + ) + if (missing.length > 0) + throw new Error( + `${SUPPLEMENTAL_FONT[style]} does not cover supplemental characters: ${missing.join(' ')}`, + ) + + const file = `hanji-rare-${style}.woff2` + styleChunks[style]!.push( + queue({ + font: SUPPLEMENTAL_FONT[style], + text: chars.join(''), + file, + names: + style === 'sans' + ? { + family: 'HJS', + fullName: 'HJS', + postscriptName: 'HJS', + } + : { + family: 'HJR', + fullName: 'HJR', + postscriptName: 'HJR', + }, + }), + ) + faces[style]!.push( + `@font-face { + font-family: '${SUPPLEMENTAL_FAMILY[style]}'; + src: url('./${file}') format('woff2'); + font-display: block; + unicode-range: ${unicodeRange(chars)}; +}`, + ) +} + /** * The interface copy gets the same treatment as the table. * @@ -429,16 +535,27 @@ export const FACE_MARKS: Record> = ${JSON.stringi await writeFile(join(ROOT, 'app/generated/face-marks.ts'), await faceMarks()) console.error('app/generated/face-marks.ts') -const banner = `/* Generated by scripts/build-fonts.ts -- do not edit. +const banners: Record = { + ui: `/* Generated by scripts/build-fonts.ts -- do not edit. * Noto Sans CJK and Noto Serif CJK (SIL OFL 1.1), subset to the characters * this app uses. Licence text is served at /notices/noto-ofl.txt. - */` + */`, + sans: `/* Generated by scripts/build-fonts.ts -- do not edit. + * Noto Sans CJK plus a data-driven Plangothic P1 supplement, both under SIL + * OFL 1.1. Notices: /notices/noto-ofl.txt and /notices/plangothic-ofl.txt. + */`, + serif: `/* Generated by scripts/build-fonts.ts -- do not edit. + * Noto Serif CJK plus a data-driven WenJin Mincho P2 supplement, both under + * SIL OFL 1.1. Notices: /notices/noto-ofl.txt and + * /notices/wenjin-mincho-ofl.md. + */`, +} // Serif ships as its own stylesheet so the ~140KB of @font-face rules only // arrive when a reader actually asks for serif. let cssBytes = 0 for (const [name, rules] of Object.entries(faces)) { - const css = `${banner}\n\n${rules.join('\n\n')}\n` + const css = `${banners[name]}\n\n${rules.join('\n\n')}\n` await writeFile(join(FONT_DIR, `fonts-${name}.css`), css) cssBytes += css.length console.error(`fonts-${name}.css ${(css.length / 1024).toFixed(0)} KB`) diff --git a/scripts/sfnt.ts b/scripts/sfnt.ts new file mode 100644 index 0000000..5900cb5 --- /dev/null +++ b/scripts/sfnt.ts @@ -0,0 +1,107 @@ +import { Buffer } from 'node:buffer' + +export interface SfntNames { + family: string + fullName?: string + postscriptName: string +} + +interface TableRecord { + directoryOffset: number + offset: number + length: number +} + +const uint32 = (value: number) => value >>> 0 + +function tableRecords(font: Buffer): Map { + const tables = new Map() + const count = font.readUInt16BE(4) + for (let index = 0; index < count; index++) { + const directoryOffset = 12 + index * 16 + const tag = font.toString('ascii', directoryOffset, directoryOffset + 4) + tables.set(tag, { + directoryOffset, + offset: font.readUInt32BE(directoryOffset + 8), + length: font.readUInt32BE(directoryOffset + 12), + }) + } + return tables +} + +function checksum(font: Buffer, offset: number, length: number): number { + let sum = 0 + const end = offset + Math.ceil(length / 4) * 4 + for (let cursor = offset; cursor < end; cursor += 4) { + let word = 0 + for (let byte = 0; byte < 4; byte++) + word = (word << 8) | (font[cursor + byte] ?? 0) + sum = uint32(sum + uint32(word)) + } + return sum +} + +function encodedName(platform: number, value: string): Buffer { + if (platform !== 0 && platform !== 3) return Buffer.from(value, 'ascii') + const encoded = Buffer.alloc(value.length * 2) + for (let index = 0; index < value.length; index++) + encoded.writeUInt16BE(value.codePointAt(index)!, index * 2) + return encoded +} + +/** + * Replace the user-facing names in an SFNT font without disturbing its + * copyright and licence records. Every replacement is shorter than the + * upstream name, so the name table can retain its existing storage offsets. + * HarfBuzz rebuilds the table when it creates the final subset. + */ +export function renameSfnt(font: Buffer, names: SfntNames): Buffer { + const output = Buffer.from(font) + const tables = tableRecords(output) + const name = tables.get('name') + const head = tables.get('head') + if (!name || !head) throw new Error('font has no name or head table') + + const count = output.readUInt16BE(name.offset + 2) + const strings = name.offset + output.readUInt16BE(name.offset + 4) + const fullName = names.fullName ?? names.family + const replacements: Partial> = { + 1: names.family, + 3: names.postscriptName, + 4: fullName, + 6: names.postscriptName, + 16: names.family, + 21: names.family, + 25: names.postscriptName.replace(/-Regular$/, ''), + } + + for (let index = 0; index < count; index++) { + const record = name.offset + 6 + index * 12 + const value = replacements[output.readUInt16BE(record + 6)] + if (!value) continue + const encoded = encodedName(output.readUInt16BE(record), value) + const capacity = output.readUInt16BE(record + 8) + if (encoded.length > capacity) + throw new Error(`replacement font name is too long: ${value}`) + const offset = strings + output.readUInt16BE(record + 10) + output.fill(0, offset, offset + capacity) + encoded.copy(output, offset) + output.writeUInt16BE(encoded.length, record + 8) + } + + // Keep the input font internally consistent before handing it to HarfBuzz. + output.writeUInt32BE( + checksum(output, name.offset, name.length), + name.directoryOffset + 4, + ) + output.writeUInt32BE(0, head.offset + 8) + output.writeUInt32BE( + checksum(output, head.offset, head.length), + head.directoryOffset + 4, + ) + output.writeUInt32BE( + uint32(0xb1b0afba - checksum(output, 0, output.length)), + head.offset + 8, + ) + return output +} diff --git a/scripts/sources.ts b/scripts/sources.ts index 5f3e609..4b6bcc7 100644 --- a/scripts/sources.ts +++ b/scripts/sources.ts @@ -59,6 +59,9 @@ const noto = (style: 'Sans' | 'Serif', region: string) => `${style}/OTF/${NOTO_DIR[region]}/Noto${style}CJK${region}-Regular.otf`, ) +const latestGitHubRelease = (repo: string, file: string) => + `https://github.com/${repo}/releases/latest/download/${file}` + /** * Cache path -> moving upstream URL. `pnpm update:sources` resolves these refs * to immutable versions and writes their checksums to SOURCE_LOCK_PATH. Builds @@ -128,6 +131,26 @@ export const ASSET_URLS: Record = { 'font/NotoSerifCJKjp-Regular.otf': noto('Serif', 'jp'), 'font/NotoSerifCJKkr-Regular.otf': noto('Serif', 'kr'), + 'font/PlangothicP1-Regular.ttf': latestGitHubRelease( + 'Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project', + 'PlangothicP1-Regular.ttf', + ), + 'font/Plangothic-OFL.txt': gh( + 'Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project', + 'main', + 'LICENSE-OFL.txt', + ), + 'font/WenJinMinchoP2-Regular.otf': gh( + 'takushun-wu/WenJinMincho', + 'main', + 'otf/WenJinMinchoP2-Regular.otf', + ), + 'font/WenJinMincho-OFL.md': gh( + 'takushun-wu/WenJinMincho', + 'main', + 'LICENSE.md', + ), + // Latin, digits and punctuation for the interface, plus the tone marks the // readings carry. Cut from faces designed for Latin rather than from the // CJK families, whose Latin is a compromise. diff --git a/scripts/subset-worker.ts b/scripts/subset-worker.ts index fdd8b22..1de45cc 100644 --- a/scripts/subset-worker.ts +++ b/scripts/subset-worker.ts @@ -13,6 +13,7 @@ import { readFile, writeFile } from 'node:fs/promises' import { join } from 'node:path' import { parentPort } from 'node:worker_threads' import subsetFont from 'subset-font' +import { renameSfnt, type SfntNames } from './sfnt.ts' import { FONT_DIR, RAW_DIR } from './sources.ts' import type { Buffer } from 'node:buffer' @@ -22,6 +23,8 @@ export interface SubsetJob { font: string text: string file: string + /** Replace Reserved Font Names before publishing a modified web subset. */ + names?: SfntNames } export interface SubsetDone { @@ -34,16 +37,18 @@ export interface SubsetDone { * The queue hands out jobs in order, so every worker is somewhere in the same * run of chunks and holding one source covers a long stretch of them. */ -let loadedName: string | undefined +let loadedKey: string | undefined let loadedFont: Buffer | undefined const port = parentPort! port.on('message', async (job: SubsetJob) => { try { - if (loadedName !== job.font) { - loadedFont = await readFile(join(RAW_DIR, job.font)) - loadedName = job.font + const key = `${job.font}\u{0}${JSON.stringify(job.names)}` + if (loadedKey !== key) { + const source = await readFile(join(RAW_DIR, job.font)) + loadedFont = job.names ? renameSfnt(source, job.names) : source + loadedKey = key } const subset = await subsetFont(loadedFont!, job.text, { targetFormat: 'woff2', diff --git a/scripts/tests/chars.test.ts b/scripts/tests/chars.test.ts index d81ec04..f9b2e30 100644 --- a/scripts/tests/chars.test.ts +++ b/scripts/tests/chars.test.ts @@ -636,12 +636,25 @@ describe('source-entry accounting', () => { expect(row('穭').chars[0]).toBe('稆') }) - it('keeps drawable entries when their mapped form cannot be rendered', () => { - for (const char of ['𫭼', '暅', '𬒗']) { - expect(row(char).chars).toEqual([char, char, char, char, char]) - expect(row(char).tier[0]).toBeGreaterThan(0) - } - }) + it.each([ + ['𡑍', '𫭼'], + ['𣈶', '暅'], + ['𥗽', '𬒗'], + ])( + 'keeps the explicit %s mapping when bundled fonts only cover %s', + (orthodox, mainland) => { + expect(row(orthodox)).toMatchObject({ + chars: [mainland, orthodox, orthodox, orthodox, orthodox], + cp: '01111', + glyph: '01111', + supplementalFont: { + sans: ['hk', 'tw', 'jp', 'kr'], + serif: ['hk', 'tw', 'jp', 'kr'], + }, + }) + expect(row(orthodox).tier[0]).toBeGreaterThan(0) + }, + ) it('does not duplicate the selected form among regional alternatives', () => { for (const entry of data.rows) diff --git a/scripts/tests/fonts.test.ts b/scripts/tests/fonts.test.ts index 1baa412..1d96e22 100644 --- a/scripts/tests/fonts.test.ts +++ b/scripts/tests/fonts.test.ts @@ -7,22 +7,26 @@ * they ever disagree, the table would be labelling differences the reader * cannot see. */ -import { readdirSync, readFileSync } from 'node:fs' +import { readdirSync, readFileSync, statSync } from 'node:fs' import { join } from 'node:path' import * as fontkit from 'fontkit' import { describe, expect, it } from 'vitest' import { frequencyRankOf } from '../../shared/frequency.ts' +import { formsOf } from '../../shared/links.ts' import { fontIndexOf, fontRegionOf, projectSignature, + usesSupplementalFont, varietyOf, } from '../../shared/row.ts' import { FREQUENCY_REGIONS, REGIONS, + STYLES, type CharsData, type FrequencyRegion, + type Style, } from '../../shared/types.ts' import { partitionSignature } from '../cmap.ts' import { DATA_DIR, FONT_DIR } from '../sources.ts' @@ -32,6 +36,41 @@ const data: CharsData = JSON.parse( ) const rows = new Map(data.rows.map((row) => [row.key, row])) +/** The data, rather than this test, decides which codepoints Noto cannot draw. */ +const supplementalChars = Object.fromEntries( + STYLES.map((style) => [style, new Set()]), +) as Record> + +for (const row of data.rows) { + for (const style of STYLES) { + for (const [index, region] of REGIONS.entries()) + if (usesSupplementalFont(row, region, style)) + supplementalChars[style].add(row.chars[index]!) + if (row.old && usesSupplementalFont(row, 'old', style)) + supplementalChars[style].add(row.old.char) + for (const form of formsOf(row)) + if (form.column && usesSupplementalFont(row, form.column, style)) + supplementalChars[style].add(form.char) + } +} + +function unicodeRange(chars: Iterable): string { + const points = [...chars] + .map((char) => char.codePointAt(0)!) + .toSorted((left, right) => left - right) + const parts: string[] = [] + for (let index = 0; index < points.length;) { + let end = index + while (end + 1 < points.length && points[end + 1] === points[end]! + 1) + end++ + const start = points[index]!.toString(16).toUpperCase() + const finish = points[end]!.toString(16).toUpperCase() + parts.push(index === end ? `U+${start}` : `U+${start}-${finish}`) + index = end + 1 + } + return parts.join(',') +} + /** Which chunk holds a character is not known up front, so open them all. */ const fontsOf = (region: string) => readdirSync(FONT_DIR) @@ -45,6 +84,14 @@ const fontsOf = (region: string) => })) const fonts = Object.fromEntries(REGIONS.map((r) => [r, fontsOf(r)])) +const supplementalFonts = Object.fromEntries( + STYLES.map((style) => [ + style, + fontkit.create( + readFileSync(join(FONT_DIR, `hanji-rare-${style}.woff2`)), + ) as fontkit.Font, + ]), +) as Record /** Glyph IDs are not comparable across subsets, so compare outlines. */ function outline(region: string, char: string): string { @@ -56,6 +103,23 @@ function outline(region: string, char: string): string { throw new Error(`no chunk of hanji-sans-${region} carries ${char}`) } +function supplementalOutline(style: Style, char: string): string { + const glyph = supplementalFonts[style].glyphForCodePoint(char.codePointAt(0)!) + if (glyph.id === 0) + throw new Error(`hanji-rare-${style}.woff2 does not carry ${char}`) + return glyph.path.toSVG() +} + +function displayedSansOutline( + row: (typeof data.rows)[number], + region: number, +): string { + const column = REGIONS[region]! + return usesSupplementalFont(row, column, 'sans') + ? supplementalOutline('sans', row.chars[region]!) + : outline(REGIONS[fontIndexOf(row, region)]!, row.chars[region]!) +} + /** Which generated chunk carries this region's character. */ function shardOf(region: string, char: string): number { const codePoint = char.codePointAt(0)! @@ -104,17 +168,51 @@ describe('subset coverage', () => { ...data.rows.filter((r) => r.glyph === '01234').slice(0, 40), ] for (const row of sample) - for (let i = 0; i < REGIONS.length; i++) - expect(() => - outline(REGIONS[fontIndexOf(row, i)], row.chars[i]), - ).not.toThrow() + for (const index of REGIONS.keys()) + expect(() => displayedSansOutline(row, index)).not.toThrow() + }) + + it('draws every codepoint assigned to a bundled supplemental font', () => { + for (const style of STYLES) { + expect(supplementalChars[style].size).toBeGreaterThan(0) + for (const char of supplementalChars[style]) + expect(() => supplementalOutline(style, char)).not.toThrow() + } + }) + + it('publishes renamed, data-driven supplemental webfonts', () => { + for (const style of STYLES) { + const file = join(FONT_DIR, `hanji-rare-${style}.woff2`) + const font = supplementalFonts[style] + const chars = supplementalChars[style] + expect(font.familyName).toBe(style === 'sans' ? 'HJS' : 'HJR') + expect(font.postscriptName).toBe(style === 'sans' ? 'HJS' : 'HJR') + expect(statSync(file).size).toBeGreaterThan(0) + for (const char of chars) + expect(font.glyphForCodePoint(char.codePointAt(0)!).id).not.toBe(0) + expect(font.characterSet.toSorted((left, right) => left - right)).toEqual( + [...chars] + .map((char) => char.codePointAt(0)!) + .toSorted((left, right) => left - right), + ) + + const css = readFileSync(join(FONT_DIR, `fonts-${style}.css`), 'utf8') + expect(css).toContain( + `font-family: 'Hanji Rare ${style === 'sans' ? 'Sans' : 'Serif'}'`, + ) + expect(css).toContain(`unicode-range: ${unicodeRange(chars)};`) + } }) it('carries the kyujitai in the Japanese font', () => { const withOld = data.rows.filter((r) => r.old).slice(0, 40) expect(withOld.length).toBeGreaterThan(0) for (const row of withOld) - expect(() => outline('jp', row.old!.char)).not.toThrow() + expect(() => + usesSupplementalFont(row, 'old', 'sans') + ? supplementalOutline('sans', row.old!.char) + : outline('jp', row.old!.char), + ).not.toThrow() }) }) diff --git a/scripts/tests/i18n.test.ts b/scripts/tests/i18n.test.ts index ecf5e9c..09882fc 100644 --- a/scripts/tests/i18n.test.ts +++ b/scripts/tests/i18n.test.ts @@ -12,6 +12,7 @@ import { zhCN } from '../../app/locales/zh-cn.ts' import { zhHK } from '../../app/locales/zh-hk.ts' import { zhTW } from '../../app/locales/zh-tw.ts' import { hanNumber } from '../../app/utils/han-number.ts' +import { SOURCES } from '../../shared/sources.ts' function leaves(value: unknown, prefix = ''): Record { if (typeof value === 'string') return { [prefix]: value } @@ -25,6 +26,51 @@ function leaves(value: unknown, prefix = ''): Record { const params = (message: string): string[] => [...message.matchAll(/\{(\w+)\}/g)].map((match) => match[1]!).toSorted() +const chineseMessages = [ + ['zh-CN', zhCN], + ['zh-TW', zhTW], + ['zh-HK', zhHK], +] as const + +const manualHanLatinSpace = + /\p{Script=Han}[ \u{A0}]+(?=[\p{Script=Latin}\p{Number}])|[\p{Script=Latin}\p{Number}][ \u{A0}]+(?=\p{Script=Han})/u + +describe('Chinese typography', () => { + it('leaves Han–Latin spacing to text-autospace', () => { + const copy: Array<[string, string]> = chineseMessages.flatMap( + ([locale, messages]) => + Object.entries(leaves(messages)).map( + ([key, value]) => [`${locale}.${key}`, value] as [string, string], + ), + ) + + for (const source of SOURCES) { + for (const [locale] of chineseMessages) { + copy.push( + [`sources.${source.id}.${locale}.use`, source.use[locale]], + [ + `sources.${source.id}.${locale}.name`, + source.localizedName?.[locale] ?? source.name, + ], + [ + `sources.${source.id}.${locale}.license`, + source.localizedLicense?.[locale] ?? source.license, + ], + ) + if (source.note) + copy.push([ + `sources.${source.id}.${locale}.note`, + source.note[locale], + ]) + } + } + + expect( + copy.filter(([, message]) => manualHanLatinSpace.test(message)), + ).toEqual([]) + }) +}) + describe('browser locale matching', () => { it.each(['ja', 'ja-JP', 'ja-Jpan-JP'])('matches %s to Japanese', (tag) => { expect(matchLocale([tag])).toBe('ja-JP') diff --git a/scripts/tests/sources.test.ts b/scripts/tests/sources.test.ts new file mode 100644 index 0000000..ccebc14 --- /dev/null +++ b/scripts/tests/sources.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it } from 'vitest' +import { ASSET_URLS, sourceLock } from '../sources.ts' + +const repo = 'Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project' +const asset = 'PlangothicP1-Regular.ttf' +const name = `font/${asset}` + +describe('updatable font sources', () => { + it('resolves Plangothic from the latest GitHub release', () => { + expect(ASSET_URLS[name]).toBe( + `https://github.com/${repo}/releases/latest/download/${asset}`, + ) + + const lock = sourceLock() + const tag = lock.revisions[`github-release:${repo}`] + expect(tag).toBeTruthy() + expect(lock.assets[name]?.url).toBe( + `https://github.com/${repo}/releases/download/${tag}/${asset}`, + ) + }) +}) diff --git a/scripts/update-sources.ts b/scripts/update-sources.ts index 5639aeb..54d4064 100644 --- a/scripts/update-sources.ts +++ b/scripts/update-sources.ts @@ -17,6 +17,10 @@ const GITHUB_RAW = /^https:\/\/raw\.githubusercontent\.com\/([^/]+)\/([^/]+)\/([^/]+)\/(.+)$/ const GITHUB_ARCHIVE = /^https:\/\/codeload\.github\.com\/([^/]+)\/([^/]+)\/zip\/refs\/heads\/(.+)$/ +const GITHUB_LATEST_RELEASE = + /^https:\/\/github\.com\/([^/]+)\/([^/]+)\/releases\/latest\/download\/(.+)$/ +const GITHUB_VERSIONED_RELEASE = + /^https:\/\/github\.com\/([^/]+)\/([^/]+)\/releases\/download\/([^/]+)\/(.+)$/ const UNICODE_LATEST = 'https://www.unicode.org/Public/UCD/latest/ucd/' const UNICODE_README = 'https://www.unicode.org/Public/UCD/latest/ReadMe.txt' @@ -25,6 +29,21 @@ interface GitHubRef { ref: string } +interface GitHubReleaseAssetRef { + repo: string + asset: string +} + +interface GitHubRelease { + tag: string + assets: Map +} + +interface ResolvedGitHubReleaseAsset extends GitHubReleaseAssetRef { + tag: string + url: string +} + const githubRef = (url: string): GitHubRef | undefined => { const raw = GITHUB_RAW.exec(url) if (raw) return { repo: `${raw[1]}/${raw[2]}`, ref: raw[3]! } @@ -34,25 +53,72 @@ const githubRef = (url: string): GitHubRef | undefined => { : undefined } -async function githubRevision(source: GitHubRef): Promise { +const githubReleaseAsset = (url: string): GitHubReleaseAssetRef | undefined => { + const match = GITHUB_LATEST_RELEASE.exec(url) + return match + ? { repo: `${match[1]}/${match[2]}`, asset: match[3]! } + : undefined +} + +const githubHeaders = (): Record => { const headers: Record = { Accept: 'application/vnd.github+json', 'User-Agent': 'hanji-source-lock', } if (process.env.GITHUB_TOKEN) headers.Authorization = `Bearer ${process.env.GITHUB_TOKEN}` + return headers +} + +async function githubRevision(source: GitHubRef): Promise { const res = await fetch( - `https://api.github.com/repos/${source.repo}/commits/${encodeURIComponent(source.ref)}`, - { headers }, + `https://api.github.com/repos/${source.repo}/git/ref/heads/${encodeURIComponent(source.ref)}`, + { headers: githubHeaders() }, ) if (!res.ok) throw new Error( `cannot resolve ${source.repo}@${source.ref}: ${res.status}`, ) - const json = (await res.json()) as { sha?: string } - if (!json.sha || !/^[a-f\d]{40}$/.test(json.sha)) + const json = (await res.json()) as { + object?: { sha?: unknown; type?: unknown } + } + if ( + json.object?.type !== 'commit' || + typeof json.object.sha !== 'string' || + !/^[a-f\d]{40}$/.test(json.object.sha) + ) throw new Error(`invalid revision for ${source.repo}@${source.ref}`) - return json.sha + return json.object.sha +} + +async function githubLatestReleaseAsset( + source: GitHubReleaseAssetRef, +): Promise { + const latest = `https://github.com/${source.repo}/releases/latest/download/${source.asset}` + // GitHub resolves this stable URL to the matching versioned release asset. + // Stop at the first redirect so the expiring storage URL is never locked. + const res = await fetch(latest, { method: 'HEAD', redirect: 'manual' }) + const location = res.headers.get('location') + if (res.status < 300 || res.status >= 400 || !location) + throw new Error( + `cannot resolve latest release asset ${source.repo}/${source.asset}: ${res.status}`, + ) + + const url = new URL(location, latest).href + const match = GITHUB_VERSIONED_RELEASE.exec(url) + if ( + !match || + `${match[1]}/${match[2]}` !== source.repo || + decodeURIComponent(match[4]!) !== source.asset + ) + throw new Error( + `invalid latest release redirect for ${source.repo}/${source.asset}: ${url}`, + ) + return { + ...source, + tag: decodeURIComponent(match[3]!), + url, + } } async function unicodeVersion(): Promise { @@ -68,6 +134,7 @@ async function unicodeVersion(): Promise { function pinnedUrl( url: string, revisions: ReadonlyMap, + releases: ReadonlyMap, unicode: string, ): string { const match = GITHUB_RAW.exec(url) @@ -84,6 +151,17 @@ function pinnedUrl( if (!revision) throw new Error(`unresolved GitHub source: ${key}`) return `https://codeload.github.com/${archive[1]}/${archive[2]}/zip/${revision}` } + const releaseAsset = githubReleaseAsset(url) + if (releaseAsset) { + const resolved = releases + .get(releaseAsset.repo) + ?.assets.get(releaseAsset.asset) + if (!resolved) + throw new Error( + `latest release of ${releaseAsset.repo} has no ${releaseAsset.asset}`, + ) + return resolved + } if (url.startsWith(UNICODE_LATEST)) return url.replace( UNICODE_LATEST, @@ -140,34 +218,57 @@ async function download(name: string, url: string): Promise { } const refs = new Map() +const releaseAssets = new Map() for (const url of Object.values(ASSET_URLS)) { const source = githubRef(url) if (source) refs.set(`${source.repo}@${source.ref}`, source) + const release = githubReleaseAsset(url) + if (release) + releaseAssets.set(`${release.repo}\u{0}${release.asset}`, release) } process.stderr.write('Resolving source versions...\n') -const [resolvedRefs, unicode, previous] = await Promise.all([ +const [resolvedRefs, resolvedReleases, unicode, previous] = await Promise.all([ Promise.all( [...refs].map( async ([key, source]) => [key, await githubRevision(source)] as const, ), ), + Promise.all([...releaseAssets.values()].map(githubLatestReleaseAsset)), unicodeVersion(), existingLock(), ]) const revisions = new Map(resolvedRefs) +const releases = new Map() +for (const resolved of resolvedReleases) { + const release = releases.get(resolved.repo) + if (release && release.tag !== resolved.tag) + throw new Error( + `latest release assets for ${resolved.repo} resolved to different tags`, + ) + const current = release ?? { tag: resolved.tag, assets: new Map() } + current.assets.set(resolved.asset, resolved.url) + releases.set(resolved.repo, current) +} const revisionRecord = { ...Object.fromEntries( [...revisions].map(([key, revision]) => [`github:${key}`, revision]), ), + ...Object.fromEntries( + [...releases].map(([repo, release]) => [ + `github-release:${repo}`, + release.tag, + ]), + ), 'unicode:ucd': unicode, } const entries = Object.entries(ASSET_URLS).map(([name, sourceUrl]) => ({ name, - url: pinnedUrl(sourceUrl, revisions, unicode), - // Direct institutional downloads have no revision in their URL. Recheck - // their bytes on an explicit source update; ordinary builds remain locked - // to the recorded SHA-256 and fail rather than accepting a silent change. + url: pinnedUrl(sourceUrl, revisions, releases, unicode), + // Direct institutional downloads and release assets can change without a + // new revision in their URL. Recheck their bytes on an explicit source + // update; ordinary builds remain locked to the recorded SHA-256 and fail + // rather than accepting a silent change. refresh: !githubRef(sourceUrl) && !sourceUrl.startsWith(UNICODE_LATEST), })) const assets: (readonly [string, LockedAsset])[] = [] diff --git a/shared/links.ts b/shared/links.ts index 599c667..f921edd 100644 --- a/shared/links.ts +++ b/shared/links.ts @@ -1,6 +1,6 @@ // @unocss-include -- DictLink.icon holds UnoCSS icon classes import { fontRegionOf } from './row.ts' -import { REGIONS, type CharRow, type Region } from './types.ts' +import { REGIONS, type CharRow, type Column, type Region } from './types.ts' export const REPO_URL = 'https://github.com/sxzz/hanji' export const ISSUES_URL = `${REPO_URL}/issues` @@ -26,6 +26,8 @@ export interface DictGroupOptions { export interface Form { char: string font: Region + /** Display column that supplied this form, when it is one of the columns. */ + column?: Column } /** @@ -43,20 +45,20 @@ export function formsOf( regions: readonly Region[] = REGIONS, ): Form[] { const out: Form[] = [] - const add = (char: string, font: Region) => { + const add = (char: string, font: Region, column?: Column) => { if (char && out.every((form) => form.char !== char)) - out.push({ char, font }) + out.push({ char, font, ...(column ? { column } : {}) }) } for (const region of regions) { const index = REGIONS.indexOf(region) - add(row.chars[index]!, fontRegionOf(row, index)) + add(row.chars[index]!, fontRegionOf(row, index), region) } // The key and the names it merged with are not always a column of their own add(row.key, 'cn') for (const name of row.aka ?? []) add(name, 'cn') for (const region of regions) for (const entry of row.alternatives?.[region] ?? []) - add(entry.char, region) + add(entry.char, region, region) return out } diff --git a/shared/row.ts b/shared/row.ts index fe9dea4..dd7adc4 100644 --- a/shared/row.ts +++ b/shared/row.ts @@ -1,4 +1,10 @@ -import { REGIONS, type CharRow, type Column, type Region } from './types.ts' +import { + REGIONS, + type CharRow, + type Column, + type Region, + type Style, +} from './types.ts' /** * Which region's font should render this cell: the earliest region in its @@ -27,6 +33,15 @@ export function fontRegionOf(row: CharRow, region: number): Region { return REGIONS[fontIndexOf(row, region)]! } +/** Whether this column uses the bundled supplemental face for this style. */ +export function usesSupplementalFont( + row: CharRow, + column: Column, + style: Style, +): boolean { + return row.supplementalFont?.[style]?.includes(column) ?? false +} + /** * Split a signature into runs, used both for the underline beneath the * regional cells and for the filter chips. diff --git a/shared/sources.ts b/shared/sources.ts index 3b30ee1..824fabc 100644 --- a/shared/sources.ts +++ b/shared/sources.ts @@ -40,6 +40,9 @@ export const SOURCES: Source[] = [ }, name: 'Adobe Source Han Sans / Serif(CMap 资源)', localizedName: { + 'zh-CN': 'Adobe Source Han Sans / Serif(CMap资源)', + 'zh-TW': 'Adobe Source Han Sans / Serif(CMap資源)', + 'zh-HK': 'Adobe Source Han Sans / Serif(CMap資源)', 'ja-JP': 'Adobe Source Han Sans / Serif(CMapリソース)', 'ko-KR': 'Adobe Source Han Sans / Serif(CMap 리소스)', }, @@ -48,11 +51,11 @@ export const SOURCES: Source[] = [ licenseUrl: 'https://openfontlicense.org/', note: { 'zh-CN': - '每套字体的五份地区 CMap 给出「码点 → CID」映射,同一字形池内 CID 相同即同一字形。判定取黑体与宋体的并集。', + '每套字体的五份地区CMap给出「码点 → CID」映射,同一字形池内CID相同即同一字形。判定取黑体与宋体的并集。', 'zh-TW': - '每套字體的五份地區 CMap 給出「碼位 → CID」對映,同一字形池內 CID 相同即同一字形。判定取黑體與宋體的聯集。', + '每套字體的五份地區CMap給出「碼位 → CID」對映,同一字形池內CID相同即同一字形。判定取黑體與宋體的聯集。', 'zh-HK': - '每套字體的五份地區 CMap 給出「碼點 → CID」對應,同一字形池內 CID 相同即同一字形。判定取黑體與宋體的並集。', + '每套字體的五份地區CMap給出「碼點 → CID」對應,同一字形池內CID相同即同一字形。判定取黑體與宋體的並集。', 'ja-JP': '各書体の5地域向けCMapには「コードポイント → CID」の対応があり、同じ字形プール内でCIDが同じなら同一字形です。判定にはゴシック体と明朝体の和集合を使います。', 'ko-KR': @@ -70,6 +73,9 @@ export const SOURCES: Source[] = [ }, name: 'Noto Sans / Noto Serif(含 CJK)', localizedName: { + 'zh-CN': 'Noto Sans / Noto Serif(含CJK)', + 'zh-TW': 'Noto Sans / Noto Serif(含CJK)', + 'zh-HK': 'Noto Sans / Noto Serif(含CJK)', 'ja-JP': 'Noto Sans / Noto Serif(CJK対応)', 'ko-KR': 'Noto Sans / Noto Serif(CJK 지원)', }, @@ -78,17 +84,72 @@ export const SOURCES: Source[] = [ licenseUrl: 'https://openfontlicense.org/', note: { 'zh-CN': - '汉字取自 CJK 版本,拉丁字母与数字取自拉丁版本,都按应用内用字子集化后自托管,OFL 声明随附于 /notices/noto-ofl.txt。', + '汉字取自CJK版本,拉丁字母与数字取自拉丁版本,都按应用内用字子集化后自托管,OFL声明随附于 /notices/noto-ofl.txt。', 'zh-TW': - '漢字取自 CJK 版本,拉丁字母與數字取自拉丁版本,都按應用內用字子集化後自行託管,OFL 聲明隨附於 /notices/noto-ofl.txt。', + '漢字取自CJK版本,拉丁字母與數字取自拉丁版本,都按應用內用字子集化後自行託管,OFL聲明隨附於 /notices/noto-ofl.txt。', 'zh-HK': - '漢字取自 CJK 版本,拉丁字母與數字取自拉丁版本,都按應用內用字子集化後自行託管,OFL 聲明隨附於 /notices/noto-ofl.txt。', + '漢字取自CJK版本,拉丁字母與數字取自拉丁版本,都按應用內用字子集化後自行託管,OFL聲明隨附於 /notices/noto-ofl.txt。', 'ja-JP': '漢字はCJK版、ラテン文字と数字はラテン版を使用し、アプリ内で使う文字だけにサブセット化してセルフホストしています。OFLの表記は/notices/noto-ofl.txtに同梱しています。', 'ko-KR': '한자는 CJK 버전, 라틴 문자와 숫자는 라틴 버전을 사용합니다. 앱에서 쓰는 문자만 서브셋으로 만들어 자체 호스팅하며, OFL 고지문은 /notices/noto-ofl.txt에 함께 제공합니다.', }, }, + { + id: 'plangothic', + use: { + 'zh-CN': '补充Noto未收录的黑体字形', + 'zh-TW': '補充Noto未收錄的黑體字形', + 'zh-HK': '補充Noto未收錄的黑體字形', + 'ja-JP': 'Noto未収録字のゴシック体補完', + 'ko-KR': 'Noto 미수록 글자의 고딕체 보완', + }, + name: 'Plangothic P1', + homepage: + 'https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project', + license: 'SIL OFL 1.1', + licenseUrl: + 'https://github.com/Fitzgerald-Porthmouth-Koenigsegg/Plangothic_Project/blob/main/LICENSE-OFL.txt', + note: { + 'zh-CN': + '按当前数据中Noto未收录但需要显示的码点,动态生成重命名后的网页字体子集;这些字形不参与地区差异判定。OFL声明随附于/notices/plangothic-ofl.txt。', + 'zh-TW': + '依目前資料中Noto未收錄但需要顯示的碼位,動態產生重新命名的網頁字型子集;這些字形不參與地區差異判定。OFL聲明隨附於/notices/plangothic-ofl.txt。', + 'zh-HK': + '按目前資料中Noto未收錄但需要顯示的碼位,動態產生重新命名的網頁字體子集;這些字形不參與地區差異判定。OFL聲明隨附於/notices/plangothic-ofl.txt。', + 'ja-JP': + '現在のデータで表示が必要ながらNotoにないコードポイントから、名称を変更したWebフォントサブセットを動的に生成します。これらの字形は地域差の判定には使いません。OFL表記は/notices/plangothic-ofl.txtに同梱しています。', + 'ko-KR': + '현재 데이터에서 표시해야 하지만 Noto에 없는 코드 포인트를 동적으로 수집해 이름을 바꾼 웹 글꼴 서브셋을 생성합니다. 이 자형은 지역 차이 판정에 사용하지 않습니다. OFL 고지문은 /notices/plangothic-ofl.txt에 있습니다.', + }, + }, + { + id: 'wenjin-mincho', + use: { + 'zh-CN': '补充Noto未收录的宋体字形', + 'zh-TW': '補充Noto未收錄的宋體字形', + 'zh-HK': '補充Noto未收錄的宋體字形', + 'ja-JP': 'Noto未収録字の明朝体補完', + 'ko-KR': 'Noto 미수록 글자의 명조체 보완', + }, + name: 'WenJin Mincho P2', + homepage: 'https://github.com/takushun-wu/WenJinMincho', + license: 'SIL OFL 1.1', + licenseUrl: + 'https://github.com/takushun-wu/WenJinMincho/blob/main/LICENSE.md', + note: { + 'zh-CN': + '按当前数据中Noto未收录但需要显示的码点,动态生成重命名后的网页字体子集;这些字形不参与地区差异判定。OFL声明随附于/notices/wenjin-mincho-ofl.md。', + 'zh-TW': + '依目前資料中Noto未收錄但需要顯示的碼位,動態產生重新命名的網頁字型子集;這些字形不參與地區差異判定。OFL聲明隨附於/notices/wenjin-mincho-ofl.md。', + 'zh-HK': + '按目前資料中Noto未收錄但需要顯示的碼位,動態產生重新命名的網頁字體子集;這些字形不參與地區差異判定。OFL聲明隨附於/notices/wenjin-mincho-ofl.md。', + 'ja-JP': + '現在のデータで表示が必要ながらNotoにないコードポイントから、名称を変更したWebフォントサブセットを動的に生成します。これらの字形は地域差の判定には使いません。OFL表記は/notices/wenjin-mincho-ofl.mdに同梱しています。', + 'ko-KR': + '현재 데이터에서 표시해야 하지만 Noto에 없는 코드 포인트를 동적으로 수집해 이름을 바꾼 웹 글꼴 서브셋을 생성합니다. 이 자형은 지역 차이 판정에 사용하지 않습니다. OFL 고지문은 /notices/wenjin-mincho-ofl.md에 있습니다.', + }, + }, { id: 'opencc', use: { @@ -100,6 +161,9 @@ export const SOURCES: Source[] = [ }, name: 'OpenCC 开放中文转换', localizedName: { + 'zh-CN': 'OpenCC开放中文转换', + 'zh-TW': 'OpenCC開放中文轉換', + 'zh-HK': 'OpenCC開放中文轉換', 'ja-JP': 'OpenCC(Open Chinese Convert)', 'ko-KR': 'OpenCC(Open Chinese Convert)', }, diff --git a/shared/types.ts b/shared/types.ts index bd5d2f8..75ef689 100644 --- a/shared/types.ts +++ b/shared/types.ts @@ -89,6 +89,12 @@ export interface CharRow { * to merge. These are display-only: they are not names or forms of the row. */ uncertain?: UncertainRelation[] + /** + * Columns whose codepoint is absent from Noto and must use the bundled + * supplemental face. Kept per style because Source Han Sans and Serif do + * not necessarily cover the same codepoints. + */ + supplementalFont?: Partial> /** Codepoint partition signature; "00000" when all five share a codepoint. */ cp: string /**