diff --git a/CLAUDE.md b/CLAUDE.md index 03296f1..8b06a58 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -8,7 +8,7 @@ Use the installed `goat` CLI for lexicon operations: `lex lint`, `lex parse`, `l - `goat lex publish` and `goat lex unpublish` write to a real PDS. Never run either; tell jmm the command to run. - Schema files are stored in goat's form: keys sorted, two-space indent, trailing newline. After editing one, run `python3 tools/lint_lexicons.py --format`. Do not hand-order keys for readability; the form is what makes a file here and a pulled copy identical. -- Do not replace `tools/lint_lexicons.py` with goat. goat does not check that a file's path matches its `id`, and does not check that references resolve. +- goat owns the spec checks. `tools/lint_lexicons.py` covers only what goat reports clean — id against path, references that resolve, empty `defs`, a union with no `refs`, the stored form. Do not add a check to it that goat already makes, and do not delete it: those five are not checked anywhere else. ## Writing descriptions diff --git a/README.md b/README.md index 5e35883..27b6b8a 100644 --- a/README.md +++ b/README.md @@ -59,7 +59,13 @@ Hooks are managed by [prek](https://prek.j178.dev/) and configured in uv tool install prek prek install --prepare-hooks -They then run on every commit; `prek run --all-files` runs them by hand. The -schema hooks are `goat lex lint` and `tools/lint_lexicons.py`, which checks the -two things goat does not: that a file's path and its `id` agree, and that every -reference resolves to a schema this repo defines. +They then run on every commit; `prek run --all-files` runs them by hand. + +The schema hooks are `goat lex lint`, which is the Lexicon spec — it parses +first, so there is no separate parse step — and `tools/lint_lexicons.py`, which +covers what goat reports clean: + +- a file whose `id` disagrees with its path +- a reference to a schema this repo does not define +- empty `defs`, or a union with no `refs` +- a file not in the form above diff --git a/TODO/01-namespace.md b/TODO/01-namespace.md index e909922..f504861 100644 --- a/TODO/01-namespace.md +++ b/TODO/01-namespace.md @@ -95,11 +95,15 @@ stages then follow the same steps without rediscovering them. hand-ordered file can never match a pulled copy; writing what goat writes means a diff between the two is a real difference rather than formatting. `tools/lint_lexicons.py --format` applies it. -- [x] **Lexicon linters in `prek.toml`.** `goat lex lint` for the spec and the - style rules that go with it, and `tools/lint_lexicons.py` for the two - checks goat does not make: that a file's `id` matches its path, and that - every reference resolves to a schema this repo defines. A file that gets - either of those wrong lints clean under goat. +- [x] **Lexicon linters in `prek.toml`.** `goat lex lint` is the Lexicon spec — + the language version, types, record keys, `required` fields that exist, + def-name syntax, a primary type only on `main` — and it parses before it + lints, so no parse hook is needed beside it. + + `tools/lint_lexicons.py` covers only what goat reports clean, measured + case by case: an `id` that disagrees with its path, a reference to a + schema nobody wrote, empty `defs`, a union with no `refs`, and a file + that is not in the stored form. Nothing is checked in both places. - [ ] **Decide the Rust type story.** Codegen from these files, or hand-written types in headquarters. Hand-written is fine while there is one schema; the decision matters when there are ten and they start changing. diff --git a/prek.toml b/prek.toml index 1767f21..5957458 100644 --- a/prek.toml +++ b/prek.toml @@ -13,9 +13,11 @@ hooks = [ { id = "trailing-whitespace" }, ] -# The schemas themselves, checked twice over. goat knows the Lexicon spec and -# the style rules that go with it; the script knows this repo's two -# conventions, which goat has no way to check. See README for goat's install. +# The schemas themselves. goat parses and lints them against the Lexicon spec +# — `lex lint` parses first, so there is no separate parse hook to run — and +# the script covers only what goat reports clean: an id that disagrees with +# its path, a reference to a schema nobody wrote, empty defs, a union with no +# refs, and a file not in the form goat writes. See README for goat's install. # # The script has no dependencies, so prek runs it with an interpreter it # fetches itself and installs nothing. @@ -23,5 +25,5 @@ hooks = [ repo = "local" hooks = [ { id = "goat-lex-lint", name = "goat lex lint", entry = "goat lex lint", language = "system", files = '^lexicons/.*\.json$' }, - { id = "lint-lexicons", name = "lint lexicons", entry = "tools/lint_lexicons.py", language = "python", files = '^lexicons/(blue|games)/.*\.json$' }, + { id = "lint-lexicons", name = "lexicon conventions", entry = "tools/lint_lexicons.py", language = "python", files = '^lexicons/(blue|games)/.*\.json$' }, ] diff --git a/tools/lint_lexicons.py b/tools/lint_lexicons.py index 2acfe4c..179a7a4 100644 --- a/tools/lint_lexicons.py +++ b/tools/lint_lexicons.py @@ -3,19 +3,22 @@ # requires-python = ">=3.11" # dependencies = [] # /// -"""Check the lexicon files in this repo. +"""Check what `goat lex lint` cannot, in the schema files in this repo. -Two kinds of check. Some are the Lexicon spec: a file says `lexicon: 1`, only -`main` holds a primary type, a record names a key type, there is no float. The -rest are this repo's conventions, and they are the reason the script still -exists: a file's path has to match its NSID, and every reference has to -resolve. `goat lex lint` runs beside this one and checks neither — a schema -whose id disagrees with its path lints clean there, and so does a reference to -a schema nobody has written. +goat does the Lexicon spec: the language version, types, record keys, required +fields that exist, def-name syntax, a primary type only on `main`. It runs +first in the commit hooks, and none of that is repeated here. + +Four things are left, and they are the reason this still exists. goat reports +a file clean when its `id` disagrees with its path, when a reference names a +schema nobody has written, when `defs` is empty, and when a union lists no +refs. It also has no opinion on how a file is formatted, and this repo does: +schemas are stored in the form goat itself writes. No dependencies on purpose. -Run by prek on commit. `python3 tools/lint_lexicons.py` checks everything. +Run by prek on commit. `python3 tools/lint_lexicons.py` checks everything, and +`--format` rewrites files into the stored form. """ from __future__ import annotations @@ -31,27 +34,7 @@ from pathlib import Path LEXICONS = Path("lexicons") ROOTS = ("blue", "games") -PRIMARY = {"record", "query", "procedure", "subscription", "permission-set"} -TYPES = PRIMARY | { - "object", - "array", - "params", - "token", - "ref", - "union", - "unknown", - "boolean", - "integer", - "string", - "bytes", - "cid-link", - "blob", - "permission", -} -KEYS = {"tid", "nsid", "any"} - NSID = re.compile(r"^[a-z][a-z0-9]*(\.[a-z][a-zA-Z0-9]*)+$") -DEF_NAME = re.compile(r"^[a-zA-Z][a-zA-Z0-9]*$") class Problems: @@ -63,6 +46,17 @@ class Problems: self.found.append(f"{self.path}: {where}: {message}") +def canonical(doc: object) -> str: + """The bytes goat writes for a schema: keys sorted, two-space indent. + + `goat lex pull` and `goat lex new` both produce this, and a published + record comes back in it whatever order it went out in. Authoring in the + same form is what makes a file here and a copy pulled from the network + the same bytes. + """ + return json.dumps(doc, indent=2, sort_keys=True, ensure_ascii=False) + "\n" + + def nsid_for(path: Path) -> str: """The NSID a file at this path must declare.""" rel = path.relative_to(LEXICONS) @@ -79,8 +73,12 @@ def is_lexicon(path: Path) -> bool: ) -def check_refs(value: str, where: str, defs: set[str], known: set[str], p: Problems) -> None: - """A reference resolves inside this file, or names a schema this repo has.""" +def check_ref(value: str, where: str, defs: set[str], known: set[str], p: Problems) -> None: + """A reference resolves inside this file, or names a schema this repo has. + + A reference to somebody else's namespace is left alone: this repo is not + where app.bsky is defined, and resolving it needs the network. + """ name, _, fragment = value.partition("#") if not name: if fragment not in defs: @@ -93,81 +91,41 @@ def check_refs(value: str, where: str, defs: set[str], known: set[str], p: Probl p.add(where, f"reference to {name}, which this repo does not define") -def check_schema(node: object, where: str, defs: set[str], known: set[str], p: Problems) -> None: - """Walk one schema node and everything nested in it.""" - if not isinstance(node, dict): - p.add(where, "is not an object") - return +def walk(node: object, where: str, defs: set[str], known: set[str], p: Problems) -> None: + """Find every reference anywhere in a schema, and the two empty shapes. - kind = node.get("type") - if kind in ("float", "number"): - p.add(where, "Lexicon has no float type; use an integer and say what the units are") + Deliberately untyped: goat has already established that each node is a + valid schema, so this walks whatever is there rather than restating the + grammar. + """ + if isinstance(node, list): + for i, child in enumerate(node): + walk(child, f"{where}[{i}]", defs, known, p) return - if kind not in TYPES: - p.add(where, f"unknown type {kind!r}") + if not isinstance(node, dict): return + kind = node.get("type") if kind == "ref": target = node.get("ref") if isinstance(target, str): - check_refs(target, where, defs, known, p) + check_ref(target, where, defs, known, p) else: p.add(where, "ref has no ref") - - if kind == "union": + elif kind == "union": refs = node.get("refs") if not isinstance(refs, list) or not refs: + # Valid Lexicon — an empty union is a closed one — but in this repo + # it is always an unfinished edit. p.add(where, "union has no refs") else: for target in refs: if isinstance(target, str): - check_refs(target, where, defs, known, p) - - if kind in ("object", "params"): - properties = node.get("properties", {}) - if not isinstance(properties, dict): - p.add(where, "properties is not an object") - return - for name, child in properties.items(): - check_schema(child, f"{where}.{name}", defs, known, p) - for name in node.get("required", []): - if name not in properties: - p.add(where, f"requires {name}, which it does not define") - - if kind == "array": - items = node.get("items") - if items is None: - p.add(where, "array has no items") - else: - check_schema(items, f"{where}[]", defs, known, p) - - if kind == "record": - key = node.get("key") - if not isinstance(key, str) or (key not in KEYS and not key.startswith("literal:")): - p.add(where, f"record key type is {key!r}; expected tid, nsid, any or literal:*") - record = node.get("record") - if record is None: - p.add(where, "record has no record") - else: - check_schema(record, f"{where}.record", defs, known, p) - - for field in ("input", "output", "message", "parameters"): - child = node.get(field) - if isinstance(child, dict): - schema = child.get("schema", child) if field != "parameters" else child - if isinstance(schema, dict) and "type" in schema: - check_schema(schema, f"{where}.{field}", defs, known, p) - - -def canonical(doc: object) -> str: - """The bytes goat writes for a schema: keys sorted, two-space indent. + check_ref(target, where, defs, known, p) - `goat lex pull` and `goat lex new` both produce this, and a published - record comes back in it whatever order it went out in. Authoring in the - same form is what makes a file here and a copy pulled from the network - the same bytes. - """ - return json.dumps(doc, indent=2, sort_keys=True, ensure_ascii=False) + "\n" + for name, child in node.items(): + if name not in ("ref", "refs", "type"): + walk(child, f"{where}.{name}", defs, known, p) def check_file(path: Path, known: set[str]) -> list[str]: @@ -185,9 +143,6 @@ def check_file(path: Path, known: set[str]) -> list[str]: if text != canonical(doc): p.add("file", "is not in goat's form; run tools/lint_lexicons.py --format") - if doc.get("lexicon") != 1: - p.add("file", f"lexicon is {doc.get('lexicon')!r}; expected 1") - nsid = doc.get("id") expected = nsid_for(path) if not isinstance(nsid, str) or not NSID.match(nsid): @@ -204,18 +159,7 @@ def check_file(path: Path, known: set[str]) -> list[str]: names = set(defs) for name, node in defs.items(): - where = f"defs.{name}" - if not DEF_NAME.match(name): - p.add(where, "is not a valid def name") - if isinstance(node, dict): - if node.get("type") in PRIMARY and name != "main": - p.add(where, f"is a {node['type']}; only main may be one") - # Only main. Published lexicons leave most fields undescribed and - # let the name and type speak, and a description forced out of - # someone is worse than none. - if name == "main" and not node.get("description"): - p.add(where, "has no description") - check_schema(node, where, names, known, p) + walk(node, f"defs.{name}", names, known, p) return p.found