From 3aac6e63cd5858243ef9eec4dd75d814fb130ea7 Mon Sep 17 00:00:00 2001 From: Matt Stavola Date: Wed, 15 Apr 2026 16:06:41 -0400 Subject: [PATCH] Introduce self {} and extension annotations --- README.md | 8 + codegen-plugins/mlf-codegen-go/src/lib.rs | 5 +- codegen-plugins/mlf-codegen-rust/src/lib.rs | 5 +- .../mlf-codegen-typescript/src/lib.rs | 5 +- mlf-cli/src/generate/lexicon.rs | 8 +- mlf-cli/src/generate/mlf.rs | 299 +++++- mlf-cli/src/generate/mlf.rs.backup | 941 ------------------ mlf-codegen/src/lib.rs | 219 +++- mlf-lang/src/ast.rs | 25 + mlf-lang/src/lexer.rs | 3 + mlf-lang/src/parser.rs | 94 ++ mlf-lang/src/workspace.rs | 12 +- mlf-lsp/src/server.rs | 21 +- mlf-lsp/src/utils.rs | 4 +- mlf-wasm/src/lib.rs | 8 +- .../const_float_stringifies/expected.json | 20 + .../expected_warnings.txt | 1 + .../lexicon/const_float_stringifies/input.mlf | 8 + .../lexicon/const_float_stringifies/test.toml | 4 + .../item_extensions_codegen/expected.json | 20 + .../lexicon/item_extensions_codegen/input.mlf | 6 + .../lexicon/item_extensions_codegen/test.toml | 4 + tests/codegen/lexicon/self_item/expected.json | 18 + tests/codegen/lexicon/self_item/input.mlf | 10 + tests/codegen/lexicon/self_item/test.toml | 4 + tests/codegen_integration.rs | 31 +- .../item_extensions/expected.mlf | 7 + .../lexicon_to_mlf/item_extensions/input.json | 19 + .../ref_hint_warning/expected.mlf | 6 + .../ref_hint_warning/expected_warnings.txt | 1 + .../ref_hint_warning/input.json | 14 + .../self_top_level_description/expected.mlf | 8 + .../self_top_level_description/input.json | 14 + .../self_top_level_extensions/expected.mlf | 11 + .../self_top_level_extensions/input.json | 17 + .../unknown_def_type_passthrough/expected.mlf | 7 + .../unknown_def_type_passthrough/input.json | 18 + tests/real_world/roundtrip.rs | 4 +- tree-sitter-mlf/grammar.js | 48 +- tree-sitter-mlf/queries/highlights.scm | 4 + tree-sitter-mlf/src/grammar.json | 196 +++- tree-sitter-mlf/src/node-types.json | 107 +- .../docs/language-guide/11-annotations.md | 84 +- .../docs/language-guide/11-lexicon-mapping.md | 56 ++ website/syntaxes/mlf.sublime-syntax | 2 +- 45 files changed, 1381 insertions(+), 1025 deletions(-) delete mode 100644 mlf-cli/src/generate/mlf.rs.backup create mode 100644 tests/codegen/lexicon/const_float_stringifies/expected.json create mode 100644 tests/codegen/lexicon/const_float_stringifies/expected_warnings.txt create mode 100644 tests/codegen/lexicon/const_float_stringifies/input.mlf create mode 100644 tests/codegen/lexicon/const_float_stringifies/test.toml create mode 100644 tests/codegen/lexicon/item_extensions_codegen/expected.json create mode 100644 tests/codegen/lexicon/item_extensions_codegen/input.mlf create mode 100644 tests/codegen/lexicon/item_extensions_codegen/test.toml create mode 100644 tests/codegen/lexicon/self_item/expected.json create mode 100644 tests/codegen/lexicon/self_item/input.mlf create mode 100644 tests/codegen/lexicon/self_item/test.toml create mode 100644 tests/lexicon_to_mlf/item_extensions/expected.mlf create mode 100644 tests/lexicon_to_mlf/item_extensions/input.json create mode 100644 tests/lexicon_to_mlf/ref_hint_warning/expected.mlf create mode 100644 tests/lexicon_to_mlf/ref_hint_warning/expected_warnings.txt create mode 100644 tests/lexicon_to_mlf/ref_hint_warning/input.json create mode 100644 tests/lexicon_to_mlf/self_top_level_description/expected.mlf create mode 100644 tests/lexicon_to_mlf/self_top_level_description/input.json create mode 100644 tests/lexicon_to_mlf/self_top_level_extensions/expected.mlf create mode 100644 tests/lexicon_to_mlf/self_top_level_extensions/input.json create mode 100644 tests/lexicon_to_mlf/unknown_def_type_passthrough/expected.mlf create mode 100644 tests/lexicon_to_mlf/unknown_def_type_passthrough/input.json diff --git a/README.md b/README.md index 5dd3562..a9b1121 100644 --- a/README.md +++ b/README.md @@ -7,6 +7,10 @@ A human-friendly DSL for ATProto Lexicons ## What it looks like ```mlf +/// Blog-style post record for the Bluesky feed. +@const("revision", 3) +self {} + record post { text!: string constrained { maxLength: 3000, @@ -22,6 +26,10 @@ def type replyRef = { }; ``` +`self {}` represents the lexicon itself — docs on it map to the +top-level `description`; `@const` / `@reference` annotations carry +non-spec or vendor extension fields through to JSON. + ## Installation Right now you can only install mlf from source: diff --git a/codegen-plugins/mlf-codegen-go/src/lib.rs b/codegen-plugins/mlf-codegen-go/src/lib.rs index e2b9392..4a7b9c0 100644 --- a/codegen-plugins/mlf-codegen-go/src/lib.rs +++ b/codegen-plugins/mlf-codegen-go/src/lib.rs @@ -190,8 +190,9 @@ impl CodeGenerator for GoGenerator { Item::Query(_) | Item::Procedure(_) | Item::Subscription(_) => { // TODO: Generate client methods } - Item::Use(_) => { - // Skip use statements + Item::Use(_) | Item::SelfItem(_) => { + // Skip `use` statements and the `self { }` item — both + // are lexicon-level metadata, not types to emit. } } } diff --git a/codegen-plugins/mlf-codegen-rust/src/lib.rs b/codegen-plugins/mlf-codegen-rust/src/lib.rs index 0d09b7a..4b69591 100644 --- a/codegen-plugins/mlf-codegen-rust/src/lib.rs +++ b/codegen-plugins/mlf-codegen-rust/src/lib.rs @@ -216,8 +216,9 @@ impl CodeGenerator for RustGenerator { Item::Query(_) | Item::Procedure(_) | Item::Subscription(_) => { // TODO: Generate client methods } - Item::Use(_) => { - // Skip use statements + Item::Use(_) | Item::SelfItem(_) => { + // Skip `use` statements and the `self { }` item — both + // are lexicon-level metadata, not types to emit. } } } diff --git a/codegen-plugins/mlf-codegen-typescript/src/lib.rs b/codegen-plugins/mlf-codegen-typescript/src/lib.rs index 98bd459..f4441d5 100644 --- a/codegen-plugins/mlf-codegen-typescript/src/lib.rs +++ b/codegen-plugins/mlf-codegen-typescript/src/lib.rs @@ -162,8 +162,9 @@ impl CodeGenerator for TypeScriptGenerator { // TODO: Generate client methods for these // Annotation idea: @clientMethod for custom generation } - Item::Use(_) => { - // Skip use statements in output + Item::Use(_) | Item::SelfItem(_) => { + // Skip `use` statements and the `self { }` item — both + // are lexicon-level metadata, not types to emit. } } } diff --git a/mlf-cli/src/generate/lexicon.rs b/mlf-cli/src/generate/lexicon.rs index 136f7e2..379b1b4 100644 --- a/mlf-cli/src/generate/lexicon.rs +++ b/mlf-cli/src/generate/lexicon.rs @@ -163,7 +163,11 @@ pub fn run(input_paths: Vec, output_dir: Option, explicit_root continue; } - let json_lexicon = mlf_codegen::generate_lexicon(&namespace, &lexicon, &workspace); + let output = mlf_codegen::generate_lexicon(&namespace, &lexicon, &workspace); + + for warning in &output.warnings { + eprintln!(" warning: {}: {}", warning.namespace, warning.message); + } let output_path = if flat { output_dir.join(format!("{}.json", namespace)) @@ -180,7 +184,7 @@ pub fn run(input_paths: Vec, output_dir: Option, explicit_root path }; - let json_str = serde_json::to_string_pretty(&json_lexicon).unwrap(); + let json_str = serde_json::to_string_pretty(&output.json).unwrap(); if let Err(source) = std::fs::write(&output_path, format!("{}\n", json_str)) { errors.push((output_path.display().to_string(), format!("Failed to write file: {}", source))); continue; diff --git a/mlf-cli/src/generate/mlf.rs b/mlf-cli/src/generate/mlf.rs index 786dfcc..fbdd73b 100644 --- a/mlf-cli/src/generate/mlf.rs +++ b/mlf-cli/src/generate/mlf.rs @@ -228,7 +228,6 @@ pub fn run(input_patterns: Vec, output_dir: Option, flat: bool) pub fn generate_mlf_from_json(json: &Value) -> Result { let mut output = String::new(); - // Extract NSID to get the last segment for "main" definitions let nsid = json .get("id") .and_then(|v| v.as_str()) @@ -250,7 +249,13 @@ pub fn generate_mlf_from_json(json: &Value) -> Result Result { - let mlf = generate_record(name, def, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "query" => { - let mlf = generate_query(name, def, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "procedure" => { - let mlf = generate_procedure(name, def, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "subscription" => { - let mlf = generate_subscription(name, def, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "token" => { - let mlf = generate_token(name, def)?; - output.push_str(&mlf); - output.push('\n'); - } - _ => { - // All other types (object, string, array, union, etc.) are treated as def type - let mlf = generate_def_type(name, def, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - } + let mlf = match def_type { + "record" => generate_record(name, def, &ctx)?, + "query" => generate_query(name, def, &ctx)?, + "procedure" => generate_procedure(name, def, &ctx)?, + "subscription" => generate_subscription(name, def, &ctx)?, + "token" => generate_token(name, def, &ctx)?, + t if is_known_def_type(t) => generate_def_type(name, def, &ctx)?, + // Unknown def type (e.g. `permission-set`): emit a + // placeholder def with `@const` annotations carrying every + // field, so the shape roundtrips byte-faithfully without + // requiring a grammar entry for every future spec type. + _ => render_unknown_def_passthrough(name, def, last_segment, &ctx), + }; + output.push_str(&mlf); + output.push('\n'); } Ok(MlfGenerateOutput { @@ -299,6 +286,230 @@ pub fn generate_mlf_from_json(json: &Value) -> Result bool { + KNOWN_DEF_TYPES.contains(&type_name) +} + +/// Spec-defined fields on each def kind. Any other key on the def's +/// JSON object becomes an `@const` annotation on the emitted item, so +/// vendor extensions (`revision`, `x-*` flags, etc.) roundtrip +/// byte-faithfully. +const RECORD_SPEC_FIELDS: &[&str] = &["type", "description", "key", "record"]; +const QUERY_SPEC_FIELDS: &[&str] = &[ + "type", "description", "parameters", "output", "errors", +]; +const PROCEDURE_SPEC_FIELDS: &[&str] = &[ + "type", "description", "parameters", "input", "output", "errors", +]; +const SUBSCRIPTION_SPEC_FIELDS: &[&str] = &[ + "type", "description", "parameters", "message", "errors", +]; +const TOKEN_SPEC_FIELDS: &[&str] = &["type", "description"]; + +/// Spec-defined fields at the top of a def-type definition. Covers +/// primitives (with their constraint keys), containers (array, object, +/// union, ref) and unifies them all — anything outside this list on a +/// def-type JSON object is treated as an extension. +const DEF_TYPE_SPEC_FIELDS: &[&str] = &[ + "type", "description", + // Constraint keys (mirror CONSTRAINT_KEYS). + "minLength", "maxLength", "minGraphemes", "maxGraphemes", + "minimum", "maximum", "format", "enum", "knownValues", + "accept", "maxSize", "default", "const", + // Container keys. + "items", "properties", "required", "nullable", + "refs", "closed", "ref", +]; + +/// Build a `self {}` item from the top-level JSON, or `None` when +/// there's nothing to emit (no description, no unknown fields). Docs +/// come from top-level `description`; extension fields become `@const` +/// annotations. +fn render_self_item(json: &Value, ctx: &ConversionContext) -> Option { + let obj = json.as_object()?; + + let description = obj.get("description").and_then(|v| v.as_str()).unwrap_or(""); + let has_extension = obj + .keys() + .any(|k| !TOP_LEVEL_SPEC_FIELDS.contains(&k.as_str())); + + if description.is_empty() && !has_extension { + return None; + } + + let mut out = String::new(); + for line in description.lines() { + out.push_str("/// "); + out.push_str(line); + out.push('\n'); + } + for (key, value) in obj { + if TOP_LEVEL_SPEC_FIELDS.contains(&key.as_str()) { + continue; + } + warn_if_reference_shaped(ctx, key, value); + out.push_str(&format!( + "@const(\"{}\", {})\n", + escape_string_for_mlf(key), + render_json_as_mlf_literal(value) + )); + } + out.push_str("self {}\n"); + Some(out) +} + +/// Heuristic: warn when a `@const` string value contains `#`, since that's +/// the ATProto local-ref shape and the author may have intended `@reference`. +/// The converter can't know intent from JSON, so it always emits `@const`; +/// the warning nudges hand-review. +fn warn_if_reference_shaped(ctx: &ConversionContext, key: &str, value: &Value) { + let Value::String(s) = value else { return }; + if !s.contains('#') { + return; + } + ctx.warn(format!( + "extension field {:?} has value {:?} which looks NSID-shaped; \ + emitted as `@const` — consider `@reference` if you intend workspace \ + name resolution when hand-editing the MLF", + key, s + )); +} + +/// Emit a placeholder `def type X = unknown;` with `@const` annotations +/// for every field — used when the def's `type` isn't in our known +/// set. Keeps the lexicon's shape roundtrippable without a dedicated +/// grammar entry. +fn render_unknown_def_passthrough( + name: &str, + def: &Value, + last_segment: &str, + ctx: &ConversionContext, +) -> String { + let obj = match def.as_object() { + Some(o) => o, + None => return format!("def type {} = unknown;\n", escape_name(name)), + }; + let mut out = String::new(); + if let Some(description) = obj.get("description").and_then(|v| v.as_str()) { + if !description.is_empty() { + for line in description.lines() { + out.push_str("/// "); + out.push_str(line); + out.push('\n'); + } + } + } + if name == "main" { + out.push_str("@main\n"); + } + for (key, value) in obj { + // `description` surfaced as the doc-comment block already; don't + // double-emit as `@const`. Every other field — including `type` + // itself — passes through as an annotation so the lexicon's + // shape is preserved verbatim. + if key == "description" { + continue; + } + warn_if_reference_shaped(ctx, key, value); + out.push_str(&format!( + "@const(\"{}\", {})\n", + escape_string_for_mlf(key), + render_json_as_mlf_literal(value) + )); + } + let def_name = if name == "main" { + escape_name(last_segment) + } else { + escape_name(name) + }; + out.push_str(&format!("def type {} = unknown;\n", def_name)); + out +} + +/// Emit `@const(key, value)` annotation lines for every field on `def` +/// that isn't listed in `spec_fields`. Each generator calls this after +/// emitting `@main` (if applicable) and before the declaration line, +/// so vendor extensions carry through in the same position the codegen +/// expects to find them when emitting JSON back. +fn render_extension_annotations( + def: &Value, + spec_fields: &[&str], + ctx: &ConversionContext, +) -> String { + let Some(obj) = def.as_object() else { + return String::new(); + }; + let mut out = String::new(); + for (key, value) in obj { + if spec_fields.contains(&key.as_str()) { + continue; + } + warn_if_reference_shaped(ctx, key, value); + out.push_str(&format!( + "@const(\"{}\", {})\n", + escape_string_for_mlf(key), + render_json_as_mlf_literal(value) + )); + } + out +} + +/// Render a JSON value as MLF source text suitable for use as an +/// annotation-value literal (the second arg of `@const`). Handles every +/// JSON shape; strings are quoted and escaped, objects use the +/// `{ "key": value, ... }` form. +fn render_json_as_mlf_literal(value: &Value) -> String { + match value { + Value::Null => "null".to_string(), + Value::Bool(b) => b.to_string(), + Value::String(s) => format!("\"{}\"", escape_string_for_mlf(s)), + Value::Number(n) => { + if let Some(i) = n.as_i64() { + i.to_string() + } else if let Some(f) = n.as_f64() { + f.to_string() + } else { + "null".to_string() + } + } + Value::Array(items) => { + let rendered: Vec = items.iter().map(render_json_as_mlf_literal).collect(); + format!("[{}]", rendered.join(", ")) + } + Value::Object(map) => { + let rendered: Vec = map + .iter() + .map(|(k, v)| { + format!( + "\"{}\": {}", + escape_string_for_mlf(k), + render_json_as_mlf_literal(v) + ) + }) + .collect(); + format!("{{ {} }}", rendered.join(", ")) + } + } +} + +fn escape_string_for_mlf(s: &str) -> String { + s.replace('\\', "\\\\").replace('"', "\\\"") +} + struct ConversionContext { current_namespace: String, /// MLF-side name of this lexicon's main def. @@ -365,6 +576,8 @@ fn generate_record(name: &str, def: &Value, ctx: &ConversionContext) -> Result Result Resul output.push_str("@main\n"); } + output.push_str(&render_extension_annotations(def, PROCEDURE_SPEC_FIELDS, ctx)); + let procedure_name = if name == "main" { escape_name(&ctx.local_main_name) } else { @@ -625,6 +842,8 @@ fn generate_subscription(name: &str, def: &Value, ctx: &ConversionContext) -> Re output.push_str("@main\n"); } + output.push_str(&render_extension_annotations(def, SUBSCRIPTION_SPEC_FIELDS, ctx)); + let subscription_name = if name == "main" { escape_name(&ctx.local_main_name) } else { @@ -680,7 +899,11 @@ fn generate_subscription(name: &str, def: &Value, ctx: &ConversionContext) -> Re Ok(output) } -fn generate_token(name: &str, def: &Value) -> Result { +fn generate_token( + name: &str, + def: &Value, + ctx: &ConversionContext, +) -> Result { let mut output = String::new(); // Add doc comment @@ -692,6 +915,8 @@ fn generate_token(name: &str, def: &Value) -> Result { } } + output.push_str(&render_extension_annotations(def, TOKEN_SPEC_FIELDS, ctx)); + let escaped_name = escape_name(name); output.push_str(&format!("token {};\n", escaped_name)); Ok(output) @@ -714,6 +939,8 @@ fn generate_def_type(name: &str, def: &Value, ctx: &ConversionContext) -> Result output.push_str("@main\n"); } + output.push_str(&render_extension_annotations(def, DEF_TYPE_SPEC_FIELDS, ctx)); + // Use last segment of NSID for "main" definitions // Keywords are now allowed by the parser, so just escape with backticks let def_name = if name == "main" { diff --git a/mlf-cli/src/generate/mlf.rs.backup b/mlf-cli/src/generate/mlf.rs.backup deleted file mode 100644 index a58612c..0000000 --- a/mlf-cli/src/generate/mlf.rs.backup +++ /dev/null @@ -1,941 +0,0 @@ -use miette::Diagnostic; -use serde_json::Value; -use std::path::PathBuf; -use thiserror::Error; - -#[derive(Error, Debug, Diagnostic)] -pub enum MlfGenerateError { - #[error("Failed to read file: {path}")] - #[diagnostic(code(mlf::generate::read_file))] - #[allow(dead_code)] - ReadFile { - path: String, - #[source] - source: std::io::Error, - }, - - #[error("Failed to parse JSON: {path}")] - #[diagnostic(code(mlf::generate::parse_json))] - #[allow(dead_code)] - ParseJson { - path: String, - #[source] - source: serde_json::Error, - }, - - #[error("Failed to write output: {path}")] - #[diagnostic(code(mlf::generate::write_output))] - WriteOutput { - path: String, - #[source] - source: std::io::Error, - }, - - #[error("Invalid lexicon format: {message}")] - #[diagnostic(code(mlf::generate::invalid_lexicon))] - InvalidLexicon { message: String }, - - #[error("Failed to expand glob pattern")] - #[diagnostic(code(mlf::generate::glob_error))] - GlobError { - #[source] - source: glob::GlobError, - }, - - #[error("Invalid glob pattern: {pattern}")] - #[diagnostic(code(mlf::generate::invalid_glob))] - InvalidGlob { - pattern: String, - #[source] - source: glob::PatternError, - }, -} - -pub fn run(input_patterns: Vec, output_dir: PathBuf) -> Result<(), MlfGenerateError> { - let mut file_paths = Vec::new(); - - for pattern in input_patterns { - if pattern.contains('*') || pattern.contains('?') { - for entry in glob::glob(&pattern).map_err(|source| MlfGenerateError::InvalidGlob { - pattern: pattern.clone(), - source, - })? { - let path = entry.map_err(|source| MlfGenerateError::GlobError { source })?; - file_paths.push(path); - } - } else { - file_paths.push(PathBuf::from(pattern)); - } - } - - std::fs::create_dir_all(&output_dir).map_err(|source| MlfGenerateError::WriteOutput { - path: output_dir.display().to_string(), - source, - })?; - - let mut errors = Vec::new(); - let mut success_count = 0; - - for file_path in file_paths { - let source = match std::fs::read_to_string(&file_path) { - Ok(s) => s, - Err(source) => { - errors.push(( - file_path.display().to_string(), - format!("Failed to read file: {}", source), - )); - continue; - } - }; - - let json: Value = match serde_json::from_str(&source) { - Ok(j) => j, - Err(source) => { - errors.push(( - file_path.display().to_string(), - format!("Failed to parse JSON: {}", source), - )); - continue; - } - }; - - let mlf_content = match generate_mlf_from_json(&json) { - Ok(content) => content, - Err(e) => { - errors.push((file_path.display().to_string(), format!("{:?}", e))); - continue; - } - }; - - // Extract namespace from JSON "id" field - let namespace = json - .get("id") - .and_then(|v| v.as_str()) - .ok_or_else(|| MlfGenerateError::InvalidLexicon { - message: "Missing 'id' field in lexicon".to_string(), - })?; - - // Create output path from namespace - let mut output_path = output_dir.clone(); - for segment in namespace.split('.') { - output_path.push(segment); - } - if let Err(source) = std::fs::create_dir_all(&output_path.parent().unwrap()) { - errors.push(( - file_path.display().to_string(), - format!("Failed to create directory: {}", source), - )); - continue; - } - output_path.set_extension("mlf"); - - if let Err(source) = std::fs::write(&output_path, mlf_content) { - errors.push(( - output_path.display().to_string(), - format!("Failed to write file: {}", source), - )); - continue; - } - - println!("Generated: {}", output_path.display()); - success_count += 1; - } - - if !errors.is_empty() { - eprintln!( - "\n{} file(s) generated successfully, {} error(s) encountered:\n", - success_count, - errors.len() - ); - for (path, error) in &errors { - eprintln!(" {} - {}", path, error); - } - eprintln!(); - return Err(MlfGenerateError::InvalidLexicon { - message: format!("{} errors total", errors.len()), - }); - } - - println!("\nSuccessfully generated {} file(s)", success_count); - Ok(()) -} - -pub fn generate_mlf_from_json(json: &Value) -> Result { - let mut output = String::new(); - - // Extract NSID to get the last segment for "main" definitions - let nsid = json - .get("id") - .and_then(|v| v.as_str()) - .ok_or_else(|| MlfGenerateError::InvalidLexicon { - message: "Missing 'id' field in lexicon".to_string(), - })?; - - let last_segment = nsid.split('.').last().unwrap_or("main"); - - let defs = json.get("defs").and_then(|v| v.as_object()).ok_or_else(|| { - MlfGenerateError::InvalidLexicon { - message: "Missing or invalid 'defs' field".to_string(), - } - })?; - - // Create a context to pass the current namespace to type generation - let ctx = ConversionContext { - current_namespace: nsid.to_string(), - }; - - // Process all definitions - for (name, def) in defs { - let def_type = def.get("type").and_then(|v| v.as_str()).ok_or_else(|| { - MlfGenerateError::InvalidLexicon { - message: format!("Missing 'type' field for definition '{}'", name), - } - })?; - - match def_type { - "record" => { - let mlf = generate_record(name, def, last_segment, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "query" => { - let mlf = generate_query(name, def, last_segment, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "procedure" => { - let mlf = generate_procedure(name, def, last_segment, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "subscription" => { - let mlf = generate_subscription(name, def, last_segment, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - "token" => { - let mlf = generate_token(name, def)?; - output.push_str(&mlf); - output.push('\n'); - } - "object" => { - let mlf = generate_def_type(name, def, last_segment, &ctx)?; - output.push_str(&mlf); - output.push('\n'); - } - _ => { - // Unknown type, skip - } - } - } - - Ok(output) -} - -struct ConversionContext { - current_namespace: String, -} - -/// Reserved words in MLF that need to be escaped -const RESERVED_WORDS: &[&str] = &[ - "main", "record", "query", "procedure", "subscription", "token", "def", "type", "use", - "pub", "alias", "namespace", "constrained", "error", "unit", "null", "boolean", - "integer", "string", "bytes", "blob", "unknown", "array", "object", "union", "ref", -]; - -/// Escape a name if it's a reserved word -fn escape_name(name: &str) -> String { - if RESERVED_WORDS.contains(&name) { - format!("`{}`", name) - } else { - name.to_string() - } -} - -fn generate_record(name: &str, def: &Value, last_segment: &str, ctx: &ConversionContext) -> Result { - let mut output = String::new(); - - // Add doc comment if present - if let Some(desc) = def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!("/// {}\n", line)); - } - } - } - - // Add @main annotation for "main" definitions - if name == "main" { - output.push_str("@main\n"); - } - - // Use last segment of NSID for "main" definitions - let record_name = if name == "main" { - escape_name(last_segment) - } else { - escape_name(name) - }; - - output.push_str(&format!("record {} {{\n", record_name)); - - // Get the record object - let record_obj = def.get("record").and_then(|v| v.as_object()).ok_or_else(|| { - MlfGenerateError::InvalidLexicon { - message: format!("Missing 'record' field in record definition '{}'", name), - } - })?; - - let properties = record_obj - .get("properties") - .and_then(|v| v.as_object()) - .ok_or_else(|| MlfGenerateError::InvalidLexicon { - message: format!("Missing 'properties' in record '{}'", name), - })?; - - let required = record_obj - .get("required") - .and_then(|v| v.as_array()) - .map(|arr| { - arr.iter() - .filter_map(|v| v.as_str()) - .collect::>() - }) - .unwrap_or_default(); - - for (field_name, field_def) in properties { - // Add field doc comment - if let Some(desc) = field_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!(" /// {}\n", line)); - } - } - } - - let is_required = required.contains(&field_name.as_str()); - let required_marker = if is_required { "!" } else { "" }; - - let field_type = generate_type(field_def)?; - let escaped_field_name = escape_name(field_name); - output.push_str(&format!( - " {}{}: {},\n", - escaped_field_name, required_marker, field_type - )); - } - - output.push_str("}\n"); - Ok(output) -} - -fn generate_query(name: &str, def: &Value, last_segment: &str, ctx: &ConversionContext) -> Result { - let mut output = String::new(); - - // Add doc comment - if let Some(desc) = def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!("/// {}\n", line)); - } - } - } - - // Add @main annotation for "main" definitions - if name == "main" { - output.push_str("@main\n"); - } - - let query_name = if name == "main" { - escape_name(last_segment) - } else { - escape_name(name) - }; - output.push_str(&format!("query {}", query_name)); - - // Parameters - output.push('('); - if let Some(params) = def.get("parameters").and_then(|v| v.as_object()) { - let properties = params.get("properties").and_then(|v| v.as_object()); - let required = params - .get("required") - .and_then(|v| v.as_array()) - .map(|arr| { - arr.iter() - .filter_map(|v| v.as_str()) - .collect::>() - }) - .unwrap_or_default(); - - if let Some(props) = properties { - let param_strs: Vec = props - .iter() - .map(|(param_name, param_def)| { - let is_required = required.contains(¶m_name.as_str()); - let required_marker = if is_required { "!" } else { "" }; - let param_type = generate_type(param_def).unwrap_or_else(|_| "unknown".to_string()); - let escaped_param_name = escape_name(param_name); - - // Add doc comment inline if present - let mut result = String::new(); - if let Some(desc) = param_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - result.push_str(&format!("\n /// {}\n ", desc)); - } - } - result.push_str(&format!("{}{}: {}", escaped_param_name, required_marker, param_type)); - result - }) - .collect(); - - if !param_strs.is_empty() { - output.push_str(¶m_strs.join(",")); - } - } - } - output.push(')'); - - // Output type - if let Some(output_obj) = def.get("output").and_then(|v| v.as_object()) { - if let Some(schema) = output_obj.get("schema") { - let return_type = generate_type(schema)?; - output.push_str(&format!(": {}", return_type)); - - // Check for errors - if let Some(errors) = output_obj.get("errors").and_then(|v| v.as_object()) { - output.push_str(" | error {\n"); - for (error_name, error_def) in errors { - if let Some(desc) = error_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - output.push_str(&format!(" /// {}\n", desc)); - } - } - output.push_str(&format!(" {},\n", error_name)); - } - output.push('}'); - } - } - } - - output.push_str(";\n"); - Ok(output) -} - -fn generate_procedure(name: &str, def: &Value, last_segment: &str, ctx: &ConversionContext) -> Result { - let mut output = String::new(); - - // Add doc comment - if let Some(desc) = def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!("/// {}\n", line)); - } - } - } - - // Add @main annotation for "main" definitions - if name == "main" { - output.push_str("@main\n"); - } - - let procedure_name = if name == "main" { - escape_name(last_segment) - } else { - escape_name(name) - }; - output.push_str(&format!("procedure {}", procedure_name)); - - // Input parameters - output.push('('); - if let Some(input) = def.get("input").and_then(|v| v.as_object()) { - if let Some(schema) = input.get("schema").and_then(|v| v.as_object()) { - let properties = schema.get("properties").and_then(|v| v.as_object()); - let required = schema - .get("required") - .and_then(|v| v.as_array()) - .map(|arr| { - arr.iter() - .filter_map(|v| v.as_str()) - .collect::>() - }) - .unwrap_or_default(); - - if let Some(props) = properties { - let param_strs: Vec = props - .iter() - .map(|(param_name, param_def)| { - let is_required = required.contains(¶m_name.as_str()); - let required_marker = if is_required { "!" } else { "" }; - let param_type = - generate_type(param_def).unwrap_or_else(|_| "unknown".to_string()); - let escaped_param_name = escape_name(param_name); - - // Add doc comment inline if present - let mut result = String::new(); - if let Some(desc) = param_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - result.push_str(&format!("\n /// {}\n ", desc)); - } - } - result.push_str(&format!( - "{}{}: {}", - escaped_param_name, required_marker, param_type - )); - result - }) - .collect(); - - if !param_strs.is_empty() { - output.push_str(¶m_strs.join(",")); - } - } - } - } - output.push(')'); - - // Output type - if let Some(output_obj) = def.get("output").and_then(|v| v.as_object()) { - if let Some(schema) = output_obj.get("schema") { - let return_type = generate_type(schema)?; - output.push_str(&format!(": {}", return_type)); - - // Check for errors - if let Some(errors) = output_obj.get("errors").and_then(|v| v.as_object()) { - output.push_str(" | error {\n"); - for (error_name, error_def) in errors { - if let Some(desc) = error_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - output.push_str(&format!(" /// {}\n", desc)); - } - } - output.push_str(&format!(" {},\n", error_name)); - } - output.push('}'); - } - } - } - - output.push_str(";\n"); - Ok(output) -} - -fn generate_subscription(name: &str, def: &Value, last_segment: &str, ctx: &ConversionContext) -> Result { - let mut output = String::new(); - - // Add doc comment - if let Some(desc) = def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!("/// {}\n", line)); - } - } - } - - // Add @main annotation for "main" definitions - if name == "main" { - output.push_str("@main\n"); - } - - let subscription_name = if name == "main" { - escape_name(last_segment) - } else { - escape_name(name) - }; - output.push_str(&format!("subscription {}", subscription_name)); - - // Parameters - output.push('('); - if let Some(params) = def.get("parameters").and_then(|v| v.as_object()) { - let properties = params.get("properties").and_then(|v| v.as_object()); - let required = params - .get("required") - .and_then(|v| v.as_array()) - .map(|arr| { - arr.iter() - .filter_map(|v| v.as_str()) - .collect::>() - }) - .unwrap_or_default(); - - if let Some(props) = properties { - let param_strs: Vec = props - .iter() - .map(|(param_name, param_def)| { - let is_required = required.contains(¶m_name.as_str()); - let required_marker = if is_required { "!" } else { "" }; - let param_type = generate_type(param_def).unwrap_or_else(|_| "unknown".to_string()); - let escaped_param_name = escape_name(param_name); - - format!("{}{}: {}", escaped_param_name, required_marker, param_type) - }) - .collect(); - - if !param_strs.is_empty() { - output.push_str(¶m_strs.join(", ")); - } - } - } - output.push(')'); - - // Message types - if let Some(message) = def.get("message").and_then(|v| v.as_object()) { - if let Some(schema) = message.get("schema") { - let message_type = generate_type(schema)?; - output.push_str(&format!(": {}", message_type)); - } - } - - output.push_str(";\n"); - Ok(output) -} - -fn generate_token(name: &str, def: &Value) -> Result { - let mut output = String::new(); - - // Add doc comment - if let Some(desc) = def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!("/// {}\n", line)); - } - } - } - - let escaped_name = escape_name(name); - output.push_str(&format!("token {};\n", escaped_name)); - Ok(output) -} - -fn generate_def_type(name: &str, def: &Value, last_segment: &str, ctx: &ConversionContext) -> Result { - let mut output = String::new(); - - // Add @main annotation for "main" definitions - if name == "main" { - output.push_str("@main\n"); - } - - // Use last segment of NSID for "main" definitions - let def_name = if name == "main" { - escape_name(last_segment) - } else { - escape_name(name) - }; - - output.push_str(&format!("def type {} = ", def_name)); - let type_str = generate_type_with_indent(def, 0)?; - output.push_str(&type_str); - output.push_str(";\n"); - - Ok(output) -} - -fn generate_type_with_indent(type_def: &Value, indent_level: usize, ctx: &ConversionContext) -> Result { - let type_name = type_def.get("type").and_then(|v| v.as_str()); - - match type_name { - Some("object") => { - let indent = " ".repeat(indent_level); - let field_indent = " ".repeat(indent_level + 1); - - let mut output = String::from("{\n"); - let properties = type_def - .get("properties") - .and_then(|v| v.as_object()) - .ok_or_else(|| MlfGenerateError::InvalidLexicon { - message: "Missing 'properties' in object type".to_string(), - })?; - - let required = type_def - .get("required") - .and_then(|v| v.as_array()) - .map(|arr| { - arr.iter() - .filter_map(|v| v.as_str()) - .collect::>() - }) - .unwrap_or_default(); - - for (field_name, field_def) in properties { - // Add field doc comment - if let Some(desc) = field_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!("{}/// {}\n", field_indent, line)); - } - } - } - - let is_required = required.contains(&field_name.as_str()); - let required_marker = if is_required { "!" } else { "" }; - let field_type = generate_type_with_indent(field_def, indent_level + 1)?; - let escaped_field_name = escape_name(field_name); - output.push_str(&format!( - "{}{}{}: {},\n", - field_indent, escaped_field_name, required_marker, field_type - )); - } - - output.push_str(&format!("{}}}", indent)); - Ok(output) - } - _ => generate_type(type_def), - } -} - -fn generate_type(type_def: &Value, ctx: &ConversionContext) -> Result { - let type_name = type_def.get("type").and_then(|v| v.as_str()); - - match type_name { - Some("null") => Ok("null".to_string()), - Some("boolean") => Ok("boolean".to_string()), - Some("integer") => { - let mut result = "integer".to_string(); - result = apply_constraints(result, type_def); - Ok(result) - } - Some("string") => { - // Check if this is a format string that maps to a prelude type - if let Some(format) = type_def.get("format").and_then(|v| v.as_str()) { - let prelude_type = match format { - "did" => "Did", - "at-uri" => "AtUri", - "at-identifier" => "AtIdentifier", - "handle" => "Handle", - "datetime" => "Datetime", - "uri" => "Uri", - "cid" => "Cid", - "nsid" => "Nsid", - "tid" => "Tid", - "record-key" => "RecordKey", - "language" => "Language", - _ => { - // Unknown format, fall through to normal string with constraints - let mut result = "string".to_string(); - result = apply_constraints(result, type_def); - return Ok(result); - } - }; - // If it's a known prelude type with only the format constraint, use the prelude type directly - // Check if there are other constraints besides format - let has_other_constraints = type_def.get("minLength").is_some() - || type_def.get("maxLength").is_some() - || type_def.get("minGraphemes").is_some() - || type_def.get("maxGraphemes").is_some() - || type_def.get("enum").is_some() - || type_def.get("knownValues").is_some() - || type_def.get("default").is_some(); - - if !has_other_constraints { - return Ok(prelude_type.to_string()); - } - } - - let mut result = "string".to_string(); - result = apply_constraints(result, type_def); - Ok(result) - } - Some("bytes") => Ok("bytes".to_string()), - Some("blob") => { - let mut result = "blob".to_string(); - result = apply_constraints(result, type_def); - Ok(result) - } - Some("unknown") => Ok("unknown".to_string()), - Some("array") => { - let items = type_def.get("items").ok_or_else(|| { - MlfGenerateError::InvalidLexicon { - message: "Missing 'items' in array type".to_string(), - } - })?; - - // Check if items have constraints - let items_obj = items.as_object(); - let has_item_constraints = items_obj.map_or(false, |obj| { - obj.contains_key("minLength") || - obj.contains_key("maxLength") || - obj.contains_key("minGraphemes") || - obj.contains_key("maxGraphemes") || - obj.contains_key("minimum") || - obj.contains_key("maximum") || - obj.contains_key("enum") || - obj.contains_key("knownValues") || - obj.contains_key("default") - }); - - let item_type = if has_item_constraints { - // If item has constraints, we need to wrap in parentheses to apply constraints before [] - // For now, just generate the base type without item constraints - // TODO: Consider generating a type alias for complex constrained items - items.get("type") - .and_then(|t| t.as_str()) - .unwrap_or("unknown") - .to_string() - } else { - generate_type(items)? - }; - - let mut result = format!("{}[]", item_type); - result = apply_constraints(result, type_def); - Ok(result) - } - Some("object") => { - let mut output = String::from("{\n"); - let properties = type_def - .get("properties") - .and_then(|v| v.as_object()) - .ok_or_else(|| MlfGenerateError::InvalidLexicon { - message: "Missing 'properties' in object type".to_string(), - })?; - - let required = type_def - .get("required") - .and_then(|v| v.as_array()) - .map(|arr| { - arr.iter() - .filter_map(|v| v.as_str()) - .collect::>() - }) - .unwrap_or_default(); - - for (field_name, field_def) in properties { - // Add field doc comment - if let Some(desc) = field_def.get("description").and_then(|v| v.as_str()) { - if !desc.is_empty() { - for line in desc.lines() { - output.push_str(&format!(" /// {}\n", line)); - } - } - } - - let is_required = required.contains(&field_name.as_str()); - let required_marker = if is_required { "!" } else { "" }; - let field_type = generate_type(field_def)?; - let escaped_field_name = escape_name(field_name); - output.push_str(&format!( - " {}{}: {},\n", - escaped_field_name, required_marker, field_type - )); - } - - output.push_str(" }"); - Ok(output) - } - Some("union") => { - let refs = type_def.get("refs").and_then(|v| v.as_array()).ok_or_else(|| { - MlfGenerateError::InvalidLexicon { - message: "Missing 'refs' in union type".to_string(), - } - })?; - - let type_strs: Vec = refs - .iter() - .map(|r| generate_type(r).unwrap_or_else(|_| "unknown".to_string())) - .collect(); - - let mut result = type_strs.join(" | "); - - // Check if closed - if type_def.get("closed").and_then(|v| v.as_bool()).unwrap_or(false) { - result.push_str(" | !"); - } - - Ok(result) - } - Some("ref") => { - if let Some(ref_str) = type_def.get("ref").and_then(|v| v.as_str()) { - // Handle references: - // "#defName" -> "defName" (local reference, same file) - // "namespace.id#defName" -> Check if same namespace, if so use "defName", else use full path - - if let Some(stripped) = ref_str.strip_prefix('#') { - // Local reference: #defName -> defName - Ok(stripped.to_string()) - } else if let Some((namespace, def_name)) = ref_str.split_once('#') { - // Check if this is the current namespace - // For now, we'll just use the def name if it's the same namespace - // Note: This requires passing context through, which we'll add - // For external refs, we keep the full NSID format - Ok(format!("{}.{}", namespace, def_name)) - } else { - // No # at all - shouldn't happen in valid lexicons, but handle gracefully - Ok(ref_str.to_string()) - } - } else { - Err(MlfGenerateError::InvalidLexicon { - message: "Missing 'ref' in ref type".to_string(), - }) - } - } - _ => Ok("unknown".to_string()), - } -} - -fn apply_constraints(mut type_str: String, type_def: &Value) -> String { - let mut constraints = Vec::new(); - - if let Some(min_length) = type_def.get("minLength").and_then(|v| v.as_i64()) { - constraints.push(format!("minLength: {}", min_length)); - } - if let Some(max_length) = type_def.get("maxLength").and_then(|v| v.as_i64()) { - constraints.push(format!("maxLength: {}", max_length)); - } - if let Some(min_graphemes) = type_def.get("minGraphemes").and_then(|v| v.as_i64()) { - constraints.push(format!("minGraphemes: {}", min_graphemes)); - } - if let Some(max_graphemes) = type_def.get("maxGraphemes").and_then(|v| v.as_i64()) { - constraints.push(format!("maxGraphemes: {}", max_graphemes)); - } - if let Some(minimum) = type_def.get("minimum").and_then(|v| v.as_i64()) { - constraints.push(format!("minimum: {}", minimum)); - } - if let Some(maximum) = type_def.get("maximum").and_then(|v| v.as_i64()) { - constraints.push(format!("maximum: {}", maximum)); - } - if let Some(format) = type_def.get("format").and_then(|v| v.as_str()) { - constraints.push(format!("format: \"{}\"", format)); - } - if let Some(enum_vals) = type_def.get("enum").and_then(|v| v.as_array()) { - let vals: Vec = enum_vals - .iter() - .filter_map(|v| v.as_str()) - .map(|s| format!("\"{}\"", s)) - .collect(); - constraints.push(format!("enum: [{}]", vals.join(", "))); - } - if let Some(known_vals) = type_def.get("knownValues").and_then(|v| v.as_array()) { - let vals: Vec = known_vals - .iter() - .filter_map(|v| v.as_str()) - .map(|s| format!("\"{}\"", s)) - .collect(); - constraints.push(format!("knownValues: [{}]", vals.join(", "))); - } - if let Some(accept) = type_def.get("accept").and_then(|v| v.as_array()) { - let mimes: Vec = accept - .iter() - .filter_map(|v| v.as_str()) - .map(|s| format!("\"{}\"", s)) - .collect(); - constraints.push(format!("accept: [{}]", mimes.join(", "))); - } - if let Some(max_size) = type_def.get("maxSize").and_then(|v| v.as_i64()) { - constraints.push(format!("maxSize: {}", max_size)); - } - if let Some(default) = type_def.get("default") { - let default_str = match default { - Value::String(s) => format!("\"{}\"", s), - Value::Number(n) => n.to_string(), - Value::Bool(b) => b.to_string(), - _ => "null".to_string(), - }; - constraints.push(format!("default: {}", default_str)); - } - - if !constraints.is_empty() { - type_str.push_str(" constrained {\n"); - for constraint in &constraints { - type_str.push_str(&format!(" {},\n", constraint)); - } - type_str.push_str(" }"); - } - - type_str -} diff --git a/mlf-codegen/src/lib.rs b/mlf-codegen/src/lib.rs index f29d0a7..f967bda 100644 --- a/mlf-codegen/src/lib.rs +++ b/mlf-codegen/src/lib.rs @@ -3,6 +3,25 @@ use mlf_lang::{ResolvedRef, Workspace}; use serde_json::{json, Map, Value}; use std::collections::HashMap; +/// A non-fatal advisory emitted during codegen. Mirrors the shape of +/// `ConversionWarning` on the JSON→MLF side so both directions use one +/// structured-warning model. +#[derive(Debug, Clone, PartialEq)] +pub struct CodegenWarning { + /// Namespace of the lexicon that emitted the warning. + pub namespace: String, + /// Human-readable description of what was coerced and why. + pub message: String, +} + +/// Bundle of codegen output: the generated lexicon JSON plus any +/// advisory warnings produced while walking the AST. +#[derive(Debug, Clone)] +pub struct CodegenOutput { + pub json: Value, + pub warnings: Vec, +} + // Re-export inventory for macros #[doc(hidden)] pub use inventory; @@ -112,43 +131,60 @@ fn get_encoding_annotation(annotations: &[Annotation], param_name: &str) -> Opti }) } -pub fn generate_lexicon(namespace: &str, lexicon: &Lexicon, workspace: &Workspace) -> Value { +pub fn generate_lexicon(namespace: &str, lexicon: &Lexicon, workspace: &Workspace) -> CodegenOutput { let usage_counts = analyze_type_usage(lexicon); let eligibility = MainEligibility::for_lexicon(namespace, lexicon); let mut defs = Map::new(); + let mut self_description = String::new(); + let mut self_extensions: Vec<(String, Value)> = Vec::new(); + let mut warnings: Vec = Vec::new(); for item in &lexicon.items { match item { Item::Record(record) => { - let value = generate_record_json(record, &usage_counts, workspace, namespace); + let mut value = generate_record_json(record, &usage_counts, workspace, namespace); + apply_extension_annotations(&mut value, &record.annotations, workspace, namespace, &mut warnings); insert_def(&mut defs, &record.name.name, eligibility.is_main(&record.name.name, &record.annotations), value); } Item::Query(query) => { - let value = generate_query_json(query, &usage_counts, workspace, namespace); + let mut value = generate_query_json(query, &usage_counts, workspace, namespace); + apply_extension_annotations(&mut value, &query.annotations, workspace, namespace, &mut warnings); insert_def(&mut defs, &query.name.name, eligibility.is_main(&query.name.name, &query.annotations), value); } Item::Procedure(procedure) => { - let value = generate_procedure_json(procedure, &usage_counts, workspace, namespace); + let mut value = generate_procedure_json(procedure, &usage_counts, workspace, namespace); + apply_extension_annotations(&mut value, &procedure.annotations, workspace, namespace, &mut warnings); insert_def(&mut defs, &procedure.name.name, eligibility.is_main(&procedure.name.name, &procedure.annotations), value); } Item::Subscription(subscription) => { - let value = generate_subscription_json(subscription, &usage_counts, workspace, namespace); + let mut value = generate_subscription_json(subscription, &usage_counts, workspace, namespace); + apply_extension_annotations(&mut value, &subscription.annotations, workspace, namespace, &mut warnings); insert_def(&mut defs, &subscription.name.name, eligibility.is_main(&subscription.name.name, &subscription.annotations), value); } Item::DefType(def_type) => { - let value = generate_def_type_json(def_type, &usage_counts, workspace, namespace); + let mut value = generate_def_type_json(def_type, &usage_counts, workspace, namespace); + apply_extension_annotations(&mut value, &def_type.annotations, workspace, namespace, &mut warnings); insert_def(&mut defs, &def_type.name.name, eligibility.is_main(&def_type.name.name, &def_type.annotations), value); } Item::Token(token) => { let mut token_obj = Map::new(); token_obj.insert("type".to_string(), json!("token")); insert_opt_str(&mut token_obj, "description", &extract_docs(&token.docs)); - defs.insert(token.name.name.clone(), Value::Object(token_obj)); + let mut value = Value::Object(token_obj); + apply_extension_annotations(&mut value, &token.annotations, workspace, namespace, &mut warnings); + defs.insert(token.name.name.clone(), value); + } + Item::SelfItem(self_item) => { + // The `self { }` item carries lexicon-level metadata. Its + // docs become the top-level `description`; its extension + // annotations become top-level JSON fields alongside + // `lexicon`, `id`, `defs`. + self_description = extract_docs(&self_item.docs); + self_extensions = collect_extension_fields(&self_item.annotations, workspace, namespace, &mut warnings); } // Inline types never appear in `defs` — they expand at their point - // of use. Other item kinds (use statements, namespace blocks) are - // structural and not emitted into the lexicon output. + // of use. Use statements are structural and not emitted. _ => {} } } @@ -157,8 +193,164 @@ pub fn generate_lexicon(namespace: &str, lexicon: &Lexicon, workspace: &Workspac root.insert("$type".to_string(), json!("com.atproto.lexicon.schema")); root.insert("lexicon".to_string(), json!(1)); root.insert("id".to_string(), json!(namespace)); + insert_opt_str(&mut root, "description", &self_description); + for (key, value) in self_extensions { + root.insert(key, value); + } root.insert("defs".to_string(), json!(defs)); - Value::Object(root) + CodegenOutput { + json: Value::Object(root), + warnings, + } +} + +/// Extract `@const` / `@reference` annotations into (key, JSON value) +/// pairs. Used by the self-item handling to populate top-level lexicon +/// fields. Annotations of other names (including generator-scoped ones +/// like `@rust:deprecated`) are ignored. +fn collect_extension_fields( + annotations: &[Annotation], + workspace: &Workspace, + current_namespace: &str, + warnings: &mut Vec, +) -> Vec<(String, Value)> { + annotations + .iter() + .filter_map(|ann| extension_field(ann, workspace, current_namespace, warnings)) + .collect() +} + +/// Mutate `value` (expected to be a JSON object) in place, adding any +/// extension fields derived from `@const` / `@reference` annotations. +/// Non-object values are left untouched. +fn apply_extension_annotations( + value: &mut Value, + annotations: &[Annotation], + workspace: &Workspace, + current_namespace: &str, + warnings: &mut Vec, +) { + let Some(obj) = value.as_object_mut() else { + return; + }; + for (key, field_value) in collect_extension_fields(annotations, workspace, current_namespace, warnings) { + obj.insert(key, field_value); + } +} + +/// If `annotation` is a recognised extension (`@const` or `@reference`), +/// return the `(key, JSON value)` it produces. Otherwise `None`. +/// +/// `@const(key, value)` — value is a literal; rendered verbatim. +/// `@reference(key, path)` — path is resolved through the workspace and +/// emitted as an NSID string. +fn extension_field( + annotation: &Annotation, + workspace: &Workspace, + current_namespace: &str, + warnings: &mut Vec, +) -> Option<(String, Value)> { + if !annotation.selectors.is_empty() { + // Generator-scoped annotation (e.g. `@rust:deprecated`); not an + // extension. + return None; + } + let name = annotation.name.name.as_str(); + let is_const = name == "const"; + let is_reference = name == "reference"; + if !is_const && !is_reference { + return None; + } + // Both forms take exactly two positional args: (key: string, value). + let mut args = annotation.args.iter(); + let key_arg = args.next()?; + let value_arg = args.next()?; + if args.next().is_some() { + return None; + } + let AnnotationArg::Positional(AnnotationValue::String(key)) = key_arg else { + return None; + }; + let AnnotationArg::Positional(value) = value_arg else { + return None; + }; + let json_value = if is_const { + annotation_value_to_json(value, key, current_namespace, warnings) + } else { + // @reference: expect a type path; emit the resolved NSID. + let AnnotationValue::Reference(path) = value else { + return None; + }; + Value::String(resolve_ref_nsid(path, workspace, current_namespace)) + }; + Some((key.clone(), json_value)) +} + +/// Literal-to-JSON conversion for `@const` values. Coerces whole-number +/// f64 to i64 (so `@const("revision", 3)` emits `3`, not `3.0`) and +/// transparently stringifies genuinely fractional numbers — ATProto's +/// data model has no floats, so the spec-compliant representation is a +/// string. `key` and `namespace` are used to tag any emitted warnings. +fn annotation_value_to_json( + value: &AnnotationValue, + key: &str, + namespace: &str, + warnings: &mut Vec, +) -> Value { + match value { + AnnotationValue::String(s) => Value::String(s.clone()), + AnnotationValue::Number(n) => { + if n.is_finite() && n.fract() == 0.0 && *n >= i64::MIN as f64 && *n <= i64::MAX as f64 { + Value::Number(serde_json::Number::from(*n as i64)) + } else if n.is_finite() { + warnings.push(CodegenWarning { + namespace: namespace.to_string(), + message: format!( + "@const({:?}, {}): ATProto's data model has no floats; \ + emitting as string {:?} to stay spec-compliant", + key, n, n.to_string() + ), + }); + Value::String(n.to_string()) + } else { + warnings.push(CodegenWarning { + namespace: namespace.to_string(), + message: format!( + "@const({:?}, {}): non-finite number is not representable in JSON; \ + emitting `null`", + key, n + ), + }); + Value::Null + } + } + AnnotationValue::Boolean(b) => Value::Bool(*b), + AnnotationValue::Null => Value::Null, + AnnotationValue::Array(items) => { + Value::Array( + items + .iter() + .map(|item| annotation_value_to_json(item, key, namespace, warnings)) + .collect(), + ) + } + AnnotationValue::Object(entries) => { + let mut obj = Map::new(); + for (entry_key, entry_value) in entries { + obj.insert( + entry_key.clone(), + annotation_value_to_json(entry_value, key, namespace, warnings), + ); + } + Value::Object(obj) + } + // A `Reference` shouldn't appear in a `@const` value position — + // it would mean the author wrote a bare identifier where they + // meant a literal. Fall back to emitting the path as-is; the + // parser could also reject this earlier, but defensively handle + // it here. + AnnotationValue::Reference(path) => Value::String(path.to_string()), + } } /// Decides whether a given def should be emitted under the key `main` or @@ -874,8 +1066,11 @@ impl CodeGenerator for JsonLexiconGenerator { } fn generate(&self, ctx: &GeneratorContext) -> Result { - let json = generate_lexicon(ctx.namespace, ctx.lexicon, ctx.workspace); - serde_json::to_string_pretty(&json) + // The plugin `CodeGenerator` trait doesn't carry a warning + // channel today; the callers that care about warnings call + // `generate_lexicon` directly instead of going through this shim. + let output = generate_lexicon(ctx.namespace, ctx.lexicon, ctx.workspace); + serde_json::to_string_pretty(&output.json) .map_err(|e| format!("Failed to serialize JSON: {}", e)) } } diff --git a/mlf-lang/src/ast.rs b/mlf-lang/src/ast.rs index beffc62..3e6e2bb 100644 --- a/mlf-lang/src/ast.rs +++ b/mlf-lang/src/ast.rs @@ -38,6 +38,11 @@ pub enum Item { Procedure(Procedure), Subscription(Subscription), Use(Use), + /// `self { }` — a declaration representing the lexicon itself. Docs + /// and annotations attached to it map to the lexicon's top-level + /// fields in JSON (e.g. `description`, vendor extensions). Body is + /// empty in V1; the `{}` shape is reserved for future contents. + SelfItem(SelfItem), } impl Spanned for Item { @@ -51,10 +56,19 @@ impl Spanned for Item { Item::Procedure(p) => p.span, Item::Subscription(s) => s.span, Item::Use(u) => u.span, + Item::SelfItem(s) => s.span, } } } +/// The lexicon-as-item. See [`Item::SelfItem`] for the semantic role. +#[derive(Debug, Clone, PartialEq)] +pub struct SelfItem { + pub docs: Vec, + pub annotations: Vec, + pub span: Span, +} + /// Documentation comment #[derive(Debug, Clone, PartialEq, Eq)] pub struct DocComment { @@ -87,6 +101,17 @@ pub enum AnnotationValue { String(String), Number(f64), Boolean(bool), + /// JSON `null`. Used by `@const` to represent explicit-null values + /// that appear in source lexicons' extension fields. + Null, + /// JSON array. Element types are freely mixed, matching JSON semantics. + Array(Vec), + /// JSON object literal — a map with string keys. Keys permit any + /// JSON-legal form (including hyphens, e.g. `"x-vendor-flag"`). + Object(Vec<(String, AnnotationValue)>), + /// A type path resolved through the workspace. Used by `@reference` + /// to name an MLF type by path; codegen resolves to an NSID string. + Reference(Path), } /// A record definition diff --git a/mlf-lang/src/lexer.rs b/mlf-lang/src/lexer.rs index c2dbdc0..61e09c9 100644 --- a/mlf-lang/src/lexer.rs +++ b/mlf-lang/src/lexer.rs @@ -29,6 +29,7 @@ pub enum Token { Procedure, Query, Record, + SelfKw, String, Subscription, Token, @@ -84,6 +85,7 @@ impl core::fmt::Display for Token { Token::Procedure => write!(f, "procedure"), Token::Query => write!(f, "query"), Token::Record => write!(f, "record"), + Token::SelfKw => write!(f, "self"), Token::String => write!(f, "string"), Token::Subscription => write!(f, "subscription"), Token::Token => write!(f, "token"), @@ -158,6 +160,7 @@ fn identifier(input: &str) -> IResult<&str, Token> { "procedure" => Token::Procedure, "query" => Token::Query, "record" => Token::Record, + "self" => Token::SelfKw, "string" => Token::String, "subscription" => Token::Subscription, "token" => Token::Token, diff --git a/mlf-lang/src/parser.rs b/mlf-lang/src/parser.rs index a713283..f9dafe9 100644 --- a/mlf-lang/src/parser.rs +++ b/mlf-lang/src/parser.rs @@ -211,6 +211,7 @@ impl Parser { LexToken::Procedure => self.parse_procedure(doc_comments, annotations), LexToken::Subscription => self.parse_subscription(doc_comments, annotations), LexToken::Use => self.parse_use(), + LexToken::SelfKw => self.parse_self(doc_comments, annotations), _ => Err(ParseError::Syntax { message: alloc::format!("Expected item definition, found {}", self.current().token), span: self.current().span, @@ -218,6 +219,36 @@ impl Parser { } } + /// Parse a `self { }` declaration — the lexicon-as-item. Body is + /// required but currently always empty; the `{}` shape is reserved + /// for future contents (see the C6 design doc). + fn parse_self( + &mut self, + docs: Vec, + annotations: Vec, + ) -> Result { + let start = self.expect(LexToken::SelfKw)?; + self.expect(LexToken::LeftBrace)?; + // V1: body must be empty. Allow whitespace/comments (already + // skipped by the tokeniser) but reject any real content. + if !matches!(self.current().token, LexToken::RightBrace) { + return Err(ParseError::Syntax { + message: alloc::format!( + "`self {{}}` body is reserved for future use; expected `}}`, found {}", + self.current().token + ), + span: self.current().span, + }); + } + let end = self.expect(LexToken::RightBrace)?; + + Ok(Item::SelfItem(SelfItem { + docs, + annotations, + span: Span::new(start.start, end.end), + })) + } + fn parse_annotations(&mut self) -> Result, ParseError> { let mut annotations = Vec::new(); @@ -326,6 +357,18 @@ impl Parser { self.advance(); Ok(AnnotationValue::Boolean(false)) } + LexToken::Null => { + self.advance(); + Ok(AnnotationValue::Null) + } + LexToken::LeftBracket => self.parse_annotation_value_array(), + LexToken::LeftBrace => self.parse_annotation_value_object(), + // Identifier → type reference path. Consumed here so `@reference` + // can carry a named type like `SomeType` or `ns.Foo`. + LexToken::Ident(_) => { + let path = self.parse_path()?; + Ok(AnnotationValue::Reference(path)) + } _ => Err(ParseError::Syntax { message: alloc::format!("Expected annotation value, found {}", current.token), span: current.span, @@ -333,6 +376,57 @@ impl Parser { } } + /// Parse `[value, value, ...]` as an annotation-value array. Trailing + /// commas are accepted. Empty `[]` is valid. + fn parse_annotation_value_array(&mut self) -> Result { + self.expect(LexToken::LeftBracket)?; + let mut items = Vec::new(); + while !matches!(self.current().token, LexToken::RightBracket) { + items.push(self.parse_annotation_value()?); + if matches!(self.current().token, LexToken::Comma) { + self.advance(); + } else { + break; + } + } + self.expect(LexToken::RightBracket)?; + Ok(AnnotationValue::Array(items)) + } + + /// Parse `{ "key": value, ... }` as an annotation-value object. Keys + /// must be string literals (not identifiers) so arbitrary JSON-legal + /// keys like `"x-vendor-flag"` work without additional grammar. + fn parse_annotation_value_object(&mut self) -> Result { + self.expect(LexToken::LeftBrace)?; + let mut entries = Vec::new(); + while !matches!(self.current().token, LexToken::RightBrace) { + let current = self.current(); + let key = match ¤t.token { + LexToken::StringLit(s) => s.clone(), + _ => { + return Err(ParseError::Syntax { + message: alloc::format!( + "Expected string literal as object key, found {}", + current.token + ), + span: current.span, + }); + } + }; + self.advance(); + self.expect(LexToken::Colon)?; + let value = self.parse_annotation_value()?; + entries.push((key, value)); + if matches!(self.current().token, LexToken::Comma) { + self.advance(); + } else { + break; + } + } + self.expect(LexToken::RightBrace)?; + Ok(AnnotationValue::Object(entries)) + } + fn parse_record(&mut self, docs: Vec, annotations: Vec) -> Result { let start = self.expect(LexToken::Record)?; let name = self.parse_ident()?; diff --git a/mlf-lang/src/workspace.rs b/mlf-lang/src/workspace.rs index b1efad2..dc4f12c 100644 --- a/mlf-lang/src/workspace.rs +++ b/mlf-lang/src/workspace.rs @@ -1115,7 +1115,7 @@ impl Workspace { Item::Query(q) => Some(q.name.name.as_str()), Item::Procedure(p) => Some(p.name.name.as_str()), Item::Subscription(s) => Some(s.name.name.as_str()), - Item::Use(_) => None, + Item::Use(_) | Item::SelfItem(_) => None, }; if let Some(name) = name { @@ -1136,7 +1136,7 @@ impl Workspace { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; errors.push(crate::error::ValidationError::ReservedName { name: name.clone(), @@ -1204,7 +1204,7 @@ impl Workspace { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; errors.push(crate::error::ValidationError::ConflictNotAllowed { name: name.clone(), @@ -1230,7 +1230,7 @@ impl Workspace { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; errors.push(crate::error::ValidationError::DuplicateDefinition { name: name.clone(), @@ -1242,7 +1242,7 @@ impl Workspace { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }, second_span: span, module_namespace: namespace.to_string(), @@ -1395,7 +1395,7 @@ impl Workspace { Item::Query(q) => self.resolve_query(namespace, q), Item::Procedure(p) => self.resolve_procedure(namespace, p), Item::Subscription(s) => self.resolve_subscription(namespace, s), - Item::Token(_) | Item::Use(_) => Ok(()), + Item::Token(_) | Item::Use(_) | Item::SelfItem(_) => Ok(()), } } diff --git a/mlf-lsp/src/server.rs b/mlf-lsp/src/server.rs index eadbc2e..0fd5084 100644 --- a/mlf-lsp/src/server.rs +++ b/mlf-lsp/src/server.rs @@ -566,7 +566,7 @@ impl MlfLanguageServer { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; self.client @@ -664,7 +664,7 @@ impl MlfLanguageServer { Item::Query(q) => Some(q.name.span), Item::Procedure(p) => Some(p.name.span), Item::Subscription(s) => Some(s.name.span), - Item::Use(_) => None, + Item::Use(_) | Item::SelfItem(_) => None, }; if let Some(span) = def_span { @@ -684,7 +684,7 @@ impl MlfLanguageServer { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; return Some((doc_uri.clone(), def_span)); @@ -908,6 +908,7 @@ impl LanguageServer for MlfLanguageServer { Item::Query(q) => Some(&q.annotations), Item::Procedure(p) => Some(&p.annotations), Item::Subscription(s) => Some(&s.annotations), + Item::SelfItem(s) => Some(&s.annotations), _ => None, }; @@ -944,6 +945,8 @@ impl LanguageServer for MlfLanguageServer { "cache" => "Defines caching strategy", "indexed" => "Marks this field as indexed", "sensitive" => "Marks this field as containing sensitive data (e.g., PII)", + "const" => "Extension field (literal). `@const(key, value)` emits `key: value` verbatim in the JSON Lexicon. Use on `self {}` for top-level fields, or any item for per-item fields.", + "reference" => "Extension field (named-type reference). `@reference(key, path)` resolves `path` through the workspace and emits the resulting NSID string under `key`.", _ => "Custom annotation", }; @@ -1022,6 +1025,7 @@ impl LanguageServer for MlfLanguageServer { Item::Procedure(_) => "procedure", Item::Subscription(_) => "subscription", Item::Use(_) => "use", + Item::SelfItem(_) => "self", }; contents.push(MarkedString::LanguageString(LanguageString { @@ -1029,6 +1033,12 @@ impl LanguageServer for MlfLanguageServer { value: format!("{} {}", kind, name), })); + if let Item::SelfItem(_) = item { + contents.push(MarkedString::String( + "Represents the lexicon itself. Doc comments emit as the top-level `description`; `@const` / `@reference` annotations emit as top-level JSON fields.".to_string() + )); + } + // Add documentation if available if !docs.is_empty() { contents.push(MarkedString::String(docs.join("\n"))); @@ -1427,6 +1437,7 @@ impl LanguageServer for MlfLanguageServer { ("procedure", CompletionItemKind::KEYWORD, "Define a procedure"), ("subscription", CompletionItemKind::KEYWORD, "Define a subscription"), ("use", CompletionItemKind::KEYWORD, "Import types from another module"), + ("self", CompletionItemKind::KEYWORD, "Lexicon-as-item — attach top-level docs and @const / @reference extensions"), ]; for (label, kind, detail) in keywords { @@ -1620,7 +1631,7 @@ impl LanguageServer for MlfLanguageServer { Item::Query(q) => q.name.span, Item::Procedure(p) => p.name.span, Item::Subscription(s) => s.name.span, - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; let range = span_to_range(&text, def_span); @@ -1745,7 +1756,7 @@ impl LanguageServer for MlfLanguageServer { SymbolKind::EVENT, span_to_range(&doc_state.text, s.span), ), - Item::Use(_) => continue, + Item::Use(_) | Item::SelfItem(_) => continue, }; #[allow(deprecated)] diff --git a/mlf-lsp/src/utils.rs b/mlf-lsp/src/utils.rs index 6b6ac20..9bb5931 100644 --- a/mlf-lsp/src/utils.rs +++ b/mlf-lsp/src/utils.rs @@ -79,6 +79,7 @@ pub fn find_item_at_offset(lexicon: &Lexicon, offset: usize) -> Option<&Item> { Item::Procedure(p) => p.span, Item::Subscription(s) => s.span, Item::Use(u) => u.span, + Item::SelfItem(s) => s.span, }; offset_in_span(offset, span) }) @@ -133,7 +134,7 @@ pub fn get_item_name(item: &Item) -> &str { Item::Query(q) => &q.name.name, Item::Procedure(p) => &p.name.name, Item::Subscription(s) => &s.name.name, - Item::Use(_) => "", + Item::Use(_) | Item::SelfItem(_) => "", } } @@ -147,6 +148,7 @@ pub fn get_item_docs(item: &Item) -> Vec { Item::Query(q) => &q.docs, Item::Procedure(p) => &p.docs, Item::Subscription(s) => &s.docs, + Item::SelfItem(s) => &s.docs, Item::Use(_) => return vec![], }; diff --git a/mlf-wasm/src/lib.rs b/mlf-wasm/src/lib.rs index 87d27cf..c95dfbb 100644 --- a/mlf-wasm/src/lib.rs +++ b/mlf-wasm/src/lib.rs @@ -139,10 +139,12 @@ pub fn generate_lexicon(source: &str, namespace: &str) -> JsValue { return serde_wasm_bindgen::to_value(&result).unwrap(); } - // Generate JSON lexicon - let json_lexicon = mlf_codegen::generate_lexicon(namespace, &lexicon, &workspace); + // Generate JSON lexicon. Codegen warnings are dropped on the floor + // in the wasm surface today — the playground doesn't have a UI for + // them yet; when it does, they flow through `output.warnings`. + let output = mlf_codegen::generate_lexicon(namespace, &lexicon, &workspace); - match serde_json::to_string_pretty(&json_lexicon) { + match serde_json::to_string_pretty(&output.json) { Ok(json_str) => { let result = GenerateResult { success: true, diff --git a/tests/codegen/lexicon/const_float_stringifies/expected.json b/tests/codegen/lexicon/const_float_stringifies/expected.json new file mode 100644 index 0000000..df16f5c --- /dev/null +++ b/tests/codegen/lexicon/const_float_stringifies/expected.json @@ -0,0 +1,20 @@ +{ + "$type": "com.atproto.lexicon.schema", + "lexicon": 1, + "id": "com.example.const_float_stringifies", + "x-threshold": "3.14", + "x-revision": 3, + "defs": { + "main": { + "type": "record", + "key": "tid", + "record": { + "type": "object", + "required": ["name"], + "properties": { + "name": {"type": "string"} + } + } + } + } +} diff --git a/tests/codegen/lexicon/const_float_stringifies/expected_warnings.txt b/tests/codegen/lexicon/const_float_stringifies/expected_warnings.txt new file mode 100644 index 0000000..08981b0 --- /dev/null +++ b/tests/codegen/lexicon/const_float_stringifies/expected_warnings.txt @@ -0,0 +1 @@ +com.example.const_float_stringifies: @const("x-threshold", 3.14): ATProto's data model has no floats; emitting as string "3.14" to stay spec-compliant diff --git a/tests/codegen/lexicon/const_float_stringifies/input.mlf b/tests/codegen/lexicon/const_float_stringifies/input.mlf new file mode 100644 index 0000000..1c5beda --- /dev/null +++ b/tests/codegen/lexicon/const_float_stringifies/input.mlf @@ -0,0 +1,8 @@ +@const("x-threshold", 3.14) +@const("x-revision", 3) +self {} + +@main +record constFloatStringifies { + name!: string, +} diff --git a/tests/codegen/lexicon/const_float_stringifies/test.toml b/tests/codegen/lexicon/const_float_stringifies/test.toml new file mode 100644 index 0000000..f63012b --- /dev/null +++ b/tests/codegen/lexicon/const_float_stringifies/test.toml @@ -0,0 +1,4 @@ +[test] +name = "const_float_stringifies" +description = "@const with a fractional numeric value stringifies to stay spec-compliant and emits a warning" +namespace = "com.example.const_float_stringifies" diff --git a/tests/codegen/lexicon/item_extensions_codegen/expected.json b/tests/codegen/lexicon/item_extensions_codegen/expected.json new file mode 100644 index 0000000..8eafe5c --- /dev/null +++ b/tests/codegen/lexicon/item_extensions_codegen/expected.json @@ -0,0 +1,20 @@ +{ + "$type": "com.atproto.lexicon.schema", + "lexicon": 1, + "id": "com.example.item_extensions_codegen", + "defs": { + "main": { + "type": "record", + "key": "tid", + "record": { + "type": "object", + "required": ["name"], + "properties": { + "name": {"type": "string"} + } + }, + "x-deprecated": true, + "x-since": "2024-01-01" + } + } +} diff --git a/tests/codegen/lexicon/item_extensions_codegen/input.mlf b/tests/codegen/lexicon/item_extensions_codegen/input.mlf new file mode 100644 index 0000000..38bbb3c --- /dev/null +++ b/tests/codegen/lexicon/item_extensions_codegen/input.mlf @@ -0,0 +1,6 @@ +@main +@const("x-deprecated", true) +@const("x-since", "2024-01-01") +record itemExtensions { + name!: string, +} diff --git a/tests/codegen/lexicon/item_extensions_codegen/test.toml b/tests/codegen/lexicon/item_extensions_codegen/test.toml new file mode 100644 index 0000000..056fab4 --- /dev/null +++ b/tests/codegen/lexicon/item_extensions_codegen/test.toml @@ -0,0 +1,4 @@ +[test] +name = "item_extensions_codegen" +description = "@const annotations on a record emit as extra JSON fields on the record def" +namespace = "com.example.item_extensions_codegen" diff --git a/tests/codegen/lexicon/self_item/expected.json b/tests/codegen/lexicon/self_item/expected.json new file mode 100644 index 0000000..6850fb0 --- /dev/null +++ b/tests/codegen/lexicon/self_item/expected.json @@ -0,0 +1,18 @@ +{ + "$type": "com.atproto.lexicon.schema", + "lexicon": 1, + "id": "com.example.self_item", + "description": "A lexicon with top-level metadata.", + "revision": 3, + "x-vendor-flag": true, + "xFallbackType": "#selfItem", + "defs": { + "main": { + "type": "object", + "required": ["value"], + "properties": { + "value": {"type": "string"} + } + } + } +} diff --git a/tests/codegen/lexicon/self_item/input.mlf b/tests/codegen/lexicon/self_item/input.mlf new file mode 100644 index 0000000..0f2691c --- /dev/null +++ b/tests/codegen/lexicon/self_item/input.mlf @@ -0,0 +1,10 @@ +/// A lexicon with top-level metadata. +@const("revision", 3) +@const("x-vendor-flag", true) +@reference("xFallbackType", selfItem) +self {} + +@main +def type selfItem = { + value!: string, +}; diff --git a/tests/codegen/lexicon/self_item/test.toml b/tests/codegen/lexicon/self_item/test.toml new file mode 100644 index 0000000..25f768b --- /dev/null +++ b/tests/codegen/lexicon/self_item/test.toml @@ -0,0 +1,4 @@ +[test] +name = "self_item" +description = "self {} with docs and @const/@reference extensions emit as top-level JSON fields" +namespace = "com.example.self_item" diff --git a/tests/codegen_integration.rs b/tests/codegen_integration.rs index f251135..c332627 100644 --- a/tests/codegen_integration.rs +++ b/tests/codegen_integration.rs @@ -38,16 +38,41 @@ fn run_lexicon_test(input_path: &Path) -> datatest_stable::Result<()> { let lexicon = ws.get_lexicon(&namespace).ok_or("Module not found")?; - let output_json = generate_lexicon(&namespace, lexicon, &ws); + let output = generate_lexicon(&namespace, lexicon, &ws); let expected_str = fs::read_to_string(test_dir.join("expected.json"))?; let expected_json: Value = serde_json::from_str(&expected_str)?; - if output_json != expected_json { + if output.json != expected_json { return Err(format!( "Output mismatch:\nExpected:\n{}\n\nGot:\n{}", serde_json::to_string_pretty(&expected_json).unwrap(), - serde_json::to_string_pretty(&output_json).unwrap() + serde_json::to_string_pretty(&output.json).unwrap() + ) + .into()); + } + + let warnings_path = test_dir.join("expected_warnings.txt"); + let expected_warnings = if warnings_path.exists() { + fs::read_to_string(&warnings_path)? + } else { + String::new() + }; + let actual_warnings = output + .warnings + .iter() + .map(|w| format!("{}: {}", w.namespace, w.message)) + .collect::>() + .join("\n"); + let actual_warnings = if actual_warnings.is_empty() { + String::new() + } else { + format!("{}\n", actual_warnings) + }; + if actual_warnings != expected_warnings { + return Err(format!( + "Warnings mismatch:\n--- expected ---\n{}\n--- got ---\n{}", + expected_warnings, actual_warnings ) .into()); } diff --git a/tests/lexicon_to_mlf/item_extensions/expected.mlf b/tests/lexicon_to_mlf/item_extensions/expected.mlf new file mode 100644 index 0000000..d07cc1c --- /dev/null +++ b/tests/lexicon_to_mlf/item_extensions/expected.mlf @@ -0,0 +1,7 @@ +@main +@const("x-deprecated", true) +@const("x-since", "2024-01-01") +record itemext { + name!: string, +} + diff --git a/tests/lexicon_to_mlf/item_extensions/input.json b/tests/lexicon_to_mlf/item_extensions/input.json new file mode 100644 index 0000000..2caffd4 --- /dev/null +++ b/tests/lexicon_to_mlf/item_extensions/input.json @@ -0,0 +1,19 @@ +{ + "lexicon": 1, + "id": "com.example.itemext", + "defs": { + "main": { + "type": "record", + "key": "tid", + "x-deprecated": true, + "x-since": "2024-01-01", + "record": { + "type": "object", + "required": ["name"], + "properties": { + "name": {"type": "string"} + } + } + } + } +} diff --git a/tests/lexicon_to_mlf/ref_hint_warning/expected.mlf b/tests/lexicon_to_mlf/ref_hint_warning/expected.mlf new file mode 100644 index 0000000..2c65f0b --- /dev/null +++ b/tests/lexicon_to_mlf/ref_hint_warning/expected.mlf @@ -0,0 +1,6 @@ +@main +@const("xFallback", "com.example.other#someType") +def type refhint = { + value: string, +}; + diff --git a/tests/lexicon_to_mlf/ref_hint_warning/expected_warnings.txt b/tests/lexicon_to_mlf/ref_hint_warning/expected_warnings.txt new file mode 100644 index 0000000..de41d1a --- /dev/null +++ b/tests/lexicon_to_mlf/ref_hint_warning/expected_warnings.txt @@ -0,0 +1 @@ +com.example.refhint: extension field "xFallback" has value "com.example.other#someType" which looks NSID-shaped; emitted as `@const` — consider `@reference` if you intend workspace name resolution when hand-editing the MLF diff --git a/tests/lexicon_to_mlf/ref_hint_warning/input.json b/tests/lexicon_to_mlf/ref_hint_warning/input.json new file mode 100644 index 0000000..3ec46b8 --- /dev/null +++ b/tests/lexicon_to_mlf/ref_hint_warning/input.json @@ -0,0 +1,14 @@ +{ + "lexicon": 1, + "id": "com.example.refhint", + "defs": { + "main": { + "type": "object", + "required": [], + "properties": { + "value": {"type": "string"} + }, + "xFallback": "com.example.other#someType" + } + } +} diff --git a/tests/lexicon_to_mlf/self_top_level_description/expected.mlf b/tests/lexicon_to_mlf/self_top_level_description/expected.mlf new file mode 100644 index 0000000..57d6587 --- /dev/null +++ b/tests/lexicon_to_mlf/self_top_level_description/expected.mlf @@ -0,0 +1,8 @@ +/// A short description of the whole lexicon. +self {} + +@main +def type described = { + name!: string, +}; + diff --git a/tests/lexicon_to_mlf/self_top_level_description/input.json b/tests/lexicon_to_mlf/self_top_level_description/input.json new file mode 100644 index 0000000..af4caf2 --- /dev/null +++ b/tests/lexicon_to_mlf/self_top_level_description/input.json @@ -0,0 +1,14 @@ +{ + "lexicon": 1, + "id": "com.example.described", + "description": "A short description of the whole lexicon.", + "defs": { + "main": { + "type": "object", + "required": ["name"], + "properties": { + "name": {"type": "string"} + } + } + } +} diff --git a/tests/lexicon_to_mlf/self_top_level_extensions/expected.mlf b/tests/lexicon_to_mlf/self_top_level_extensions/expected.mlf new file mode 100644 index 0000000..b1f1049 --- /dev/null +++ b/tests/lexicon_to_mlf/self_top_level_extensions/expected.mlf @@ -0,0 +1,11 @@ +/// With extensions. +@const("revision", 3) +@const("x-vendor-flag", true) +@const("x-tags", ["alpha", "beta"]) +self {} + +@main +def type extended = { + value: string, +}; + diff --git a/tests/lexicon_to_mlf/self_top_level_extensions/input.json b/tests/lexicon_to_mlf/self_top_level_extensions/input.json new file mode 100644 index 0000000..61fd940 --- /dev/null +++ b/tests/lexicon_to_mlf/self_top_level_extensions/input.json @@ -0,0 +1,17 @@ +{ + "lexicon": 1, + "id": "com.example.extended", + "description": "With extensions.", + "revision": 3, + "x-vendor-flag": true, + "x-tags": ["alpha", "beta"], + "defs": { + "main": { + "type": "object", + "required": [], + "properties": { + "value": {"type": "string"} + } + } + } +} diff --git a/tests/lexicon_to_mlf/unknown_def_type_passthrough/expected.mlf b/tests/lexicon_to_mlf/unknown_def_type_passthrough/expected.mlf new file mode 100644 index 0000000..98d53be --- /dev/null +++ b/tests/lexicon_to_mlf/unknown_def_type_passthrough/expected.mlf @@ -0,0 +1,7 @@ +@main +@const("type", "permission-set") +@const("title", "Example Access") +@const("detail", "Access to example resources.") +@const("permissions", [{ "type": "permission", "resource": "repo", "collection": ["com.example.one", "com.example.two"] }]) +def type permission = unknown; + diff --git a/tests/lexicon_to_mlf/unknown_def_type_passthrough/input.json b/tests/lexicon_to_mlf/unknown_def_type_passthrough/input.json new file mode 100644 index 0000000..e1c767b --- /dev/null +++ b/tests/lexicon_to_mlf/unknown_def_type_passthrough/input.json @@ -0,0 +1,18 @@ +{ + "lexicon": 1, + "id": "com.example.permission", + "defs": { + "main": { + "type": "permission-set", + "title": "Example Access", + "detail": "Access to example resources.", + "permissions": [ + { + "type": "permission", + "resource": "repo", + "collection": ["com.example.one", "com.example.two"] + } + ] + } + } +} diff --git a/tests/real_world/roundtrip.rs b/tests/real_world/roundtrip.rs index 8910769..471414c 100644 --- a/tests/real_world/roundtrip.rs +++ b/tests/real_world/roundtrip.rs @@ -168,8 +168,8 @@ fn regenerate_lexicons_from_mlf(mlf_dir: &Path) -> Result seq( + repeat($.annotation), + 'self', + '{', + '}' ), // Comments @@ -77,7 +89,41 @@ module.exports = grammar({ annotation_value: $ => choice( $.string, $.number, - $.boolean + $.boolean, + $.null_literal, + $.annotation_array, + $.annotation_object, + // A bare type path — used by the `@reference` extension annotation + // to name a type that codegen resolves through the workspace. + $.type_path + ), + + null_literal: $ => 'null', + + annotation_array: $ => seq( + '[', + optional(seq( + $.annotation_value, + repeat(seq(',', $.annotation_value)), + optional(',') + )), + ']' + ), + + annotation_object: $ => seq( + '{', + optional(seq( + $.annotation_object_entry, + repeat(seq(',', $.annotation_object_entry)), + optional(',') + )), + '}' + ), + + annotation_object_entry: $ => seq( + field('key', $.string), + ':', + field('value', $.annotation_value) ), // Use statements diff --git a/tree-sitter-mlf/queries/highlights.scm b/tree-sitter-mlf/queries/highlights.scm index 06d23d9..3fe2781 100644 --- a/tree-sitter-mlf/queries/highlights.scm +++ b/tree-sitter-mlf/queries/highlights.scm @@ -13,8 +13,12 @@ "subscription" "error" "constrained" + "self" ] @keyword +; Null literal (in annotation values) +(null_literal) @constant.builtin + ; Primitive types [ "null" diff --git a/tree-sitter-mlf/src/grammar.json b/tree-sitter-mlf/src/grammar.json index 969c204..44a8ffd 100644 --- a/tree-sitter-mlf/src/grammar.json +++ b/tree-sitter-mlf/src/grammar.json @@ -1,4 +1,5 @@ { + "$schema": "https://tree-sitter.github.io/tree-sitter/assets/schemas/grammar.schema.json", "name": "mlf", "rules": { "source_file": { @@ -42,6 +43,34 @@ { "type": "SYMBOL", "name": "subscription_definition" + }, + { + "type": "SYMBOL", + "name": "self_definition" + } + ] + }, + "self_definition": { + "type": "SEQ", + "members": [ + { + "type": "REPEAT", + "content": { + "type": "SYMBOL", + "name": "annotation" + } + }, + { + "type": "STRING", + "value": "self" + }, + { + "type": "STRING", + "value": "{" + }, + { + "type": "STRING", + "value": "}" } ] }, @@ -265,6 +294,167 @@ { "type": "SYMBOL", "name": "boolean" + }, + { + "type": "SYMBOL", + "name": "null_literal" + }, + { + "type": "SYMBOL", + "name": "annotation_array" + }, + { + "type": "SYMBOL", + "name": "annotation_object" + }, + { + "type": "SYMBOL", + "name": "type_path" + } + ] + }, + "null_literal": { + "type": "STRING", + "value": "null" + }, + "annotation_array": { + "type": "SEQ", + "members": [ + { + "type": "STRING", + "value": "[" + }, + { + "type": "CHOICE", + "members": [ + { + "type": "SEQ", + "members": [ + { + "type": "SYMBOL", + "name": "annotation_value" + }, + { + "type": "REPEAT", + "content": { + "type": "SEQ", + "members": [ + { + "type": "STRING", + "value": "," + }, + { + "type": "SYMBOL", + "name": "annotation_value" + } + ] + } + }, + { + "type": "CHOICE", + "members": [ + { + "type": "STRING", + "value": "," + }, + { + "type": "BLANK" + } + ] + } + ] + }, + { + "type": "BLANK" + } + ] + }, + { + "type": "STRING", + "value": "]" + } + ] + }, + "annotation_object": { + "type": "SEQ", + "members": [ + { + "type": "STRING", + "value": "{" + }, + { + "type": "CHOICE", + "members": [ + { + "type": "SEQ", + "members": [ + { + "type": "SYMBOL", + "name": "annotation_object_entry" + }, + { + "type": "REPEAT", + "content": { + "type": "SEQ", + "members": [ + { + "type": "STRING", + "value": "," + }, + { + "type": "SYMBOL", + "name": "annotation_object_entry" + } + ] + } + }, + { + "type": "CHOICE", + "members": [ + { + "type": "STRING", + "value": "," + }, + { + "type": "BLANK" + } + ] + } + ] + }, + { + "type": "BLANK" + } + ] + }, + { + "type": "STRING", + "value": "}" + } + ] + }, + "annotation_object_entry": { + "type": "SEQ", + "members": [ + { + "type": "FIELD", + "name": "key", + "content": { + "type": "SYMBOL", + "name": "string" + } + }, + { + "type": "STRING", + "value": ":" + }, + { + "type": "FIELD", + "name": "value", + "content": { + "type": "SYMBOL", + "name": "annotation_value" + } } ] }, @@ -1436,6 +1626,6 @@ "precedences": [], "externals": [], "inline": [], - "supertypes": [] -} - + "supertypes": [], + "reserved": {} +} \ No newline at end of file diff --git a/tree-sitter-mlf/src/node-types.json b/tree-sitter-mlf/src/node-types.json index ff85253..fbb8301 100644 --- a/tree-sitter-mlf/src/node-types.json +++ b/tree-sitter-mlf/src/node-types.json @@ -76,6 +76,62 @@ ] } }, + { + "type": "annotation_array", + "named": true, + "fields": {}, + "children": { + "multiple": true, + "required": false, + "types": [ + { + "type": "annotation_value", + "named": true + } + ] + } + }, + { + "type": "annotation_object", + "named": true, + "fields": {}, + "children": { + "multiple": true, + "required": false, + "types": [ + { + "type": "annotation_object_entry", + "named": true + } + ] + } + }, + { + "type": "annotation_object_entry", + "named": true, + "fields": { + "key": { + "multiple": false, + "required": true, + "types": [ + { + "type": "string", + "named": true + } + ] + }, + "value": { + "multiple": false, + "required": true, + "types": [ + { + "type": "annotation_value", + "named": true + } + ] + } + } + }, { "type": "annotation_selectors", "named": true, @@ -99,10 +155,22 @@ "multiple": false, "required": true, "types": [ + { + "type": "annotation_array", + "named": true + }, + { + "type": "annotation_object", + "named": true + }, { "type": "boolean", "named": true }, + { + "type": "null_literal", + "named": true + }, { "type": "number", "named": true @@ -110,6 +178,10 @@ { "type": "string", "named": true + }, + { + "type": "type_path", + "named": true } ] } @@ -460,6 +532,10 @@ "type": "record_definition", "named": true }, + { + "type": "self_definition", + "named": true + }, { "type": "subscription_definition", "named": true @@ -506,6 +582,11 @@ ] } }, + { + "type": "null_literal", + "named": true, + "fields": {} + }, { "type": "object_type", "named": true, @@ -754,9 +835,25 @@ ] } }, + { + "type": "self_definition", + "named": true, + "fields": {}, + "children": { + "multiple": true, + "required": false, + "types": [ + { + "type": "annotation", + "named": true + } + ] + } + }, { "type": "source_file", "named": true, + "root": true, "fields": {}, "children": { "multiple": true, @@ -996,7 +1093,8 @@ }, { "type": "comment", - "named": true + "named": true, + "extra": true }, { "type": "constrained", @@ -1008,7 +1106,8 @@ }, { "type": "doc_comment", - "named": true + "named": true, + "extra": true }, { "type": "error", @@ -1050,6 +1149,10 @@ "type": "record", "named": false }, + { + "type": "self", + "named": false + }, { "type": "string", "named": false diff --git a/website/content/docs/language-guide/11-annotations.md b/website/content/docs/language-guide/11-annotations.md index c755e64..7f8a5ec 100644 --- a/website/content/docs/language-guide/11-annotations.md +++ b/website/content/docs/language-guide/11-annotations.md @@ -30,8 +30,11 @@ record example { Arguments can be: - **Strings**: `"value"` -- **Numbers**: `42`, `3.14` +- **Numbers**: `42`, `3.14`, `-10` - **Booleans**: `true`, `false` +- **Null**: `null` +- **Arrays / objects** of the above (JSON-shaped literals) +- **Type paths**: `com.example.other.thing` (used by `@reference`) ### Named Arguments @@ -190,6 +193,85 @@ procedure convert(data!: bytes): object; - `"*/*"` - Any MIME type - Custom MIME types as needed +## Lexicon-Level Metadata: `self { }` + +ATProto Lexicon JSON has four top-level fields: `lexicon`, `id`, `description`, and `defs`. The first two are derived mechanically from the file, and `defs` is filled by your record/query/etc. definitions. The last — `description`, plus any non-spec top-level fields vendors commonly add — needs somewhere to live in MLF source. + +That somewhere is `self { }`: a first-class item that represents the lexicon itself. Doc comments on `self { }` become the top-level `description`, and extension annotations on it become top-level JSON fields. + +```mlf +/// Blog-style post record for the Bluesky feed. +@const("revision", 3) +@const("x-vendor-flag", true) +self {} + +record post { + text!: string, + createdAt!: Datetime, +} +``` + +Generates: + +```json +{ + "$type": "com.atproto.lexicon.schema", + "lexicon": 1, + "id": "...", + "description": "Blog-style post record for the Bluesky feed.", + "revision": 3, + "x-vendor-flag": true, + "defs": { "main": { "type": "record", ... } } +} +``` + +**Rules:** +- Optional. Files without `self { }` behave exactly as today. +- At most one per file, at the top level only. +- Body is always empty (`{}`) in V1; the shape is reserved for future contents. + +## Extension Annotations + +Annotations parse uniformly — any `@name(args...)` is grammatically valid. Two annotation names are recognized by the lexicon generator as carrying JSON extension fields; everything else is metadata that codegen ignores. Both attach to `self { }` for top-level fields, or to any record / query / procedure / subscription / def / token for per-item extra fields. + +### `@const(key, value)` — literal + +Emitted verbatim as JSON. Takes any annotation literal: string, integer, boolean, null, array, or nested object. This is the form used for lexicon-level metadata and is also what the JSON-to-MLF converter emits when preserving non-spec fields. + +```mlf +@const("revision", 3) +@const("x-tags", ["alpha", "beta"]) +@const("x-meta", { "team": "platform", "critical": true }) +self {} +``` + +**On fractional numbers:** ATProto's data model has no floats. If you pass a fractional value to `@const`, the lexicon generator transparently stringifies it (`@const("x-threshold", 3.14)` emits `"x-threshold": "3.14"`) and emits a warning. Integers and whole-number floats emit as JSON numbers unchanged. + +### `@reference(key, path)` — named-type reference + +Resolves the type path through the workspace and emits the resolved NSID string. Useful when you want an extension field to point at another defined type without hardcoding its NSID. + +```mlf +@reference("xFallbackType", com.example.other.thing) +record foo { + name!: string, +} +``` + +Generates (on the `foo` record): + +```json +{ + "type": "record", + "xFallbackType": "com.example.other.thing", + "record": { ... } +} +``` + +### Other annotations + +Any `@whatever(args)` the generator doesn't recognize is left untouched — it's metadata for whoever reads the AST (linters, codegen plugins, documentation tools). `@const` and `@reference` are the only two that influence JSON output. + ## Annotation Processing Annotations are preserved in the MLF AST and can be accessed by: diff --git a/website/content/docs/language-guide/11-lexicon-mapping.md b/website/content/docs/language-guide/11-lexicon-mapping.md index 643fb8c..fa695dd 100644 --- a/website/content/docs/language-guide/11-lexicon-mapping.md +++ b/website/content/docs/language-guide/11-lexicon-mapping.md @@ -489,6 +489,62 @@ images: Uri[] constrained { } ``` +## Lexicon-Level Metadata + +The `self { }` item carries lexicon-level metadata: docs become the top-level `description`, and `@const` / `@reference` annotations on it become top-level JSON fields. See [Annotations → Lexicon-Level Metadata](/docs/language-guide/annotations/#lexicon-level-metadata-self) for the full story. + +**MLF:** +```mlf +/// A short description of the whole lexicon. +@const("revision", 3) +self {} + +record post { + text!: string, +} +``` + +**Generated JSON:** +```json +{ + "$type": "com.atproto.lexicon.schema", + "lexicon": 1, + "id": "com.example.post", + "description": "A short description of the whole lexicon.", + "revision": 3, + "defs": { + "main": { "type": "record", ... } + } +} +``` + +## Per-Item Extension Fields + +`@const` and `@reference` attached to a record / query / procedure / subscription / def / token emit as extra fields on that item's JSON object. This is how the JSON-to-MLF converter preserves non-spec fields on individual defs. + +**MLF:** +```mlf +@const("x-deprecated", true) +@reference("xFallbackType", com.example.other.thing) +record post { + text!: string, +} +``` + +**Generated JSON:** +```json +{ + "defs": { + "main": { + "type": "record", + "x-deprecated": true, + "xFallbackType": "com.example.other.thing", + "record": { "type": "object", ... } + } + } +} +``` + ## Complete Example Comparison Here's a full lexicon showing MLF and its JSON output: diff --git a/website/syntaxes/mlf.sublime-syntax b/website/syntaxes/mlf.sublime-syntax index 1476061..e5851d5 100644 --- a/website/syntaxes/mlf.sublime-syntax +++ b/website/syntaxes/mlf.sublime-syntax @@ -83,7 +83,7 @@ contexts: pop: true keywords: - - match: '\b(namespace|use|as|record|inline|def|type|token|query|procedure|subscription|throws|constrained|error)\b' + - match: '\b(namespace|use|as|record|inline|def|type|token|query|procedure|subscription|throws|constrained|error|self)\b' scope: keyword.control.mlf - match: '\b(main|defs)\b' scope: keyword.other.mlf -- 2.51.2