diff --git a/Cargo.lock b/Cargo.lock index ebcc8264..fd988848 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2288,7 +2288,7 @@ dependencies = [ [[package]] name = "jacquard-api" version = "0.8.0" -source = "git+https://tangled.org/@nonbinary.computer/jacquard#988f0eedfc499d0e2cdd667f6adab086465984e2" +source = "git+https://tangled.org/@nonbinary.computer/jacquard#4eb26a04129dd5a71c7a2282d43bef50f713e335" dependencies = [ "bon", "bytes", @@ -2378,7 +2378,7 @@ dependencies = [ [[package]] name = "jacquard-common" version = "0.8.0" -source = "git+https://tangled.org/@nonbinary.computer/jacquard#988f0eedfc499d0e2cdd667f6adab086465984e2" +source = "git+https://tangled.org/@nonbinary.computer/jacquard#4eb26a04129dd5a71c7a2282d43bef50f713e335" dependencies = [ "base64 0.22.1", "bon", @@ -2431,7 +2431,7 @@ dependencies = [ [[package]] name = "jacquard-derive" version = "0.8.0" -source = "git+https://tangled.org/@nonbinary.computer/jacquard#988f0eedfc499d0e2cdd667f6adab086465984e2" +source = "git+https://tangled.org/@nonbinary.computer/jacquard#4eb26a04129dd5a71c7a2282d43bef50f713e335" dependencies = [ "heck 0.5.0", "jacquard-lexicon 0.8.0 (git+https://tangled.org/@nonbinary.computer/jacquard)", @@ -2468,7 +2468,7 @@ dependencies = [ [[package]] name = "jacquard-identity" version = "0.8.0" -source = "git+https://tangled.org/@nonbinary.computer/jacquard#988f0eedfc499d0e2cdd667f6adab086465984e2" +source = "git+https://tangled.org/@nonbinary.computer/jacquard#4eb26a04129dd5a71c7a2282d43bef50f713e335" dependencies = [ "bon", "bytes", @@ -2542,7 +2542,7 @@ dependencies = [ [[package]] name = "jacquard-lexicon" version = "0.8.0" -source = "git+https://tangled.org/@nonbinary.computer/jacquard#988f0eedfc499d0e2cdd667f6adab086465984e2" +source = "git+https://tangled.org/@nonbinary.computer/jacquard#4eb26a04129dd5a71c7a2282d43bef50f713e335" dependencies = [ "glob", "heck 0.5.0", diff --git a/crates/jacquard-lexgen/examples/extract_inventory.rs b/crates/jacquard-lexgen/examples/extract_inventory.rs new file mode 100644 index 00000000..d24f0058 --- /dev/null +++ b/crates/jacquard-lexgen/examples/extract_inventory.rs @@ -0,0 +1,79 @@ +//! Extract AT Protocol lexicon schemas from compiled Rust types via inventory +//! +//! This example discovers types with `#[derive(LexiconSchema)]` via inventory +//! and generates lexicon JSON files. This approach requires types to be linked +//! into the binary at compile time. +//! +//! For workspace-wide schema extraction without linking requirements, +//! use the `extract-schemas` binary or `WorkspaceDiscovery` API instead. + +use clap::Parser; +use jacquard_lexgen::schema_extraction::{self, ExtractOptions, SchemaExtractor}; +use miette::Result; + +/// Extract lexicon schemas from compiled Rust types via inventory +#[derive(Parser, Debug)] +#[command(name = "extract-inventory")] +#[command(about = "Extract AT Protocol lexicon schemas from linked Rust types")] +#[command(long_about = r#" +Discovers types implementing LexiconSchema via inventory and generates +lexicon JSON files. The binary only discovers types that are linked, +so you need to import your schema types in this binary or a custom one. + +For workspace-wide extraction, use the extract-schemas binary instead. + +See: https://docs.rs/jacquard-lexgen/latest/jacquard_lexgen/schema_extraction/ +"#)] +struct Args { + /// Output directory for generated schema files + #[arg(short, long, default_value = "lexicons")] + output: String, + + /// Verbose output + #[arg(short, long)] + verbose: bool, + + /// Filter by NSID prefix (e.g., "app.bsky") + #[arg(short, long)] + filter: Option, + + /// Validate schemas before writing + #[arg(short = 'V', long, default_value = "true")] + validate: bool, + + /// Pretty-print JSON output + #[arg(short, long, default_value = "true")] + pretty: bool, + + /// Watch mode - regenerate on changes + #[arg(short, long)] + watch: bool, +} + +fn main() -> Result<()> { + let args = Args::parse(); + + // Simple case: use convenience function + if !args.watch && args.filter.is_none() && args.validate && args.pretty { + return schema_extraction::run(&args.output, args.verbose); + } + + // Advanced case: use full options + let options = ExtractOptions { + output_dir: args.output.into(), + verbose: args.verbose, + filter: args.filter, + validate: args.validate, + pretty: args.pretty, + }; + + let extractor = SchemaExtractor::new(options); + + if args.watch { + extractor.watch()?; + } else { + extractor.extract_all()?; + } + + Ok(()) +} diff --git a/crates/jacquard-lexgen/examples/workspace_discovery.rs b/crates/jacquard-lexgen/examples/workspace_discovery.rs deleted file mode 100644 index 92923f01..00000000 --- a/crates/jacquard-lexgen/examples/workspace_discovery.rs +++ /dev/null @@ -1,50 +0,0 @@ -#!/usr/bin/env cargo -//! Example: Discover schemas across the workspace without link-time discovery -//! -//! Run with: cargo run --example workspace_discovery - -use jacquard_lexgen::schema_discovery::WorkspaceDiscovery; - -fn main() -> miette::Result<()> { - println!("Workspace Schema Discovery Example\n"); - - // Create workspace discovery - let discovery = WorkspaceDiscovery::new().verbose(true); - - // Scan workspace - let schemas = discovery.scan()?; - - println!("\n━━━ Results ━━━"); - println!("Discovered {} schema types:\n", schemas.len()); - - // Group by crate - use std::collections::HashMap; - let mut by_crate: HashMap> = HashMap::new(); - - for schema in &schemas { - let crate_name = schema - .source_path - .components() - .find_map(|c| { - let s = c.as_os_str().to_str()?; - if s.starts_with("jacquard-") || s == "jacquard" { - Some(s.to_string()) - } else { - None - } - }) - .unwrap_or_else(|| "unknown".to_string()); - - by_crate.entry(crate_name).or_default().push(schema); - } - - for (crate_name, crate_schemas) in by_crate { - println!("📦 {} ({} schemas)", crate_name, crate_schemas.len()); - for schema in crate_schemas { - println!(" • {} ({})", schema.nsid, schema.type_name); - } - println!(); - } - - Ok(()) -} diff --git a/crates/jacquard-lexgen/src/bin/extract_schemas.rs b/crates/jacquard-lexgen/src/bin/extract_schemas.rs index 430554a5..858675cf 100644 --- a/crates/jacquard-lexgen/src/bin/extract_schemas.rs +++ b/crates/jacquard-lexgen/src/bin/extract_schemas.rs @@ -1,23 +1,25 @@ -//! Extract AT Protocol lexicon schemas from compiled Rust types +//! Extract AT Protocol lexicon schemas via workspace discovery //! -//! This binary discovers types with `#[derive(LexiconSchema)]` via inventory -//! and generates lexicon JSON files. See the `schema_extraction` module docs -//! for usage patterns and integration examples. +//! This binary scans the workspace for types with `#[derive(LexiconSchema)]` +//! and generates lexicon JSON files. Unlike inventory-based extraction, this +//! discovers schemas across the entire workspace without requiring linking. use clap::Parser; -use jacquard_lexgen::schema_extraction::{self, ExtractOptions, SchemaExtractor}; +use jacquard_lexgen::schema_discovery::WorkspaceDiscovery; use miette::Result; -/// Extract lexicon schemas from compiled Rust types +/// Extract lexicon schemas from workspace source files #[derive(Parser, Debug)] #[command(name = "extract-schemas")] -#[command(about = "Extract AT Protocol lexicon schemas from Rust types")] +#[command(about = "Extract AT Protocol lexicon schemas from workspace")] #[command(long_about = r#" -Discovers types implementing LexiconSchema via inventory and generates -lexicon JSON files. The binary only discovers types that are linked, -so you need to import your schema types in this binary or a custom one. +Scans workspace source files for types with #[derive(LexiconSchema)] and +generates lexicon JSON files. This discovers all schemas in the workspace +without requiring types to be linked into the binary. -See: https://docs.rs/jacquard-lexgen/latest/jacquard_lexgen/schema_extraction/ +For inventory-based extraction (link-time discovery), see the extract_inventory example. + +See: https://docs.rs/jacquard-lexgen/latest/jacquard_lexgen/schema_discovery/ "#)] struct Args { /// Output directory for generated schema files @@ -27,48 +29,15 @@ struct Args { /// Verbose output #[arg(short, long)] verbose: bool, - - /// Filter by NSID prefix (e.g., "app.bsky") - #[arg(short, long)] - filter: Option, - - /// Validate schemas before writing - #[arg(short = 'V', long, default_value = "true")] - validate: bool, - - /// Pretty-print JSON output - #[arg(short, long, default_value = "true")] - pretty: bool, - - /// Watch mode - regenerate on changes - #[arg(short, long)] - watch: bool, } fn main() -> Result<()> { let args = Args::parse(); - // Simple case: use convenience function - if !args.watch && args.filter.is_none() && args.validate && args.pretty { - return schema_extraction::run(&args.output, args.verbose); - } - - // Advanced case: use full options - let options = ExtractOptions { - output_dir: args.output.into(), - verbose: args.verbose, - filter: args.filter, - validate: args.validate, - pretty: args.pretty, - }; - - let extractor = SchemaExtractor::new(options); + let discovery = WorkspaceDiscovery::new() + .verbose(args.verbose); - if args.watch { - extractor.watch()?; - } else { - extractor.extract_all()?; - } + discovery.generate_and_write(args.output)?; Ok(()) } diff --git a/crates/jacquard-lexgen/src/lib.rs b/crates/jacquard-lexgen/src/lib.rs index c45d2d59..6087a39e 100644 --- a/crates/jacquard-lexgen/src/lib.rs +++ b/crates/jacquard-lexgen/src/lib.rs @@ -36,7 +36,5 @@ pub mod cli; pub mod fetch; pub mod schema_discovery; pub mod schema_extraction; -#[cfg(any(test, debug_assertions))] -pub mod test_schemas; pub use fetch::{Config, Fetcher}; diff --git a/crates/jacquard-lexgen/src/schema_discovery.rs b/crates/jacquard-lexgen/src/schema_discovery.rs index 9e16b901..7c09a792 100644 --- a/crates/jacquard-lexgen/src/schema_discovery.rs +++ b/crates/jacquard-lexgen/src/schema_discovery.rs @@ -155,11 +155,10 @@ impl WorkspaceDiscovery { // Use schema builder based on kind let built = match schema_info.kind { - SchemaKind::Struct => { - jacquard_lexicon::schema::from_ast::build_struct_schema(&ast)? - } + SchemaKind::Struct => jacquard_lexicon::schema::from_ast::build_struct_schema(&ast) + .into_diagnostic()?, SchemaKind::Enum => { - jacquard_lexicon::schema::from_ast::build_enum_schema(&ast)? + jacquard_lexicon::schema::from_ast::build_enum_schema(&ast).into_diagnostic()? } }; @@ -210,8 +209,11 @@ impl WorkspaceDiscovery { } /// Group schemas by base NSID (strip fragment suffix) - fn group_by_base_nsid(&self, schemas: &[GeneratedSchema]) -> BTreeMap> { - let mut groups: BTreeMap> = BTreeMap::new(); + fn group_by_base_nsid<'a>( + &self, + schemas: &'a [GeneratedSchema], + ) -> BTreeMap> { + let mut groups: BTreeMap> = BTreeMap::new(); for schema in schemas { // Split on # to get base NSID @@ -297,7 +299,7 @@ impl WorkspaceDiscovery { /// Serialize a lexicon doc with "main" def first fn serialize_with_main_first(&self, doc: &LexiconDoc) -> Result { - use serde_json::{json, Map, Value}; + use serde_json::{Map, Value, json}; // Build defs map with main first let mut defs_map = Map::new(); @@ -533,7 +535,8 @@ impl WorkspaceDiscovery { lex_attrs.key = Some(lit.value()); } Ok(()) - }).into_diagnostic()?; + }) + .into_diagnostic()?; } } diff --git a/crates/jacquard-lexgen/src/test_schemas.rs b/crates/jacquard-lexgen/src/test_schemas.rs deleted file mode 100644 index c4f96f8b..00000000 --- a/crates/jacquard-lexgen/src/test_schemas.rs +++ /dev/null @@ -1,31 +0,0 @@ -// Test schemas for verifying extraction works -// These are only compiled in tests/dev builds - -use jacquard_common::CowStr; -use jacquard_derive::LexiconSchema; - -#[derive(LexiconSchema)] -#[lexicon(nsid = "com.example.testRecord", record, key = "tid")] -pub struct TestRecord<'a> { - #[lexicon(max_length = 100)] - pub text: CowStr<'a>, - pub count: i64, -} - -#[derive(LexiconSchema)] -#[lexicon(nsid = "com.example.testRecord#fragment")] -pub struct TestFragment { - pub field: i64, -} - -#[derive(LexiconSchema)] -#[lexicon(nsid = "com.example.testDefs.defs#defOne")] -pub struct DefOne { - pub value: String, -} - -#[derive(LexiconSchema)] -#[lexicon(nsid = "com.example.testDefs.defs#defTwo")] -pub struct DefTwo { - pub number: i64, -} diff --git a/crates/jacquard-lexgen/tests/schema_extraction.rs b/crates/jacquard-lexgen/tests/schema_extraction.rs deleted file mode 100644 index 0df8eb2b..00000000 --- a/crates/jacquard-lexgen/tests/schema_extraction.rs +++ /dev/null @@ -1,82 +0,0 @@ -use jacquard_lexgen::schema_extraction::{ExtractOptions, SchemaExtractor}; -use tempfile::TempDir; - -#[test] -fn test_extract_all_creates_output_dir() { - let temp_dir = TempDir::new().unwrap(); - - let options = ExtractOptions { - output_dir: temp_dir.path().to_path_buf(), - verbose: false, - filter: None, - validate: true, - pretty: true, - }; - - let extractor = SchemaExtractor::new(options); - - // This will discover any schemas registered via inventory in the binary - // In a minimal test environment, this might be 0 - let result = extractor.extract_all(); - - // Should succeed even if no schemas found - assert!(result.is_ok()); - - // Directory should exist - assert!(temp_dir.path().exists()); -} - -#[test] -fn test_extract_with_filter() { - let temp_dir = TempDir::new().unwrap(); - - let options = ExtractOptions { - output_dir: temp_dir.path().to_path_buf(), - verbose: false, - filter: Some("com.example.nonexistent".into()), - validate: true, - pretty: true, - }; - - let extractor = SchemaExtractor::new(options); - let result = extractor.extract_all(); - - // Should succeed (just won't write any files) - assert!(result.is_ok()); -} - -#[test] -fn test_extract_with_verbose() { - let temp_dir = TempDir::new().unwrap(); - - let options = ExtractOptions { - output_dir: temp_dir.path().to_path_buf(), - verbose: true, - filter: None, - validate: true, - pretty: true, - }; - - let extractor = SchemaExtractor::new(options); - let result = extractor.extract_all(); - - assert!(result.is_ok()); -} - -#[test] -fn test_extract_compact_json() { - let temp_dir = TempDir::new().unwrap(); - - let options = ExtractOptions { - output_dir: temp_dir.path().to_path_buf(), - verbose: false, - filter: None, - validate: true, - pretty: false, // Compact JSON - }; - - let extractor = SchemaExtractor::new(options); - let result = extractor.extract_all(); - - assert!(result.is_ok()); -}