diff --git a/mlf-cli/src/fetch.rs b/mlf-cli/src/fetch.rs index c31beee..e45f4a5 100644 --- a/mlf-cli/src/fetch.rs +++ b/mlf-cli/src/fetch.rs @@ -231,16 +231,27 @@ pub fn fetch_lexicon(nsid: &str, project_root: &std::path::Path) -> Result<(), F let cache_file = mlf_dir.join(".lexicon-cache.toml"); let mut cache = LexiconCache::load(&cache_file)?; - // Check if already cached - if cache.lexicons.contains_key(nsid) { + // Validate NSID format: must be specific (3+ segments) or use wildcard + validate_nsid_format(nsid)?; + + // Check if it's a wildcard pattern + let is_wildcard = nsid.ends_with(".*"); + let nsid_pattern = if is_wildcard { + nsid.strip_suffix(".*").unwrap() + } else { + nsid + }; + + // Check if already cached (for specific NSIDs only) + if !is_wildcard && cache.lexicons.contains_key(nsid) { println!("Lexicon '{}' is already cached. Skipping fetch.", nsid); println!(" (Use --force to re-fetch)"); return Ok(()); } - // Extract authority from NSID (e.g., "stream.place" from "stream.place.foo") - let authority = extract_authority(nsid)?; - println!("Fetching lexicons for authority: {}", authority); + // Extract authority from NSID (e.g., "place.stream" from "place.stream.key") + let authority = extract_authority(nsid_pattern)?; + println!("Fetching lexicons for pattern: {}", nsid); // Step 1: DNS TXT lookup let did = resolve_lexicon_did(&authority)?; @@ -257,17 +268,28 @@ pub fn fetch_lexicon(nsid: &str, project_root: &std::path::Path) -> Result<(), F ))); } + let mut processed_count = 0; + // Step 3: Process each record for record in records { // Extract NSID from record URI or value let record_nsid = extract_nsid_from_record(&record)?; - // Only process records that match or are under the requested NSID authority - if !record_nsid.starts_with(&authority) { + // Match against pattern + let matches = if is_wildcard { + // Wildcard: match all records starting with the pattern + record_nsid.starts_with(nsid_pattern) && record_nsid.len() > nsid_pattern.len() + } else { + // Specific: exact match only + record_nsid == nsid + }; + + if !matches { continue; } println!(" Processing: {}", record_nsid); + processed_count += 1; // Save JSON file with directory structure // e.g., "place.stream.key" -> "place/stream/key.json" @@ -316,20 +338,45 @@ pub fn fetch_lexicon(nsid: &str, project_root: &std::path::Path) -> Result<(), F // Save cache cache.save(&cache_file)?; - println!("✓ Successfully fetched lexicons for {}", nsid); + if processed_count == 0 { + return Err(FetchError::HttpError(format!( + "No lexicons matched pattern: {}", + nsid + ))); + } + + println!("✓ Successfully fetched {} lexicon(s) for {}", processed_count, nsid); + Ok(()) +} + +fn validate_nsid_format(nsid: &str) -> Result<(), FetchError> { + // Remove wildcard suffix for validation + let nsid_base = nsid.strip_suffix(".*").unwrap_or(nsid); + + let parts: Vec<&str> = nsid_base.split('.').collect(); + + // NSID must have at least 3 segments (authority + name) + // e.g., "place.stream.key" or "place.stream.*" + if parts.len() < 3 { + return Err(FetchError::InvalidNsid(format!( + "NSID must have at least 3 segments or use wildcard (e.g., 'place.stream.key' or 'place.stream.*'): {}", + nsid + ))); + } + Ok(()) } -fn extract_authority(nsid: &str) -> Result { +fn extract_authority(nsid_pattern: &str) -> Result { // NSID format: authority.name(.name)* - // For "stream.place" or "stream.place.foo", authority is "stream.place" + // For "place.stream.key", authority is "place.stream" // Typically authority is the first 2 segments (reversed domain) - let parts: Vec<&str> = nsid.split('.').collect(); + let parts: Vec<&str> = nsid_pattern.split('.').collect(); if parts.len() < 2 { return Err(FetchError::InvalidNsid(format!( "NSID must have at least 2 segments: {}", - nsid + nsid_pattern ))); } diff --git a/website/content/docs/cli/07-fetch.md b/website/content/docs/cli/07-fetch.md index e81547d..d4fe352 100644 --- a/website/content/docs/cli/07-fetch.md +++ b/website/content/docs/cli/07-fetch.md @@ -12,18 +12,24 @@ The `mlf fetch` command downloads ATProto lexicons from remote repositories and # Fetch all dependencies from mlf.toml mlf fetch -# Fetch a specific namespace -mlf fetch +# Fetch a specific lexicon +mlf fetch + +# Fetch all lexicons matching a wildcard +mlf fetch # Fetch and save to dependencies -mlf fetch --save +mlf fetch --save ``` **Arguments:** -- `[NAMESPACE]` - Optional namespace to fetch (e.g., `stream.place`, `app.bsky`) +- `[NSID]` - Optional NSID or pattern to fetch: + - Specific lexicon: `com.example.forum.post` + - Wildcard pattern: `com.example.forum.*` + - Real-world example: `app.bsky.feed.*` **Options:** -- `--save` - Add the namespace to dependencies in `mlf.toml` +- `--save` - Add the NSID/pattern to dependencies in `mlf.toml` ## How It Works @@ -42,7 +48,7 @@ With an `mlf.toml` file: ```toml [dependencies] -dependencies = ["stream.place", "app.bsky"] +dependencies = ["com.example.forum.*", "com.example.social.*"] ``` Run: @@ -55,39 +61,50 @@ mlf fetch ``` Fetching 2 dependencies... -Fetching: stream.place -Fetching lexicons for authority: stream.place - → Resolved DID: did:web:stream.place - → Using PDS: https://stream.place - → Found 5 lexicon record(s) - Processing: stream.place.thread - → Saved JSON to .mlf/lexicons/json/stream.place.thread.json - → Converted to MLF at .mlf/lexicons/mlf/stream.place.thread.mlf -✓ Successfully fetched lexicons for stream.place - -Fetching: app.bsky +Fetching: com.example.forum.* +Fetching lexicons for pattern: com.example.forum.* + → Resolved DID: did:web:example.com + → Using PDS: https://example.com + → Found 3 lexicon record(s) + Processing: com.example.forum.post + → Saved JSON to .mlf/lexicons/json/com/example/forum/post.json + → Converted to MLF at .mlf/lexicons/mlf/com/example/forum/post.mlf + Processing: com.example.forum.thread + → Saved JSON to .mlf/lexicons/json/com/example/forum/thread.json + → Converted to MLF at .mlf/lexicons/mlf/com/example/forum/thread.mlf +✓ Successfully fetched 2 lexicon(s) for com.example.forum.* + +Fetching: com.example.social.* ... ✓ Successfully fetched all 2 dependencies ``` -### Fetch Specific Namespace +### Fetch Specific Lexicon + +```bash +mlf fetch com.example.forum.post +``` + +This downloads only the `com.example.forum.post` lexicon. + +### Fetch with Wildcard ```bash -mlf fetch stream.place +mlf fetch com.example.forum.* ``` -This downloads all lexicons under the `stream.place` authority. +This downloads all lexicons under the `com.example.forum` namespace. ### Fetch and Save ```bash -mlf fetch stream.place --save +mlf fetch com.example.forum.* --save ``` This: -1. Downloads the lexicons -2. Adds `"stream.place"` to the dependencies array in `mlf.toml` +1. Downloads all `com.example.forum.*` lexicons +2. Adds `"com.example.forum.*"` to the dependencies array in `mlf.toml` 3. Creates `mlf.toml` if it doesn't exist ## Storage Structure @@ -100,11 +117,29 @@ Fetched lexicons are stored in `.mlf/lexicons/`: ├── .lexicon-cache.toml # Cache metadata └── lexicons/ ├── json/ # Original JSON lexicons - │ ├── stream.place.thread.json - │ └── app.bsky.actor.profile.json + │ ├── com/ + │ │ └── example/ + │ │ ├── forum/ + │ │ │ ├── post.json + │ │ │ └── thread.json + │ │ └── social/ + │ │ └── profile.json + │ └── app/ + │ └── bsky/ + │ └── feed/ + │ └── post.json └── mlf/ # Converted MLF format - ├── stream.place.thread.mlf - └── app.bsky.actor.profile.mlf + ├── com/ + │ └── example/ + │ ├── forum/ + │ │ ├── post.mlf + │ │ └── thread.mlf + │ └── social/ + │ └── profile.mlf + └── app/ + └── bsky/ + └── feed/ + └── post.mlf ``` ### Cache File @@ -112,31 +147,32 @@ Fetched lexicons are stored in `.mlf/lexicons/`: The `.lexicon-cache.toml` tracks what's been fetched: ```toml -[[lexicons.stream.place.thread]] -nsid = "stream.place.thread" +[[lexicons."com.example.forum.post"]] +nsid = "com.example.forum.post" fetched_at = "2024-01-15T10:30:00Z" -did = "did:web:stream.place" +did = "did:web:example.com" +hash = "abc123..." ``` ## DNS Resolution -For a namespace like `stream.place.thread`: +For an NSID like `com.example.forum.post`: -1. Extract authority: `stream.place` -2. Reverse for DNS: `place.stream` -3. Query TXT record: `_lexicon.place.stream` +1. Extract authority: `com.example` (first 2 segments) +2. Reverse for DNS: `example.com` +3. Query TXT record: `_lexicon.example.com` 4. Parse `did=did:web:...` or `did=did:plc:...` **Example DNS record:** ``` -_lexicon.place.stream. 300 IN TXT "did=did:web:stream.place" +_lexicon.example.com. 300 IN TXT "did=did:web:example.com" ``` ## DID Resolution ### did:web -For `did:web:stream.place`, the PDS is `https://stream.place` +For `did:web:example.com`, the PDS is `https://example.com` ### did:plc @@ -149,12 +185,12 @@ Once fetched, lexicons in `.mlf/lexicons/mlf/` are automatically available for: ### Type References ```mlf -use stream.place.thread; +use com.example.forum.post; -def Reply = { - thread!: stream.place.thread, +record reply { + post!: com.example.forum.post, text!: string, -}; +} ``` ### Code Generation @@ -178,7 +214,7 @@ The check command loads fetched lexicons for type resolution. If you don't have an `mlf.toml`, the fetch command will offer to create one: ```bash -$ mlf fetch stream.place +$ mlf fetch com.example.forum.* No mlf.toml found in current or parent directories. Would you like to create one in the current directory? (y/n) y @@ -188,18 +224,18 @@ Created mlf.toml in /path/to/current/dir ## Re-fetching -If a namespace is already cached, fetch skips it: +If a lexicon is already cached, fetch skips it: ```bash -$ mlf fetch stream.place -Lexicon 'stream.place.thread' is already cached. Skipping fetch. +$ mlf fetch com.example.forum.post +Lexicon 'com.example.forum.post' is already cached. Skipping fetch. (Use --force to re-fetch) ``` To re-fetch: ```bash -mlf fetch stream.place --force # Not yet implemented +mlf fetch com.example.forum.post --force # Not yet implemented ``` ## Error Handling @@ -207,7 +243,7 @@ mlf fetch stream.place --force # Not yet implemented ### DNS Errors ``` -✗ DNS lookup failed: No TXT record found for _lexicon.place.stream +✗ DNS lookup failed: No TXT record found for _lexicon.example.com ``` **Causes:** @@ -229,14 +265,24 @@ mlf fetch stream.place --force # Not yet implemented ### No Records Found ``` -✗ No lexicon records found for stream.place +✗ No lexicons matched pattern: com.example.forum.* ``` **Causes:** -- Namespace exists but has no published lexicons -- Wrong namespace (typo) +- No lexicons exist matching the pattern +- Wrong NSID (typo) - PDS doesn't support lexicon publishing +### Invalid NSID Format + +``` +✗ NSID must have at least 3 segments or use wildcard (e.g., 'com.example.forum.post' or 'com.example.forum.*'): com.example +``` + +**Solution:** +- Use a specific NSID: `com.example.forum.post` +- Or use a wildcard: `com.example.forum.*` + ## Best Practices 1. **Fetch before work** - Always fetch dependencies before coding @@ -245,22 +291,6 @@ mlf fetch stream.place --force # Not yet implemented 4. **Check DNS** - Verify TXT records before fetching 5. **Version dependencies** - Consider tracking lexicon versions (future feature) -## CI/CD Integration - -In your CI pipeline: - -```yaml -# GitHub Actions example -- name: Fetch ATProto Lexicons - run: | - mlf fetch - -- name: Generate Code - run: | - mlf generate -``` - -This ensures builds have access to the latest lexicons. ## Comparison with npm/cargo @@ -279,18 +309,19 @@ The fetch command is similar to package managers: ```bash # Check DNS resolution -dig TXT _lexicon.place.stream +dig TXT _lexicon.example.com # Test DID resolution curl https://plc.directory/did:plc:abc123 ``` -### Invalid Namespace +### Invalid NSID Format -Make sure you're using the correct namespace format: -- ✓ `stream.place` -- ✓ `app.bsky` -- ✗ `stream.place.thread` (too specific) +Make sure you're using the correct format: +- ✓ `com.example.forum.post` (specific lexicon) +- ✓ `com.example.forum.*` (wildcard) +- ✓ `app.bsky.feed.*` (real-world wildcard) +- ✗ `com.example` (must be specific or use wildcard) ### Permission Errors