diff --git a/mlf-cli/src/config.rs b/mlf-cli/src/config.rs index 20e099e..8dd7047 100644 --- a/mlf-cli/src/config.rs +++ b/mlf-cli/src/config.rs @@ -1,4 +1,5 @@ use serde::{Deserialize, Serialize}; +use std::collections::HashMap; use std::path::{Path, PathBuf}; use thiserror::Error; @@ -54,12 +55,28 @@ pub struct OutputConfig { pub struct DependenciesConfig { #[serde(default)] pub dependencies: Vec, + + #[serde(default = "default_allow_transitive_deps")] + pub allow_transitive_deps: bool, + + #[serde(default = "default_optimize_transitive_fetches")] + pub optimize_transitive_fetches: bool, +} + +fn default_allow_transitive_deps() -> bool { + true +} + +fn default_optimize_transitive_fetches() -> bool { + false } impl Default for DependenciesConfig { fn default() -> Self { Self { dependencies: vec![], + allow_transitive_deps: default_allow_transitive_deps(), + optimize_transitive_fetches: default_optimize_transitive_fetches(), } } } @@ -136,13 +153,69 @@ pub fn init_mlf_cache(project_root: &Path) -> std::io::Result<()> { std::fs::write(&gitignore_path, "*\n!.gitignore\n")?; } - // Create or touch .lexicon-cache.toml - let cache_file = mlf_dir.join(".lexicon-cache.toml"); - if !cache_file.exists() { - std::fs::write(&cache_file, "# Lexicon cache metadata\n")?; + Ok(()) +} + +/// Lock file format for tracking resolved lexicons +#[derive(Debug, Serialize, Deserialize, Default)] +pub struct LockFile { + /// Lock file format version + pub version: u32, + + /// All resolved lexicons (both direct and transitive dependencies) + #[serde(default)] + pub lexicons: HashMap, +} + +/// A single locked lexicon entry +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct LockedLexicon { + /// The NSID of this lexicon + pub nsid: String, + + /// The DID of the repository this was fetched from + pub did: String, + + /// SHA-256 checksum of the JSON content + pub checksum: String, + + /// List of NSIDs this lexicon depends on (external references) + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub dependencies: Vec, +} + +impl LockFile { + pub fn new() -> Self { + Self { + version: 1, + lexicons: HashMap::new(), + } + } + + pub fn load(path: &Path) -> Result { + if !path.exists() { + return Ok(Self::new()); + } + + let content = std::fs::read_to_string(path)?; + toml::from_str(&content).map_err(|e| ConfigError::ParseError(e)) } - Ok(()) + pub fn save(&self, path: &Path) -> Result<(), ConfigError> { + let content = toml::to_string_pretty(self) + .map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e))?; + std::fs::write(path, content)?; + Ok(()) + } + + pub fn add_lexicon(&mut self, nsid: String, did: String, checksum: String, dependencies: Vec) { + self.lexicons.insert(nsid.clone(), LockedLexicon { + nsid, + did, + checksum, + dependencies, + }); + } } #[cfg(test)] @@ -156,4 +229,24 @@ mod tests { assert!(config.output.is_empty()); assert!(config.dependencies.dependencies.is_empty()); } + + #[test] + fn test_lockfile_basic() { + let mut lockfile = LockFile::new(); + assert_eq!(lockfile.version, 1); + assert!(lockfile.lexicons.is_empty()); + + lockfile.add_lexicon( + "app.bsky.actor.profile".to_string(), + "did:plc:test".to_string(), + "sha256:abc123".to_string(), + vec![], + ); + + assert_eq!(lockfile.lexicons.len(), 1); + let locked = lockfile.lexicons.get("app.bsky.actor.profile").unwrap(); + assert_eq!(locked.nsid, "app.bsky.actor.profile"); + assert_eq!(locked.did, "did:plc:test"); + assert_eq!(locked.checksum, "sha256:abc123"); + } } diff --git a/mlf-cli/src/fetch.rs b/mlf-cli/src/fetch.rs index 479b5f9..2ee0a36 100644 --- a/mlf-cli/src/fetch.rs +++ b/mlf-cli/src/fetch.rs @@ -1,12 +1,10 @@ -use crate::config::{find_project_root, get_mlf_cache_dir, init_mlf_cache, ConfigError, MlfConfig}; -use chrono::{DateTime, Utc}; +use crate::config::{find_project_root, get_mlf_cache_dir, init_mlf_cache, ConfigError, MlfConfig, LockFile}; use hickory_resolver::config::*; use hickory_resolver::Resolver; use miette::Diagnostic; -use serde::{Deserialize, Serialize}; +use serde::Deserialize; use sha2::{Digest, Sha256}; -use std::collections::HashMap; -use std::path::Path; +use std::collections::HashSet; use thiserror::Error; #[derive(Error, Debug, Diagnostic)] @@ -43,61 +41,11 @@ pub enum FetchError { #[diagnostic(code(mlf::fetch::io_error))] IoError(#[from] std::io::Error), - #[error("Failed to load cache: {0}")] - #[diagnostic(code(mlf::fetch::cache_error))] - CacheError(String), - #[error("Invalid NSID format: {0}")] #[diagnostic(code(mlf::fetch::invalid_nsid))] InvalidNsid(String), } -#[derive(Debug, Serialize, Deserialize)] -pub struct LexiconCache { - #[serde(default)] - pub lexicons: HashMap, -} - -#[derive(Debug, Serialize, Deserialize, Clone)] -pub struct CacheEntry { - pub nsid: String, - pub fetched_at: DateTime, - pub did: String, - #[serde(default)] - pub hash: String, -} - -impl LexiconCache { - pub fn load(path: &Path) -> Result { - if !path.exists() { - return Ok(Self { - lexicons: HashMap::new(), - }); - } - - let content = std::fs::read_to_string(path)?; - toml::from_str(&content).map_err(|e| FetchError::CacheError(e.to_string())) - } - - pub fn save(&self, path: &Path) -> Result<(), FetchError> { - let content = - toml::to_string_pretty(self).map_err(|e| FetchError::CacheError(e.to_string()))?; - std::fs::write(path, content)?; - Ok(()) - } - - pub fn add_entry(&mut self, nsid: String, did: String, hash: String) { - self.lexicons.insert( - nsid.clone(), - CacheEntry { - nsid, - fetched_at: Utc::now(), - did, - hash, - }, - ); - } -} #[derive(Debug, Deserialize)] struct AtProtoRecord { @@ -106,7 +54,14 @@ struct AtProtoRecord { } /// Main entry point for fetch command -pub fn run_fetch(nsid: Option, save: bool) -> Result<(), FetchError> { +pub fn run_fetch(nsid: Option, save: bool, update: bool, locked: bool) -> Result<(), FetchError> { + // Validate flags + if update && locked { + return Err(FetchError::HttpError( + "Cannot use --update and --locked together".to_string() + )); + } + // Find project root let current_dir = std::env::current_dir()?; let project_root = ensure_project_root(¤t_dir)?; @@ -125,7 +80,7 @@ pub fn run_fetch(nsid: Option, save: bool) -> Result<(), FetchError> { } None => { // Fetch all dependencies from mlf.toml - fetch_all_dependencies(&project_root) + fetch_all_dependencies(&project_root, update, locked) } } } @@ -156,7 +111,7 @@ fn ensure_project_root(current_dir: &std::path::Path) -> Result Result<(), FetchError> { +fn fetch_all_dependencies(project_root: &std::path::Path, update: bool, locked: bool) -> Result<(), FetchError> { // Load mlf.toml let config_path = project_root.join("mlf.toml"); let config = MlfConfig::load(&config_path).map_err(FetchError::NoProjectRoot)?; @@ -166,16 +121,60 @@ fn fetch_all_dependencies(project_root: &std::path::Path) -> Result<(), FetchErr return Ok(()); } - println!("Fetching {} dependencies...", config.dependencies.dependencies.len()); + let allow_transitive = config.dependencies.allow_transitive_deps; + + // Load or create lockfile + let lockfile_path = project_root.join("mlf-lock.toml"); + let existing_lockfile = LockFile::load(&lockfile_path).map_err(FetchError::NoProjectRoot)?; + let has_existing_lockfile = lockfile_path.exists() && !existing_lockfile.lexicons.is_empty(); + + // Handle --locked mode + if locked { + if !has_existing_lockfile { + return Err(FetchError::HttpError( + "No lockfile found. Run `mlf fetch` first to create mlf-lock.toml".to_string() + )); + } + + // In locked mode, we use the lockfile and verify nothing needs updating + // For now, we'll just use the lockfile - verification can be enhanced later + println!("Using locked dependencies from mlf-lock.toml"); + return fetch_from_lockfile(project_root, &existing_lockfile); + } + + // Determine fetch mode + let mode = if update { + "update (ignoring lockfile)" + } else if has_existing_lockfile { + "lockfile" + } else { + "fresh" + }; + + println!("Fetching {} dependencies... (mode: {}, transitive deps: {})", + config.dependencies.dependencies.len(), + mode, + if allow_transitive { "enabled" } else { "disabled" }); + + // In update mode or if no lockfile, do full fetch + // In normal mode with lockfile, use lockfile for cached entries + let mut lockfile = if update || !has_existing_lockfile { + LockFile::new() + } else { + existing_lockfile + }; let mut errors = Vec::new(); let mut success_count = 0; + let mut fetched_nsids = HashSet::new(); + // Fetch initial dependencies for dep in &config.dependencies.dependencies { println!("\nFetching: {}", dep); - match fetch_lexicon(dep, project_root) { + match fetch_lexicon_with_lock(dep, project_root, &mut lockfile) { Ok(()) => { success_count += 1; + fetched_nsids.insert(dep.clone()); } Err(e) => { errors.push((dep.clone(), format!("{}", e))); @@ -183,6 +182,129 @@ fn fetch_all_dependencies(project_root: &std::path::Path) -> Result<(), FetchErr } } + // If transitive dependencies are enabled, iteratively fetch missing deps + if allow_transitive { + let mut iteration = 0; + let max_iterations = 10; // Prevent infinite loops + + loop { + iteration += 1; + if iteration > max_iterations { + eprintln!("\nWarning: Reached maximum iteration limit for transitive dependencies"); + break; + } + + // Collect unresolved references + let unresolved = match collect_unresolved_references(project_root) { + Ok(refs) => refs, + Err(e) => { + eprintln!("\nWarning: Failed to analyze dependencies: {}", e); + break; + } + }; + + // Filter out NSIDs we've already fetched or tried to fetch + let new_deps: HashSet = unresolved + .into_iter() + .filter(|nsid| !fetched_nsids.contains(nsid)) + .collect(); + + if new_deps.is_empty() { + break; + } + + // Determine whether to optimize transitive fetches + let should_optimize = config.dependencies.optimize_transitive_fetches; + + if should_optimize { + // Optimize the fetch patterns to reduce number of fetches + let optimized_patterns = optimize_fetch_patterns(&new_deps); + + println!("\n→ Found {} unresolved reference(s), fetching {} optimized pattern(s)...", + new_deps.len(), optimized_patterns.len()); + + // Track which patterns are wildcards and their constituent NSIDs + let mut wildcard_failures: Vec<(String, Vec)> = Vec::new(); + + for pattern in optimized_patterns { + let is_wildcard = pattern.ends_with(".*"); + println!("\nFetching transitive dependency: {}", pattern); + fetched_nsids.insert(pattern.clone()); + + match fetch_lexicon_with_lock(&pattern, project_root, &mut lockfile) { + Ok(()) => { + success_count += 1; + } + Err(e) => { + eprintln!(" Warning: Failed to fetch {}: {}", pattern, e); + + // If this was a wildcard that failed, collect the individual NSIDs for retry + if is_wildcard { + let pattern_prefix = pattern.strip_suffix(".*").unwrap(); + let matching_nsids: Vec = new_deps.iter() + .filter(|nsid| nsid.starts_with(pattern_prefix)) + .cloned() + .collect(); + + if !matching_nsids.is_empty() { + wildcard_failures.push((pattern.clone(), matching_nsids)); + } + } + } + } + } + + // Retry failed wildcards with individual NSIDs + if !wildcard_failures.is_empty() { + println!("\n→ Retrying failed wildcard patterns with individual NSIDs..."); + + for (failed_pattern, nsids) in wildcard_failures { + println!(" Retrying {} NSIDs from failed pattern: {}", nsids.len(), failed_pattern); + + for nsid in nsids { + if !fetched_nsids.contains(&nsid) { + println!(" Fetching: {}", nsid); + fetched_nsids.insert(nsid.clone()); + + match fetch_lexicon_with_lock(&nsid, project_root, &mut lockfile) { + Ok(()) => { + success_count += 1; + } + Err(e) => { + eprintln!(" Warning: Failed to fetch {}: {}", nsid, e); + } + } + } + } + } + } + } else { + // Fetch individually without optimization (safer, more predictable) + println!("\n→ Found {} unresolved reference(s), fetching individually...", + new_deps.len()); + + for nsid in &new_deps { + println!("\nFetching transitive dependency: {}", nsid); + fetched_nsids.insert(nsid.clone()); + + match fetch_lexicon_with_lock(nsid, project_root, &mut lockfile) { + Ok(()) => { + success_count += 1; + } + Err(e) => { + // Don't fail the entire fetch for transitive deps + eprintln!(" Warning: Failed to fetch {}: {}", nsid, e); + } + } + } + } + } + } + + // Save the lockfile + lockfile.save(&lockfile_path).map_err(FetchError::NoProjectRoot)?; + println!("\n→ Updated mlf-lock.toml"); + if !errors.is_empty() { eprintln!( "\n{} dependency(ies) fetched successfully, {} error(s):", @@ -202,6 +324,122 @@ fn fetch_all_dependencies(project_root: &std::path::Path) -> Result<(), FetchErr Ok(()) } +/// Fetch dependencies using the lockfile +/// This refetches each lexicon from its recorded DID and verifies the checksum +fn fetch_from_lockfile(project_root: &std::path::Path, lockfile: &LockFile) -> Result<(), FetchError> { + if lockfile.lexicons.is_empty() { + println!("Lockfile is empty"); + return Ok(()); + } + + println!("Fetching {} lexicon(s) from lockfile...", lockfile.lexicons.len()); + + let mut errors = Vec::new(); + let mut success_count = 0; + + // Fetch each lexicon from its DID + for (nsid, locked) in &lockfile.lexicons { + println!("\nRefetching: {}", nsid); + + // Fetch the lexicon using the DID from lockfile + match fetch_specific_lexicon(nsid, &locked.did, &locked.checksum, project_root) { + Ok(()) => { + success_count += 1; + } + Err(e) => { + errors.push((nsid.clone(), format!("{}", e))); + } + } + } + + if !errors.is_empty() { + eprintln!( + "\n{} lexicon(s) fetched successfully, {} error(s):", + success_count, + errors.len() + ); + for (nsid, error) in &errors { + eprintln!(" {} - {}", nsid, error); + } + return Err(FetchError::HttpError(format!( + "Failed to fetch {} lexicons", + errors.len() + ))); + } + + println!("\n✓ Successfully fetched all {} lexicons", success_count); + Ok(()) +} + +/// Fetch a specific lexicon by NSID from a known DID, verifying checksum +fn fetch_specific_lexicon( + nsid: &str, + did: &str, + expected_checksum: &str, + project_root: &std::path::Path, +) -> Result<(), FetchError> { + // Initialize .mlf directory + init_mlf_cache(project_root).map_err(FetchError::InitFailed)?; + let mlf_dir = get_mlf_cache_dir(project_root); + + // Fetch records from the DID + let records = fetch_lexicon_records(did)?; + + // Find the specific NSID + for record in records { + let record_nsid = extract_nsid_from_record(&record)?; + + if record_nsid == nsid { + // Found it! Process and verify checksum + let json_str = serde_json::to_string_pretty(&record.value)?; + let hash = calculate_hash(&json_str); + + if hash != expected_checksum { + return Err(FetchError::HttpError(format!( + "Checksum mismatch for {}: expected {}, got {}", + nsid, expected_checksum, hash + ))); + } + + // Save JSON + let mut json_path = mlf_dir.join("lexicons/json"); + for segment in nsid.split('.') { + json_path.push(segment); + } + json_path.set_extension("json"); + + if let Some(parent) = json_path.parent() { + std::fs::create_dir_all(parent)?; + } + std::fs::write(&json_path, &json_str)?; + println!(" → Saved JSON (checksum verified)"); + + // Convert to MLF + let mlf_content = crate::generate::mlf::generate_mlf_from_json(&record.value) + .map_err(|e| FetchError::ConversionError(format!("{:?}", e)))?; + + let mut mlf_path = mlf_dir.join("lexicons/mlf"); + for segment in nsid.split('.') { + mlf_path.push(segment); + } + mlf_path.set_extension("mlf"); + + if let Some(parent) = mlf_path.parent() { + std::fs::create_dir_all(parent)?; + } + std::fs::write(&mlf_path, mlf_content)?; + println!(" → Converted to MLF"); + + return Ok(()); + } + } + + Err(FetchError::HttpError(format!( + "Lexicon {} not found in repo {}", + nsid, did + ))) +} + fn save_dependency(project_root: &std::path::Path, nsid: &str) -> Result<(), FetchError> { let config_path = project_root.join("mlf.toml"); let mut config = MlfConfig::load(&config_path).map_err(FetchError::NoProjectRoot)?; @@ -219,12 +457,15 @@ fn save_dependency(project_root: &std::path::Path, nsid: &str) -> Result<(), Fet } pub fn fetch_lexicon(nsid: &str, project_root: &std::path::Path) -> Result<(), FetchError> { + let mut lockfile = LockFile::new(); + fetch_lexicon_with_lock(nsid, project_root, &mut lockfile) +} + +fn fetch_lexicon_with_lock(nsid: &str, project_root: &std::path::Path, lockfile: &mut LockFile) -> Result<(), FetchError> { // Initialize .mlf directory init_mlf_cache(project_root).map_err(FetchError::InitFailed)?; let mlf_dir = get_mlf_cache_dir(&project_root); - let cache_file = mlf_dir.join(".lexicon-cache.toml"); - let mut cache = LexiconCache::load(&cache_file)?; // Validate NSID format: must be specific (3+ segments) or use wildcard validate_nsid_format(nsid)?; @@ -237,13 +478,6 @@ pub fn fetch_lexicon(nsid: &str, project_root: &std::path::Path) -> Result<(), F nsid }; - // Check if already cached (for specific NSIDs only) - if !is_wildcard && cache.lexicons.contains_key(nsid) { - println!("Lexicon '{}' is already cached. Skipping fetch.", nsid); - println!(" (Use --force to re-fetch)"); - return Ok(()); - } - // Extract authority and name segments from NSID // For "app.bsky.actor.profile", authority is "app.bsky", name is "actor.profile" // For DNS lookup, we need "_lexicon.actor.bsky.app" @@ -339,12 +573,12 @@ pub fn fetch_lexicon(nsid: &str, project_root: &std::path::Path) -> Result<(), F // Calculate hash of JSON content let hash = calculate_hash(&json_str); - // Update cache - cache.add_entry(record_nsid.clone(), did.clone(), hash); - } + // Extract dependencies from JSON + let dependencies = extract_dependencies_from_json(&record.value); - // Save cache - cache.save(&cache_file)?; + // Update lockfile + lockfile.add_lexicon(record_nsid.clone(), did.clone(), hash.clone(), dependencies); + } if processed_count == 0 { return Err(FetchError::HttpError(format!( @@ -587,5 +821,216 @@ fn extract_nsid_from_record(record: &AtProtoRecord) -> Result String { let mut hasher = Sha256::new(); hasher.update(content.as_bytes()); - format!("{:x}", hasher.finalize()) + format!("sha256:{:x}", hasher.finalize()) +} + +/// Extract external references from a lexicon JSON +/// Returns a list of NSIDs that this lexicon depends on +fn extract_dependencies_from_json(json: &serde_json::Value) -> Vec { + let mut deps = HashSet::new(); + + fn visit_value(value: &serde_json::Value, deps: &mut HashSet) { + match value { + serde_json::Value::Object(map) => { + // Check if this is a ref object + if let Some(ref_val) = map.get("ref") { + if let Some(ref_str) = ref_val.as_str() { + // External refs are multi-segment NSIDs + if ref_str.contains('.') { + deps.insert(ref_str.to_string()); + } + } + } + + // Recurse into all values + for val in map.values() { + visit_value(val, deps); + } + } + serde_json::Value::Array(arr) => { + for val in arr { + visit_value(val, deps); + } + } + _ => {} + } + } + + visit_value(json, &mut deps); + let mut result: Vec = deps.into_iter().collect(); + result.sort(); + result +} + +/// Extract external references from MLF files that need to be resolved +/// Returns a set of namespace patterns (not full NSIDs) that need to be fetched +fn collect_unresolved_references(project_root: &std::path::Path) -> Result, FetchError> { + use mlf_lang::{parser, workspace::Workspace}; + + let mlf_dir = get_mlf_cache_dir(project_root); + let mlf_lexicons_dir = mlf_dir.join("lexicons/mlf"); + + if !mlf_lexicons_dir.exists() { + return Ok(HashSet::new()); + } + + // Build a workspace from all fetched MLF files + let mut workspace = Workspace::new(); + let mut unresolved = HashSet::new(); + + // Recursively find all .mlf files + fn collect_mlf_files(dir: &std::path::Path, files: &mut Vec) -> std::io::Result<()> { + if dir.is_dir() { + for entry in std::fs::read_dir(dir)? { + let entry = entry?; + let path = entry.path(); + if path.is_dir() { + collect_mlf_files(&path, files)?; + } else if path.extension().and_then(|s| s.to_str()) == Some("mlf") { + files.push(path); + } + } + } + Ok(()) + } + + let mut mlf_files = Vec::new(); + collect_mlf_files(&mlf_lexicons_dir, &mut mlf_files)?; + + // Parse each MLF file and add to workspace + for mlf_file in mlf_files { + let content = std::fs::read_to_string(&mlf_file)?; + + // Extract namespace from file path + // e.g., ".mlf/lexicons/mlf/place/stream/key.mlf" -> "place.stream.key" + let relative_path = mlf_file.strip_prefix(&mlf_lexicons_dir) + .map_err(|_| FetchError::IoError(std::io::Error::new( + std::io::ErrorKind::Other, + "Failed to compute relative path" + )))?; + + let namespace = relative_path + .with_extension("") + .to_string_lossy() + .replace(std::path::MAIN_SEPARATOR, "."); + + // Parse the lexicon + if let Ok(lexicon) = parser::parse_lexicon(&content) { + let _ = workspace.add_module(namespace, lexicon); + } + } + + // Resolve to find undefined references + if let Err(errors) = workspace.resolve() { + for error in errors.errors { + if let mlf_lang::error::ValidationError::UndefinedReference { name, .. } = error { + // Only collect multi-segment NSIDs (external references) + // Single-segment names are likely local typos + if name.contains('.') { + // Convert type reference to namespace pattern + // e.g., "app.bsky.actor.defs.profileViewBasic" -> "app.bsky.actor.*" + // We fetch the whole namespace since we don't know which specific + // lexicon file contains the type definition + let namespace_pattern = extract_namespace_pattern(&name); + unresolved.insert(namespace_pattern); + } + } + } + } + + Ok(unresolved) +} + +/// Extract the namespace pattern from a type reference +/// For "app.bsky.actor.defs.profileViewBasic" returns "app.bsky.actor.*" +/// This handles the common ATProto pattern where defs are in a separate namespace +fn extract_namespace_pattern(type_ref: &str) -> String { + let parts: Vec<&str> = type_ref.split('.').collect(); + + // For references with 3+ segments, use the first 3 segments as the namespace + // e.g., "app.bsky.actor.defs.profileViewBasic" -> "app.bsky.actor.*" + // e.g., "com.atproto.repo.strongRef" -> "com.atproto.repo.*" + if parts.len() >= 3 { + format!("{}.{}.{}.*", parts[0], parts[1], parts[2]) + } else if parts.len() == 2 { + // For 2-segment refs like "place.stream", fetch everything under that authority + format!("{}.*", type_ref) + } else { + // Single segment or empty, just return as-is (shouldn't happen) + type_ref.to_string() + } +} + +/// Optimize a set of NSIDs by collapsing them into the minimal set of fetch patterns +/// For example: ["app.bsky.actor.foo", "app.bsky.actor.bar"] -> ["app.bsky.actor.*"] +/// This function tries multiple grouping strategies to find the most efficient pattern +fn optimize_fetch_patterns(nsids: &HashSet) -> Vec { + use std::collections::BTreeMap; + + if nsids.is_empty() { + return Vec::new(); + } + + // Strategy 1: Try grouping by authority (first 2 segments) + // e.g., ["app.bsky.actor.foo", "app.bsky.feed.bar"] -> ["app.bsky.*"] + let mut authority_groups: BTreeMap> = BTreeMap::new(); + + for nsid in nsids { + let parts: Vec<&str> = nsid.split('.').collect(); + if parts.len() >= 2 { + let authority = format!("{}.{}", parts[0], parts[1]); + authority_groups.entry(authority).or_insert_with(Vec::new).push(nsid.clone()); + } + } + + // Strategy 2: Try grouping by namespace prefix (all but last segment) + // e.g., ["app.bsky.actor.foo", "app.bsky.actor.bar"] -> ["app.bsky.actor.*"] + let mut prefix_groups: BTreeMap> = BTreeMap::new(); + + for nsid in nsids { + let parts: Vec<&str> = nsid.split('.').collect(); + if parts.len() >= 3 { + let prefix = parts[..parts.len() - 1].join("."); + prefix_groups.entry(prefix).or_insert_with(Vec::new).push(nsid.clone()); + } + } + + let mut result = Vec::new(); + let mut handled_nsids = HashSet::new(); + + // First pass: Apply namespace-level grouping (more specific) + for (prefix, group) in &prefix_groups { + if group.len() >= 2 && !handled_nsids.contains(&group[0]) { + result.push(format!("{}.*", prefix)); + for nsid in group { + handled_nsids.insert(nsid.clone()); + } + } + } + + // Second pass: For remaining NSIDs, consider authority-level grouping + // Only use authority wildcard if we have 3+ different namespaces under same authority + for (authority, group) in &authority_groups { + let unhandled: Vec<&String> = group.iter() + .filter(|nsid| !handled_nsids.contains(*nsid)) + .collect(); + + if unhandled.len() >= 3 { + result.push(format!("{}.*", authority)); + for nsid in &unhandled { + handled_nsids.insert((*nsid).clone()); + } + } + } + + // Third pass: Add remaining individual NSIDs + for nsid in nsids { + if !handled_nsids.contains(nsid) { + result.push(nsid.clone()); + } + } + + // Sort for consistent output + result.sort(); + result } diff --git a/mlf-cli/src/main.rs b/mlf-cli/src/main.rs index 20af32c..3ea284a 100644 --- a/mlf-cli/src/main.rs +++ b/mlf-cli/src/main.rs @@ -63,6 +63,12 @@ enum Commands { #[arg(long, help = "Add namespace to dependencies in mlf.toml")] save: bool, + + #[arg(long, help = "Update dependencies to latest versions (ignores lockfile)")] + update: bool, + + #[arg(long, help = "Require lockfile and fail if dependencies need updating")] + locked: bool, }, } @@ -134,8 +140,8 @@ fn main() { generate::run_all().into_diagnostic() } }, - Commands::Fetch { nsid, save } => { - fetch::run_fetch(nsid, save).into_diagnostic() + Commands::Fetch { nsid, save, update, locked } => { + fetch::run_fetch(nsid, save, update, locked).into_diagnostic() } }; diff --git a/website/content/docs/cli/02-configuration.md b/website/content/docs/cli/02-configuration.md index d359f57..8204a15 100644 --- a/website/content/docs/cli/02-configuration.md +++ b/website/content/docs/cli/02-configuration.md @@ -91,9 +91,22 @@ dependencies = [ "app.bsky", "com.atproto" ] + +# Enable/disable transitive dependency resolution (default: true) +allow_transitive_deps = true + +# Enable/disable fetch optimization (default: false) +# When true, tries to collapse similar NSIDs into wildcards +optimize_transitive_fetches = false ``` -These dependencies are fetched when you run `mlf fetch` without arguments. +**Options:** + +- `dependencies` - List of NSID patterns to fetch (supports wildcards like `app.bsky.*`) +- `allow_transitive_deps` - Automatically fetch dependencies of dependencies (default: `true`) +- `optimize_transitive_fetches` - Group similar NSIDs into wildcards to reduce fetch count (default: `false`) + +These dependencies are fetched when you run `mlf fetch` without arguments. MLF automatically resolves transitive dependencies unless `allow_transitive_deps` is set to `false`. See the [Fetch Command](../07-fetch/#transitive-dependencies) for more details. ## Commands Using Configuration @@ -257,7 +270,6 @@ When you fetch lexicons, they're stored in `.mlf/lexicons/`: ``` .mlf/ ├── .gitignore # Automatically created -├── .lexicon-cache.toml # Metadata about fetched lexicons └── lexicons/ ├── json/ # Original JSON lexicons │ ├── app.bsky.actor.profile.json @@ -269,14 +281,30 @@ When you fetch lexicons, they're stored in `.mlf/lexicons/`: The `.mlf` directory is automatically added to `.gitignore`, so fetched lexicons won't be committed to your repository. +## Lockfile + +MLF tracks resolved lexicons in `mlf-lock.toml` at your project root: + +```toml +version = 1 + +[lexicons."app.bsky.actor.profile"] +nsid = "app.bsky.actor.profile" +did = "did:plc:4v4y5r3lwsbtmsxhile2ljac" +checksum = "sha256:abc123..." +dependencies = ["com.atproto.repo.strongRef"] +``` + +**Always commit `mlf-lock.toml`** to version control to ensure reproducible builds. See the [Fetch Command](../07-fetch/#lockfile-mlf-lock-toml) documentation for details. + ## Best Practices -1. **Commit `mlf.toml`** - Version control your configuration -2. **Don't commit `.mlf/`** - Let each developer fetch dependencies +1. **Commit `mlf.toml` and `mlf-lock.toml`** - Version control your configuration and lockfile +2. **Don't commit `.mlf/`** - Let each developer fetch dependencies independently 3. **Use semantic namespaces** - Organize lexicons by domain 4. **Set consistent root** - Keep your source directory as the root for namespace calculation 5. **Multiple outputs** - Generate both lexicons and code simultaneously -6. **CI/CD integration** - Run `mlf check` in your CI pipeline +6. **CI/CD integration** - Run `mlf check` and `mlf fetch --locked` in your CI pipeline ## Override Configuration diff --git a/website/content/docs/cli/03-init.md b/website/content/docs/cli/03-init.md index 43ab2a7..6f5636a 100644 --- a/website/content/docs/cli/03-init.md +++ b/website/content/docs/cli/03-init.md @@ -36,7 +36,6 @@ Running `mlf init` creates: ``` .mlf/ ├── .gitignore # Ignores all files except itself - ├── .lexicon-cache.toml # Metadata about fetched lexicons └── lexicons/ ├── json/ # Original JSON lexicons └── mlf/ # Converted MLF format @@ -44,6 +43,8 @@ Running `mlf init` creates: The `.mlf` directory is automatically added to `.gitignore` so fetched lexicons aren't committed to version control. +When you fetch dependencies, a **mlf-lock.toml** lockfile is created at the project root to track resolved lexicon versions. This lockfile should be committed to version control. + ## Interactive Mode Without `--yes`, init prompts for confirmation: @@ -197,6 +198,7 @@ Use `mlf init --yes` in CI pipelines: 1. **Always start with init** - It sets up the correct structure 2. **Use --yes in scripts** - Avoids hanging on prompts -3. **Commit mlf.toml** - Track your project configuration -4. **Don't commit .mlf/** - Let each developer fetch dependencies +3. **Commit mlf.toml and mlf-lock.toml** - Track your project configuration and lockfile +4. **Don't commit .mlf/** - Let each developer fetch dependencies independently 5. **Customize after init** - Edit `mlf.toml` to add outputs and dependencies +6. **Use --locked in CI** - Run `mlf fetch --locked` for reproducible builds diff --git a/website/content/docs/cli/07-fetch.md b/website/content/docs/cli/07-fetch.md index d4fe352..a54c293 100644 --- a/website/content/docs/cli/07-fetch.md +++ b/website/content/docs/cli/07-fetch.md @@ -9,9 +9,15 @@ The `mlf fetch` command downloads ATProto lexicons from remote repositories and ## Usage ```bash -# Fetch all dependencies from mlf.toml +# Fetch all dependencies from mlf.toml (use lockfile if present) mlf fetch +# Fetch and update all dependencies to latest versions +mlf fetch --update + +# Strict mode: fetch from lockfile only (for CI/CD) +mlf fetch --locked + # Fetch a specific lexicon mlf fetch @@ -30,6 +36,72 @@ mlf fetch --save **Options:** - `--save` - Add the NSID/pattern to dependencies in `mlf.toml` +- `--update` - Update dependencies to latest versions (ignores lockfile) +- `--locked` - Require lockfile and fail if dependencies need updating (for CI/CD) + +## Lockfile (`mlf-lock.toml`) + +MLF uses a lockfile to ensure reproducible builds, similar to `package-lock.json` (npm) or `Cargo.lock` (Rust). + +### Lockfile Format + +```toml +version = 1 + +[lexicons."place.stream.richtext.facet"] +nsid = "place.stream.richtext.facet" +did = "did:web:stream.place" +checksum = "sha256:72c8986132821c7c6e3bd30d697f017861d77867b358e3c7850c19baef0a50d5" +dependencies = ["app.bsky.richtext.facet#byteSlice"] + +[lexicons."app.bsky.richtext.facet"] +nsid = "app.bsky.richtext.facet" +did = "did:plc:4v4y5r3lwsbtmsxhile2ljac" +checksum = "sha256:db59d218c482774e617bb5d90d19ab75e2557f8cdebafe798be01b37d957d336" +``` + +### Fetch Modes + +| Mode | Command | Behavior | +|------|---------|----------| +| **Fresh** | `mlf fetch` | No lockfile exists, performs full DNS lookup and fetch, creates lockfile | +| **Lockfile** | `mlf fetch` | Uses existing lockfile to guide fetch, updates lockfile if dependencies change | +| **Update** | `mlf fetch --update` | Ignores lockfile, refetches everything, updates lockfile with latest versions | +| **Locked** | `mlf fetch --locked` | Strict CI mode, uses only lockfile, verifies checksums, fails if no lockfile exists | + +### When to Use Each Mode + +- **Development**: Use `mlf fetch` (default) - respects lockfile for consistency +- **Update deps**: Use `mlf fetch --update` - gets latest versions +- **CI/Production**: Use `mlf fetch --locked` - ensures reproducible builds + +## Transitive Dependencies + +MLF automatically resolves and fetches transitive dependencies (dependencies of dependencies). + +### Example + +If `place.stream.richtext.facet` depends on `app.bsky.richtext.facet#byteSlice`, MLF will: +1. Fetch `place.stream.richtext.*` (your explicit dependency) +2. Parse the lexicons to find external references +3. Automatically fetch `app.bsky.richtext.*` (transitive dependency) +4. Record both in `mlf-lock.toml` + +### Configuration + +Control transitive dependency resolution in `mlf.toml`: + +```toml +[dependencies] +dependencies = ["place.stream.*"] + +# Enable/disable transitive dependency resolution (default: true) +allow_transitive_deps = true + +# Enable/disable fetch optimization (default: false) +# When true, tries to collapse similar NSIDs into wildcards +optimize_transitive_fetches = false +``` ## How It Works @@ -39,6 +111,7 @@ The fetch command follows the ATProto lexicon discovery protocol: 2. **DID Resolution** - Resolves the DID to a PDS endpoint 3. **Fetch Records** - Queries `com.atproto.repo.listRecords` for lexicon schemas 4. **Save & Convert** - Saves JSON and converts to MLF format +5. **Update Lockfile** - Records NSIDs, DIDs, checksums, and dependencies ## Examples @@ -59,7 +132,7 @@ mlf fetch **Output:** ``` -Fetching 2 dependencies... +Fetching 2 dependencies... (mode: fresh, transitive deps: enabled) Fetching: com.example.forum.* Fetching lexicons for pattern: com.example.forum.* @@ -69,17 +142,45 @@ Fetching lexicons for pattern: com.example.forum.* Processing: com.example.forum.post → Saved JSON to .mlf/lexicons/json/com/example/forum/post.json → Converted to MLF at .mlf/lexicons/mlf/com/example/forum/post.mlf - Processing: com.example.forum.thread - → Saved JSON to .mlf/lexicons/json/com/example/forum/thread.json - → Converted to MLF at .mlf/lexicons/mlf/com/example/forum/thread.mlf ✓ Successfully fetched 2 lexicon(s) for com.example.forum.* -Fetching: com.example.social.* -... +→ Updated mlf-lock.toml ✓ Successfully fetched all 2 dependencies ``` +### Update to Latest Versions + +```bash +mlf fetch --update +``` + +This ignores the lockfile and fetches the latest versions of all dependencies. + +### CI/CD with Locked Mode + +```bash +mlf fetch --locked +``` + +**Output:** +``` +Using locked dependencies from mlf-lock.toml +Fetching 2 lexicon(s) from lockfile... + +Refetching: place.stream.richtext.facet + → Using PDS: https://stream.place + → Saved JSON (checksum verified) + → Converted to MLF + +✓ Successfully fetched all 2 lexicons +``` + +If no lockfile exists: +``` +✗ No lockfile found. Run `mlf fetch` first to create mlf-lock.toml +``` + ### Fetch Specific Lexicon ```bash @@ -114,7 +215,6 @@ Fetched lexicons are stored in `.mlf/lexicons/`: ``` .mlf/ ├── .gitignore # Auto-generated -├── .lexicon-cache.toml # Cache metadata └── lexicons/ ├── json/ # Original JSON lexicons │ ├── com/ @@ -142,17 +242,7 @@ Fetched lexicons are stored in `.mlf/lexicons/`: └── post.mlf ``` -### Cache File - -The `.lexicon-cache.toml` tracks what's been fetched: - -```toml -[[lexicons."com.example.forum.post"]] -nsid = "com.example.forum.post" -fetched_at = "2024-01-15T10:30:00Z" -did = "did:web:example.com" -hash = "abc123..." -``` +**Note:** The lockfile (`mlf-lock.toml`) lives at the project root, sibling to `mlf.toml`. ## DNS Resolution @@ -222,22 +312,6 @@ Created mlf.toml in /path/to/current/dir ... ``` -## Re-fetching - -If a lexicon is already cached, fetch skips it: - -```bash -$ mlf fetch com.example.forum.post -Lexicon 'com.example.forum.post' is already cached. Skipping fetch. - (Use --force to re-fetch) -``` - -To re-fetch: - -```bash -mlf fetch com.example.forum.post --force # Not yet implemented -``` - ## Error Handling ### DNS Errors @@ -276,31 +350,50 @@ mlf fetch com.example.forum.post --force # Not yet implemented ### Invalid NSID Format ``` -✗ NSID must have at least 3 segments or use wildcard (e.g., 'com.example.forum.post' or 'com.example.forum.*'): com.example +✗ NSID must have at least 2 segments or use wildcard: com ``` **Solution:** - Use a specific NSID: `com.example.forum.post` - Or use a wildcard: `com.example.forum.*` -## Best Practices +### Checksum Mismatch (--locked mode) + +``` +✗ Checksum mismatch for place.stream.facet: expected sha256:abc123, got sha256:def456 +``` + +**Causes:** +- Lexicon was updated on the server +- Lock file is out of date + +**Solution:** +```bash +mlf fetch --update # Update lockfile with new checksums +``` -1. **Fetch before work** - Always fetch dependencies before coding -2. **Use --save** - Keep `mlf.toml` up to date with dependencies -3. **Don't commit `.mlf/`** - Let each developer fetch independently -4. **Check DNS** - Verify TXT records before fetching -5. **Version dependencies** - Consider tracking lexicon versions (future feature) +## Best Practices +1. **Commit lockfile** - Always commit `mlf-lock.toml` to version control +2. **Use --locked in CI** - Ensures reproducible builds in CI/CD pipelines +3. **Fetch before work** - Always fetch dependencies before coding +4. **Use --save** - Keep `mlf.toml` up to date with dependencies +5. **Don't commit `.mlf/`** - Let each developer fetch independently +6. **Check DNS** - Verify TXT records before fetching +7. **Update explicitly** - Use `mlf fetch --update` when you want latest versions ## Comparison with npm/cargo -The fetch command is similar to package managers: +The fetch command follows patterns from popular package managers: -| Command | npm | cargo | mlf | -|---------|-----|-------|-----| -| Install deps | `npm install` | `cargo fetch` | `mlf fetch` | -| Add dep | `npm install pkg --save` | `cargo add pkg` | `mlf fetch ns --save` | +| Aspect | npm | Cargo | MLF | +|--------|-----|-------|-----| +| Install deps | `npm install` | `cargo build` | `mlf fetch` | +| Update deps | `npm update` | `cargo update` | `mlf fetch --update` | +| Strict mode | `npm ci` | `cargo build --locked` | `mlf fetch --locked` | +| Add dep | `npm install pkg` | `cargo add pkg` | `mlf fetch ns --save` | | Config file | `package.json` | `Cargo.toml` | `mlf.toml` | +| Lock file | `package-lock.json` | `Cargo.lock` | `mlf-lock.toml` | | Cache | `node_modules/` | `~/.cargo/` | `.mlf/` | ## Troubleshooting @@ -321,8 +414,18 @@ Make sure you're using the correct format: - ✓ `com.example.forum.post` (specific lexicon) - ✓ `com.example.forum.*` (wildcard) - ✓ `app.bsky.feed.*` (real-world wildcard) -- ✗ `com.example` (must be specific or use wildcard) +- ✗ `com` (must have at least 2 segments) ### Permission Errors -Ensure you have write permissions for the project directory to create `.mlf/`. +Ensure you have write permissions for the project directory to create `.mlf/` and `mlf-lock.toml`. + +### Conflicting Flags + +``` +✗ Cannot use --update and --locked together +``` + +Choose one mode: +- Use `--update` to get latest versions +- Use `--locked` for strict reproducible builds