Something went wrong. Try again.
This repository has no description
Something went wrong. Try again.
8.7 kB · 248 lines
Rust
at main
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249//! `fileset` — parse a gitignore-like manifest that decides which files to include,//! and copy the included files out of a directory tree.//!//! A fileset manifest is one glob per line, compiled by `globset` with//! `literal_separator(true)` and anchored gitignore-style: a pattern with a `/` in it//! is anchored to the tree root, a pattern without one matches at any depth, and a//! trailing `/` matches a directory's whole subtree. A leading `!` re-includes;//! `@import <path>` inlines another manifest (path relative to the importing file).//! Paths are matched tree-root-relative, later rules win, and an unmatched path//! defaults to "included". See the README for the source-of-truth format docs.
use std::fs;use std::path::{Path, PathBuf};
use anyhow::{bail, Context, Result};use globset::{GlobBuilder, GlobMatcher};use walkdir::WalkDir;
/// One rule: the glob(s) a single pattern expands to, plus whether matching them/// re-includes (`!`) or excludes. The globs are OR'd (any match = the rule fires).struct Rule { negated: bool, matchers: Vec<GlobMatcher>,}
/// An ordered list of rules resolved from a fileset file (imports inlined in place).pub struct Fileset { rules: Vec<Rule>,}
impl Fileset { /// Load a fileset file, recursively inlining `@import`ed files. pub fn load(path: &Path) -> Result<Fileset> { let mut rules = Vec::new(); parse_file(path, &mut rules, &mut Vec::new())?; Ok(Fileset { rules }) }
/// Whether a tree-root-relative path is included. Default include; scan rules in /// order and let the last match win (exclude drops, `!` re-includes). pub fn includes(&self, rel: &Path) -> bool { let mut included = true; for rule in &self.rules { if rule.matchers.iter().any(|m| m.is_match(rel)) { included = rule.negated; } } included }
#[cfg(test)] fn from_str(content: &str) -> Result<Fileset> { let mut rules = Vec::new(); parse_lines(content, Path::new("."), &mut rules, &mut Vec::new())?; Ok(Fileset { rules }) }}
/// Copy every file under `src` that `fileset` includes into `out`, preserving relative/// structure. Skips symlinks and prunes `.git`/`.jj` before descending into them./// Returns the number of files copied.pub fn filter_tree(src: &Path, out: &Path, fileset: &Fileset) -> Result<usize> { let src_root = src .canonicalize() .with_context(|| format!("resolving source dir {}", src.display()))?;
// Prune VCS metadata dirs before walking them (they're huge and also excluded). let walker = WalkDir::new(&src_root).into_iter().filter_entry(|e| { !(e.file_type().is_dir() && matches!(e.file_name().to_str(), Some(".git" | ".jj"))) });
let mut copied = 0usize; for entry in walker { let entry = entry?; // Regular files only: symlinks and directories are skipped. if !entry.file_type().is_file() { continue; } let rel = entry .path() .strip_prefix(&src_root) .expect("walked path is under src_root"); if !fileset.includes(rel) { continue; } let dest = out.join(rel); if let Some(parent) = dest.parent() { fs::create_dir_all(parent).with_context(|| format!("mkdir {}", parent.display()))?; } fs::copy(entry.path(), &dest) .with_context(|| format!("copy {} -> {}", entry.path().display(), dest.display()))?; copied += 1; } Ok(copied)}
/// Read a fileset file and parse its lines. `visited` holds canonicalized paths/// already parsed, breaking `@import` cycles.fn parse_file(path: &Path, rules: &mut Vec<Rule>, visited: &mut Vec<PathBuf>) -> Result<()> { let canon = path .canonicalize() .with_context(|| format!("resolving fileset {}", path.display()))?; if visited.contains(&canon) { return Ok(()); } visited.push(canon.clone());
let content = std::fs::read_to_string(&canon) .with_context(|| format!("reading fileset {}", canon.display()))?; let dir = canon.parent().unwrap_or_else(|| Path::new(".")); parse_lines(&content, dir, rules, visited)}
/// Parse the lines of a fileset, resolving `@import` relative to `dir`.fn parse_lines( content: &str, dir: &Path, rules: &mut Vec<Rule>, visited: &mut Vec<PathBuf>,) -> Result<()> { for (idx, raw) in content.lines().enumerate() { let line = raw.trim(); if line.is_empty() || line.starts_with('#') { continue; } if let Some(rest) = line.strip_prefix("@import") { let target = rest.trim(); if target.is_empty() { bail!("line {}: `@import` needs a path", idx + 1); } parse_file(&dir.join(target), rules, visited)?; continue; } let (negated, pat) = match line.strip_prefix('!') { Some(rest) => (true, rest.trim()), None => (false, line), }; let matchers = compile_pattern(pat) .with_context(|| format!("line {}: invalid pattern {:?}", idx + 1, pat))?; rules.push(Rule { negated, matchers }); } Ok(())}
/// Expand one gitignore-style pattern into the globset globs it should match, then/// compile them. Anchoring rules mirror gitignore: a `/` anywhere anchors to the/// root, no `/` matches at any depth, a trailing `/` matches a directory subtree.fn compile_pattern(raw: &str) -> Result<Vec<GlobMatcher>> { let (dir_only, pat) = match raw.strip_suffix('/') { Some(p) => (true, p), None => (false, raw), }; let (leading_anchor, pat) = match pat.strip_prefix('/') { Some(p) => (true, p), None => (false, pat), }; if pat.is_empty() { bail!("empty pattern"); } let anchored = leading_anchor || pat.contains('/');
let mut globs: Vec<String> = Vec::new(); if anchored { if !dir_only { globs.push(pat.to_string()); } globs.push(format!("{pat}/**")); } else { if !dir_only { globs.push(pat.to_string()); globs.push(format!("**/{pat}")); } globs.push(format!("{pat}/**")); globs.push(format!("**/{pat}/**")); }
globs .iter() .map(|g| { Ok(GlobBuilder::new(g) .literal_separator(true) .build()? .compile_matcher()) }) .collect()}
#[cfg(test)]mod tests { use super::*; use std::path::Path;
fn included(fs: &Fileset, p: &str) -> bool { fs.includes(Path::new(p)) }
#[test] fn default_is_included() { let fs = Fileset::from_str("secrets/\n").unwrap(); assert!(included(&fs, "venus/foo.nix")); }
#[test] fn directory_exclude_covers_contents_at_any_depth() { let fs = Fileset::from_str("secrets/\n").unwrap(); assert!(!included(&fs, "secrets/passwords.yaml")); assert!(!included(&fs, "secrets/personal/jupiter.env")); assert!(!included(&fs, "nested/secrets/x")); // slashless -> any depth assert!(included(&fs, "secretsX/thing")); // respects the name boundary }
#[test] fn anchored_pattern_respects_literal_separator() { // A `/` anchors to root; `*` never crosses a path separator. let fs = Fileset::from_str("a/*.md\n").unwrap(); assert!(!included(&fs, "a/x.md")); assert!(included(&fs, "a/b/x.md")); // `*` does not span a directory assert!(included(&fs, "x.md")); // anchored: only under a/ }
#[test] fn slashless_pattern_matches_any_depth() { let fs = Fileset::from_str("*.loro\n").unwrap(); assert!(!included(&fs, "snap.loro")); assert!(!included(&fs, "deep/dir/snap.loro")); }
#[test] fn negation_re_includes_and_last_match_wins() { let fs = Fileset::from_str("build/\n!build/keep/\n").unwrap(); assert!(!included(&fs, "build/out.js")); assert!(included(&fs, "build/keep/y.js")); }
#[test] fn lockfiles_excluded_but_flake_lock_kept() { let fs = Fileset::from_str("*.lock\n!flake.lock\n").unwrap(); assert!(!included(&fs, "Cargo.lock")); assert!(!included(&fs, "a/b/pnpm-lock.lock")); assert!(included(&fs, "flake.lock")); assert!(included(&fs, "experimental/owl/flake.lock")); }
#[test] fn import_directive_needs_a_path() { assert!(Fileset::from_str("@import\n").is_err()); }}