diff --git a/Cargo.lock b/Cargo.lock index 657728c04..73fd9e995 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -40,7 +40,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "531a9155a481e2ee699d4f98f43c0ca4ff8ee1bfd55c31e9e98fb29d2b176fe0" dependencies = [ "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "serde", ] @@ -59,6 +59,17 @@ dependencies = [ "syn", ] +[[package]] +name = "console" +version = "0.16.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4fe5f465a4f6fee88fad41b85d990f84c835335e85b5d9e6e63e0d06d28cba7c" +dependencies = [ + "encode_unicode", + "libc", + "windows-sys 0.61.2", +] + [[package]] name = "convert_case" version = "0.11.0" @@ -126,6 +137,12 @@ version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7fcaabb2fef8c910e7f4c7ce9f67a1283a1715879a7c230ca9d6d1ae31f16d91" +[[package]] +name = "encode_unicode" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" + [[package]] name = "errno" version = "0.3.9" @@ -241,14 +258,14 @@ dependencies = [ [[package]] name = "globset" -version = "0.4.17" +version = "0.4.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eab69130804d941f8075cfd713bf8848a2c3b3f201a9457a11e6f87e1ab62305" +checksum = "07c34a9410465b45bd9787443bc7370f37735bad04b0f0cd57ff1a3186c98988" dependencies = [ "aho-corasick", "bstr", "log", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "regex-syntax 0.8.5", ] @@ -273,7 +290,7 @@ dependencies = [ "globset", "log", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "same-file", "walkdir", "winapi-util", @@ -281,7 +298,7 @@ dependencies = [ [[package]] name = "ignore" -version = "0.4.24" +version = "0.4.33" dependencies = [ "bstr", "crossbeam-channel", @@ -290,12 +307,24 @@ dependencies = [ "globset", "log", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "same-file", "walkdir", "winapi-util", ] +[[package]] +name = "insta" +version = "1.48.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "86f0f8fee8c926415c58d6ae43a08523a26faccb2323f5e6b644fe7dd4ef6b82" +dependencies = [ + "console", + "once_cell", + "similar", + "tempfile", +] + [[package]] name = "itertools" version = "0.11.0" @@ -445,9 +474,9 @@ dependencies = [ [[package]] name = "once_cell" -version = "1.19.0" +version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fdb12b2476b595f9358c5161aa467c2438859caa136dec86c26fdd2efe17b92" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "overload" @@ -517,7 +546,7 @@ checksum = "b544ef1b4eac5dc2db33ea63606ae9ffcfac26c1416a2806ae0bf5f56b201191" dependencies = [ "aho-corasick", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "regex-syntax 0.8.5", ] @@ -532,9 +561,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.8" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "368758f23274712b504848e9d5a6f010445cc8b87a7cdb4d7cbee666c1288da3" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -602,6 +631,12 @@ dependencies = [ "lazy_static", ] +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + [[package]] name = "slab" version = "0.4.12" @@ -646,7 +681,8 @@ dependencies = [ "dunce", "fast-glob", "globwalk", - "ignore 0.4.24", + "ignore 0.4.33", + "insta", "log", "pretty_assertions", "rayon", @@ -832,6 +868,15 @@ dependencies = [ "windows-targets", ] +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + [[package]] name = "windows-targets" version = "0.52.6" diff --git a/crates/ignore/Cargo.toml b/crates/ignore/Cargo.toml index bd0352d15..d0af38ba9 100644 --- a/crates/ignore/Cargo.toml +++ b/crates/ignore/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "ignore" -version = "0.4.24" #:version +version = "0.4.33" #:version authors = ["Andrew Gallant "] description = """ A fast library for efficiently matching ignore files such as `.gitignore` @@ -12,7 +12,10 @@ repository = "https://github.com/BurntSushi/ripgrep/tree/master/crates/ignore" readme = "README.md" keywords = ["glob", "ignore", "gitignore", "pattern", "file"] license = "Unlicense OR MIT" +# CHANGED: Use an explicit edition instead of `edition.workspace = true` since this crate is +# vendored into the Tailwind CSS workspace. edition = "2024" +rust-version = "1.88" [lib] name = "ignore" @@ -20,15 +23,17 @@ bench = false [dependencies] crossbeam-deque = "0.8.3" -globset = "0.4.17" +# CHANGED: Use the published globset crate instead of a path dependency. +globset = "0.4.20" log = "0.4.20" memchr = "2.6.3" same-file = "1.0.6" walkdir = "2.4.0" +# CHANGED: Added `dunce` to canonicalize paths without UNC prefixes on Windows. dunce = "1.0.5" [dependencies.regex-automata] -version = "0.4.0" +version = "0.4.18" default-features = false features = ["std", "perf", "syntax", "meta", "nfa", "hybrid", "dfa-onepass"] diff --git a/crates/ignore/examples/walk.rs b/crates/ignore/examples/walk.rs index 9c627dc3e..c61d0515e 100644 --- a/crates/ignore/examples/walk.rs +++ b/crates/ignore/examples/walk.rs @@ -18,9 +18,7 @@ fn main() { let stdout_thread = std::thread::spawn(move || { let mut stdout = std::io::BufWriter::new(std::io::stdout()); for dent in rx { - stdout - .write_all(&Vec::from_path_lossy(dent.path())) - .unwrap(); + stdout.write_all(&Vec::from_path_lossy(dent.path())).unwrap(); stdout.write_all(b"\n").unwrap(); } }); diff --git a/crates/ignore/src/default_types.rs b/crates/ignore/src/default_types.rs index 4e060b76a..6b5bba0f0 100644 --- a/crates/ignore/src/default_types.rs +++ b/crates/ignore/src/default_types.rs @@ -47,6 +47,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["cml"], &["*.cml"]), (&["coffeescript"], &["*.coffee"]), (&["config"], &["*.cfg", "*.conf", "*.config", "*.ini"]), + (&["container"], &["*Containerfile*", "*Dockerfile*"]), (&["coq"], &["*.v"]), (&["cpp"], &[ "*.[ChH]", "*.cc", "*.[ch]pp", "*.[ch]xx", "*.hh", "*.inl", @@ -109,6 +110,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["hbs"], &["*.hbs"]), (&["hs"], &["*.hs", "*.lhs"]), (&["html"], &["*.htm", "*.html", "*.ejs"]), + (&["hurl"], &["*.hurl"]), (&["hy"], &["*.hy"]), (&["idris"], &["*.idr", "*.lidr"]), (&["janet"], &["*.janet"]), @@ -185,6 +187,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["mint"], &["*.mint"]), (&["mk"], &["mkfile"]), (&["ml"], &["*.ml"]), + (&["mojo"], &["*.mojo"]), (&["motoko"], &["*.mo"]), (&["msbuild"], &[ "*.csproj", "*.fsproj", "*.vcxproj", "*.proj", "*.props", "*.targets", @@ -206,11 +209,12 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ "*.php", "*.php3", "*.php4", "*.php5", "*.php7", "*.php8", "*.pht", "*.phtml" ]), + (&["pkgbuild"], &["PKGBUILD"]), (&["po"], &["*.po"]), (&["pod"], &["*.pod"]), (&["postscript"], &["*.eps", "*.ps"]), (&["prolog"], &["*.pl", "*.pro", "*.prolog", "*.P"]), - (&["protobuf"], &["*.proto"]), + (&["proto", "protobuf"], &["*.proto"]), (&["ps"], &["*.cdxml", "*.ps1", "*.ps1xml", "*.psd1", "*.psm1"]), (&["puppet"], &["*.epp", "*.erb", "*.pp", "*.rb"]), (&["purs"], &["*.purs"]), @@ -231,6 +235,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["red"], &["*.r", "*.red", "*.reds"]), (&["rescript"], &["*.res", "*.resi"]), (&["robot"], &["*.robot"]), + (&["rocq"], &["*.v"]), (&["rst"], &["*.rst"]), (&["ruby"], &[ // Idiomatic files @@ -274,6 +279,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["spark"], &["*.spark"]), (&["spec"], &["*.spec"]), (&["sql"], &["*.sql", "*.psql"]), + (&["ssa"], &["*.ssa"]), (&["stylus"], &["*.styl"]), (&["sv"], &["*.v", "*.vg", "*.sv", "*.svh", "*.h"]), (&["svelte"], &["*.svelte", "*.svelte.ts"]), @@ -359,4 +365,14 @@ mod tests { previous_name = name; } } + + #[test] + fn default_types_aliases_are_sorted() { + for (aliases, _) in DEFAULT_TYPES.iter() { + assert!( + aliases.is_sorted(), + "this alias list is not sorted: {aliases:?}", + ); + } + } } diff --git a/crates/ignore/src/dir.rs b/crates/ignore/src/dir.rs index 11b58f8ca..6bee724c8 100644 --- a/crates/ignore/src/dir.rs +++ b/crates/ignore/src/dir.rs @@ -16,7 +16,7 @@ use std::{ collections::HashMap, ffi::{OsStr, OsString}, - fs::{File, FileType}, + fs::{self, File, FileType}, io::{self, BufRead}, path::{Path, PathBuf}, sync::{Arc, RwLock, Weak}, @@ -25,7 +25,7 @@ use std::{ use crate::{ gitignore::{self, Gitignore, GitignoreBuilder}, overrides::{self, Override}, - pathutil::{is_hidden, strip_prefix}, + pathutil::{is_hidden_entry, strip_prefix}, types::{self, Types}, walk::DirEntry, {Error, Match, PartialErrorBuilder}, @@ -91,7 +91,25 @@ struct IgnoreOptions { /// Ignore is a matcher useful for recursively walking one or more directories. #[derive(Clone, Debug)] -pub(crate) struct Ignore(Arc); +pub(crate) struct Ignore { + inner: Arc, + // Parent matchers are cached independently of the path being walked, but + // matching them still needs the canonicalized path originally passed to + // `add_parents`. For example, when walking `/tmp/project/src`, parent + // matchers use `/tmp/project/src` to rewrite `/tmp/project/src/foo.py` + // before matching it against ignore files from `/tmp/project` and its + // ancestors. + // + // For ripgrep itself, this means that `rg pat src tests` must rewrite + // `src/foo` relative to `.../src`, and not whatever root was prepared + // first. + // + // See: https://github.com/BurntSushi/ripgrep/pull/3420 + // See: https://github.com/BurntSushi/ripgrep/issues/3376 + // See: https://github.com/BurntSushi/ripgrep/issues/3419 + // See: https://github.com/BurntSushi/ripgrep/issues/3320 + absolute_base: Option>, +} #[derive(Clone, Debug)] struct IgnoreInner { @@ -112,12 +130,9 @@ struct IgnoreInner { /// /// If this is the root directory or there are otherwise no more /// directories to match, then `parent` is `None`. - parent: Option, + parent: Option>, /// Whether this is an absolute parent matcher, as added by add_parent. is_absolute_parent: bool, - /// The absolute base path of this matcher. Populated only if parent - /// directories are added. - absolute_base: Option>, /// The directory that gitignores should be interpreted relative to. /// /// Usually this is the directory containing the gitignore file. But in @@ -152,34 +167,36 @@ struct IgnoreInner { impl Ignore { /// Return the directory path of this matcher. + #[cfg(test)] pub(crate) fn path(&self) -> &Path { - &self.0.dir + &self.inner.dir } /// Return true if this matcher has no parent. pub(crate) fn is_root(&self) -> bool { - self.0.parent.is_none() - } - - /// Returns true if this matcher was added via the `add_parents` method. - pub(crate) fn is_absolute_parent(&self) -> bool { - self.0.is_absolute_parent + self.inner.parent.is_none() } /// Return this matcher's parent, if one exists. pub(crate) fn parent(&self) -> Option { - self.0.parent.clone() + self.inner.parent.as_ref().map(|parent| Ignore { + inner: parent.clone(), + absolute_base: self.absolute_base.clone(), + }) } /// Create a new `Ignore` matcher with the parent directories of `dir`. /// /// Note that this can only be called on an `Ignore` matcher with no /// parents (i.e., `is_root` returns `true`). This will panic otherwise. - pub(crate) fn add_parents>(&self, path: P) -> (Ignore, Option) { - if !self.0.opts.parents - && !self.0.opts.git_ignore - && !self.0.opts.git_exclude - && !self.0.opts.git_global + pub(crate) fn add_parents>( + &self, + path: P, + ) -> (Ignore, Option) { + if !self.inner.opts.parents + && !self.inner.opts.git_ignore + && !self.inner.opts.git_exclude + && !self.inner.opts.git_global { // If we never need info from parent directories, then don't do // anything. @@ -209,25 +226,34 @@ impl Ignore { let mut errs = PartialErrorBuilder::default(); let mut ig = self.clone(); for parent in parents.into_iter().rev() { - let mut compiled = self.0.compiled.write().unwrap(); + let mut compiled = self.inner.compiled.write().unwrap(); if let Some(weak) = compiled.get(parent.as_os_str()) { if let Some(prebuilt) = weak.upgrade() { - ig = Ignore(prebuilt); + ig = Ignore { + inner: prebuilt, + absolute_base: Some(absolute_base.clone()), + }; continue; } } let (mut igtmp, err) = ig.add_child_path(parent); errs.maybe_push(err); igtmp.is_absolute_parent = true; - igtmp.absolute_base = Some(absolute_base.clone()); - igtmp.has_git = if self.0.opts.require_git && self.0.opts.git_ignore { - parent.join(".git").exists() || parent.join(".jj").exists() - } else { - false - }; + igtmp.has_git = + if self.inner.opts.require_git && self.inner.opts.git_ignore { + parent.join(".git").exists() || parent.join(".jj").exists() + } else { + false + }; let ig_arc = Arc::new(igtmp); - ig = Ignore(ig_arc.clone()); - compiled.insert(parent.as_os_str().to_os_string(), Arc::downgrade(&ig_arc)); + ig = Ignore { + inner: ig_arc.clone(), + absolute_base: Some(absolute_base.clone()), + }; + compiled.insert( + parent.as_os_str().to_os_string(), + Arc::downgrade(&ig_arc), + ); } (ig, errs.into_error_option()) } @@ -240,60 +266,161 @@ impl Ignore { /// returned if it exists. /// /// Note that all I/O errors are completely ignored. - pub(crate) fn add_child>(&self, dir: P) -> (Ignore, Option) { + pub(crate) fn add_child>( + &self, + dir: P, + ) -> (Ignore, Option) { let (ig, err) = self.add_child_path(dir.as_ref()); - (Ignore(Arc::new(ig)), err) + ( + Ignore { + inner: Arc::new(ig), + absolute_base: self.absolute_base.clone(), + }, + err, + ) + } + + /// Like add_child, but uses successful read_dir entries to reduce + /// probing when discovering ignore files. + pub(crate) fn add_child_with_entries>( + &self, + dir: P, + entries: &[fs::DirEntry], + ) -> (Ignore, Option) { + let files = self.collect_ignore_files(entries); + let (ig, err) = self.add_child_path_with_found_ignore_files( + dir.as_ref(), + Some(&files), + ); + ( + Ignore { + inner: Arc::new(ig), + absolute_base: self.absolute_base.clone(), + }, + err, + ) } /// Like add_child, but takes a full path and returns an IgnoreInner. fn add_child_path(&self, dir: &Path) -> (IgnoreInner, Option) { - let git_type = - if self.0.opts.require_git && (self.0.opts.git_ignore || self.0.opts.git_exclude) { - dir.join(".git").metadata().ok().map(|md| md.file_type()) - } else { - None - }; - let has_git = git_type.is_some() || dir.join(".jj").exists(); + self.add_child_path_with_found_ignore_files(dir, None) + } + + fn collect_ignore_files( + &self, + entries: &[fs::DirEntry], + ) -> IgnoreFilesFound { + let custom_ignore_filenames = &self.inner.custom_ignore_filenames; + let mut files = IgnoreFilesFound { + has_ignore: false, + has_git_ignore: false, + has_git_dir: false, + has_jj_dir: false, + custom_ignore_files: vec![false; custom_ignore_filenames.len()], + }; + for entry in entries { + let file_name = entry.file_name(); + if file_name == OsStr::new(".ignore") { + files.has_ignore = true; + } else if file_name == OsStr::new(".gitignore") { + files.has_git_ignore = true; + } else if file_name == OsStr::new(".git") { + files.has_git_dir = true; + } else if file_name == OsStr::new(".jj") { + files.has_jj_dir = true; + } + for (i, name) in custom_ignore_filenames.iter().enumerate() { + if file_name == name.as_os_str() { + files.custom_ignore_files[i] = true; + } + } + } + files + } + + fn add_child_path_with_found_ignore_files( + &self, + dir: &Path, + ignore_files_list: Option<&IgnoreFilesFound>, + ) -> (IgnoreInner, Option) { + let check_vcs_dir = self.inner.opts.require_git + && (self.inner.opts.git_ignore || self.inner.opts.git_exclude); + let git_type = if check_vcs_dir + && ignore_files_list.is_none_or(|i| i.has_git_dir) + { + dir.join(".git").metadata().ok().map(|md| md.file_type()) + } else { + None + }; + let has_jj = check_vcs_dir + && ignore_files_list.is_none_or(|i| i.has_jj_dir) + && dir.join(".jj").exists(); + let has_git = check_vcs_dir && (git_type.is_some() || has_jj); let mut errs = PartialErrorBuilder::default(); - let custom_ig_matcher = if self.0.custom_ignore_filenames.is_empty() { + let custom_ig_matcher = if self + .inner + .custom_ignore_filenames + .is_empty() + { Gitignore::empty() } else { - let (m, err) = create_gitignore( - &dir, - &dir, - &self.0.custom_ignore_filenames, - self.0.opts.ignore_case_insensitive, - ); - errs.maybe_push(err); - m + let custom_ignore_names: Vec<&OsString> = match ignore_files_list { + None => self.inner.custom_ignore_filenames.iter().collect(), + Some(m) => self + .inner + .custom_ignore_filenames + .iter() + .zip(m.custom_ignore_files.iter()) + .filter(|&(_, &matched)| matched) + .map(|(name, _)| name) + .collect(), + }; + if custom_ignore_names.is_empty() { + Gitignore::empty() + } else { + let (m, err) = create_gitignore( + &dir, + &dir, + &custom_ignore_names, + self.inner.opts.ignore_case_insensitive, + ); + errs.maybe_push(err); + m + } }; - let ig_matcher = if !self.0.opts.ignore { + let ig_matcher = if !self.inner.opts.ignore + || !ignore_files_list.is_none_or(|i| i.has_ignore) + { Gitignore::empty() } else { let (m, err) = create_gitignore( &dir, &dir, &[".ignore"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m }; - let gi_matcher = if !self.0.opts.git_ignore { + let gi_matcher = if !self.inner.opts.git_ignore + || !ignore_files_list.is_none_or(|i| i.has_git_ignore) + { Gitignore::empty() } else { let (m, err) = create_gitignore( &dir, &dir, &[".gitignore"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m }; - let gi_exclude_matcher = if !self.0.opts.git_exclude { + let gi_exclude_matcher = if !self.inner.opts.git_exclude + || !ignore_files_list.is_none_or(|i| i.has_git_dir) + { Gitignore::empty() } else { match resolve_git_commondir(dir, git_type) { @@ -302,7 +429,7 @@ impl Ignore { &dir, &git_dir, &["info/exclude"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m @@ -314,32 +441,38 @@ impl Ignore { } }; let ig = IgnoreInner { - compiled: self.0.compiled.clone(), + compiled: self.inner.compiled.clone(), dir: dir.to_path_buf(), - overrides: self.0.overrides.clone(), - types: self.0.types.clone(), - parent: Some(self.clone()), + overrides: self.inner.overrides.clone(), + types: self.inner.types.clone(), + parent: Some(self.inner.clone()), is_absolute_parent: false, - absolute_base: self.0.absolute_base.clone(), - global_gitignores_relative_to: self.0.global_gitignores_relative_to.clone(), - explicit_ignores: self.0.explicit_ignores.clone(), - custom_ignore_filenames: self.0.custom_ignore_filenames.clone(), + global_gitignores_relative_to: self + .inner + .global_gitignores_relative_to + .clone(), + explicit_ignores: self.inner.explicit_ignores.clone(), + custom_ignore_filenames: self + .inner + .custom_ignore_filenames + .clone(), custom_ignore_matcher: custom_ig_matcher, ignore_matcher: ig_matcher, - git_global_matcher: self.0.git_global_matcher.clone(), + git_global_matcher: self.inner.git_global_matcher.clone(), git_ignore_matcher: gi_matcher, git_exclude_matcher: gi_exclude_matcher, has_git, - opts: self.0.opts, + opts: self.inner.opts, }; (ig, errs.into_error_option()) } /// Returns true if at least one type of ignore rule should be matched. fn has_any_ignore_rules(&self) -> bool { - let opts = self.0.opts; - let has_custom_ignore_files = !self.0.custom_ignore_filenames.is_empty(); - let has_explicit_ignores = !self.0.explicit_ignores.is_empty(); + let opts = self.inner.opts; + let has_custom_ignore_files = + !self.inner.custom_ignore_filenames.is_empty(); + let has_explicit_ignores = !self.inner.explicit_ignores.is_empty(); opts.ignore || opts.git_global @@ -350,9 +483,12 @@ impl Ignore { } /// Like `matched`, but works with a directory entry instead. - pub(crate) fn matched_dir_entry<'a>(&'a self, dent: &DirEntry) -> Match> { + pub(crate) fn matched_dir_entry<'a>( + &'a self, + dent: &DirEntry, + ) -> Match> { let m = self.matched(dent.path(), dent.is_dir()); - if m.is_none() && self.0.opts.hidden && is_hidden(dent) { + if m.is_none() && self.inner.opts.hidden && is_hidden_entry(dent) { return Match::Ignore(IgnoreMatch::hidden()); } m @@ -362,7 +498,11 @@ impl Ignore { /// ignored or not. /// /// The match contains information about its origin. - fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub(crate) fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { // We need to be careful with our path. If it has a leading ./, then // strip it because it causes nothing but trouble. let mut path = path.as_ref(); @@ -373,9 +513,9 @@ impl Ignore { // regardless of whether it's whitelist/ignore, then we quit and // return that result immediately. Overrides have the highest // precedence. - if !self.0.overrides.is_empty() { + if !self.inner.overrides.is_empty() { let mat = self - .0 + .inner .overrides .matched(path, is_dir) .map(IgnoreMatch::overrides); @@ -392,8 +532,9 @@ impl Ignore { whitelisted = mat; } } - if !self.0.types.is_empty() { - let mat = self.0.types.matched(path, is_dir).map(IgnoreMatch::types); + if !self.inner.types.is_empty() { + let mat = + self.inner.types.matched(path, is_dir).map(IgnoreMatch::types); if mat.is_ignore() { return mat; } else if mat.is_whitelist() { @@ -405,98 +546,143 @@ impl Ignore { /// Performs matching only on the ignore files for this directory and /// all parent directories. - fn matched_ignore<'a>(&'a self, path: &Path, is_dir: bool) -> Match> { - let (mut m_custom_ignore, mut m_ignore, mut m_gi, mut m_gi_exclude, mut m_explicit) = ( - Match::None, - Match::None, - Match::None, - Match::None, - Match::None, - ); - let any_git = !self.0.opts.require_git || self.parents().any(|ig| ig.0.has_git); + pub(crate) fn matched_ignore<'a>( + &'a self, + path: &Path, + is_dir: bool, + ) -> Match> { + let ( + mut m_custom_ignore, + mut m_ignore, + mut m_gi, + mut m_gi_exclude, + mut m_explicit, + ) = (Match::None, Match::None, Match::None, Match::None, Match::None); + let any_git = !self.inner.opts.require_git + || self.parents().any(|ig| ig.inner.has_git); let mut saw_git = false; - for ig in self.parents().take_while(|ig| !ig.0.is_absolute_parent) { + for ig in self.parents().take_while(|ig| !ig.inner.is_absolute_parent) + { if m_custom_ignore.is_none() { - m_custom_ignore = - ig.0.custom_ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_custom_ignore = ig + .inner + .custom_ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if m_ignore.is_none() { - m_ignore = - ig.0.ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_ignore = ig + .inner + .ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi.is_none() { - m_gi = - ig.0.git_ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi = ig + .inner + .git_ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi_exclude.is_none() { - m_gi_exclude = - ig.0.git_exclude_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi_exclude = ig + .inner + .git_exclude_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } - saw_git = saw_git || ig.0.has_git; + saw_git = saw_git || ig.inner.has_git; } - if self.0.opts.parents { - if let Some(_) = self.absolute_base() { - // CHANGED: We removed a code path that rewrote the `path` to be relative to - // `self.absolute_base()` because it assumed that the every path is inside the base - // which is not the case for us as we use `WalkBuilder#add` to add roots outside of the - // base. - for ig in self.parents().skip_while(|ig| !ig.0.is_absolute_parent) { + if self.inner.opts.parents { + if let Some(abs_parent_path) = self.absolute_base() { + // What we want to do here is take the absolute base path of + // this directory and join it with the path we're searching. + // The main issue we want to avoid is accidentally duplicating + // directory components, so we try to strip any common prefix + // off of `path`. Overall, this seems a little ham-fisted, but + // it does fix a nasty bug. It should do fine until we overhaul + // this crate. + let path = abs_parent_path.join( + self.parents() + .take_while(|ig| !ig.inner.is_absolute_parent) + .last() + .map_or(path, |ig| { + // This is a weird special case when ripgrep users + // search with just a `.`, as some tools do + // automatically (like consult). In this case, if + // we don't bail out now, the code below will strip + // a leading `.` from `path`, which might mangle + // a hidden file name! + if ig.inner.dir.as_path() == Path::new(".") { + return path; + } + let without_dot_slash = strip_if_is_prefix( + "./", + ig.inner.dir.as_path(), + ); + let relative_base = + strip_if_is_prefix(without_dot_slash, path); + strip_if_is_prefix("/", relative_base) + }), + ); + + for ig in self + .parents() + .skip_while(|ig| !ig.inner.is_absolute_parent) + { if m_custom_ignore.is_none() { - m_custom_ignore = - ig.0.custom_ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_custom_ignore = ig + .inner + .custom_ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if m_ignore.is_none() { - m_ignore = - ig.0.ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_ignore = ig + .inner + .ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi.is_none() { - m_gi = - ig.0.git_ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi = ig + .inner + .git_ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi_exclude.is_none() { - m_gi_exclude = - ig.0.git_exclude_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi_exclude = ig + .inner + .git_exclude_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } - saw_git = saw_git || ig.0.has_git; + saw_git = saw_git || ig.inner.has_git; } } } - for gi in self.0.explicit_ignores.iter().rev() { - // CHANGED: We need to make sure that the explicit gitignore rules apply to the path - // - // path = Is the current file/folder we are traversing - // gi.path() = Is the path of the custom gitignore file - // - // E.g.: If we have a custom rule for `/src/utils` with `**/*`, and we are looking at - // just `/src`, then the `**/*` rules do not apply to this folder, so we can - // ignore the current custom gitignore file. - // - if !path.starts_with(gi.path()) { - continue; - } + for gi in self.inner.explicit_ignores.iter().rev() { if !m_explicit.is_none() { break; } + // CHANGED: We need to make sure that the explicit gitignore rules + // apply to the path + // + // path = Is the current file/folder we are traversing + // gi.path() = Is the path of the custom gitignore file + // + // E.g.: If we have a custom rule for `/src/utils` with `**/*`, and + // we are looking at just `/src`, then the `**/*` rules do + // not apply to this folder, so we can ignore the current + // custom gitignore file. + if !path.starts_with(gi.path()) { + continue; + } m_explicit = gi.matched(&path, is_dir).map(IgnoreMatch::gitignore); } let m_global = if any_git { - self.0 + self.inner .git_global_matcher .matched(&path, is_dir) .map(IgnoreMatch::gitignore) @@ -504,59 +690,90 @@ impl Ignore { Match::None }; - // CHANGED: We added logic to configure an order in which the ignore files are respected and - // allowed a whitelist in a later file to overrule a block on an earlier file. + // CHANGED: We added logic to configure an order in which the ignore + // files are respected. Explicitly added ignores (via + // `WalkBuilder::add_gitignore`) take precedence over all ignore files + // found on disk, and the first source with a definitive answer wins. let order = [ // Manually added ignores - &m_explicit, + m_explicit, // .custom-ignore - &m_custom_ignore, + m_custom_ignore, // .ignore - &m_ignore, + m_ignore, // .gitignore - &m_gi, + m_gi, // .git/info/exclude - &m_gi_exclude, + m_gi_exclude, // Global gitignore - &m_global, + m_global, ]; - for check in order.into_iter() { - if check.is_none() { - continue; + if !check.is_none() { + return check; } - - return check.clone(); } - - m_explicit + Match::None } /// Returns an iterator over parent ignore matchers, including this one. pub(crate) fn parents(&self) -> Parents<'_> { - Parents(Some(self)) + Parents(Some(IgnoreRef { inner: &self.inner })) } /// Returns the first absolute path of the first absolute parent, if /// one exists. fn absolute_base(&self) -> Option<&Path> { - self.0.absolute_base.as_ref().map(|p| &***p) + self.absolute_base.as_ref().map(|p| &***p) + } +} + +/// State for tracking what kinds of files ripgrep is interested in for a +/// given directory. +/// +/// This is computed over the entire set of files in a directory instead of +/// trying to stat each file individually. If a file is present, it's only then +/// that we stat it for more information, instead of relying on the stat to +/// determine its existence. +#[derive(Debug)] +struct IgnoreFilesFound { + has_ignore: bool, + has_git_ignore: bool, + has_git_dir: bool, + has_jj_dir: bool, + custom_ignore_files: Vec, +} + +#[derive(Clone, Copy)] +pub(crate) struct IgnoreRef<'a> { + inner: &'a IgnoreInner, +} + +impl IgnoreRef<'_> { + pub(crate) fn path(&self) -> &Path { + &self.inner.dir + } + + pub(crate) fn is_absolute_parent(&self) -> bool { + self.inner.is_absolute_parent } } /// An iterator over all parents of an ignore matcher, including itself. -/// -/// The lifetime `'a` refers to the lifetime of the initial `Ignore` matcher. -pub(crate) struct Parents<'a>(Option<&'a Ignore>); +pub(crate) struct Parents<'a>(Option>); impl<'a> Iterator for Parents<'a> { - type Item = &'a Ignore; + type Item = IgnoreRef<'a>; - fn next(&mut self) -> Option<&'a Ignore> { + fn next(&mut self) -> Option> { match self.0.take() { None => None, Some(ig) => { - self.0 = ig.0.parent.as_ref(); + self.0 = ig + .inner + .parent + .as_deref() + .map(|inner| IgnoreRef { inner }); Some(ig) } } @@ -645,33 +862,42 @@ impl IgnoreBuilder { } gi } else { - log::debug!("ignoring global gitignore file because CWD is not known"); + log::debug!( + "ignoring global gitignore file because CWD is not known" + ); Gitignore::empty() }; - Ignore(Arc::new(IgnoreInner { - compiled: Arc::new(RwLock::new(HashMap::new())), - dir: self.dir.clone(), - overrides: self.overrides.clone(), - types: self.types.clone(), - parent: None, - is_absolute_parent: true, + Ignore { + inner: Arc::new(IgnoreInner { + compiled: Arc::new(RwLock::new(HashMap::new())), + dir: self.dir.clone(), + overrides: self.overrides.clone(), + types: self.types.clone(), + parent: None, + is_absolute_parent: true, + global_gitignores_relative_to, + explicit_ignores: Arc::new(self.explicit_ignores.clone()), + custom_ignore_filenames: Arc::new( + self.custom_ignore_filenames.clone(), + ), + custom_ignore_matcher: Gitignore::empty(), + ignore_matcher: Gitignore::empty(), + git_global_matcher: Arc::new(git_global_matcher), + git_ignore_matcher: Gitignore::empty(), + git_exclude_matcher: Gitignore::empty(), + has_git: false, + opts: self.opts, + }), absolute_base: None, - global_gitignores_relative_to, - explicit_ignores: Arc::new(self.explicit_ignores.clone()), - custom_ignore_filenames: Arc::new(self.custom_ignore_filenames.clone()), - custom_ignore_matcher: Gitignore::empty(), - ignore_matcher: Gitignore::empty(), - git_global_matcher: Arc::new(git_global_matcher), - git_ignore_matcher: Gitignore::empty(), - git_exclude_matcher: Gitignore::empty(), - has_git: false, - opts: self.opts, - })) + } } /// Set the current directory used for matching global gitignores. - pub(crate) fn current_dir(&mut self, cwd: impl Into) -> &mut IgnoreBuilder { + pub(crate) fn current_dir( + &mut self, + cwd: impl Into, + ) -> &mut IgnoreBuilder { self.global_gitignores_relative_to = Some(cwd.into()); self } @@ -681,7 +907,10 @@ impl IgnoreBuilder { /// By default, no override matcher is used. /// /// This overrides any previous setting. - pub(crate) fn overrides(&mut self, overrides: Override) -> &mut IgnoreBuilder { + pub(crate) fn overrides( + &mut self, + overrides: Override, + ) -> &mut IgnoreBuilder { self.overrides = Arc::new(overrides); self } @@ -712,8 +941,7 @@ impl IgnoreBuilder { &mut self, file_name: S, ) -> &mut IgnoreBuilder { - self.custom_ignore_filenames - .push(file_name.as_ref().to_os_string()); + self.custom_ignore_filenames.push(file_name.as_ref().to_os_string()); self } @@ -725,6 +953,11 @@ impl IgnoreBuilder { self } + /// Whether ignoring hidden files is enabled or not. + pub(crate) fn is_hidden(&self) -> bool { + self.opts.hidden + } + /// Enables reading `.ignore` files. /// /// `.ignore` files have the same semantics as `gitignore` files and are @@ -795,7 +1028,10 @@ impl IgnoreBuilder { /// Process ignore files case insensitively /// /// This is disabled by default. - pub(crate) fn ignore_case_insensitive(&mut self, yes: bool) -> &mut IgnoreBuilder { + pub(crate) fn ignore_case_insensitive( + &mut self, + yes: bool, + ) -> &mut IgnoreBuilder { self.opts.ignore_case_insensitive = yes; self } @@ -854,7 +1090,10 @@ pub(crate) fn create_gitignore>( /// them when multiple repositories are searched. /// /// Some I/O errors are ignored. -fn resolve_git_commondir(dir: &Path, git_type: Option) -> Result> { +fn resolve_git_commondir( + dir: &Path, + git_type: Option, +) -> Result> { let git_dir_path = || dir.join(".git"); let git_dir = git_dir_path(); if !git_type.map_or(false, |ft| ft.is_file()) { @@ -899,15 +1138,20 @@ fn resolve_git_commondir(dir: &Path, git_type: Option) -> Result + ?Sized>(prefix: &'a P, path: &'a Path) -> &'a Path { +fn strip_if_is_prefix<'a, P: AsRef + ?Sized>( + prefix: &'a P, + path: &'a Path, +) -> &'a Path { strip_prefix(prefix, path).map_or(path, |p| p) } #[cfg(test)] mod tests { - use std::{io::Write, path::Path}; + use std::{io::Write, path::Path, sync::Arc}; - use crate::{Error, dir::IgnoreBuilder, gitignore::Gitignore, tests::TempDir}; + use crate::{ + Error, dir::IgnoreBuilder, gitignore::Gitignore, tests::TempDir, + }; fn wfile>(path: P, contents: &str) { let mut file = std::fs::File::create(path).unwrap(); @@ -936,11 +1180,11 @@ mod tests { let (gi, err) = Gitignore::new(td.path().join("not-an-ignore")); assert!(err.is_none()); - let (ig, err) = IgnoreBuilder::new() - .add_ignore(gi) - .build() - .add_child(td.path()); + let (ig, err) = + IgnoreBuilder::new().add_ignore(gi).build().add_child(td.path()); assert!(err.is_none()); + // CHANGED: Explicit ignores only apply to paths inside the directory + // of the ignore file, so we have to match against full paths. assert!(ig.matched(td.path().join("foo"), false).is_ignore()); assert!(ig.matched(td.path().join("bar"), false).is_whitelist()); assert!(ig.matched(td.path().join("baz"), false).is_none()); @@ -1211,15 +1455,152 @@ mod tests { let (ig2, err) = ig1.add_child("src"); assert!(err.is_none()); - // CHANGED: These test cases do not make sense for us as we never call the Ignore with - // relative paths. - assert!(ig1.matched("llvm", true).is_ignore()); - assert!(ig2.matched("llvm", true).is_ignore()); + assert!(ig1.matched("llvm", true).is_none()); + assert!(ig2.matched("llvm", true).is_none()); assert!(ig2.matched("src/llvm", true).is_none()); assert!(ig2.matched("foo", false).is_ignore()); assert!(ig2.matched("src/foo", false).is_ignore()); } + #[test] + fn absolute_parent_matchers_are_cached_across_roots() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("src/build")); + mkdirp(td.path().join("tests/build")); + wfile(td.path().join(".gitignore"), "tests/**/build/\n"); + + let ig0 = IgnoreBuilder::new().build(); + let (src_parents, err) = ig0.add_parents(td.path().join("src")); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(td.path().join("src")); + assert!(err.is_none()); + let (tests_parents, err) = ig0.add_parents(td.path().join("tests")); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(td.path().join("tests")); + assert!(err.is_none()); + + assert!(Arc::ptr_eq(&src_parents.inner, &tests_parents.inner)); + assert!(src.matched("build", true).is_none()); + assert!(tests.matched("build", true).is_ignore()); + } + + /// Parent matchers are shared across search roots, but path rewriting for + /// absolute parents must use each root's own base path. Otherwise a rule + /// like `src/invalid` is matched against the wrong absolute path when + /// `src` is searched before a sibling root (e.g. `tests`). + /// + /// Paths passed to `matched` use the same relative layout as `Walk` when + /// roots are given as relative directory names. + /// + /// Regression for: https://github.com/BurntSushi/ripgrep/issues/3376 + /// and https://github.com/BurntSushi/ripgrep/issues/3419 + #[test] + fn multi_root_gitignore_order_independent() { + let td = tmpdir(); + let cwd = std::env::current_dir().unwrap(); + // Use paths relative to CWD like the CLI walk does for `rg pat src tests`. + let root = td.path().strip_prefix(&cwd).unwrap_or(td.path()); + let src_root = root.join("src"); + let tests_root = root.join("tests"); + + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("src")); + mkdirp(td.path().join("tests")); + wfile(td.path().join(".gitignore"), "src/invalid\n"); + wfile(td.path().join("src/invalid"), "x"); + wfile(td.path().join("src/valid"), "x"); + wfile(td.path().join("tests/valid"), "x"); + + let ig0 = IgnoreBuilder::new().build(); + + // Historically buggy order: search `src` first, then `tests`. + let (src_parents, err) = ig0.add_parents(&src_root); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(&src_root); + assert!(err.is_none()); + let (tests_parents, err) = ig0.add_parents(&tests_root); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(&tests_root); + assert!(err.is_none()); + + assert!(Arc::ptr_eq(&src_parents.inner, &tests_parents.inner)); + // Each root must carry its own absolute_base even though inners are shared. + assert_ne!( + src.absolute_base.as_ref().unwrap().as_path(), + tests.absolute_base.as_ref().unwrap().as_path() + ); + assert!( + src.matched(src_root.join("invalid"), false).is_ignore(), + "parent .gitignore must apply for the src root even when \ + another root was prepared in the same process" + ); + assert!(src.matched(src_root.join("valid"), false).is_none()); + assert!(tests.matched(tests_root.join("valid"), false).is_none()); + + // Reverse order should behave the same way. + let ig0 = IgnoreBuilder::new().build(); + let (tests_parents, err) = ig0.add_parents(&tests_root); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(&tests_root); + assert!(err.is_none()); + let (src_parents, err) = ig0.add_parents(&src_root); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(&src_root); + assert!(err.is_none()); + + assert!(src.matched(src_root.join("invalid"), false).is_ignore()); + assert!(src.matched(src_root.join("valid"), false).is_none()); + assert!(tests.matched(tests_root.join("valid"), false).is_none()); + } + + /// Same multi-root / order issue for non-git ignore files (e.g. `.rgignore` + /// via custom ignore names). + /// + /// Regression for: https://github.com/BurntSushi/ripgrep/issues/3320 + #[test] + fn multi_root_custom_ignore_order_independent() { + let td = tmpdir(); + let cwd = std::env::current_dir().unwrap(); + let root = td.path().strip_prefix(&cwd).unwrap_or(td.path()); + let alpha_root = root.join("alpha"); + let beta_root = root.join("beta"); + + mkdirp(td.path().join("alpha")); + mkdirp(td.path().join("beta")); + wfile(td.path().join(".rgignore"), "beta/**/*.svg\n"); + wfile(td.path().join("alpha/a.txt"), "x"); + wfile(td.path().join("beta/x.svg"), "x"); + + let ig0 = IgnoreBuilder::new() + .add_custom_ignore_filename(".rgignore") + .ignore(false) + .git_ignore(false) + .git_global(false) + .git_exclude(false) + .build(); + + let (alpha_parents, err) = ig0.add_parents(&alpha_root); + assert!(err.is_none()); + let (alpha, err) = alpha_parents.add_child(&alpha_root); + assert!(err.is_none()); + let (beta_parents, err) = ig0.add_parents(&beta_root); + assert!(err.is_none()); + let (beta, err) = beta_parents.add_child(&beta_root); + assert!(err.is_none()); + + assert_ne!( + alpha.absolute_base.as_ref().unwrap().as_path(), + beta.absolute_base.as_ref().unwrap().as_path() + ); + assert!(alpha.matched(alpha_root.join("a.txt"), false).is_none()); + assert!( + beta.matched(beta_root.join("x.svg"), false).is_ignore(), + "parent .rgignore must apply for the beta root regardless of \ + which root was set up first" + ); + } + #[test] fn git_info_exclude_in_linked_worktree() { let td = tmpdir(); @@ -1227,16 +1608,14 @@ mod tests { mkdirp(git_dir.join("info")); wfile(git_dir.join("info/exclude"), "ignore_me"); mkdirp(git_dir.join("worktrees/linked-worktree")); - let commondir_path = || git_dir.join("worktrees/linked-worktree/commondir"); + let commondir_path = + || git_dir.join("worktrees/linked-worktree/commondir"); mkdirp(td.path().join("linked-worktree")); let worktree_git_dir_abs = format!( "gitdir: {}", git_dir.join("worktrees/linked-worktree").to_str().unwrap(), ); - wfile( - td.path().join("linked-worktree/.git"), - &worktree_git_dir_abs, - ); + wfile(td.path().join("linked-worktree/.git"), &worktree_git_dir_abs); // relative commondir wfile(commondir_path(), "../.."); diff --git a/crates/ignore/src/gitignore.rs b/crates/ignore/src/gitignore.rs index f822d8390..8824139b5 100644 --- a/crates/ignore/src/gitignore.rs +++ b/crates/ignore/src/gitignore.rs @@ -102,7 +102,9 @@ impl Gitignore { /// /// Note that I/O errors are ignored. For more granular control over /// errors, use `GitignoreBuilder`. - pub fn new>(gitignore_path: P) -> (Gitignore, Option) { + pub fn new>( + gitignore_path: P, + ) -> (Gitignore, Option) { let path = gitignore_path.as_ref(); let parent = path.parent().unwrap_or(Path::new("/")); let mut builder = GitignoreBuilder::new(parent); @@ -123,6 +125,17 @@ impl Gitignore { /// The global config file path is specified by git's `core.excludesFile` /// config option. /// + /// # Behavior + /// + /// This routine does its best to discover any global git exclude files. + /// This will try to parse out the `excludesFile` config option in your + /// global git configuration, if necessary. + /// + /// The specific things this routine tries (which are subject to change + /// based on how git behaves) are: + /// + /// + /// /// Git's config file location is `$HOME/.gitconfig`. If `$HOME/.gitconfig` /// does not exist or does not specify `core.excludesFile`, then /// `$XDG_CONFIG_HOME/git/ignore` is read. If `$XDG_CONFIG_HOME` is not @@ -145,7 +158,8 @@ impl Gitignore { num_ignores: 0, num_whitelists: 0, matches: None, - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: false, } } @@ -190,7 +204,11 @@ impl Gitignore { /// determined by a common suffix of the directory containing this /// gitignore) is stripped. If there is no common suffix/prefix overlap, /// then `path` is assumed to be relative to this matcher. - pub fn matched>(&self, path: P, is_dir: bool) -> Match<&Glob> { + pub fn matched>( + &self, + path: P, + is_dir: bool, + ) -> Match<&Glob> { if self.is_empty() { return Match::None; } @@ -243,11 +261,16 @@ impl Gitignore { } /// Like matched, but takes a path that has already been stripped. - fn matched_stripped>(&self, path: P, is_dir: bool) -> Match<&Glob> { + fn matched_stripped>( + &self, + path: P, + is_dir: bool, + ) -> Match<&Glob> { if self.is_empty() { return Match::None; } - // CHANGED: Rules marked as only_on_files can not match against directories. + // CHANGED: Rules marked as only_on_files can not match against + // directories. if self.only_on_files && is_dir { return Match::None; } @@ -270,7 +293,10 @@ impl Gitignore { /// Strips the given path such that it's suitable for matching with this /// gitignore matcher. - fn strip<'a, P: 'a + AsRef + ?Sized>(&'a self, path: &'a P) -> &'a Path { + fn strip<'a, P: 'a + AsRef + ?Sized>( + &'a self, + path: &'a P, + ) -> &'a Path { let mut path = path.as_ref(); // A leading ./ is completely superfluous. We also strip it from // our gitignore root path, so we need to strip it from our candidate @@ -326,7 +352,8 @@ impl GitignoreBuilder { globs: vec![], case_insensitive: false, allow_unclosed_class: true, - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: false, } } @@ -337,18 +364,21 @@ impl GitignoreBuilder { pub fn build(&self) -> Result { let nignore = self.globs.iter().filter(|g| !g.is_whitelist()).count(); let nwhite = self.globs.iter().filter(|g| g.is_whitelist()).count(); - let set = self.builder.build().map_err(|err| Error::Glob { - glob: None, - err: err.to_string(), - })?; + let set = self + .builder + .build() + .map_err(|err| Error::Glob { glob: None, err: err.to_string() })?; Ok(Gitignore { set, root: self.root.clone(), globs: self.globs.clone(), num_ignores: nignore as u64, num_whitelists: nwhite as u64, - matches: Some(Arc::new(Pool::new(|| vec![]))), - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + matches: Some(Arc::new( + Pool::with_available_parallelism_capacity(|| vec![]), + )), + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: self.only_on_files, }) } @@ -411,11 +441,8 @@ impl GitignoreBuilder { // Match Git's handling of .gitignore files that begin with the Unicode BOM const UTF8_BOM: &str = "\u{feff}"; - let line = if i == 0 { - line.trim_start_matches(UTF8_BOM) - } else { - &line - }; + let line = + if i == 0 { line.trim_start_matches(UTF8_BOM) } else { &line }; if let Err(err) = self.add_line(Some(path.to_path_buf()), &line) { errs.push(err.tagged(path, lineno)); @@ -537,7 +564,10 @@ impl GitignoreBuilder { /// affected. /// /// This is disabled by default. - pub fn case_insensitive(&mut self, yes: bool) -> Result<&mut GitignoreBuilder, Error> { + pub fn case_insensitive( + &mut self, + yes: bool, + ) -> Result<&mut GitignoreBuilder, Error> { // TODO: This should not return a `Result`. Fix this in the next semver // release. self.case_insensitive = yes; @@ -556,7 +586,10 @@ impl GitignoreBuilder { /// modes since the glob parser becomes more permissive. You might want to /// enable this when compatibility (e.g., with POSIX glob implementations) /// is more important than good error messages. - pub fn allow_unclosed_class(&mut self, yes: bool) -> &mut GitignoreBuilder { + pub fn allow_unclosed_class( + &mut self, + yes: bool, + ) -> &mut GitignoreBuilder { self.allow_unclosed_class = yes; self } @@ -576,32 +609,56 @@ impl GitignoreBuilder { /// /// Note that the file path returned may not exist. pub fn gitconfig_excludes_path() -> Option { - // git supports $HOME/.gitconfig and $XDG_CONFIG_HOME/git/config. Notably, - // both can be active at the same time, where $HOME/.gitconfig takes - // precedent. So if $HOME/.gitconfig defines a `core.excludesFile`, then - // we're done. - match gitconfig_home_contents().and_then(|x| parse_excludes_file(&x)) { - Some(path) => return Some(path), - None => {} + // When GIT_CONFIG_GLOBAL is set, it replaces both $HOME/.gitconfig and + // $XDG_CONFIG_HOME/git/config (per git 2.32+). Otherwise, git supports + // $HOME/.gitconfig and $XDG_CONFIG_HOME/git/config simultaneously, where + // $HOME/.gitconfig takes precedent. + gitconfig_global_env_contents() + .and_then(|x| parse_excludes_file(&x)) + .or_else(|| { + gitconfig_home_contents().and_then(|x| parse_excludes_file(&x)) + }) + .or_else(|| { + gitconfig_xdg_contents().and_then(|x| parse_excludes_file(&x)) + }) + // System-level config has the lowest priority for core.excludesFile. + // GIT_CONFIG_SYSTEM overrides the default /etc/gitconfig path. + .or_else(|| { + gitconfig_system_contents().and_then(|x| parse_excludes_file(&x)) + }) + .or_else(excludes_file_default) +} + +/// Returns the file contents of git's global config file from the path +/// specified by the `GIT_CONFIG_GLOBAL` environment variable. +fn gitconfig_global_env_contents() -> Option> { + let path = std::env::var_os("GIT_CONFIG_GLOBAL").map(PathBuf::from)?; + if path.as_os_str().is_empty() { + return None; } - match gitconfig_xdg_contents().and_then(|x| parse_excludes_file(&x)) { - Some(path) => return Some(path), - None => {} - } - excludes_file_default() + let mut file = BufReader::new(File::open(path).ok()?); + let mut contents = vec![]; + file.read_to_end(&mut contents).ok().map(|_| contents) +} + +/// Returns the file contents of git's system-level config file. +/// +/// Checks `GIT_CONFIG_SYSTEM` first, then falls back to `/etc/gitconfig`. +fn gitconfig_system_contents() -> Option> { + let path = std::env::var_os("GIT_CONFIG_SYSTEM") + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) + .unwrap_or_else(|| PathBuf::from("/etc/gitconfig")); + let mut file = BufReader::new(File::open(path).ok()?); + let mut contents = vec![]; + file.read_to_end(&mut contents).ok().map(|_| contents) } /// Returns the file contents of git's global config file, if one exists, in /// the user's home directory. fn gitconfig_home_contents() -> Option> { - let home = match home_dir() { - None => return None, - Some(home) => home, - }; - let mut file = match File::open(home.join(".gitconfig")) { - Err(_) => return None, - Ok(file) => BufReader::new(file), - }; + let home = home_dir()?; + let mut file = BufReader::new(File::open(home.join(".gitconfig")).ok()?); let mut contents = vec![]; file.read_to_end(&mut contents).ok().map(|_| contents) } @@ -610,19 +667,11 @@ fn gitconfig_home_contents() -> Option> { /// the user's XDG_CONFIG_HOME directory. fn gitconfig_xdg_contents() -> Option> { let path = std::env::var_os("XDG_CONFIG_HOME") - .and_then(|x| { - if x.is_empty() { - None - } else { - Some(PathBuf::from(x)) - } - }) + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) .or_else(|| home_dir().map(|p| p.join(".config"))) - .map(|x| x.join("git/config")); - let mut file = match path.and_then(|p| File::open(p).ok()) { - None => return None, - Some(file) => BufReader::new(file), - }; + .map(|x| x.join("git/config"))?; + let mut file = BufReader::new(File::open(path).ok()?); let mut contents = vec![]; file.read_to_end(&mut contents).ok().map(|_| contents) } @@ -632,13 +681,8 @@ fn gitconfig_xdg_contents() -> Option> { /// Specifically, this respects XDG_CONFIG_HOME. fn excludes_file_default() -> Option { std::env::var_os("XDG_CONFIG_HOME") - .and_then(|x| { - if x.is_empty() { - None - } else { - Some(PathBuf::from(x)) - } - }) + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) .or_else(|| home_dir().map(|p| p.join(".config"))) .map(|x| x.join("git/ignore")) } @@ -667,9 +711,7 @@ fn parse_excludes_file(data: &[u8]) -> Option { re.captures(data, &mut caps); let span = caps.get_group(1)?; let candidate = &data[span]; - std::str::from_utf8(candidate) - .ok() - .map(|s| PathBuf::from(expand_tilde(s))) + std::str::from_utf8(candidate).ok().map(|s| PathBuf::from(expand_tilde(s))) } /// Expands ~ in file paths to the value of $HOME. @@ -831,7 +873,10 @@ mod tests { fn parse_excludes_file4() { let data = bytes("[core]\nexcludesFile = \"~/foo/bar\""); let got = super::parse_excludes_file(&data); - assert_eq!(path_string(got.unwrap()), super::expand_tilde("~/foo/bar")); + assert_eq!( + path_string(got.unwrap()), + super::expand_tilde("~/foo/bar") + ); } #[test] diff --git a/crates/ignore/src/incremental.rs b/crates/ignore/src/incremental.rs new file mode 100644 index 000000000..5030b0d00 --- /dev/null +++ b/crates/ignore/src/incremental.rs @@ -0,0 +1,1286 @@ +use std::{ + collections::HashMap, + path::{Path, PathBuf}, + sync::OnceLock, +}; + +use crate::{ + Error, Match, PartialErrorBuilder, dir::Ignore, pathutil::is_hidden_path, +}; + +/// A cached matcher for checking paths against hierarchical ignore files. +/// +/// An `IncrementalIgnore` is built from a [`crate::WalkBuilder`]. Unlike a +/// recursive walk, it can check individual paths while still respecting the +/// ignore files in every relevant parent directory. Matchers for directories +/// are compiled on first use and then retained for later queries. +/// Each matcher corresponds to exactly one root configured on the builder, +/// and paths passed to it are interpreted relative to that root. +/// A matcher for the special `-` root representing standard input is inert +/// and always returns a non-match. +/// +/// The matcher checks path-based filters in the same precedence order as +/// a traversal. This includes glob overrides, `.ignore`, `.gitignore`, +/// `.git/info/exclude`, global and explicitly added ignore files, custom +/// ignore file names and file type selections. It does not apply filters that +/// require a directory entry or other traversal state, such as custom entry +/// predicates. Hidden-file detection, minimum and maximum depth limits and the +/// maximum file size are applied. +/// +/// A matcher is a snapshot at directory granularity. Once the ignore files in +/// a directory have been loaded, edits to those files are not observed. Build +/// a new matcher to reload them. +/// +/// # Warning +/// +/// The incremental path checking here necessarily needs to do a lot more work +/// per path matched. Callers should _not_ use this to run directory traversal. +/// This is intended to avoid the work of re-traversing an entire directory +/// tree when only a few changes are detected. (For example, in response to +/// file additions or deletions.) +/// +/// # Example +/// +/// ```rust,no_run +/// use ignore::WalkBuilder; +/// +/// let mut builder = WalkBuilder::new("."); +/// builder.add_custom_ignore_filename(".rgignore"); +/// let mut matchers = builder.build_matchers(); +/// let matcher = &mut matchers[0]; +/// +/// if matcher.matched("src/generated.rs", false).is_ignore() { +/// println!("ignored"); +/// } +/// ``` +#[derive(Clone, Debug)] +pub struct IncrementalIgnore { + /// The root exactly as it was given to `WalkBuilder`. + root: PathBuf, + /// The normalized root used only by the opt-in normalization routine. + normalized_root: OnceLock>, + /// The matcher for the configured root directory, loaded on first use. + ignore: RootIgnore, + /// Directory paths relative to `root`, excluding the root itself. + dirs: HashMap, + /// Options for additional filtering beyond gitignore. + options: IncrementalIgnoreOptions, +} + +/// The options for a matcher, mostly meant to duplicate as much as we can from +/// `WalkParallel`. +#[derive(Clone, Debug)] +pub(crate) struct IncrementalIgnoreOptions { + pub(crate) min_depth: Option, + pub(crate) max_depth: Option, + pub(crate) max_filesize: Option, + pub(crate) hidden: bool, + pub(crate) follow_links: bool, +} + +#[derive(Clone, Debug)] +enum RootIgnore { + Unloaded(Ignore), + Loaded(Ignore), + NotDirectory, + Stdin, +} + +/// Cached traversal state for a directory relative to the configured root. +/// +/// The presence of an entry means that the directory and every ancestor +/// between it and the root have already been checked. +#[derive(Clone, Debug)] +enum CachedDir { + /// The directory may be descended into. The matcher includes the ignore + /// rules loaded through this directory and is therefore the matcher to use + /// for its children. + Allowed(Ignore), + /// The directory may not be descended into because it was ignored by a + /// path rule or hidden-file filtering. Every descendant is consequently + /// ignored, and ignore files inside this directory are not loaded. + Ignored, +} + +impl IncrementalIgnore { + pub(crate) fn new( + root: PathBuf, + ignore: Ignore, + options: IncrementalIgnoreOptions, + ) -> IncrementalIgnore { + // File traversal special cases `-` to search stdin, so we recognize + // it here for completeness too. In particular, we really want + // `WalkBuilder::build_matchers` to return a matcher for every root, + // even when it's a simple file (handled automatically) or when it's + // stdin (necessarily special cased). + // + // If callers need to search a file or directory named `-`, then they + // can use `./-`. As is the case for file traversal too. + let ignore = if root == Path::new("-") { + RootIgnore::Stdin + } else { + RootIgnore::Unloaded(ignore) + }; + IncrementalIgnore { + root, + normalized_root: OnceLock::new(), + ignore, + dirs: HashMap::new(), + options, + } + } + + /// Return the root that paths matched by this matcher are relative to. + pub fn root(&self) -> &Path { + &self.root + } + + /// Normalize `path` and return it relative to this matcher's root. + /// + /// This returns `None` when `path` cannot be made absolute or + /// when it is known to be outside this matcher's root. Unlike + /// [`IncrementalIgnore::matched`], this performs absolute path conversion, + /// lexical normalization and allocation. It is intended as an opt-in + /// convenience for callers that do not already have root-relative paths. + /// + /// Note that `.` is interpreted relative to the process level current + /// working directory. It is _not_ interpreted relative to the root of + /// this matcher. + /// + /// Note also that this may reject paths that only differ in casing. For + /// example, if the root path for this matcher is `/FOO` but the provided + /// path is `/foo/bar`, then this may return `None`. Callers must ensure + /// casing is consistent between the path provided and the root path for + /// this matcher. + pub fn normalize>(&self, path: P) -> Option { + if matches!(self.ignore, RootIgnore::Stdin) { + return None; + } + let path = normalize_absolute(path.as_ref())?; + let root = self + .normalized_root + .get_or_init(|| normalize_absolute(&self.root)) + .as_ref()?; + path.strip_prefix(root).ok().map(Path::to_path_buf) + } + + /// Match a root-relative path against ignore files in its directory and + /// all relevant parent directories. + /// + /// `is_dir` should be true when `path` should be matched as a directory. + /// + /// For the return value, use [`IncrementalMatch::is_ignore`], + /// [`IncrementalMatch::is_whitelist`] or [`IncrementalMatch::is_none`] to + /// inspect it. + /// + /// Matchers for previously unseen directories are loaded and cached during + /// this call. Errors encountered while loading ignore files are logged. To + /// receive those errors, use [`IncrementalIgnore::matched_with_errors`]. + /// + /// `path` must be relative to this matcher's root and must not contain a + /// parent directory (`..`) component. Behavior is unspecified when these + /// preconditions are violated. Callers with an absolute path or with a + /// path containing `.` or `..` may use [`IncrementalIgnore::normalize`] + /// to get a path satisfying these preconditions. In all cases, a relative + /// path is *assumed* to be relative to the root of this ignore matcher. + /// + /// In general, it is intended that callers doing recursive directory + /// traversal on the root of this matcher may provide relative paths to + /// this routine *without* calling [`IncrementalIgnore::normalize`]. + /// + /// The empty path represents the explicitly configured root and also + /// returns non-match, consistent with recursive traversal where a root is + /// always treated as being at depth zero. + pub fn matched>( + &mut self, + path: P, + is_dir: bool, + ) -> IncrementalMatch { + let (matched, err) = self.matched_with_errors(path, is_dir); + if let Some(err) = err { + log::debug!("error while loading ignore files: {err}"); + } + matched + } + + /// Match a root-relative path and return errors encountered while loading + /// ignore files. + /// + /// This is equivalent to [`IncrementalIgnore::matched`], except that + /// it returns any errors from newly loaded ignore files. Loading can + /// partially succeed, so valid rules are always applied to the returned + /// match even when an error is present. + pub fn matched_with_errors>( + &mut self, + path: P, + is_dir: bool, + ) -> (IncrementalMatch, Option) { + let mut errs = PartialErrorBuilder::default(); + let matched = + self.matched_with_errors_impl(path.as_ref(), is_dir, &mut errs); + (matched, errs.into_error_option()) + } + + fn matched_with_errors_impl( + &mut self, + relative: &Path, + is_dir: bool, + errs: &mut PartialErrorBuilder, + ) -> IncrementalMatch { + // We short-circuit here when our matcher corresponds to `Stdin` in + // order to always return a non-match. This is somewhat redundant with + // `root_ignore()` which does this too, but we do it here so that it + // always happens, e.g., before depth filtering. + if relative.is_absolute() || matches!(self.ignore, RootIgnore::Stdin) { + return IncrementalMatch::none(is_dir); + } + + let (mut satisfies_min, mut edge_max) = (true, false); + if self.options.min_depth.is_some() || self.options.max_depth.is_some() + { + let depth = relative + .components() + .filter_map(|component| match component { + std::path::Component::CurDir => None, + component => Some(component), + }) + .count(); + satisfies_min = + self.options.min_depth.is_none_or(|min| depth >= min); + let satisfies_max = + self.options.max_depth.is_none_or(|max| depth <= max); + edge_max = self.options.max_depth.is_some_and(|max| depth == max); + // When we have a file that isn't past our min depth, we can give + // up right away. + if !is_dir && !satisfies_min { + return IncrementalMatch::ignore().not_within_depth(); + } + // Same for *anything* that exceeds our max depth. + if !satisfies_max { + return IncrementalMatch::ignore().not_within_depth(); + } + } + + // When the path is invalid in some way, we bail out early to avoid + // potentially doing a stat call below. + let (mut mat, valid) = self + .matched_with_errors_ignore(relative, is_dir, errs) + .map(|mat| (mat, true)) + .unwrap_or_else(|| (IncrementalMatch::none(is_dir), false)); + + if !is_dir + && !mat.is_ignore() + && valid + && let Some(max_filesize) = self.options.max_filesize + { + let path = self.root().join(relative); + let result = if self.options.follow_links { + path.metadata() + } else { + path.symlink_metadata() + }; + match result { + Ok(md) if md.len() > max_filesize => { + return IncrementalMatch::ignore(); + } + Ok(_) => {} + Err(err) => { + // Record the error but otherwise fall through + let err = Error::from(err); + errs.push(err.with_path(path)); + } + } + } + + // We still need to tag our `mat` if it doesn't pass the depth filter, + // or if it's a directory on the edge of a depth filter. This can only + // happen when the path is reported as a directory. Otherwise regular + // files are always handled above. + if is_dir { + if !satisfies_min { + mat = mat.not_within_depth(); + } + if edge_max { + mat = mat.no_descent(); + } + } + mat + } + + fn matched_with_errors_ignore( + &mut self, + relative: &Path, + is_dir: bool, + errs: &mut PartialErrorBuilder, + ) -> Option { + let mut components = relative + .components() + .filter_map(|component| match component { + std::path::Component::CurDir => None, + component => Some(component), + }) + .peekable(); + components.peek()?; + + // If the exact parent is cached, then all of its ancestors have + // already been checked. An allowed cache entry is the matcher to use + // for children of that directory, while an ignored cache entry is + // terminal for every descendant. In the usual case, this avoids both + // walking every component and allocating a relative directory path. + // + // Only try the fast path when the final component is a normal path + // component. Invalid paths are handled by the component walk below. + let has_normal_final_component = + relative.components().next_back().is_some_and(|component| { + matches!(component, std::path::Component::Normal(_)) + }); + if has_normal_final_component && let Some(parent) = relative.parent() { + match self.dirs.get(parent) { + Some(CachedDir::Allowed(ignore)) => { + return Some(self.match_path(ignore, relative, is_dir)); + } + Some(CachedDir::Ignored) => { + return Some(IncrementalMatch::ignore()); + } + None => {} + } + } + + let mut ignore = self.root_ignore(errs)?; + let mut dir = PathBuf::new(); + while let Some(component) = components.next() { + match component { + std::path::Component::ParentDir + | std::path::Component::RootDir + | std::path::Component::Prefix(_) => { + return None; + } + std::path::Component::CurDir => continue, + std::path::Component::Normal(_) => {} + } + if components.peek().is_none() { + break; + } + dir.push(component.as_os_str()); + match self.dirs.get(&dir) { + Some(CachedDir::Allowed(cached)) => { + ignore = cached.clone(); + continue; + } + Some(CachedDir::Ignored) => { + return Some(IncrementalMatch::ignore()); + } + None => {} + } + + let path = self.root.join(&dir); + let mat = ignore.matched(&path, true); + let is_hidden = + self.options.hidden && mat.is_none() && is_hidden_path(&path); + if mat.is_ignore() || is_hidden { + self.dirs.insert(dir.clone(), CachedDir::Ignored); + return Some(IncrementalMatch::ignore()); + } + let (child, err) = ignore.add_child(&path); + errs.maybe_push(err); + self.dirs.insert(dir.clone(), CachedDir::Allowed(child.clone())); + ignore = child; + } + + Some(self.match_path(&ignore, relative, is_dir)) + } + + fn match_path( + &self, + ignore: &Ignore, + relative: &Path, + is_dir: bool, + ) -> IncrementalMatch { + let path = self.root.join(relative); + let mut mat = IncrementalMatch::from_match( + ignore.matched(&path, is_dir).map(|_| ()), + is_dir, + ); + // Whether a file is hidden or not has low precedence in filtering. We + // only check it if we haven't matched anything above. This permits + // callers to whitelist hidden files or directories. + if self.options.hidden && mat.is_none() && is_hidden_path(&path) { + mat = IncrementalMatch::ignore(); + } + mat + } + + fn root_ignore( + &mut self, + errs: &mut PartialErrorBuilder, + ) -> Option { + let ignore = match self.ignore { + RootIgnore::Unloaded(ref ignore) => ignore.clone(), + RootIgnore::Loaded(ref ignore) => return Some(ignore.clone()), + RootIgnore::NotDirectory | RootIgnore::Stdin => return None, + }; + if !self.root.is_dir() { + self.ignore = RootIgnore::NotDirectory; + return None; + } + + let (parents, err) = ignore.add_parents(&self.root); + errs.maybe_push(err); + let (root, err) = parents.add_child(&self.root); + errs.maybe_push(err); + self.ignore = RootIgnore::Loaded(root.clone()); + Some(root) + } +} + +/// The result of an incremental match. +/// +/// This is similar to [`Match`] in that it reports whether a file path should +/// be ignored, whitelisted or didn't match anything at all. It also has extra +/// data, such as whether a directory matched but should not be descended into. +/// +/// Generally speaking, callers that only care about specific files can stick +/// to the `is_none()`, `is_ignore()` or `is_whitelist()` predicates. Directory +/// entries are more complicated because we sometimes want to yield directories +/// to descend into, but not actually visit (e.g., for the minimum depth +/// filter). Or, we may want to yield a directory to visit but not descend into +/// (e.g., for the maximum depth filter). +#[derive(Clone, Debug)] +pub struct IncrementalMatch { + mat: Match<()>, + should_descend: bool, + is_within_depth: bool, +} + +impl IncrementalMatch { + fn none(is_dir: bool) -> IncrementalMatch { + IncrementalMatch { + mat: Match::None, + should_descend: is_dir, + is_within_depth: true, + } + } + + fn ignore() -> IncrementalMatch { + IncrementalMatch { + mat: Match::Ignore(()), + should_descend: false, + is_within_depth: true, + } + } + + fn from_match(mat: Match<()>, is_dir: bool) -> IncrementalMatch { + let should_descend = is_dir && !mat.is_ignore(); + IncrementalMatch { mat, should_descend, is_within_depth: true } + } + + fn no_descent(self) -> IncrementalMatch { + IncrementalMatch { should_descend: false, ..self } + } + + fn not_within_depth(self) -> IncrementalMatch { + IncrementalMatch { is_within_depth: false, ..self } + } + + /// Returns true if the match result didn't match anything. + pub fn is_none(&self) -> bool { + self.mat.is_none() + } + + /// Returns true if the match result implies the path should be ignored. + pub fn is_ignore(&self) -> bool { + self.mat.is_ignore() + } + + /// Returns true if the match result implies the path should be + /// whitelisted. + pub fn is_whitelist(&self) -> bool { + self.mat.is_whitelist() + } + + /// Returns true only when this match corresponds to a directory *and* + /// whether the caller should look inside this directory for additional + /// results. + /// + /// It is possible for a match to report `false` for + /// [`IncrementalMatch::is_ignore`] _and_ `false` for + /// [`IncrementalMatch::should_descend`]. This can occur, for example, when + /// a maximum depth setting allows a directory through, but where none of + /// its children entries should be visited. + /// + /// This is always `false` for a file path that does *not* correspond to a + /// directory. + pub fn should_descend(&self) -> bool { + self.should_descend + } + + /// Returns true only when this result corresponds to an entry that is + /// within the depth filter. + /// + /// This is `false` when a path corresponds to a directory and is less than + /// the minimum depth. In this case, callers should continue looking inside + /// that directory. + /// + /// This is always `true` for a match corresponding to a file path that + /// isn't a directory. + pub fn is_within_depth(&self) -> bool { + self.is_within_depth + } + + /// Inverts the match so that `Ignore` becomes `Whitelist` and + /// `Whitelist` becomes `Ignore`. A non-match remains the same. + pub fn invert(self) -> IncrementalMatch { + IncrementalMatch { mat: self.mat.invert(), ..self } + } +} + +/// Return a lexically normalized absolute representation of `path`. +/// +/// This collapses `.` and `..`, but intentionally does not canonicalize or +/// resolve symlinks. Ignore rules apply to the lexical path, and resolving a +/// symlink below a root could move the result outside of that root. +fn normalize_absolute(path: &Path) -> Option { + let absolute = std::path::absolute(path).ok()?; + let mut normalized = PathBuf::new(); + for component in absolute.components() { + match component { + std::path::Component::CurDir => {} + std::path::Component::ParentDir => { + normalized.pop(); + } + _ => normalized.push(component.as_os_str()), + } + } + Some(normalized) +} + +#[cfg(test)] +mod tests { + use std::{ + fs::{self, File}, + io::Write, + path::{Path, PathBuf}, + }; + + use crate::{ + IncrementalIgnore, IncrementalMatch, WalkBuilder, + overrides::OverrideBuilder, tests::TempDir, types::TypesBuilder, + }; + + use super::CachedDir; + + fn wfile>(path: P, contents: &str) { + let mut file = File::create(path).unwrap(); + file.write_all(contents.as_bytes()).unwrap(); + } + + fn mkdirp>(path: P) { + fs::create_dir_all(path).unwrap(); + } + + fn tmpdir() -> TempDir { + TempDir::new().unwrap() + } + + fn builder>(path: P) -> WalkBuilder { + let mut builder = WalkBuilder::new(path); + builder.git_global(false); + builder + } + + fn one_matcher(builder: &WalkBuilder) -> IncrementalIgnore { + let mut matchers = builder.build_matchers(); + assert_eq!(matchers.len(), 1); + matchers.pop().unwrap() + } + + fn builders, P: AsRef>( + paths: I, + ) -> WalkBuilder { + let mut builder = WalkBuilder::from_iter(paths); + builder.git_global(false); + builder + } + + fn matchers(builder: &WalkBuilder) -> Vec { + builder.build_matchers() + } + + fn matchedf>( + matcher: &mut IncrementalIgnore, + path: P, + ) -> IncrementalMatch { + let (matched, err) = matcher.matched_with_errors(path, false); + assert!(err.is_none(), "unexpected matcher error: {err:?}"); + matched + } + + fn matchedd>( + matcher: &mut IncrementalIgnore, + path: P, + ) -> IncrementalMatch { + let (matched, err) = matcher.matched_with_errors(path, true); + assert!(err.is_none(), "unexpected matcher error: {err:?}"); + matched + } + + // Test that multiple parent ignore files, when nested, are respected. + #[test] + fn nested_parent_gitignores() { + let td = tmpdir(); + let root = td.path().join("project/work"); + mkdirp(td.path().join(".git")); + mkdirp(root.join("src")); + wfile(td.path().join(".gitignore"), "*.tmp\n"); + wfile(td.path().join("project/.gitignore"), "!keep.tmp\nnested.log\n"); + + let mut m = one_matcher(&builder(&root)); + assert_eq!(m.root(), root); + assert!(matchedf(&mut m, "src/drop.tmp").is_ignore()); + assert!(matchedf(&mut m, "src/keep.tmp").is_whitelist()); + assert!(matchedf(&mut m, "src/nested.log").is_ignore()); + assert!(matchedf(&mut m, "src/ok.rs").is_none()); + } + + // Test that an anchored rule in a child directory is matched relative to + // that directory, not relative to the configured root. + #[test] + fn anchored_child_rule_uses_child_root() { + let td = tmpdir(); + let root = td.path().join("root"); + mkdirp(root.join("a")); + wfile(root.join("a/.ignore"), "/foo\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "a/foo").is_ignore()); + assert!(matchedf(&mut m, "a/b/foo").is_none()); + } + + // Test that a leading `./` impacts how the rules are matched. + #[cfg(not(windows))] + #[test] + fn leading_dot_slash_impacts_matching() { + let td = tmpdir(); + let root = td.path().join("root"); + mkdirp(root.join("a")); + wfile(root.join(".ignore"), "/foo\n"); + wfile(root.join("a/.ignore"), "/foo\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "foo").is_ignore()); + assert!(matchedf(&mut m, "./foo").is_none()); + assert!(matchedf(&mut m, "a/allowed").is_none()); + assert!(matchedf(&mut m, "a/foo").is_ignore()); + assert!(matchedf(&mut m, "./a/foo").is_none()); + } + + // Test that custom ignore files are respected. + #[test] + fn parent_ignore_and_custom_ignore() { + let td = tmpdir(); + let root = td.path().join("project/work"); + mkdirp(root.join("src")); + wfile(td.path().join(".ignore"), "*.cache\n"); + wfile(td.path().join(".rgignore"), "*.svg\n"); + wfile(td.path().join("project/.ignore"), "!keep.cache\n"); + wfile(td.path().join("project/.rgignore"), "!keep.svg\n"); + + let mut builder = builder(&root); + builder.add_custom_ignore_filename(".rgignore"); + let mut m = one_matcher(&builder); + assert!(matchedf(&mut m, "src/drop.cache").is_ignore()); + assert!(matchedf(&mut m, "src/keep.cache").is_whitelist()); + assert!(matchedf(&mut m, "src/drop.svg").is_ignore()); + assert!(matchedf(&mut m, "src/keep.svg").is_whitelist()); + } + + // Test that glob overrides take precedence over ignore files. + #[test] + fn glob_overrides_are_applied() { + let td = tmpdir(); + wfile(td.path().join(".ignore"), "keep.rs\n!drop.rs\n"); + + let mut overrides = OverrideBuilder::new(td.path()); + overrides.add("keep.rs").unwrap(); + overrides.add("!drop.rs").unwrap(); + let mut b = builder(td.path()); + b.overrides(overrides.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "keep.rs").is_whitelist()); + assert!(matchedf(&mut m, "drop.rs").is_ignore()); + } + + // Test that an override-ignored directory prevents matching rules below + // it, just as it prevents a traversal from descending into the directory. + #[test] + fn glob_overrides_apply_to_ancestors() { + let td = tmpdir(); + mkdirp(td.path().join("blocked")); + wfile(td.path().join("blocked/.ignore"), "!keep.rs\n"); + + let mut overrides = OverrideBuilder::new(td.path()); + overrides.add("!blocked/").unwrap(); + let mut b = builder(td.path()); + b.overrides(overrides.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "blocked/keep.rs").is_ignore()); + } + + // Test that file type selections are applied to files, but not + // directories. + #[test] + fn file_types_are_applied() { + let td = tmpdir(); + mkdirp(td.path().join("src")); + + let mut types = TypesBuilder::new(); + types.add("rust", "*.rs").unwrap(); + types.select("rust"); + let mut b = builder(td.path()); + b.types(types.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedd(&mut m, "src").is_none()); + assert!(matchedf(&mut m, "src/lib.rs").is_whitelist()); + assert!(matchedf(&mut m, "README.md").is_ignore()); + } + + #[test] + fn directly_ignored_directory_is_not_descended() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("blocked")); + wfile(td.path().join(".gitignore"), "blocked/\n"); + + let mut m = one_matcher(&builder(td.path())); + let dir = matchedd(&mut m, "blocked"); + assert!(dir.is_ignore()); + assert!(!dir.should_descend()); + } + + // Test that when a directory is ignored, anything below it always ignored + // even when there are explicit whitelist rules. This matches directory + // traversal semantics. + #[test] + fn ignored_ancestor_wins() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("blocked")); + mkdirp(td.path().join("open")); + // This is the key line: since the entire `blocked` directory is + // ignored, a proper file traversal won't ever descend into it. So + // `blocked/keep.rs` should be ignored even if there are ignore rules + // "beneath" it that whitelist it. + wfile(td.path().join(".gitignore"), "blocked/\n!blocked/keep.rs\n"); + wfile(td.path().join("blocked/.gitignore"), "!keep.rs\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "blocked/keep.rs").is_ignore()); + assert!(matchedf(&mut m, "blocked/other.rs").is_ignore()); + assert!(matchedf(&mut m, "open/keep.rs").is_none()); + } + + // Test that we respect git boundaries. And that we don't respect git + // boundaries when not configured to do so. + #[test] + fn respects_git_repository_boundaries() { + let td = tmpdir(); + let root = td.path().join("repo/src"); + mkdirp(td.path().join("repo/.git")); + mkdirp(&root); + wfile(td.path().join(".gitignore"), "outside-rule\n"); + wfile(td.path().join(".ignore"), "tool-rule\n"); + wfile(td.path().join("repo/.gitignore"), "inside-rule\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "inside-rule").is_ignore()); + assert!(matchedf(&mut m, "outside-rule").is_none()); + assert!(matchedf(&mut m, "tool-rule").is_ignore()); + + let mut no_git_required = builder(&root); + no_git_required.require_git(false); + let mut m = one_matcher(&no_git_required); + assert!(matchedf(&mut m, "outside-rule").is_ignore()); + } + + // Test that when a gitignore matcher in a parent directory is created, we + // reuse that matcher from memory even if it's changed on disk. + #[test] + fn compiled_matchers_are_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a")); + wfile(td.path().join("a/.ignore"), "*.tmp\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/first.tmp").is_ignore()); + + // Below demonstrates that this new ignore file contents + // aren't actually picked up because it was already loaded. + wfile(td.path().join("a/.ignore"), "!*.tmp\n*.rs\n"); + assert!(matchedf(&mut m, "a/first.tmp").is_ignore()); + assert!(matchedf(&mut m, "a/second.tmp").is_ignore()); + assert!(matchedf(&mut m, "a/keep.rs").is_none()); + // To get the new ignore file, we need to rebuild the matcher. + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/first.tmp").is_whitelist()); + assert!(matchedf(&mut m, "a/second.tmp").is_whitelist()); + assert!(matchedf(&mut m, "a/keep.rs").is_ignore()); + } + + #[test] + fn cached_allowed_parent_matches_children() { + let td = tmpdir(); + mkdirp(td.path().join("a/b")); + wfile(td.path().join("a/b/.ignore"), "ignored\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/b/allowed").is_none()); + assert!(matches!( + m.dirs.get(Path::new("a/b")), + Some(CachedDir::Allowed(_)) + )); + assert!(matchedf(&mut m, "a/b/ignored").is_ignore()); + } + + #[test] + fn cached_ignored_parent_matches_children() { + let td = tmpdir(); + mkdirp(td.path().join("blocked")); + wfile(td.path().join(".ignore"), "blocked/\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "blocked/first").is_ignore()); + assert!(matches!( + m.dirs.get(Path::new("blocked")), + Some(CachedDir::Ignored) + )); + assert!(matchedf(&mut m, "blocked/second").is_ignore()); + } + + #[test] + fn cached_parent_matcher_is_used_for_directory() { + let td = tmpdir(); + mkdirp(td.path().join("a/b")); + wfile(td.path().join("a/b/.ignore"), "**\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/b/file").is_ignore()); + assert!(matches!( + m.dirs.get(Path::new("a/b")), + Some(CachedDir::Allowed(_)) + )); + let dir = matchedd(&mut m, "a/b"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + // Like `compiled_matchers_are_reused`, but with multiple roots. + #[test] + fn compiled_multi_matchers_are_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c")); + + wfile(td.path().join("a/.ignore"), "*.tmp\n"); + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + assert!(matchedf(&mut mats[0], "first.tmp").is_ignore()); + assert!(matchedf(&mut mats[0], "c/first.tmp").is_ignore()); + + // Even though we haven't used the second matcher, it should still + // reuse the `a/.ignore` above. If it didn't, then `a/b/c/first.tmp` + // below would be whitelisted. + wfile(td.path().join("a/.ignore"), "!*.tmp\n"); + assert!(matchedf(&mut mats[1], "first.tmp").is_ignore()); + } + + #[test] + fn compiled_multi_child_matchers_are_not_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c/d/e")); + + wfile(td.path().join("a/b/c/d/e/.ignore"), "*.tmp\n"); + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + // Some sanity checking first. + assert!(matchedf(&mut mats[0], "c/first.tmp").is_none()); + assert!(matchedf(&mut mats[0], "c/d/first.tmp").is_none()); + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_ignore()); + + // Now write a new ignore file at the same location as above + // and check that the other matcher still uses the "stale" data. + wfile(td.path().join("a/b/c/d/e/.ignore"), "!*.tmp\n"); + assert!(matchedf(&mut mats[1], "first.tmp").is_none()); + assert!(matchedf(&mut mats[1], "d/first.tmp").is_none()); + // This is the punch line: because we didn't load `mats[1]` before + // changing the ignore file, it loads it here and thus this path gets + // whitelisted. + assert!(matchedf(&mut mats[1], "d/e/first.tmp").is_whitelist()); + // ... but `mats[0]` still uses the stale gitignore matcher cached in + // memory! + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_ignore()); + + // If we rebuilder the matcher... then we force reloading and they're + // now consistent with one another. + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_whitelist()); + assert!(matchedf(&mut mats[1], "d/e/first.tmp").is_whitelist()); + } + + // Tests that even when there is an error with a glob pattern, we still + // respect other glob patterns that are valid. + #[test] + fn partial_errors_keep_valid_rules() { + let td = tmpdir(); + let root = td.path().join("work"); + mkdirp(&root); + wfile(td.path().join(".ignore"), "{bad\n*.tmp\n"); + + let mut builder = builder(&root); + builder.git_ignore(false).git_exclude(false); + let mut m = one_matcher(&builder); + let (matched, err) = m.matched_with_errors("drop.tmp", false); + assert!(err.is_some()); + assert!(matched.is_ignore()); + } + + // Tests that parent rules are not respected if the matcher is configured + // not to do so. + #[test] + fn parent_loading_can_be_disabled() { + let td = tmpdir(); + let root = td.path().join("work"); + mkdirp(&root); + wfile(td.path().join(".ignore"), "parent-rule\n"); + wfile(root.join(".ignore"), "root-rule\n"); + + let mut b = builder(&root); + b.parents(false); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "parent-rule").is_none()); + assert!(matchedf(&mut m, "root-rule").is_ignore()); + + // Sanity check that without `parents(false)`, the parent rule is + // respected. + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "parent-rule").is_ignore()); + assert!(matchedf(&mut m, "root-rule").is_ignore()); + } + + // Tests that we can normalize a file path that isn't already in "normal" + // relative form, and then use that to match on ignore files. + #[test] + fn paths_are_relative_to_the_root() { + let td = tmpdir(); + let root = td.path().join("root"); + let outside = td.path().join("outside"); + mkdirp(&root); + mkdirp(&outside); + wfile(root.join(".ignore"), "file\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedd(&mut m, "").is_none()); + assert!(matchedd(&mut m, ".").is_none()); + assert!(matchedf(&mut m, "file").is_ignore()); + // Doesn't work because it isn't relative to the root. It's absolute. + assert!(matchedf(&mut m, root.join("file")).is_none()); + // Also doesn't work because while it's relative, it contains `..`. + assert!(matchedf(&mut m, "dir/../file").is_none()); + + let norm = m.normalize(root.join("dir/../file")).unwrap(); + assert_eq!(norm, Path::new("file")); + assert!(matchedf(&mut m, "file").is_ignore()); + + // Doesn't normalize because its an absolute path outside of our root. + assert_eq!(m.normalize(outside.join("file")), None); + } + + // Tests that two matchers in two different directories correctly interpret + // the same parent gitignore file. + #[test] + fn multiple_roots_keep_their_own_context() { + let td = tmpdir(); + let root_a = td.path().join("a"); + let root_b = td.path().join("b"); + mkdirp(td.path().join(".git")); + mkdirp(&root_a); + mkdirp(&root_b); + wfile(td.path().join(".gitignore"), "/a/*.tmp\n/b/*.log\n"); + + let mut builder = + builders([root_a.as_path(), Path::new("-"), root_b.as_path()]); + builder.git_global(false); + let mut ms = matchers(&builder); + assert_eq!(ms.len(), 3); + assert_eq!(ms[0].root(), root_a); + assert_eq!(ms[1].root(), Path::new("-")); + assert_eq!(ms[2].root(), root_b); + assert!(matchedf(&mut ms[0], "drop.tmp").is_ignore()); + assert!(matchedf(&mut ms[0], "keep.log").is_none()); + assert!(matchedf(&mut ms[1], "anything").is_none()); + assert_eq!(ms[1].normalize("anything"), None); + assert!(matchedf(&mut ms[2], "keep.tmp").is_none()); + assert!(matchedf(&mut ms[2], "drop.log").is_ignore()); + } + + // Tests that only the exact `-` root represents standard input. + #[test] + fn dot_dash_root_is_not_stdin() { + let m = one_matcher(&builder("./-")); + assert_eq!(m.root(), Path::new("./-")); + assert_eq!(m.normalize("./-/file"), Some(PathBuf::from("file"))); + } + + #[test] + fn stdin_is_inert_with_depth_limits() { + let mut b = builder("-"); + b.min_depth(Some(2)); + let mut m = one_matcher(&b); + let mat = matchedf(&mut m, "file"); + assert!(mat.is_none()); + assert!(mat.is_within_depth()); + + let mut b = builder("-"); + b.max_depth(Some(0)); + let mut m = one_matcher(&b); + let mat = matchedd(&mut m, "dir"); + assert!(mat.is_none()); + assert!(mat.is_within_depth()); + assert!(mat.should_descend()); + } + + // Tests that ignore matching works lexically, and doesn't accidentally + // resolve symbolic links. + #[cfg(unix)] + #[test] + fn symlink_path_stays_under_lexical_root() { + use std::os::unix::fs::symlink; + + let td = tmpdir(); + let root = td.path().join("root"); + let outside = td.path().join("outside"); + mkdirp(&root); + mkdirp(&outside); + wfile(root.join(".ignore"), "link\n"); + wfile(outside.join("target"), ""); + symlink(outside.join("target"), root.join("link")).unwrap(); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "link").is_ignore()); + } + + #[test] + fn depth_limits() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c/d")); + + let mut b = builder(td.path()); + b.min_depth(Some(2)).max_depth(Some(3)); + let mut m = one_matcher(&b); + + let dir = matchedd(&mut m, "a"); + assert!(!dir.is_within_depth()); + assert!(dir.should_descend()); + assert!(matchedf(&mut m, "file").is_ignore()); + + let dir = matchedd(&mut m, "a/b"); + assert!(dir.is_within_depth()); + assert!(dir.should_descend()); + assert!(!matchedf(&mut m, "a/file").is_ignore()); + + let dir = matchedd(&mut m, "a/b/c"); + assert!(dir.is_within_depth()); + assert!(!dir.should_descend()); + assert!(!matchedf(&mut m, "a/b/file").is_ignore()); + assert!(matchedf(&mut m, "a/b/c/file").is_ignore()); + + let dir = matchedd(&mut m, "a/b/c/d"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + assert!(!dir.should_descend()); + assert!(matchedf(&mut m, "a/b/c/d/file").is_ignore()); + } + + #[test] + fn depth_limits_apply_to_root() { + let td = tmpdir(); + + let mut b = builder(td.path()); + b.min_depth(Some(1)); + let mut m = one_matcher(&b); + for path in ["", "."] { + let root = matchedd(&mut m, path); + assert!(root.is_none()); + assert!(!root.is_within_depth()); + assert!(root.should_descend()); + } + + let mut b = builder(td.path()); + b.max_depth(Some(0)); + let mut m = one_matcher(&b); + for path in ["", "."] { + let root = matchedd(&mut m, path); + assert!(root.is_none()); + assert!(root.is_within_depth()); + assert!(!root.should_descend()); + } + } + + #[test] + fn max_filesize() { + let td = tmpdir(); + mkdirp(td.path().join("dir")); + wfile(td.path().join("empty"), ""); + wfile(td.path().join("at-limit"), "12345"); + wfile(td.path().join("over-limit"), "123456"); + + let mut b = builder(td.path()); + b.max_filesize(Some(5)); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "empty").is_none()); + assert!(matchedf(&mut m, "at-limit").is_none()); + assert!(matchedf(&mut m, "over-limit").is_ignore()); + + let dir = matchedd(&mut m, "dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + #[test] + fn max_filesize_does_not_stat_ignored_file() { + let td = tmpdir(); + wfile(td.path().join(".ignore"), "ignored\n"); + + let mut b = builder(td.path()); + b.max_filesize(Some(0)); + let mut m = one_matcher(&b); + let (mat, err) = m.matched_with_errors("ignored", false); + assert!(mat.is_ignore()); + assert!(err.is_none(), "ignored missing file was statted: {err:?}"); + } + + #[cfg(unix)] + #[test] + fn max_filesize_respects_follow_links() { + use std::os::unix::fs::symlink; + + let td = tmpdir(); + wfile( + td.path().join("target"), + "target contents are much longer than the size limit", + ); + symlink("target", td.path().join("link")).unwrap(); + + let mut b = builder(td.path()); + b.max_filesize(Some(10)); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "link").is_none()); + + b.follow_links(true); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "link").is_ignore()); + } + + #[test] + fn hidden_files_and_directories() { + let td = tmpdir(); + mkdirp(td.path().join(".hidden-dir")); + mkdirp(td.path().join("visible-dir")); + wfile(td.path().join(".hidden-file"), ""); + wfile(td.path().join("visible-file"), ""); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden-file").is_ignore()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_ignore()); + assert!(!dir.should_descend()); + assert!(matchedf(&mut m, "visible-file").is_none()); + let dir = matchedd(&mut m, "visible-dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + + let mut b = builder(td.path()); + b.hidden(false); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, ".hidden-file").is_none()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + #[test] + fn gitignore_whitelist_overrides_hidden_filter() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden-dir")); + wfile(td.path().join(".hidden-file"), ""); + wfile(td.path().join(".gitignore"), "!.hidden-file\n!.hidden-dir/\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden-file").is_whitelist()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_whitelist()); + assert!(dir.should_descend()); + } + + #[test] + fn descendants_of_hidden_directories_are_ignored() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden/nested")); + mkdirp(td.path().join("visible/.hidden")); + wfile(td.path().join(".hidden/file"), ""); + wfile(td.path().join(".hidden/nested/file"), ""); + wfile(td.path().join("visible/.hidden/file"), ""); + wfile(td.path().join(".hidden/.gitignore"), "!file\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden/file").is_ignore()); + assert!(matchedf(&mut m, ".hidden/nested/file").is_ignore()); + assert!(matchedf(&mut m, "visible/.hidden/file").is_ignore()); + } + + #[test] + fn whitelisted_hidden_directory_allows_descendants() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden")); + wfile(td.path().join(".hidden/file"), ""); + wfile(td.path().join(".gitignore"), "!.hidden/\n"); + + let mut m = one_matcher(&builder(td.path())); + let dir = matchedd(&mut m, ".hidden"); + assert!(dir.is_whitelist()); + assert!(dir.should_descend()); + assert!(matchedf(&mut m, ".hidden/file").is_none()); + } + + #[test] + fn hidden_directories_outside_depth_limits() { + let td = tmpdir(); + mkdirp(td.path().join("a/.hidden")); + + let mut b = builder(td.path()); + b.min_depth(Some(3)); + let mut m = one_matcher(&b); + let dir = matchedd(&mut m, "a/.hidden"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + + let mut b = builder(td.path()); + b.max_depth(Some(1)); + let mut m = one_matcher(&b); + let dir = matchedd(&mut m, "a/.hidden"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + } +} diff --git a/crates/ignore/src/lib.rs b/crates/ignore/src/lib.rs index 609004c4e..b9c9a3ece 100644 --- a/crates/ignore/src/lib.rs +++ b/crates/ignore/src/lib.rs @@ -48,13 +48,16 @@ See the documentation for `WalkBuilder` for many other options. use std::path::{Path, PathBuf}; +pub use crate::incremental::{IncrementalIgnore, IncrementalMatch}; pub use crate::walk::{ - DirEntry, ParallelVisitor, ParallelVisitorBuilder, Walk, WalkBuilder, WalkParallel, WalkState, + DirEntry, ParallelVisitor, ParallelVisitorBuilder, Walk, WalkBuilder, + WalkParallel, WalkState, }; mod default_types; mod dir; pub mod gitignore; +mod incremental; pub mod overrides; mod pathutil; pub mod types; @@ -120,34 +123,31 @@ impl Clone for Error { fn clone(&self) -> Error { match *self { Error::Partial(ref errs) => Error::Partial(errs.clone()), - Error::WithLineNumber { line, ref err } => Error::WithLineNumber { - line, - err: err.clone(), - }, - Error::WithPath { ref path, ref err } => Error::WithPath { - path: path.clone(), - err: err.clone(), - }, - Error::WithDepth { depth, ref err } => Error::WithDepth { - depth, - err: err.clone(), - }, - Error::Loop { - ref ancestor, - ref child, - } => Error::Loop { + Error::WithLineNumber { line, ref err } => { + Error::WithLineNumber { line, err: err.clone() } + } + Error::WithPath { ref path, ref err } => { + Error::WithPath { path: path.clone(), err: err.clone() } + } + Error::WithDepth { depth, ref err } => { + Error::WithDepth { depth, err: err.clone() } + } + Error::Loop { ref ancestor, ref child } => Error::Loop { ancestor: ancestor.clone(), child: child.clone(), }, Error::Io(ref err) => match err.raw_os_error() { Some(e) => Error::Io(std::io::Error::from_raw_os_error(e)), - None => Error::Io(std::io::Error::new(err.kind(), err.to_string())), + None => { + Error::Io(std::io::Error::new(err.kind(), err.to_string())) + } }, - Error::Glob { ref glob, ref err } => Error::Glob { - glob: glob.clone(), - err: err.clone(), - }, - Error::UnrecognizedFileType(ref err) => Error::UnrecognizedFileType(err.clone()), + Error::Glob { ref glob, ref err } => { + Error::Glob { glob: glob.clone(), err: err.clone() } + } + Error::UnrecognizedFileType(ref err) => { + Error::UnrecognizedFileType(err.clone()) + } Error::InvalidDefinition => Error::InvalidDefinition, } } @@ -269,19 +269,14 @@ impl Error { /// Turn an error into a tagged error with the given depth. fn with_depth(self, depth: usize) -> Error { - Error::WithDepth { - depth, - err: Box::new(self), - } + Error::WithDepth { depth, err: Box::new(self) } } /// Turn an error into a tagged error with the given file path and line /// number. If path is empty, then it is omitted from the error. fn tagged>(self, path: P, lineno: u64) -> Error { - let errline = Error::WithLineNumber { - line: lineno, - err: Box::new(self), - }; + let errline = + Error::WithLineNumber { line: lineno, err: Box::new(self) }; if path.as_ref().as_os_str().is_empty() { return errline; } @@ -301,12 +296,12 @@ impl Error { }; } let path = err.path().map(|p| p.to_path_buf()); - let mut ig_err = Error::Io(std::io::Error::from(err)); + let mut ig_err = Error::WithDepth { + depth, + err: Box::new(Error::Io(std::io::Error::from(err))), + }; if let Some(path) = path { - ig_err = Error::WithPath { - path, - err: Box::new(ig_err), - }; + ig_err = Error::WithPath { path, err: Box::new(ig_err) }; } ig_err } @@ -333,7 +328,8 @@ impl std::fmt::Display for Error { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { match *self { Error::Partial(ref errs) => { - let msgs: Vec = errs.iter().map(|err| err.to_string()).collect(); + let msgs: Vec = + errs.iter().map(|err| err.to_string()).collect(); write!(f, "{}", msgs.join("\n")) } Error::WithLineNumber { line, ref err } => { @@ -343,10 +339,7 @@ impl std::fmt::Display for Error { write!(f, "{}: {}", path.display(), err) } Error::WithDepth { ref err, .. } => err.fmt(f), - Error::Loop { - ref ancestor, - ref child, - } => write!( + Error::Loop { ref ancestor, ref child } => write!( f, "File system loop found: \ {} points to an ancestor {}", @@ -354,14 +347,8 @@ impl std::fmt::Display for Error { ancestor.display() ), Error::Io(ref err) => err.fmt(f), - Error::Glob { - glob: None, - ref err, - } => write!(f, "{}", err), - Error::Glob { - glob: Some(ref glob), - ref err, - } => { + Error::Glob { glob: None, ref err } => write!(f, "{}", err), + Error::Glob { glob: Some(ref glob), ref err } => { write!(f, "error parsing glob '{}': {}", glob, err) } Error::UnrecognizedFileType(ref ty) => { @@ -507,7 +494,8 @@ mod tests { }; /// A convenient result type alias. - pub(crate) type Result = std::result::Result>; + pub(crate) type Result = + std::result::Result>; macro_rules! err { ($($tt:tt)*) => { @@ -545,8 +533,9 @@ mod tests { if path.is_dir() { continue; } - fs::create_dir_all(&path) - .map_err(|e| err!("failed to create {}: {}", path.display(), e))?; + fs::create_dir_all(&path).map_err(|e| { + err!("failed to create {}: {}", path.display(), e) + })?; return Ok(TempDir(path)); } Err(err!("failed to create temp dir after {} tries", TRIES)) diff --git a/crates/ignore/src/overrides.rs b/crates/ignore/src/overrides.rs index afb9f16ce..005cae8f2 100644 --- a/crates/ignore/src/overrides.rs +++ b/crates/ignore/src/overrides.rs @@ -94,7 +94,11 @@ impl Override { /// given) is stripped. If there is no common suffix/prefix overlap, then /// `path` is assumed to reside in the same directory as the root path for /// this set of overrides. - pub fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { if self.is_empty() { return Match::None; } @@ -146,7 +150,10 @@ impl OverrideBuilder { /// affected. /// /// This is disabled by default. - pub fn case_insensitive(&mut self, yes: bool) -> Result<&mut OverrideBuilder, Error> { + pub fn case_insensitive( + &mut self, + yes: bool, + ) -> Result<&mut OverrideBuilder, Error> { // TODO: This should not return a `Result`. Fix this in the next semver // release. self.builder.case_insensitive(yes)?; @@ -276,11 +283,8 @@ mod tests { #[test] fn default_case_sensitive() { - let ov = OverrideBuilder::new(ROOT) - .add("*.html") - .unwrap() - .build() - .unwrap(); + let ov = + OverrideBuilder::new(ROOT).add("*.html").unwrap().build().unwrap(); assert!(ov.matched("foo.html", false).is_whitelist()); assert!(ov.matched("foo.HTML", false).is_ignore()); assert!(ov.matched("foo.htm", false).is_ignore()); diff --git a/crates/ignore/src/pathutil.rs b/crates/ignore/src/pathutil.rs index 0ceb5a356..5de8c106c 100644 --- a/crates/ignore/src/pathutil.rs +++ b/crates/ignore/src/pathutil.rs @@ -2,55 +2,89 @@ use std::{ffi::OsStr, path::Path}; use crate::walk::DirEntry; -/// Returns true if and only if this entry is considered to be hidden. +/// Returns true if and only if this path is considered to be hidden. /// -/// This only returns true if the base name of the path starts with a `.`. +/// # Platform behavior /// -/// On Unix, this implements a more optimized check. -#[cfg(unix)] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - use std::os::unix::ffi::OsStrExt; - - if let Some(name) = file_name(dent.path()) { - name.as_bytes().get(0) == Some(&b'.') - } else { - false - } -} - -/// Returns true if and only if this entry is considered to be hidden. +/// ## Windows /// -/// On Windows, this returns true if one of the following is true: +/// This returns true if one of the following is true: /// /// * The base name of the path starts with a `.`. /// * The file attributes have the `HIDDEN` property set. -#[cfg(windows)] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - use std::os::windows::fs::MetadataExt; - use winapi_util::file; - - // This looks like we're doing an extra stat call, but on Windows, the - // directory traverser reuses the metadata retrieved from each directory - // entry and stores it on the DirEntry itself. So this is "free." - if let Ok(md) = dent.metadata() { - if file::is_hidden(md.file_attributes() as u64) { - return true; - } - } - if let Some(name) = file_name(dent.path()) { - name.to_str().map(|s| s.starts_with(".")).unwrap_or(false) - } else { - false - } -} - -/// Returns true if and only if this entry is considered to be hidden. +/// +/// ## All other platforms /// /// This only returns true if the base name of the path starts with a `.`. -#[cfg(not(any(unix, windows)))] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - if let Some(name) = file_name(dent.path()) { - name.to_str().map(|s| s.starts_with(".")).unwrap_or(false) +pub(crate) fn is_hidden_path(dent: &Path) -> bool { + #[cfg(not(windows))] + fn imp(path: &Path) -> bool { + is_hidden_path_only(path) + } + + #[cfg(windows)] + fn imp(path: &Path) -> bool { + use std::os::windows::fs::MetadataExt; + use winapi_util::file; + + if let Ok(md) = path.metadata() { + if file::is_hidden(md.file_attributes() as u64) { + return true; + } + } + is_hidden_path_only(path) + } + + imp(dent) +} + +/// Returns true if and only if this directory entry is considered to be +/// hidden. +/// +/// # Platform behavior +/// +/// ## Windows +/// +/// This returns true if one of the following is true: +/// +/// * The base name of the path starts with a `.`. +/// * The file attributes have the `HIDDEN` property set. +/// +/// ## All other platforms +/// +/// This only returns true if the base name of the path starts with a `.`. +pub(crate) fn is_hidden_entry(dent: &DirEntry) -> bool { + #[cfg(not(windows))] + fn imp(dent: &DirEntry) -> bool { + is_hidden_path_only(dent.path()) + } + + #[cfg(windows)] + fn imp(dent: &DirEntry) -> bool { + use std::os::windows::fs::MetadataExt; + use winapi_util::file; + + // This looks like we're doing an extra stat call, but on Windows, the + // directory traverser reuses the metadata retrieved from each directory + // entry and stores it on the DirEntry itself. So this is "free." + if let Ok(md) = dent.metadata() { + if file::is_hidden(md.file_attributes() as u64) { + return true; + } + } + is_hidden_path_only(dent.path()) + } + + imp(dent) +} + +/// Returns true if and only if this path is considered to be hidden from only +/// the path itself. +/// +/// This has the same behavior on all platforms. +fn is_hidden_path_only(path: &Path) -> bool { + if let Some(name) = file_name(path) { + name.as_encoded_bytes().starts_with(b".") } else { false } @@ -59,83 +93,79 @@ pub(crate) fn is_hidden(dent: &DirEntry) -> bool { /// Strip `prefix` from the `path` and return the remainder. /// /// If `path` doesn't have a prefix `prefix`, then return `None`. -#[cfg(unix)] pub(crate) fn strip_prefix<'a, P: AsRef + ?Sized>( prefix: &'a P, path: &'a Path, ) -> Option<&'a Path> { - use std::os::unix::ffi::OsStrExt; + #[cfg(unix)] + fn imp<'a>(prefix: &'a Path, path: &'a Path) -> Option<&'a Path> { + use std::os::unix::ffi::OsStrExt; - let prefix = prefix.as_ref().as_os_str().as_bytes(); - let path = path.as_os_str().as_bytes(); - if prefix.len() > path.len() || prefix != &path[0..prefix.len()] { - None - } else { - Some(&Path::new(OsStr::from_bytes(&path[prefix.len()..]))) + let prefix = prefix.as_os_str().as_bytes(); + let path = path.as_os_str().as_bytes(); + if prefix.len() > path.len() || prefix != &path[0..prefix.len()] { + None + } else { + Some(&Path::new(OsStr::from_bytes(&path[prefix.len()..]))) + } } -} -/// Strip `prefix` from the `path` and return the remainder. -/// -/// If `path` doesn't have a prefix `prefix`, then return `None`. -#[cfg(not(unix))] -pub(crate) fn strip_prefix<'a, P: AsRef + ?Sized>( - prefix: &'a P, - path: &'a Path, -) -> Option<&'a Path> { - path.strip_prefix(prefix).ok() + #[cfg(not(unix))] + fn imp<'a>(prefix: &'a Path, path: &'a Path) -> Option<&'a Path> { + path.strip_prefix(prefix).ok() + } + + imp(prefix.as_ref(), path) } /// Returns true if this file path is just a file name. i.e., Its parent is /// the empty string. -#[cfg(unix)] pub(crate) fn is_file_name>(path: P) -> bool { - use std::os::unix::ffi::OsStrExt; - - use memchr::memchr; - - let path = path.as_ref().as_os_str().as_bytes(); - memchr(b'/', path).is_none() -} - -/// Returns true if this file path is just a file name. i.e., Its parent is -/// the empty string. -#[cfg(not(unix))] -pub(crate) fn is_file_name>(path: P) -> bool { - path.as_ref() - .parent() - .map(|p| p.as_os_str().is_empty()) - .unwrap_or(false) -} - -/// The final component of the path, if it is a normal file. -/// -/// If the path terminates in ., .., or consists solely of a root of prefix, -/// file_name will return None. -#[cfg(unix)] -pub(crate) fn file_name<'a, P: AsRef + ?Sized>(path: &'a P) -> Option<&'a OsStr> { - use memchr::memrchr; - use std::os::unix::ffi::OsStrExt; - - let path = path.as_ref().as_os_str().as_bytes(); - if path.is_empty() { - return None; - } else if path.len() == 1 && path[0] == b'.' { - return None; - } else if path.last() == Some(&b'.') { - return None; - } else if path.len() >= 2 && &path[path.len() - 2..] == &b".."[..] { - return None; + #[cfg(unix)] + { + memchr::memchr(b'/', path.as_ref().as_os_str().as_encoded_bytes()) + .is_none() + } + #[cfg(not(unix))] + { + path.as_ref() + .parent() + .map(|p| p.as_os_str().is_empty()) + .unwrap_or(false) } - let last_slash = memrchr(b'/', path).map(|i| i + 1).unwrap_or(0); - Some(OsStr::from_bytes(&path[last_slash..])) } /// The final component of the path, if it is a normal file. /// -/// If the path terminates in ., .., or consists solely of a root of prefix, -/// file_name will return None. -#[cfg(not(unix))] -pub(crate) fn file_name<'a, P: AsRef + ?Sized>(path: &'a P) -> Option<&'a OsStr> { - path.as_ref().file_name() +/// If the path terminates in `.`, `..`, or consists solely of a root of +/// prefix, this will return `None`. +pub(crate) fn file_name<'a, P: AsRef + ?Sized>( + path: &'a P, +) -> Option<&'a OsStr> { + #[cfg(unix)] + fn imp(path: &Path) -> Option<&OsStr> { + use std::os::unix::ffi::OsStrExt; + + use memchr::memrchr; + + let path = path.as_os_str().as_bytes(); + if path.is_empty() { + return None; + } else if path.len() == 1 && path[0] == b'.' { + return None; + } else if path.last() == Some(&b'.') { + return None; + } else if path.len() >= 2 && &path[path.len() - 2..] == &b".."[..] { + return None; + } + let last_slash = memrchr(b'/', path).map(|i| i + 1).unwrap_or(0); + Some(OsStr::from_bytes(&path[last_slash..])) + } + + #[cfg(not(unix))] + fn imp(path: &Path) -> Option<&OsStr> { + path.file_name() + } + + imp(path.as_ref()) } diff --git a/crates/ignore/src/types.rs b/crates/ignore/src/types.rs index aa23999c0..313cf5c0e 100644 --- a/crates/ignore/src/types.rs +++ b/crates/ignore/src/types.rs @@ -204,8 +204,12 @@ impl Selection { fn map U>(self, f: F) -> Selection { match self { - Selection::Select(name, inner) => Selection::Select(name, f(inner)), - Selection::Negate(name, inner) => Selection::Negate(name, f(inner)), + Selection::Select(name, inner) => { + Selection::Select(name, f(inner)) + } + Selection::Negate(name, inner) => { + Selection::Negate(name, f(inner)) + } } } @@ -227,7 +231,9 @@ impl Types { has_selected: false, glob_to_selection: vec![], set: GlobSetBuilder::new().build().unwrap(), - matches: Arc::new(Pool::new(|| vec![])), + matches: Arc::new(Pool::with_available_parallelism_capacity( + || vec![], + )), } } @@ -254,7 +260,11 @@ impl Types { /// The path is considered ignored if it matches a negated file type. /// If at least one file type is selected and `path` doesn't match, then /// the path is also considered ignored. - pub fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { // File types don't apply to directories, and we can't do anything // if our glob set is empty. if is_dir || self.set.is_empty() { @@ -306,10 +316,7 @@ impl TypesBuilder { /// of default type definitions can be added with `add_defaults`, and /// additional type definitions can be added with `select` and `negate`. pub fn new() -> TypesBuilder { - TypesBuilder { - types: HashMap::new(), - selections: vec![], - } + TypesBuilder { types: HashMap::new(), selections: vec![] } } /// Build the current set of file type definitions *and* selections into @@ -343,17 +350,18 @@ impl TypesBuilder { } selections.push(selection.clone().map(move |_| def)); } - let set = build_set.build().map_err(|err| Error::Glob { - glob: None, - err: err.to_string(), - })?; + let set = build_set + .build() + .map_err(|err| Error::Glob { glob: None, err: err.to_string() })?; Ok(Types { defs, selections, has_selected, glob_to_selection, set, - matches: Arc::new(Pool::new(|| vec![])), + matches: Arc::new(Pool::with_available_parallelism_capacity( + || vec![], + )), }) } @@ -377,12 +385,10 @@ impl TypesBuilder { pub fn select(&mut self, name: &str) -> &mut TypesBuilder { if name == "all" { for name in self.types.keys() { - self.selections - .push(Selection::Select(name.to_string(), ())); + self.selections.push(Selection::Select(name.to_string(), ())); } } else { - self.selections - .push(Selection::Select(name.to_string(), ())); + self.selections.push(Selection::Select(name.to_string(), ())); } self } @@ -393,12 +399,10 @@ impl TypesBuilder { pub fn negate(&mut self, name: &str) -> &mut TypesBuilder { if name == "all" { for name in self.types.keys() { - self.selections - .push(Selection::Negate(name.to_string(), ())); + self.selections.push(Selection::Negate(name.to_string(), ())); } } else { - self.selections - .push(Selection::Negate(name.to_string(), ())); + self.selections.push(Selection::Negate(name.to_string(), ())); } self } @@ -453,7 +457,10 @@ impl TypesBuilder { 3 => { let name = parts[0]; let types_string = parts[2]; - if name.is_empty() || parts[1] != "include" || types_string.is_empty() { + if name.is_empty() + || parts[1] != "include" + || types_string.is_empty() + { return Err(Error::InvalidDefinition); } let types = types_string.split(','); @@ -463,7 +470,8 @@ impl TypesBuilder { return Err(Error::InvalidDefinition); } for type_name in types { - let globs = self.types.get(type_name).unwrap().globs.clone(); + let globs = + self.types.get(type_name).unwrap().globs.clone(); for glob in globs { self.add(name, &glob)?; } @@ -549,30 +557,9 @@ mod tests { matched!(not, matchnot1, types(), vec!["rust"], vec![], "index.html"); matched!(not, matchnot2, types(), vec![], vec!["rust"], "main.rs"); - matched!( - not, - matchnot3, - types(), - vec!["foo"], - vec!["rust"], - "main.rs" - ); - matched!( - not, - matchnot4, - types(), - vec!["rust"], - vec!["foo"], - "main.rs" - ); - matched!( - not, - matchnot5, - types(), - vec!["rust"], - vec!["foo"], - "main.foo" - ); + matched!(not, matchnot3, types(), vec!["foo"], vec!["rust"], "main.rs"); + matched!(not, matchnot4, types(), vec!["rust"], vec!["foo"], "main.rs"); + matched!(not, matchnot5, types(), vec!["rust"], vec!["foo"], "main.foo"); matched!(not, matchnot6, types(), vec!["combo"], vec![], "leftpad.js"); matched!(not, matchnot7, types(), vec!["py"], vec![], "index.html"); matched!(not, matchnot8, types(), vec!["python"], vec![], "doc.md"); diff --git a/crates/ignore/src/walk.rs b/crates/ignore/src/walk.rs index ebf20947a..2dff78852 100644 --- a/crates/ignore/src/walk.rs +++ b/crates/ignore/src/walk.rs @@ -17,7 +17,9 @@ use { use crate::{ Error, PartialErrorBuilder, dir::{Ignore, IgnoreBuilder}, + // CHANGED: Also import `Gitignore` for `WalkBuilder::add_gitignore`. gitignore::{Gitignore, GitignoreBuilder}, + incremental::{IncrementalIgnore, IncrementalIgnoreOptions}, overrides::Override, types::Types, }; @@ -104,24 +106,15 @@ impl DirEntry { } fn new_stdin() -> DirEntry { - DirEntry { - dent: DirEntryInner::Stdin, - err: None, - } + DirEntry { dent: DirEntryInner::Stdin, err: None } } fn new_walkdir(dent: walkdir::DirEntry, err: Option) -> DirEntry { - DirEntry { - dent: DirEntryInner::Walkdir(dent), - err, - } + DirEntry { dent: DirEntryInner::Walkdir(dent), err } } fn new_raw(dent: DirEntryRaw, err: Option) -> DirEntry { - DirEntry { - dent: DirEntryInner::Raw(dent), - err, - } + DirEntry { dent: DirEntryInner::Raw(dent), err } } } @@ -187,9 +180,11 @@ impl DirEntryInner { )); Err(err.with_path("")) } - Walkdir(ref x) => x - .metadata() - .map_err(|err| Error::Io(io::Error::from(err)).with_path(x.path())), + Walkdir(ref x) => x.metadata().map_err(|err| { + Error::Io(io::Error::from(err)) + .with_depth(x.depth()) + .with_path(x.path()) + }), Raw(ref x) => x.metadata(), } } @@ -308,7 +303,9 @@ impl DirEntryRaw { } else { fs::symlink_metadata(&self.path) } - .map_err(|err| Error::Io(io::Error::from(err)).with_path(&self.path)) + .map_err(|err| { + Error::Io(err).with_depth(self.depth).with_path(&self.path) + }) } fn file_type(&self) -> FileType { @@ -316,9 +313,7 @@ impl DirEntryRaw { } fn file_name(&self) -> &OsStr { - self.path - .file_name() - .unwrap_or_else(|| self.path.as_os_str()) + self.path.file_name().unwrap_or_else(|| self.path.as_os_str()) } fn depth(&self) -> usize { @@ -330,13 +325,13 @@ impl DirEntryRaw { self.ino } - fn from_entry(depth: usize, ent: &fs::DirEntry) -> Result { + fn from_entry( + depth: usize, + ent: &fs::DirEntry, + ) -> Result { let ty = ent.file_type().map_err(|err| { - let err = Error::Io(io::Error::from(err)).with_path(ent.path()); - Error::WithDepth { - depth, - err: Box::new(err), - } + let err = Error::Io(err).with_depth(depth).with_path(ent.path()); + Error::WithDepth { depth, err: Box::new(err) } })?; DirEntryRaw::from_entry_os(depth, ent, ty) } @@ -348,11 +343,8 @@ impl DirEntryRaw { ty: fs::FileType, ) -> Result { let md = ent.metadata().map_err(|err| { - let err = Error::Io(io::Error::from(err)).with_path(ent.path()); - Error::WithDepth { - depth, - err: Box::new(err), - } + let err = Error::Io(err).with_depth(depth).with_path(ent.path()); + Error::WithDepth { depth, err: Box::new(err) } })?; Ok(DirEntryRaw { path: ent.path(), @@ -395,8 +387,13 @@ impl DirEntryRaw { } #[cfg(windows)] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { - let md = fs::metadata(&pb).map_err(|err| Error::Io(err).with_path(&pb))?; + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { + let md = fs::metadata(&pb) + .map_err(|err| Error::Io(err).with_depth(depth).with_path(&pb))?; Ok(DirEntryRaw { path: pb, ty: md.file_type(), @@ -407,10 +404,15 @@ impl DirEntryRaw { } #[cfg(unix)] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { use std::os::unix::fs::MetadataExt; - let md = fs::metadata(&pb).map_err(|err| Error::Io(err).with_path(&pb))?; + let md = fs::metadata(&pb) + .map_err(|err| Error::Io(err).with_depth(depth).with_path(&pb))?; Ok(DirEntryRaw { path: pb, ty: md.file_type(), @@ -423,7 +425,11 @@ impl DirEntryRaw { // Placeholder implementation to allow compiling on non-standard platforms // (e.g. wasm32). #[cfg(not(any(windows, unix)))] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { Err(Error::Io(io::Error::new( io::ErrorKind::Other, "unsupported platform", @@ -502,7 +508,8 @@ pub struct WalkBuilder { /// /// When `None`, the CWD is fetched from `std::env::current_dir()`. If /// that fails, then global gitignores are ignored (an error is logged). - global_gitignores_relative_to: OnceLock>>, + global_gitignores_relative_to: + OnceLock>>, } #[derive(Clone)] @@ -544,8 +551,16 @@ impl WalkBuilder { /// is better to call `add` on this builder than to create multiple /// `Walk` values. pub fn new>(path: P) -> WalkBuilder { + WalkBuilder::from_iter([path]) + } + + /// Create an empty builder to which paths can be added. + /// + /// Note that if you call `build` on this instance before calling `add` + /// on it, it will return exactly zero items during iteration. + pub fn empty() -> WalkBuilder { WalkBuilder { - paths: vec![path.as_ref().to_path_buf()], + paths: vec![], ig_builder: IgnoreBuilder::new(), max_depth: None, min_depth: None, @@ -560,6 +575,21 @@ impl WalkBuilder { } } + /// Create a new builder for a recursive directory iterator from the + /// sequence of paths. + /// + /// Note that if the iterator is empty, this is the same as + /// `WalkBuilder::empty`. + pub fn from_iter, I: IntoIterator>( + paths: I, + ) -> WalkBuilder { + let mut builder = WalkBuilder::empty(); + for path in paths.into_iter() { + builder.add(path); + } + builder + } + /// Build a new `Walk` iterator. pub fn build(&self) -> Walk { let follow_links = self.follow_links; @@ -585,10 +615,14 @@ impl WalkBuilder { if let Some(ref sorter) = sorter { match sorter.clone() { Sorter::ByName(cmp) => { - wd = wd.sort_by(move |a, b| cmp(a.file_name(), b.file_name())); + wd = wd.sort_by(move |a, b| { + cmp(a.file_name(), b.file_name()) + }); } Sorter::ByPath(cmp) => { - wd = wd.sort_by(move |a, b| cmp(a.path(), b.path())); + wd = wd.sort_by(move |a, b| { + cmp(a.path(), b.path()) + }); } } } @@ -597,31 +631,76 @@ impl WalkBuilder { }) .collect::>() .into_iter(); - let ig_root = self - .get_or_set_current_dir() - .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) - .unwrap_or_else(|| self.ig_builder.build()); + let ig_root = self.build_ignore(); Walk { its, it: None, ig_root: ig_root.clone(), ig: ig_root.clone(), + max_depth: self.max_depth, max_filesize: self.max_filesize, skip: self.skip.clone(), filter: self.filter.clone(), } } + /// Build matchers for checking paths against ignore files without + /// recursively walking the configured roots. + /// + /// The returned matchers use the path-based filtering configuration + /// on this builder, including glob overrides, file type selections, + /// parent ignore files, `.ignore`, `.gitignore`, global Git + /// ignore files, explicitly added ignore files and custom ignore + /// file names. For example, ripgrep configures `.rgignore` via + /// [`WalkBuilder::add_custom_ignore_filename`]. Minimum and maximum depth + /// limits, maximum file size and hidden-file filtering are also applied. + /// Other options that only control traversal or require a directory entry, + /// such as custom entry predicates, are not applied. + /// + /// One matcher is returned for each configured path, in the same order as + /// the paths were added to this builder. Each matcher accepts paths + /// relative to its own [`IncrementalIgnore::root`]. The matcher for the + /// special `-` path representing standard input always returns a non-match + /// for all inputs. + /// + /// Ignore matchers are loaded lazily and cached by directory. + /// Thus, the first query may read ignore files from the root and + /// its parents, while later queries reuse the compiled matchers. + /// Errors encountered while loading ignore files are returned by + /// [`IncrementalIgnore::matched_with_errors`]. Once an ignore file has + /// been loaded, changes to it are not observed. Build new matchers to + /// reload changed ignore files. + /// + /// Matchers built together share the builder's base ignore configuration + /// and compiled parent matchers. + pub fn build_matchers(&self) -> Vec { + let ignore = self.build_ignore(); + let options = IncrementalIgnoreOptions { + min_depth: self.min_depth, + max_depth: self.max_depth, + max_filesize: self.max_filesize, + hidden: self.ig_builder.is_hidden(), + follow_links: self.follow_links, + }; + self.paths + .iter() + .map(move |path| { + IncrementalIgnore::new( + path.clone(), + ignore.clone(), + options.clone(), + ) + }) + .collect() + } + /// Build a new `WalkParallel` iterator. /// /// Note that this *doesn't* return something that implements `Iterator`. /// Instead, the returned value must be run with a closure. e.g., /// `builder.build_parallel().run(|| |path| { println!("{path:?}"); WalkState::Continue })`. pub fn build_parallel(&self) -> WalkParallel { - let ig_root = self - .get_or_set_current_dir() - .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) - .unwrap_or_else(|| self.ig_builder.build()); + let ig_root = self.build_ignore(); WalkParallel { paths: self.paths.clone().into_iter(), ig_root, @@ -651,7 +730,10 @@ impl WalkBuilder { /// The default, `None`, imposes no depth restriction. pub fn max_depth(&mut self, depth: Option) -> &mut WalkBuilder { self.max_depth = depth; - if self.min_depth.is_some() && self.max_depth.is_some() && self.max_depth < self.min_depth { + if self.min_depth.is_some() + && self.max_depth.is_some() + && self.max_depth < self.min_depth + { self.max_depth = self.min_depth; } self @@ -662,7 +744,10 @@ impl WalkBuilder { /// The default, `None`, imposes no minimum depth restriction. pub fn min_depth(&mut self, depth: Option) -> &mut WalkBuilder { self.min_depth = depth; - if self.max_depth.is_some() && self.min_depth.is_some() && self.min_depth > self.max_depth { + if self.max_depth.is_some() + && self.min_depth.is_some() + && self.min_depth > self.max_depth + { self.min_depth = self.max_depth; } self @@ -705,7 +790,12 @@ impl WalkBuilder { /// An error will also occur if this walker could not get the current /// working directory (and `WalkBuilder::current_dir` isn't set). pub fn add_ignore>(&mut self, path: P) -> Option { - // CHANGED: Dropped this code + // CHANGED: Root the ignore file at `""` instead of the current working + // directory. Explicit ignores are scoped to the directory of the + // ignore file (see `matched_ignore`), and a root of `""` makes the + // rules apply to every walked path regardless of the walk root. This + // also avoids depending on the current working directory entirely. + // // let path = path.as_ref(); // let Some(cwd) = self.get_or_set_current_dir() else { // let err = std::io::Error::other(format!( @@ -729,7 +819,11 @@ impl WalkBuilder { errs.into_error_option() } - /// CHANGED: Add a Gitignore to the builder. + /// CHANGED: Add a prebuilt Gitignore to the builder. + /// + /// Like the ignore file added via `add_ignore`, these rules are matched + /// against the full path of each walked entry, scoped to the `Gitignore`'s + /// root path. pub fn add_gitignore(&mut self, gi: Gitignore) { self.ig_builder.add_ignore(gi); } @@ -982,7 +1076,10 @@ impl WalkBuilder { /// /// Global gitignore files come from things like a user's git configuration /// or from gitignore files added via [`WalkBuilder::add_ignore`]. - pub fn current_dir(&mut self, cwd: impl Into) -> &mut WalkBuilder { + pub fn current_dir( + &mut self, + cwd: impl Into, + ) -> &mut WalkBuilder { let cwd = cwd.into(); self.ig_builder.current_dir(cwd.clone()); if let Err(cwd) = self.global_gitignores_relative_to.set(Ok(cwd)) { @@ -1002,7 +1099,10 @@ impl WalkBuilder { let result = std::env::current_dir().map_err(Arc::new); match result { Ok(ref path) => { - log::trace!("automatically discovered CWD: {}", path.display()); + log::trace!( + "automatically discovered CWD: {}", + path.display() + ); } Err(ref err) => { log::debug!( @@ -1016,6 +1116,13 @@ impl WalkBuilder { }); result.as_ref().ok().map(|path| &**path) } + + /// Build the root ignore matcher shared by all consumers of this builder. + fn build_ignore(&self) -> Ignore { + self.get_or_set_current_dir() + .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) + .unwrap_or_else(|| self.ig_builder.build()) + } } /// Walk is a recursive directory iterator over file paths in one or more @@ -1029,6 +1136,7 @@ pub struct Walk { it: Option, ig_root: Ignore, ig: Ignore, + max_depth: Option, max_filesize: Option, skip: Option>, filter: Option, @@ -1044,6 +1152,17 @@ impl Walk { WalkBuilder::new(path).build() } + /// Create a new recursive directory iterator from the sequence of paths + /// given. + /// + /// Note that if the provided iterator is empty, then `Walk` is guaranteed + /// to yield zero entries. + pub fn from_iter, I: IntoIterator>( + paths: I, + ) -> Walk { + WalkBuilder::from_iter(paths).build() + } + fn skip_entry(&self, ent: &DirEntry) -> Result { if ent.depth() == 0 { return Ok(false); @@ -1128,12 +1247,17 @@ impl Iterator for Walk { self.it.as_mut().unwrap().it.skip_current_dir(); // Still need to push this on the stack because // we'll get a WalkEvent::Exit event for this dir. - // We don't care if it errors though. - let (igtmp, _) = self.ig.add_child(ent.path()); + // Its ignore files cannot apply to any visited entry. + let (igtmp, _) = + self.ig.add_child_with_entries(ent.path(), &[]); self.ig = igtmp; continue; } - let (igtmp, err) = self.ig.add_child(ent.path()); + let (igtmp, err) = if self.max_depth == Some(ent.depth()) { + self.ig.add_child_with_entries(ent.path(), &[]) + } else { + self.ig.add_child(ent.path()) + }; self.ig = igtmp; ent.err = err; return Some(Ok(ent)); @@ -1175,11 +1299,7 @@ enum WalkEvent { impl From for WalkEventIter { fn from(it: WalkDir) -> WalkEventIter { - WalkEventIter { - depth: 0, - it: it.into_iter(), - next: None, - } + WalkEventIter { depth: 0, it: it.into_iter(), next: None } } } @@ -1252,7 +1372,9 @@ pub trait ParallelVisitorBuilder<'s> { fn build(&mut self) -> Box; } -impl<'a, 's, P: ParallelVisitorBuilder<'s>> ParallelVisitorBuilder<'s> for &'a mut P { +impl<'a, 's, P: ParallelVisitorBuilder<'s>> ParallelVisitorBuilder<'s> + for &'a mut P +{ fn build(&mut self) -> Box { (**self).build() } @@ -1273,14 +1395,17 @@ struct FnBuilder { builder: F, } -impl<'s, F: FnMut() -> FnVisitor<'s>> ParallelVisitorBuilder<'s> for FnBuilder { +impl<'s, F: FnMut() -> FnVisitor<'s>> ParallelVisitorBuilder<'s> + for FnBuilder +{ fn build(&mut self) -> Box { let visitor = (self.builder)(); Box::new(FnVisitorImp { visitor }) } } -type FnVisitor<'s> = Box) -> WalkState + Send + 's>; +type FnVisitor<'s> = + Box) -> WalkState + Send + 's>; struct FnVisitorImp<'s> { visitor: FnVisitor<'s>, @@ -1370,7 +1495,9 @@ impl WalkParallel { } }; match DirEntryRaw::from_path(0, path, false) { - Ok(dent) => (DirEntry::new_raw(dent, None), root_device), + Ok(dent) => { + (DirEntry::new_raw(dent, None), root_device) + } Err(err) => { if visitor.visit(Err(err)).is_quit() { return; @@ -1394,21 +1521,28 @@ impl WalkParallel { let quit_now = Arc::new(AtomicBool::new(false)); let active_workers = Arc::new(AtomicUsize::new(threads)); let stacks = Stack::new_for_each_thread(threads, stack); + // Collect all of the workers first. In the case that + // `builder.build()` panics, we want that to happen and + // propagate before we actually start to run any of the + // workers. + let workers: Vec<_> = stacks + .into_iter() + .map(|stack| Worker { + visitor: builder.build(), + stack, + quit_now: quit_now.clone(), + active_workers: active_workers.clone(), + max_depth: self.max_depth, + min_depth: self.min_depth, + max_filesize: self.max_filesize, + follow_links: self.follow_links, + skip: self.skip.clone(), + filter: self.filter.clone(), + }) + .collect(); std::thread::scope(|s| { - let handles: Vec<_> = stacks + let handles: Vec<_> = workers .into_iter() - .map(|stack| Worker { - visitor: builder.build(), - stack, - quit_now: quit_now.clone(), - active_workers: active_workers.clone(), - max_depth: self.max_depth, - min_depth: self.min_depth, - max_filesize: self.max_filesize, - follow_links: self.follow_links, - skip: self.skip.clone(), - filter: self.filter.clone(), - }) .map(|worker| s.spawn(|| worker.run())) .collect(); for handle in handles { @@ -1419,9 +1553,7 @@ impl WalkParallel { fn threads(&self) -> usize { if self.threads == 0 { - std::thread::available_parallelism() - .map_or(1, |n| n.get()) - .min(12) + std::thread::available_parallelism().map_or(1, |n| n.get()).min(12) } else { self.threads } @@ -1452,6 +1584,12 @@ struct Work { root_device: Option, } +#[derive(Default)] +struct ReadDirResult { + entries: Vec, + errors: Vec, +} + impl Work { /// Returns true if and only if this work item is a directory. fn is_dir(&self) -> bool { @@ -1478,6 +1616,13 @@ impl Work { err } + /// Adds ignore rules for this directory without reading its contents. + fn add_ignore(&mut self) { + let (ig, err) = self.ignore.add_child(self.dent.path()); + self.ignore = ig; + self.dent.err = err; + } + /// Reads the directory contents of this work item and adds ignore /// rules for this directory. /// @@ -1485,7 +1630,7 @@ impl Work { /// an error is returned. If there was a problem reading the ignore /// rules for this directory, then the error is attached to this /// work item's directory entry. - fn read_dir(&mut self) -> Result { + fn read_dir(&mut self) -> Result { let readdir = match fs::read_dir(self.dent.path()) { Ok(readdir) => readdir, Err(err) => { @@ -1495,10 +1640,24 @@ impl Work { return Err(err); } }; - let (ig, err) = self.ignore.add_child(self.dent.path()); + // Actually descend into the directory and read its contents + let mut result = ReadDirResult::default(); + for entry in readdir { + match entry { + Ok(entry) => result.entries.push(entry), + Err(err) => result.errors.push( + Error::from(err) + .with_path(self.dent.path()) + .with_depth(self.dent.depth() + 1), + ), + } + } + let (ig, err) = self + .ignore + .add_child_with_entries(self.dent.path(), &result.entries); self.ignore = ig; self.dent.err = err; - Ok(readdir) + Ok(result) } } @@ -1522,11 +1681,11 @@ impl Stack { // breadth-first. We do depth-first because a breadth first traversal // on wide directories with a lot of gitignores is disastrous (for // example, searching a directory tree containing all of crates.io). - let deques: Vec> = std::iter::repeat_with(Deque::new_lifo) - .take(threads) - .collect(); - let stealers = - Arc::<[Stealer]>::from(deques.iter().map(Deque::stealer).collect::>()); + let deques: Vec> = + std::iter::repeat_with(Deque::new_lifo).take(threads).collect(); + let stealers = Arc::<[Stealer]>::from( + deques.iter().map(Deque::stealer).collect::>(), + ); let stacks: Vec = deques .into_iter() .enumerate() @@ -1668,8 +1827,13 @@ impl<'s> Worker<'s> { // have sufficient read permissions to list the directory. // In that case we still want to provide the closure with a valid // entry before passing the error value. - let readdir = work.read_dir(); let depth = work.dent.depth(); + let readdir = if descend && self.max_depth.is_none_or(|m| depth < m) { + Some(work.read_dir()) + } else { + work.add_ignore(); + None + }; if should_visit { let state = self.visitor.visit(Ok(work.dent)); if !state.is_continue() { @@ -1680,6 +1844,10 @@ impl<'s> Worker<'s> { return WalkState::Skip; } + let readdir = match readdir { + Some(readdir) => readdir, + None => return WalkState::Skip, + }; let readdir = match readdir { Ok(readdir) => readdir, Err(err) => { @@ -1687,11 +1855,19 @@ impl<'s> Worker<'s> { } }; - if self.max_depth.map_or(false, |max| depth >= max) { - return WalkState::Skip; + for result in readdir.entries { + let state = self.generate_work( + &work.ignore, + depth + 1, + work.root_device, + result, + ); + if state.is_quit() { + return state; + } } - for result in readdir { - let state = self.generate_work(&work.ignore, depth + 1, work.root_device, result); + for err in readdir.errors { + let state = self.visitor.visit(Err(err)); if state.is_quit() { return state; } @@ -1717,14 +1893,8 @@ impl<'s> Worker<'s> { ig: &Ignore, depth: usize, root_device: Option, - result: Result, + fs_dent: fs::DirEntry, ) -> WalkState { - let fs_dent = match result { - Ok(fs_dent) => fs_dent, - Err(err) => { - return self.visitor.visit(Err(Error::from(err).with_depth(depth))); - } - }; let mut dent = match DirEntryRaw::from_entry(depth, &fs_dent) { Ok(dent) => DirEntry::new_raw(dent, None), Err(err) => { @@ -1760,26 +1930,24 @@ impl<'s> Worker<'s> { return WalkState::Continue; } } - let should_skip_filesize = if self.max_filesize.is_some() && !dent.is_dir() { - skip_filesize( - self.max_filesize.unwrap(), - dent.path(), - &dent.metadata().ok(), - ) - } else { - false - }; - let should_skip_filtered = if let Some(Filter(predicate)) = &self.filter { - !predicate(&dent) - } else { - false - }; + let should_skip_filesize = + if self.max_filesize.is_some() && !dent.is_dir() { + skip_filesize( + self.max_filesize.unwrap(), + dent.path(), + &dent.metadata().ok(), + ) + } else { + false + }; + let should_skip_filtered = + if let Some(Filter(predicate)) = &self.filter { + !predicate(&dent) + } else { + false + }; if !should_skip_filesize && !should_skip_filtered { - self.send(Work { - dent, - ignore: ig.clone(), - root_device, - }); + self.send(Work { dent, ignore: ig.clone(), root_device }); } WalkState::Continue } @@ -1820,6 +1988,9 @@ impl<'s> Worker<'s> { } // Wait for next `Work` or `Quit` message. loop { + if self.is_quit_now() { + return None; + } if let Some(v) = self.recv() { self.activate_worker(); value = Some(v); @@ -1873,24 +2044,25 @@ impl<'s> Worker<'s> { } } +impl<'s> Drop for Worker<'s> { + fn drop(&mut self) { + if std::thread::panicking() { + self.quit_now(); + } + } +} + fn check_symlink_loop( ig_parent: &Ignore, child_path: &Path, child_depth: usize, ) -> Result<(), Error> { let hchild = Handle::from_path(child_path).map_err(|err| { - Error::from(err) - .with_path(child_path) - .with_depth(child_depth) + Error::from(err).with_path(child_path).with_depth(child_depth) })?; - for ig in ig_parent - .parents() - .take_while(|ig| !ig.is_absolute_parent()) - { + for ig in ig_parent.parents().take_while(|ig| !ig.is_absolute_parent()) { let h = Handle::from_path(ig.path()).map_err(|err| { - Error::from(err) - .with_path(child_path) - .with_depth(child_depth) + Error::from(err).with_path(child_path).with_depth(child_depth) })?; if hchild == h { return Err(Error::Loop { @@ -1905,7 +2077,11 @@ fn check_symlink_loop( // Before calling this function, make sure that you ensure that is really // necessary as the arguments imply a file stat. -fn skip_filesize(max_filesize: u64, path: &Path, ent: &Option) -> bool { +fn skip_filesize( + max_filesize: u64, + path: &Path, + ent: &Option, +) -> bool { let filesize = match *ent { Some(ref md) => Some(md.len()), None => None, @@ -1977,9 +2153,9 @@ fn path_equals(dent: &DirEntry, handle: &Handle) -> Result { if dent.is_stdin() || never_equal(dent, handle) { return Ok(false); } - Handle::from_path(dent.path()) - .map(|h| &h == handle) - .map_err(|err| Error::Io(err).with_path(dent.path())) + Handle::from_path(dent.path()).map(|h| &h == handle).map_err(|err| { + Error::Io(err).with_depth(dent.depth()).with_path(dent.path()) + }) } /// Returns true if the given walkdir entry corresponds to a directory. @@ -1997,16 +2173,14 @@ fn walkdir_is_dir(dent: &walkdir::DirEntry) -> bool { if !dent.file_type().is_symlink() || dent.depth() > 0 { return false; } - dent.path() - .metadata() - .ok() - .map_or(false, |md| md.file_type().is_dir()) + dent.path().metadata().ok().map_or(false, |md| md.file_type().is_dir()) } /// Returns true if and only if the given path is on the same device as the /// given root device. fn is_same_file_system(root_device: u64, path: &Path) -> Result { - let dent_device = device_num(path).map_err(|err| Error::Io(err).with_path(path))?; + let dent_device = + device_num(path).map_err(|err| Error::Io(err).with_path(path))?; Ok(root_device == dent_device) } @@ -2065,11 +2239,7 @@ mod tests { } fn normal_path(unix: &str) -> String { - if cfg!(windows) { - unix.replace("\\", "/") - } else { - unix.to_string() - } + if cfg!(windows) { unix.replace("\\", "/") } else { unix.to_string() } } fn walk_collect(prefix: &Path, builder: &WalkBuilder) -> Vec { @@ -2089,7 +2259,10 @@ mod tests { paths } - fn walk_collect_parallel(prefix: &Path, builder: &WalkBuilder) -> Vec { + fn walk_collect_parallel( + prefix: &Path, + builder: &WalkBuilder, + ) -> Vec { let mut paths = vec![]; for dent in walk_collect_entries_parallel(builder) { let path = dent.path().strip_prefix(prefix).unwrap(); @@ -2278,6 +2451,27 @@ mod tests { ); } + #[test] + fn max_depth_does_not_load_unreachable_ignore_files() { + let td = tmpdir(); + let leaf = td.path().join("leaf"); + mkdirp(&leaf); + wfile(leaf.join(".ignore"), "{invalid\n"); + + let mut builder = WalkBuilder::new(td.path()); + builder.max_depth(Some(1)); + let entry = builder + .build() + .find_map(|result| { + let entry = result.unwrap(); + (entry.path() == leaf).then_some(entry) + }) + .unwrap(); + + assert!(entry.error().is_none()); + assert_paths(td.path(), &builder, &["leaf"]); + } + #[test] fn min_depth() { let td = tmpdir(); @@ -2388,7 +2582,9 @@ mod tests { assert_eq!(1, dents.len()); assert!(!dents[0].path_is_symlink()); - let dents = walk_collect_entries_parallel(&WalkBuilder::new(td.path().join("foo"))); + let dents = walk_collect_entries_parallel(&WalkBuilder::new( + td.path().join("foo"), + )); assert_eq!(1, dents.len()); assert!(!dents[0].path_is_symlink()); } @@ -2474,8 +2670,88 @@ mod tests { assert_paths( td.path(), - &WalkBuilder::new(td.path()).filter_entry(|entry| entry.file_name() != OsStr::new("a")), + &WalkBuilder::new(td.path()) + .filter_entry(|entry| entry.file_name() != OsStr::new("a")), &["x", "x/y", "x/y/foo"], ); } + + #[test] + fn empty() { + let td = tmpdir(); + assert_paths(td.path(), &WalkBuilder::empty(), &[]); + + let empty_paths: Vec<&OsStr> = Vec::new(); + assert_paths(td.path(), &WalkBuilder::from_iter(empty_paths), &[]); + } + + #[test] + fn from_iter() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c")); + mkdirp(td.path().join("d/e/f")); + mkdirp(td.path().join("x/y")); + wfile(td.path().join("a/b/foo"), ""); + wfile(td.path().join("d/e/f/foo"), ""); + wfile(td.path().join("x/y/foo"), ""); + + let paths = vec![ + td.path().join("a"), + td.path().join("d"), + td.path().join("x"), + ]; + + assert_paths( + td.path(), + &WalkBuilder::from_iter(paths), + &[ + "x", + "x/y", + "x/y/foo", + "d", + "d/e", + "d/e/f", + "d/e/f/foo", + "a", + "a/b", + "a/b/foo", + "a/b/c", + ], + ); + } + + // This should always panic and never hang. + // + // Ref: https://github.com/BurntSushi/ripgrep/issues/3009 + #[test] + #[should_panic] + fn panic_in_parallel() { + let td = tmpdir(); + wfile(td.path().join("foo.txt"), ""); + + WalkBuilder::new(td.path()) + .threads(40) + .build_parallel() + .run(|| Box::new(|_| panic!("oops!"))); + } + + // This should always panic and never hang. The first call to the visitor + // builder is used while processing the root paths. Previously, a panic on + // the third call occurred after the first worker had already been spawned, + // leaving it waiting indefinitely for workers that were never created. + #[test] + #[should_panic(expected = "builder panic")] + fn panic_in_parallel_builder() { + let td = tmpdir(); + wfile(td.path().join("foo.txt"), ""); + + let mut builds = 0; + WalkBuilder::new(td.path()).threads(2).build_parallel().run(|| { + builds += 1; + if builds == 3 { + panic!("builder panic"); + } + Box::new(|_| WalkState::Continue) + }); + } } diff --git a/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs b/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs index b7b7c6f95..ecb7b47e3 100644 --- a/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs +++ b/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs @@ -2,7 +2,8 @@ use std::path::Path; use ignore::gitignore::{Gitignore, GitignoreBuilder}; -const IGNORE_FILE: &'static str = "tests/gitignore_matched_path_or_any_parents_tests.gitignore"; +const IGNORE_FILE: &'static str = + "tests/gitignore_matched_path_or_any_parents_tests.gitignore"; fn get_gitignore() -> Gitignore { let mut builder = GitignoreBuilder::new("ROOT"); @@ -23,7 +24,9 @@ fn test_path_should_be_under_root() { #[test] fn test_files_in_root() { let gitignore = get_gitignore(); - let m = |path: &str| gitignore.matched_path_or_any_parents(Path::new(path), false); + let m = |path: &str| { + gitignore.matched_path_or_any_parents(Path::new(path), false) + }; // 0x assert!(m("ROOT/file_root_00").is_ignore()); @@ -53,7 +56,9 @@ fn test_files_in_root() { #[test] fn test_files_in_deep() { let gitignore = get_gitignore(); - let m = |path: &str| gitignore.matched_path_or_any_parents(Path::new(path), false); + let m = |path: &str| { + gitignore.matched_path_or_any_parents(Path::new(path), false) + }; // 0x assert!(m("ROOT/parent_dir/file_deep_00").is_ignore()); @@ -83,8 +88,9 @@ fn test_files_in_deep() { #[test] fn test_dirs_in_root() { let gitignore = get_gitignore(); - let m = - |path: &str, is_dir: bool| gitignore.matched_path_or_any_parents(Path::new(path), is_dir); + let m = |path: &str, is_dir: bool| { + gitignore.matched_path_or_any_parents(Path::new(path), is_dir) + }; // 00 assert!(m("ROOT/dir_root_00", true).is_ignore()); @@ -186,20 +192,25 @@ fn test_dirs_in_root() { #[test] fn test_dirs_in_deep() { let gitignore = get_gitignore(); - let m = - |path: &str, is_dir: bool| gitignore.matched_path_or_any_parents(Path::new(path), is_dir); + let m = |path: &str, is_dir: bool| { + gitignore.matched_path_or_any_parents(Path::new(path), is_dir) + }; // 00 assert!(m("ROOT/parent_dir/dir_deep_00", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_00/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_00/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_00/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_00/child_dir/file", false).is_ignore() + ); // 01 assert!(m("ROOT/parent_dir/dir_deep_01", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_01/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_01/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_01/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_01/child_dir/file", false).is_ignore() + ); // 02 assert!(m("ROOT/parent_dir/dir_deep_02", true).is_none()); @@ -241,51 +252,67 @@ fn test_dirs_in_deep() { assert!(m("ROOT/parent_dir/dir_deep_20", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_20/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_20/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_20/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_20/child_dir/file", false).is_ignore() + ); // 21 assert!(m("ROOT/parent_dir/dir_deep_21", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_21/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_21/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_21/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_21/child_dir/file", false).is_ignore() + ); // 22 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_22", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_22/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_22/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_22/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_22/child_dir/file", false).is_ignore() + ); // 23 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_23", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_23/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_23/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_23/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_23/child_dir/file", false).is_ignore() + ); // 30 assert!(m("ROOT/parent_dir/dir_deep_30", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_30/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_30/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_30/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_30/child_dir/file", false).is_ignore() + ); // 31 assert!(m("ROOT/parent_dir/dir_deep_31", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_31/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_31/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_31/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_31/child_dir/file", false).is_ignore() + ); // 32 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_32", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_32/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_32/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_32/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_32/child_dir/file", false).is_ignore() + ); // 33 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_33", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_33/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_33/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_33/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_33/child_dir/file", false).is_ignore() + ); } diff --git a/crates/oxide/Cargo.toml b/crates/oxide/Cargo.toml index 0965320c9..f8a5f6137 100644 --- a/crates/oxide/Cargo.toml +++ b/crates/oxide/Cargo.toml @@ -20,6 +20,7 @@ ignore = { path = "../ignore" } regex = "1.11.1" [dev-dependencies] +insta = "1.48.0" tempfile = "3.13.0" pretty_assertions = "1.4.1" unicode-width = "0.2.0" diff --git a/crates/oxide/src/scanner/auto_source_detection.rs b/crates/oxide/src/scanner/auto_source_detection.rs index 0d723a1ab..7c0ac0b98 100644 --- a/crates/oxide/src/scanner/auto_source_detection.rs +++ b/crates/oxide/src/scanner/auto_source_detection.rs @@ -48,7 +48,7 @@ static IGNORED_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| ) }); -pub static BINARY_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { +static BINARY_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { format!( "*.{{{}}}", include_str!("fixtures/binary-extensions.txt") diff --git a/crates/oxide/src/scanner/fixtures/ignored-files.txt b/crates/oxide/src/scanner/fixtures/ignored-files.txt index 1e8f1029f..fe9aeb1bf 100644 --- a/crates/oxide/src/scanner/fixtures/ignored-files.txt +++ b/crates/oxide/src/scanner/fixtures/ignored-files.txt @@ -2,5 +2,6 @@ package-lock.json pnpm-lock.yaml bun.lockb .gitignore +.ignore .env .env.* diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index ab61b11b3..2abae99f5 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -10,11 +10,13 @@ use crate::scanner::sources::{ public_source_entries_to_private_source_entries, PublicSourceEntry, SourceEntry, Sources, }; use crate::GlobEntry; -use auto_source_detection::BINARY_EXTENSIONS_GLOB; use bstr::ByteSlice; use fast_glob::glob_match; use fxhash::{FxHashMap, FxHashSet}; -use ignore::{gitignore::GitignoreBuilder, WalkBuilder}; +use ignore::{ + gitignore::{Gitignore, GitignoreBuilder}, + WalkBuilder, +}; use init_tracing::{init_tracing, SHOULD_TRACE}; use rayon::prelude::*; use std::path::{Path, PathBuf}; @@ -22,16 +24,71 @@ use std::sync::{Arc, Mutex}; use std::time::SystemTime; use tracing::event; -// @source "some/folder"; // This is auto source detection -// @source "some/folder/**/*"; // This is auto source detection -// @source "some/folder/*.html"; // This is just a glob, but new files matching this should be included -// @source "node_modules/my-ui-lib"; // Auto source detection but since node_modules is explicit we allow it -// // Maybe could be considered `external(…)` automatically if: -// // 1. It's git ignored but listed explicitly -// // 2. It exists outside of the current working directory (do we know that?) +// # `@source` semantics // -// @source "do-include-me.bin"; // `.bin` is typically ignored, but now it's explicit so should be included -// @source "git-ignored.html"; // A git ignored file that is listed explicitly, should be scanned +// Every `@source` directive is classified as one of: +// +// - `Auto`: `@source "some/folder"` or `@source "some/folder/**/*"` — auto source detection. +// The folder is scanned recursively while respecting `.gitignore` files and the default rules +// (skip `node_modules`/`.git`/…, skip binary and irrelevant extensions, skip lock files, …). +// +// - `External`: an `Auto` source whose folder is itself ignored (by a `.gitignore` or because +// it's a default-ignored directory like `node_modules`), e.g. +// `@source "node_modules/my-ui-lib"`. Since the folder was listed explicitly, its ignoredness +// is bypassed: everything inside is scanned as if it were an `Auto` source, except that +// `.gitignore` files from at or above the folder no longer apply — they (including the +// self-ignoring `*` file that generators typically place inside such folders) are what made +// it ignored in the first place. `.gitignore` files *deeper inside* the folder still apply, +// and so do the default rules: nested `node_modules`, binary extensions, etc. stay ignored. +// +// - `Pattern`: `@source "some/folder/*.html"` — an explicit glob. Only files matching the glob +// are scanned. The *static* prefix of the glob (`some/folder`) is the explicit part: it is +// reached even when it is git ignored or hidden behind a default-ignored directory +// (`@source "node_modules/lib/dist/*.html"` works). The *wildcard* part is not explicit: +// while expanding it we still respect `.gitignore` files inside the walked subtree and the +// default-ignored directories (`@source "./**/*.html"` does not descend into `node_modules` +// or a git ignored `dist/`; `@source "./dist/**/*.html"` does descend into `dist/`). +// Individual *files* matching the glob are always included, even when git ignored — you were +// explicit about wanting files of that shape (`@source "git-ignored.html"` and +// `@source "*.styl"` work). Files that are ignored by default are only included when the +// pattern is explicit about them: it names a concrete file (`@source "do-include-me.bin"`, +// `@source ".env"`) or pins an extension (`@source "logo.{jpg,png}"`). A glob that does +// neither (e.g. `@source "blog/*/post/**/*"`) still applies the default file rules. +// +// - `Ignored`: `@source not "…"` — excludes matching files/folders, even when a `.gitignore` +// or another `@source` allows them. +// +// Later directives win over earlier ones on conflict: `@source not "./x"` followed by +// `@source "./x/keep.html"` scans `keep.html`, and vice versa excludes it. The same holds for +// directory sources: `@source not "./x"` followed by `@source "./x"` re-includes `./x` (with +// the normal auto source detection rules applied inside). +// +// # Implementation +// +// All sources are scanned in a single file system walk. The vendored `ignore` crate only +// provides the traversal itself (parallel walking, symlink loop handling); its built-in +// gitignore handling is disabled, because it computes one global verdict per path while the +// `@source` semantics are per source: the same directory can be pruned for an auto source but +// walkable for a pattern source, and a file can be gitignored for an auto source but rescued +// by a glob. +// +// Instead, the [`Resolver`] implements the semantics in the walker's `filter_entry` callback, +// backed by its own lazily-loaded cache of the on-disk ignore files (`.gitignore`, `.ignore`, +// and the repository's `.git/info/exclude`, applied up to the git repository root): +// +// - The walk roots are the "maximal" source bases; nested bases are reached by walking, and +// the resolver keeps the static path towards an explicitly listed base open even through +// ignored directories. +// +// - A directory is entered when at least one source can contribute files inside of it, where +// each source kind applies its own rules: auto sources check the default rules and the full +// gitignore chain, external sources only the default rules, and pattern sources check the +// default rules, the `.gitignore` files at or below their base, and whether the glob can +// match anything inside the directory. Directories that cannot contribute anything are +// never descended into. +// +// - A file is kept when at least one source includes it, honoring the directive order of +// `@source not` (the later directive wins). #[derive(Debug, Clone)] pub enum ChangedContent { @@ -63,6 +120,9 @@ pub struct Scanner { /// The walker to detect all files that we have to scan walker: Option, + /// The resolver implementing the `@source` semantics for the walker + resolver: Option>, + /// All found extensions extensions: FxHashSet, @@ -111,11 +171,13 @@ impl Scanner { } } - let walker = create_walker(&sources); + let resolver = Arc::new(Resolver::new(&sources)); + let walker = create_walker(resolver.clone()); Self { sources, walker, + resolver: Some(resolver), ..Default::default() } } @@ -408,7 +470,16 @@ impl Scanner { for entry in all_entries { match entry { WalkEntry::Dir(path) => { - self.dirs.insert(path); + // Directories that are only walked to reach an explicitly listed base are + // not part of any source's content: they must not widen the generated + // file watcher globs. + let contributes = self + .resolver + .as_ref() + .is_some_and(|resolver| resolver.contributes_dir(&path)); + if contributes { + self.dirs.insert(path); + } } WalkEntry::File { path, @@ -707,205 +778,639 @@ fn walk_parallel(walker: &mut WalkBuilder) -> Vec { Arc::try_unwrap(collected).unwrap().into_inner().unwrap() } -/// Sets up a WalkBuilder with all source roots, gitignore rules, and source pattern matching. +/// Sets up the single walker for all sources. /// -/// This is the common setup shared between the full walker (with mtime tracking for re-scans) -/// and the parallel walker (without mtime tracking for the initial scan). -fn create_walker(sources: &Sources) -> Option { - let mut other_roots: FxHashSet<&PathBuf> = FxHashSet::default(); - let mut first_root: Option<&PathBuf> = None; +/// The walker is only used for the (parallel, symlink aware) traversal itself: all ignore +/// semantics are implemented by the [`Resolver`] in the `filter_entry` callback. The crate's +/// built-in gitignore handling can't be used because it computes one global verdict per path, +/// while the `@source` semantics are per source (see the spec at the top of this file): the +/// same directory can be pruned for an auto source but walkable for a pattern source, and a +/// file can be gitignored for an auto source but rescued by a glob. +/// +/// The walk roots are the "maximal" source bases: a base contained in another base is reached +/// by walking, which the resolver allows even through ignored directories (the static path to +/// an explicit base bypasses ignore rules). +fn create_walker(resolver: Arc) -> Option { + let mut roots = resolver.walk_roots().into_iter(); + let first_root = roots.next()?; - let mut ignores: Vec<(&PathBuf, Vec)> = Default::default(); - let mut emit = |base, pattern| match ignores.last_mut() { - Some((prev_base, patterns)) if *prev_base == base => { - patterns.push(pattern); - } - _ => { - ignores.push((base, vec![pattern])); - } - }; - - for source in sources.iter() { - match source { - SourceEntry::Auto { base } => { - if first_root.is_none() { - first_root = Some(base); - } else { - other_roots.insert(base); - } - } - SourceEntry::Pattern { base, pattern } => { - let pattern = pattern.to_owned(); - - if first_root.is_none() { - first_root = Some(base); - } else { - other_roots.insert(base); - } - - if !pattern.contains("**") { - // Specific patterns should take precedence even over git-ignored files: - emit(base, format!("!{}", pattern)); - } else { - // Assumption: the pattern we receive will already be brace expanded. So - // `*.{html,jsx}` will result in two separate patterns: `*.html` and `*.jsx`. - if let Some(extension) = Path::new(&pattern).extension() { - // Extend auto source detection to include the extension - emit(base, format!("!*.{}", extension.to_string_lossy())); - } - } - } - SourceEntry::Ignored { base, pattern } => { - emit(base, pattern.to_owned()); - } - SourceEntry::External { base } => { - if first_root.is_none() { - first_root = Some(base); - } else { - other_roots.insert(base); - } - - // External sources should take precedence even over git-ignored files: - emit(base, "!/**/*".to_owned()); - - // External sources should still disallow binary extensions: - emit(base, BINARY_EXTENSIONS_GLOB.clone()); - } - } + let mut builder = WalkBuilder::new(first_root); + for root in roots { + builder.add(root); } - let mut builder = WalkBuilder::new(first_root?); - // We have to follow symlinks builder.follow_links(true); - // Scan hidden files / directories - builder.hidden(false); + // Disable all of the built-in filtering (hidden files, .gitignore files, parent + // directories, global gitignore files, …): the resolver implements the ignore semantics. + builder.standard_filters(false); - // Don't respect global gitignore files - builder.git_global(false); - - // By default, allow .gitignore files to be used regardless of whether or not - // a .git directory is present. This is an optimization for when projects - // are first created and may not be in a git repo yet. - builder.require_git(false); - - // If we are in a git repo then require it to ensure that only rules within - // the repo are used. For example, we don't want to consider a .gitignore file - // in the user's home folder if we're in a git repo. - // - // The alternative is using a call like `.parents(false)` but that will - // prevent looking at parent directories for .gitignore files from within - // the repo and that's not what we want. - // - // For example, in a project with this structure: - // - // home - // .gitignore - // my-project - // .gitignore - // apps - // .gitignore - // web - // {root} - // - // We do want to consider all .gitignore files listed: - // - home/.gitignore - // - my-project/.gitignore - // - my-project/apps/.gitignore - // - // However, if a repo is initialized inside my-project then only the following - // make sense for consideration: - // - my-project/.gitignore - // - my-project/apps/.gitignore - // - // Setting the require_git(true) flag conditionally allows us to do this. - for parent in first_root?.ancestors() { - if parent.join(".git").exists() { - builder.require_git(true); - break; - } - } - - for root in other_roots { - builder.add(root); - } - - // Setup auto source detection rules - for ignore in auto_source_detection::RULES.iter() { - builder.add_gitignore(ignore.clone()); - } - - // Setup ignores based on `@source` definitions - for (base, patterns) in ignores { - let mut ignore_builder = GitignoreBuilder::new(base); - for pattern in patterns { - ignore_builder.add_line(None, &pattern).unwrap(); - } - let ignore = ignore_builder.build().unwrap(); - builder.add_gitignore(ignore); - } - - // Pre-compute source matching data to avoid allocations in the hot filter_entry path - let auto_bases: Vec = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base.clone()), - _ => None, - }) - .collect(); - - let pattern_sources: Vec<(PathBuf, String)> = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Pattern { base, pattern } => Some((base.into(), pattern.into())), - _ => None, - }) - .collect(); - - // Source pattern matching filter (lock-free, safe for parallel walking) - builder.filter_entry(move |entry| { - let path = entry.path(); - - // Ensure the entries are matching any of the provided source patterns (this is - // necessary for manual-patterns that can filter the file extension) - if path.is_file() { - let mut matches = false; - - for base in &auto_bases { - if path.starts_with(base) { - matches = true; - break; - } - } - - if !matches { - for (base, pattern) in &pattern_sources { - let remainder = path.strip_prefix(base); - if remainder.is_ok_and(|remainder| { - let mut path_str = remainder.to_string_lossy().to_string(); - if !path_str.starts_with("/") { - path_str = format!("/{path_str}"); - } - glob_match(pattern, path_str.as_bytes()) - }) { - matches = true; - break; - } - } - } - - if !matches { - return false; - } - } - - true - }); + builder.filter_entry(move |entry| resolver.keep(entry)); Some(builder) } +/// Implements the `@source` semantics for a single file system walk. +/// +/// For every walked entry, the resolver computes the union of the per-source verdicts: +/// +/// - a directory is entered when at least one source can contribute files inside of it +/// - a file is kept when at least one source includes it, honoring the directive order of +/// `@source not` (the later directive wins) +/// +/// The per-source verdicts require gitignore decisions relative to different anchors (see +/// [`Boundary`]: each source kind respects a different part of the `.gitignore` chain), so the +/// resolver maintains its own lazily-loaded cache of ignore files instead of using the +/// walker's built-in handling. +#[derive(Debug)] +struct Resolver { + /// `Auto` sources: directive position and base + autos: Vec<(usize, PathBuf)>, + + /// `External` sources: directive position and base + externals: Vec<(usize, PathBuf)>, + + /// `Pattern` sources, grouped by base + patterns: Vec, + + /// `@source not` directives + nots: Vec, + + /// Lazily loaded ignore files (`.gitignore`, `.ignore`, `.git/info/exclude`) per directory + ignore_files: IgnoreFiles, + + /// Memoized "is this directory reachable for an auto source": every directory on the path + /// from an auto base down to it passes the gitignore chain and the default rules + auto_reachable: Mutex>, + + /// Memoized "is this directory reachable for an external source": like `auto_reachable`, + /// but only `.gitignore` files strictly below the external base apply + external_reachable: Mutex>, +} + +/// All `Pattern` sources sharing a base, e.g. `@source "src/*.{html,jsx}"` produces the +/// patterns `/*.html` and `/*.jsx` for the base `src`. Patterns carry the position of their +/// `@source` directive to resolve conflicts with `@source not` directives: the later +/// directive wins. +#[derive(Debug)] +struct PatternGroup { + base: PathBuf, + + /// The glob patterns, relative to `base`, with their directive positions + patterns: Vec<(usize, String)>, + + /// Memoized "is this directory reachable for this group": like + /// `Resolver::auto_reachable`, but only `.gitignore` files at or below the base apply — + /// the static base is explicit, so everything above it is bypassed + reachable: Mutex>, +} + +impl Resolver { + fn new(sources: &Sources) -> Self { + let mut autos = vec![]; + let mut externals = vec![]; + let mut patterns: Vec = vec![]; + let mut nots = vec![]; + + for (idx, source) in sources.iter().enumerate() { + match source { + SourceEntry::Auto { base } => autos.push((idx, base.clone())), + SourceEntry::External { base } => externals.push((idx, base.clone())), + SourceEntry::Pattern { base, pattern } => { + match patterns.iter_mut().find(|group| &group.base == base) { + Some(group) => group.patterns.push((idx, pattern.clone())), + None => patterns.push(PatternGroup { + base: base.clone(), + patterns: vec![(idx, pattern.clone())], + reachable: Default::default(), + }), + } + } + SourceEntry::Ignored { base, pattern } => nots.push(NotRule { + idx, + base: base.clone(), + pattern: pattern.clone(), + }), + } + } + + Self { + autos, + externals, + patterns, + nots, + ignore_files: IgnoreFiles::default(), + auto_reachable: Default::default(), + external_reachable: Default::default(), + } + } + + /// All source bases + fn bases(&self) -> impl Iterator { + self.autos + .iter() + .map(|(_, base)| base) + .chain(self.externals.iter().map(|(_, base)| base)) + .chain(self.patterns.iter().map(|group| &group.base)) + } + + /// The walk roots: all bases that are not contained in another base. Nested bases are + /// reached by walking (the resolver keeps the path to an explicit base open). + fn walk_roots(&self) -> Vec { + let mut roots: Vec = vec![]; + for base in self.bases() { + if self + .bases() + .any(|other| other != base && base.starts_with(other)) + { + continue; + } + if !roots.contains(base) { + roots.push(base.clone()); + } + } + roots + } + + /// Whether to keep the given walk entry + fn keep(&self, entry: &ignore::DirEntry) -> bool { + // Always keep the walk roots themselves; they are explicitly listed bases + if entry.depth() == 0 { + return true; + } + + let path = entry.path(); + let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); + + if is_dir { + self.keep_dir(path) + } else { + self.keep_file(path) + } + } + + /// Whether to keep walking the given directory: either a source can contribute files + /// inside of it, or it is on the static path towards an explicitly listed base. + fn keep_dir(&self, dir: &Path) -> bool { + // A directory on the static path to an explicit base can always be entered (the base + // is explicit, so its ignoredness is bypassed), unless everything below it is excluded + // again by a later `@source not` directive. + let leads_to_base = |idx: usize, base: &PathBuf| { + base != dir + && base.starts_with(dir) + && !self + .nots + .iter() + .any(|not| not.idx > idx && not.matches(base)) + }; + if self.autos.iter().any(|(idx, base)| leads_to_base(*idx, base)) + || self + .externals + .iter() + .any(|(idx, base)| leads_to_base(*idx, base)) + || self.patterns.iter().any(|group| { + group + .patterns + .iter() + .any(|(idx, _)| leads_to_base(*idx, &group.base)) + }) + { + return true; + } + + self.contributes_dir(dir) + } + + /// Whether at least one source can contribute files inside the given directory. Unlike + /// [`Resolver::keep_dir`], directories that are only walked to reach an explicitly listed + /// base don't count: they are not part of any source's content, e.g. for the purpose of + /// generating file watcher globs. + fn contributes_dir(&self, dir: &Path) -> bool { + // Some source must be able to contribute files inside the directory, and not be + // overridden by a later `@source not` directive. + let not_after = |idx: usize| { + self.nots + .iter() + .any(|not| not.idx > idx && not.matches(dir)) + }; + + if self + .autos + .iter() + .any(|(idx, _)| self.auto_reachable(dir) && !not_after(*idx)) + { + return true; + } + + if self + .externals + .iter() + .any(|(idx, _)| self.external_reachable(dir) && !not_after(*idx)) + { + return true; + } + + self.patterns.iter().any(|group| { + dir.strip_prefix(&group.base).is_ok_and(|remainder| { + group.patterns.iter().any(|(idx, pattern)| { + dir_could_contain_matches(pattern, remainder) && !not_after(*idx) + }) && self.pattern_reachable(group, dir) + }) + }) + } + + /// Whether at least one source includes the given file + fn keep_file(&self, file: &Path) -> bool { + let Some(parent) = file.parent() else { + return false; + }; + + let not_after = |idx: usize| { + self.nots + .iter() + .any(|not| not.idx > idx && not.matches(file)) + }; + + // Auto sources: the file must pass the default rules and the gitignore chain + if self.autos.iter().any(|(idx, base)| { + file.starts_with(base) + && self.auto_reachable(parent) + && !is_ignored_by_default_rules(file, false) + && !self + .ignore_files + .is_ignored(file, false, parent, Boundary::None) + && !not_after(*idx) + }) { + return true; + } + + // External sources: like auto sources, but only `.gitignore` files strictly below the + // base apply. Rules from at or above the base are bypassed — they are what made the + // directory ignored, and it was listed explicitly anyway — while `.gitignore` files + // deeper inside the external tree still apply. + if self.externals.iter().any(|(idx, base)| { + file.starts_with(base) + && self.external_reachable(parent) + && !is_ignored_by_default_rules(file, false) + && !self + .ignore_files + .is_ignored(file, false, parent, Boundary::Inside(base)) + && !not_after(*idx) + }) { + return true; + } + + // Pattern sources: the file must match a glob. A match beats file-level gitignore + // rules — you were explicit about wanting files of that shape — and the default file + // rules only apply when the pattern isn't explicit about the file's shape. + self.patterns.iter().any(|group| { + file.strip_prefix(&group.base).is_ok_and(|remainder| { + let remainder = rooted_posix(remainder); + group.patterns.iter().any(|(idx, pattern)| { + glob_match(pattern, remainder.as_bytes()) + && (pattern_bypasses_default_file_rules(pattern) + || !is_ignored_by_default_rules(file, false)) + && !not_after(*idx) + }) && self.pattern_reachable(group, parent) + }) + }) + } + + /// Whether the given directory is reachable for an auto source: every directory on the + /// path from the auto base down to it passes the default rules and the gitignore chain. + fn auto_reachable(&self, dir: &Path) -> bool { + reachable( + &self.auto_reachable, + dir, + |dir| self.autos.iter().any(|(_, base)| base == dir), + |dir| { + !is_ignored_by_default_rules(dir, true) + && dir.parent().is_some_and(|parent| { + !self.ignore_files.is_ignored(dir, true, parent, Boundary::None) + }) + }, + ) + } + + /// Like [`Resolver::auto_reachable`], but for external sources: only `.gitignore` files + /// strictly below the external base apply (see [`Boundary`]). Rules from at or above the + /// base are bypassed — they are what made the directory ignored, and it was listed + /// explicitly anyway. + /// + /// When external bases are nested, the deepest base containing the directory bounds the + /// chain: reachability from an outer base implies reachability from a nested base (the + /// outer chain checks a superset of the ignore files), so this computes the union of the + /// per-base verdicts. + fn external_reachable(&self, dir: &Path) -> bool { + reachable( + &self.external_reachable, + dir, + |dir| self.externals.iter().any(|(_, base)| base == dir), + |dir| { + !is_ignored_by_default_rules(dir, true) + && dir.parent().is_some_and(|parent| { + let boundary = self + .externals + .iter() + .map(|(_, base)| base) + .filter(|base| dir.starts_with(base)) + .max_by_key(|base| base.components().count()) + .map_or(Boundary::None, |base| Boundary::Inside(base)); + !self.ignore_files.is_ignored(dir, true, parent, boundary) + }) + }, + ) + } + + /// Like [`Resolver::auto_reachable`], but for a pattern group: only `.gitignore` files at + /// or below the base apply — the static base is explicit, so everything above it is + /// bypassed. + fn pattern_reachable(&self, group: &PatternGroup, dir: &Path) -> bool { + reachable( + &group.reachable, + dir, + |dir| dir == group.base, + |dir| { + !is_ignored_by_default_rules(dir, true) + && dir.parent().is_some_and(|parent| { + !self + .ignore_files + .is_ignored(dir, true, parent, Boundary::At(&group.base)) + }) + }, + ) + } +} + +/// Whether every directory on the path from a source base down to `dir` passes the source's +/// `enter` rule. Bases themselves are always reachable: they are explicitly listed (and an +/// auto base that is itself ignored would have been promoted to an external source). Verdicts +/// are memoized per directory. +fn reachable( + memo: &Mutex>, + dir: &Path, + is_base: impl Fn(&Path) -> bool, + enter: impl Fn(&Path) -> bool, +) -> bool { + // Walk up to the nearest base or directory with a memoized verdict… + let mut pending = vec![]; + let mut current = dir; + let mut reachable = loop { + if is_base(current) { + break true; + } + if let Some(reachable) = memo.lock().unwrap().get(current) { + break *reachable; + } + pending.push(current.to_path_buf()); + match current.parent() { + Some(parent) => current = parent, + // Reached the file system root without finding a base + None => break false, + } + }; + + // …then fill in the verdicts back down towards `dir` + for dir in pending.into_iter().rev() { + reachable = reachable && enter(&dir); + memo.lock().unwrap().insert(dir, reachable); + } + + reachable +} + +/// A lazily-loaded cache of the on-disk ignore files (`.gitignore`, `.ignore` and the +/// repository's `.git/info/exclude`). +#[derive(Debug, Default)] +struct IgnoreFiles { + /// The matcher chains per directory: all matchers that apply to paths inside the + /// directory, deepest first, from the directory itself up to the repository root (or the + /// file system root outside of a git repository, matching `git init`-less projects where + /// all ancestor `.gitignore` files apply) + chains: Mutex>>>>, +} + +/// Which part of the ignore file chain applies to a source, anchored at its base: +/// +/// - `Auto` sources respect the full chain, up to the git repository root +/// - `Pattern` sources respect ignore files at or below their base (the static base is +/// explicit, everything above it is bypassed) +/// - `External` sources respect ignore files strictly below their base: the base's own +/// ignore file is part of its bypassed ignoredness — inside ignored trees it is typically +/// the self-ignoring `*` file that generators drop into the directory — while deeper ignore +/// files are deliberate signals about specific contents and still apply +#[derive(Debug, Clone, Copy)] +enum Boundary<'a> { + None, + At(&'a Path), + Inside(&'a Path), +} + +impl Boundary<'_> { + /// Whether an ignore file rooted at the given directory applies + fn applies_to(&self, dir: &Path) -> bool { + match self { + Boundary::None => true, + Boundary::At(base) => dir.starts_with(base), + Boundary::Inside(base) => dir != *base && dir.starts_with(base), + } + } +} + +impl IgnoreFiles { + /// Whether the ignore files definitively ignore the given path. `dir` is the directory + /// containing the path, and `boundary` restricts which ignore files of the chain apply. + /// + /// The deepest ignore file with a definitive answer wins, matching git's precedence, so a + /// path that a deeper ignore file re-includes via a `!` pattern is not ignored. + fn is_ignored(&self, path: &Path, is_dir: bool, dir: &Path, boundary: Boundary) -> bool { + for matcher in self.chain(dir).iter() { + if !boundary.applies_to(matcher.path()) { + // Chains are ordered deepest first, so nothing below the boundary can follow + break; + } + + match matcher.matched(path, is_dir) { + ignore::Match::Ignore(_) => return true, + ignore::Match::Whitelist(_) => return false, + ignore::Match::None => {} + } + } + + false + } + + /// The matcher chain for paths inside the given directory: the directory's own matcher + /// first, then its parents' matchers, up to and including the git repository root. + fn chain(&self, dir: &Path) -> Arc>> { + if let Some(chain) = self.chains.lock().unwrap().get(dir) { + return chain.clone(); + } + + let is_repo_root = dir.join(".git").exists(); + + let mut chain = vec![]; + if let Some(matcher) = load_ignore_files(dir, is_repo_root) { + chain.push(matcher); + } + + // Stop at the git repository root so that ignore files outside of the repository are + // not considered. Without a repository, all ancestor ignore files apply. + if !is_repo_root { + if let Some(parent) = dir.parent() { + chain.extend(self.chain(parent).iter().cloned()); + } + } + + let chain = Arc::new(chain); + self.chains + .lock() + .unwrap() + .insert(dir.to_path_buf(), chain.clone()); + chain + } +} + +/// Compile the ignore rules of the given directory, combining (from low to high precedence) +/// the repository's `.git/info/exclude`, the `.gitignore` file, and the `.ignore` file. +fn load_ignore_files(dir: &Path, is_repo_root: bool) -> Option> { + let mut builder = GitignoreBuilder::new(dir); + let mut any = false; + + let mut add = |file: PathBuf| { + if file.is_file() { + // I/O errors and partially invalid ignore files are ignored, matching the + // walker's behavior. + let _ = builder.add(file); + any = true; + } + }; + + if is_repo_root { + add(dir.join(".git").join("info").join("exclude")); + } + add(dir.join(".gitignore")); + add(dir.join(".ignore")); + + if any { + builder.build().ok().map(Arc::new) + } else { + None + } +} + +/// Whether a path is ignored by the default auto source detection rules: directories like +/// `node_modules`, binary and irrelevant extensions, lock files, … For directories only the +/// directory's own name is checked; the path towards it is checked by the reachability +/// helpers one directory at a time. +fn is_ignored_by_default_rules(path: &Path, is_dir: bool) -> bool { + auto_source_detection::RULES + .iter() + .any(|ignore| ignore.matched(path, is_dir).is_ignore()) +} + +/// An `@source not` directive, together with its position. +#[derive(Debug, Clone)] +struct NotRule { + /// Position of the `@source not` directive + idx: usize, + + base: PathBuf, + + /// The glob pattern, relative to `base`, e.g. `/ignored/**/*` + pattern: String, +} + +impl NotRule { + /// Whether this directive excludes the given path: a file, or a directory and thereby + /// everything inside of it. + /// + /// Like a gitignore rule, the pattern excludes a whole subtree when it matches a + /// directory, so besides the path itself every ancestor directory (up to the directive's + /// base) is tested as well. E.g. `@source not "./src/ba*"` excludes `src/bar/index.html` + /// because `/ba*` matches the `src/bar` directory. Note that directory-shaped directives + /// (`@source not "./some/dir"`) are normalized to a `/**/*` pattern with the directory as + /// its base, which matches everything inside the directory directly. + fn matches(&self, path: &Path) -> bool { + let Ok(remainder) = path.strip_prefix(&self.base) else { + return false; + }; + + // A directory-shaped directive (normalized to a `/**/*` pattern) also excludes the + // base directory itself, not just its contents, so the directory can be pruned. + if remainder.as_os_str().is_empty() { + return self.pattern == "/**/*"; + } + + remainder.ancestors().any(|prefix| { + !prefix.as_os_str().is_empty() + && glob_match(&self.pattern, rooted_posix(prefix).as_bytes()) + }) + } +} + +/// Serialize a path relative to some base as a `/`-rooted posix style string, e.g. +/// `/ba*/index.html`, matching how source patterns are stored. +fn rooted_posix(path: &Path) -> String { + let posix = crate::scanner::sources::path_to_posix_string(path); + if posix.starts_with('/') { + posix + } else { + format!("/{posix}") + } +} + +/// Whether a directory (relative to the pattern's base) can contain files matching the pattern. +/// Used to prune directories that can never contribute, e.g. for `/ba*/*.html` only `ba*` +/// directories are entered. +fn dir_could_contain_matches(pattern: &str, dir: &Path) -> bool { + let pattern_components: Vec<&str> = pattern + .trim_start_matches('/') + .split('/') + .filter(|c| !c.is_empty()) + .collect(); + + for (i, component) in dir.components().enumerate() { + let component = component.as_os_str().to_string_lossy(); + + // Once we see a `**` everything nested can contain matches + match pattern_components.get(i) { + Some(&"**") => return true, + // The last pattern component matches files, not directories. A directory nested + // deeper than the pattern's directory part can never contain matches. + Some(_) if i + 1 >= pattern_components.len() => return false, + Some(pattern_component) => { + if !glob_match(pattern_component, component.as_bytes()) { + return false; + } + } + None => return false, + } + } + + true +} + +/// Whether a pattern is explicit enough to bypass the default file rules. +/// +/// A pattern without any wildcards names a concrete file, e.g. `/.env` or +/// `/do-include-me.bin` — you asked for exactly this file, so the default rules never apply. +/// A pattern that pins a specific extension, e.g. `/*.html` or `/**/*.bin`, bypasses them as +/// well. Patterns that do neither (e.g. `/blog/*/**/*`) keep the default file rules applied. +fn pattern_bypasses_default_file_rules(pattern: &str) -> bool { + // Concrete file, no wildcards (braces have already been expanded away) + if !pattern.contains(['*', '?', '[']) { + return true; + } + + // Pinned extension + match Path::new(pattern).extension().and_then(|ext| ext.to_str()) { + Some(ext) => !ext.contains(['*', '?', '[']), + None => false, + } +} + #[cfg(test)] mod tests { use super::{ChangedContent, Scanner}; diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 53c09387f..15ca9ae54 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -1,6 +1,5 @@ -use crate::GlobEntry; use bexpand::Expression; -use fxhash::{FxHashMap, FxHashSet}; +use fxhash::FxHashMap; use ignore::gitignore::Gitignore; use std::path::{Component, Path, PathBuf}; use tracing::{event, Level}; @@ -31,6 +30,24 @@ pub enum SourceEntry { /// ``` Auto { base: PathBuf }, + /// An `Auto` source whose directory is itself ignored (by the default rules, e.g. + /// `node_modules`, or by a `.gitignore`) but was explicitly listed anyway. + /// + /// Represented by: + /// + /// ```css + /// @source "../node_modules/my-lib";` + /// @source "../node_modules/my-lib/**/*";` + /// ``` + /// + /// Being explicit bypasses the ignoredness of the directory: everything inside is scanned + /// as if it were a regular auto source, except that `.gitignore` files from at or above + /// the directory no longer apply — they (including the self-ignoring `*` file that + /// generators typically place inside such directories) are what made it ignored in the + /// first place. `.gitignore` files *deeper inside* the directory still apply, and so do + /// the default rules, so e.g. nested `node_modules` stay ignored. + External { base: PathBuf }, + /// Explicit source pattern regardless of any auto source detection rules /// /// Represented by: @@ -48,18 +65,11 @@ pub enum SourceEntry { /// @source not "src";` /// @source not "src/**/*.html";` /// ``` + /// + /// Note that directory-shaped directives (`@source not "src"`) are normalized to + /// `base: "src", pattern: "/**/*"`, which is semantically identical: everything under the + /// directory is ignored. Ignored { base: PathBuf, pattern: String }, - - /// External sources are directories that are ignored (by us or .gitignore rules), but should be - /// included bypassing the default ignore rules. - /// - /// Represented by: - /// - /// ```css - /// @source "../node_modules/my-lib";` - /// @source "../node_modules/my-lib/**/*";` - /// ``` - External { base: PathBuf }, } #[derive(Debug, Clone, Default)] @@ -77,164 +87,6 @@ impl Sources { } } -/// When dealing with a pattern, then it could be that we end up with: -/// -/// ```json -/// { base: '/some/folder', pattern: 'foo.ts' } -/// ``` -/// -/// If we just emit `!foo.ts` for the `/some/folder` path, then _everything_ else in that folder -/// would still be walked (but the result will be ignored). -/// -/// Instead, we have to ensure that we ignore everything in that folder _except_ for the `foo.ts` -/// pattern. -/// -/// This should be equivalent to: -/// ```gitignore -/// * -/// !foo.ts -/// ``` -/// -/// However, we have to be careful that we don't start ignoring files/folders that already exist. -/// ```css -/// @source "./some/folder/foo.ts"; -/// @source "./some/folder/bar.ts"; -/// ``` -/// Would result in: -/// ```json -/// { base: '/some/folder', pattern: 'foo.ts' } -/// { base: '/some/folder', pattern: 'bar.ts' } -/// ``` -/// -/// If we were to blindly emit `*` for each pattern, then the `.gitignore` equivalent would look like -/// this: -/// ```gitignore -/// * -/// !foo.ts -/// * -/// !bar.ts -/// ``` -/// -/// This would result in ignoring the `foo.ts` file as well. Therefore we only want to insert -/// this `*` pattern when nothing else exists yet. -/// -/// There is another problem that we need to solve. Let's say you have a pattern that contains a `*` -/// in the pattern: -/// ```css -/// @source './src/ba*/*.html'; -/// ``` -/// -/// This would result in -/// ```json -/// { base: '/src', pattern: '/ba*/*.html' } -/// ``` -/// -/// If we now inject the `*` pattern for the `/src` folder, then we wouldn't scan any `ba*` folders -/// (e.g. `bar` or `baz`). For this, we have to make sure that we add inverse patterns for these -/// folders. This would essentially result in: -/// ```gitignore -/// * ← ignore everything -/// !/ba*/ ← except for the `ba*/` pattern, so we scan these folders -/// !/ba*/*.html ← then ensure we scan the `*.html` files in it as well -/// ``` -/// -fn expand_restricted_patterns(sources: Vec) -> Vec { - let unrestricted_roots = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base.clone()), - SourceEntry::Pattern { base, pattern } if pattern.contains("**") => Some(base.clone()), - _ => None, - }) - .collect::>(); - - // Bases of restricted patterns. Each of these becomes its own walk root with its own - // `*` + `!` rules, so an ancestor base must not ignore them recursively. - let pattern_roots = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Pattern { base, .. } => Some(base.clone()), - _ => None, - }) - .collect::>(); - - let mut restricted_roots: FxHashSet = FxHashSet::default(); - let mut expanded = vec![]; - - for source in sources { - let SourceEntry::Pattern { base, pattern } = &source else { - expanded.push(source); - continue; - }; - - // `base` is already included by another `@source` that we know should be walked. This - // includes the case where `base` is _nested_ inside such a root, because everything under - // an unrestricted root is walked already. Restricting it would incorrectly hide siblings - // that the broader source is supposed to pick up. - if unrestricted_roots.iter().any(|root| base.starts_with(root)) { - expanded.push(source); - continue; - } - - // Ignore everything in the directory. We will later add the specific patterns we are - // interested in. - if restricted_roots.insert(base.clone()) { - // When another source root is nested inside this base — an unrestricted root, or the - // base of another restricted pattern (which is walked from its own root with its own - // rules) — only ignore direct children so the nested root can still be walked. - let has_nested_root = unrestricted_roots.iter().any(|root| root.starts_with(base)) - || pattern_roots - .iter() - .any(|root| root != base && root.starts_with(base)); - - let pattern = if has_nested_root { "/*" } else { "*" }; - - expanded.push(SourceEntry::Ignored { - base: base.clone(), - pattern: pattern.to_owned(), - }); - } - - // Ensure to _include_ parent paths, otherwise the `*` from above would block walking the - // folders that need to be walked. - // - // ```css - // @source './src/ba*/*.html'; - // ``` - // - // ```gitignore - // * ← added by the above rule - // !/ba*/ ← this is what we're focusing on in this block - // !/ba*/*.html ← this is added later - // ``` - { - let mut dir = PathBuf::new(); - let mut components = Path::new(pattern).components().peekable(); - - while let Some(component) = components.next() { - if components.peek().is_none() { - break; - } - - match component { - Component::Prefix(_) | Component::RootDir | Component::CurDir => continue, - Component::ParentDir | Component::Normal(_) => dir.push(component), - } - - expanded.push(SourceEntry::Ignored { - base: base.clone(), - pattern: format!("!/{}/", path_to_posix_string(&dir).trim_start_matches('/')), - }); - } - } - - // Track the original source - expanded.push(source); - } - - expanded -} - impl PublicSourceEntry { /// Optimize the PublicSourceEntry by trying to move all the static parts of the pattern to the /// base of the PublicSourceEntry. @@ -340,10 +192,16 @@ impl PublicSourceEntry { else if !self.pattern.starts_with("/") { self.pattern = format!("/{}", self.pattern); } + + // `src/**` means everything underneath `src`, just like `src/**/*` and `src` do. + // Normalize it so all three are classified as auto source detection. + if self.pattern == "/**" { + self.pattern = "/**/*".to_owned(); + } } } -fn path_to_posix_string(path: &Path) -> String { +pub(crate) fn path_to_posix_string(path: &Path) -> String { let mut parts = Vec::new(); let mut is_rooted = false; @@ -474,65 +332,32 @@ mod tests { } #[test] - fn concrete_patterns_are_expanded_to_restrict_their_base() { + fn optimize_normalizes_double_star_to_auto_source_detection() { let dir = tempdir().unwrap(); fs::create_dir_all(dir.path().join("src")).unwrap(); - let base = dunce::canonicalize(dir.path().join("src")).unwrap(); + let mut source = PublicSourceEntry { + base: dir.path().to_string_lossy().to_string(), + pattern: "src/**".to_string(), + negated: false, + }; + + source.optimize(); + + assert_eq!(source.pattern, "/**/*"); + + // …and therefore `src/**` is classified as an auto source, like `src/**/*` and `src` + let base = dunce::canonicalize(dir.path().join("src")).unwrap(); let sources = public_source_entries_to_private_source_entries(vec![PublicSourceEntry { base: dir.path().to_string_lossy().to_string(), - pattern: "src/foo.html".to_string(), + pattern: "src/**".to_string(), negated: false, }]); - - assert_eq!( - sources, - vec![ - SourceEntry::Ignored { - base: base.clone(), - pattern: "*".to_string(), - }, - SourceEntry::Pattern { - base, - pattern: "/foo.html".to_string(), - }, - ] - ); + assert_eq!(sources, vec![SourceEntry::Auto { base }]); } #[test] - fn restricted_patterns_include_parent_directory_allow_rules() { - let dir = tempdir().unwrap(); - fs::create_dir_all(dir.path().join("src")).unwrap(); - let base = dunce::canonicalize(dir.path().join("src")).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![PublicSourceEntry { - base: dir.path().to_string_lossy().to_string(), - pattern: "src/ef*/*.html".to_string(), - negated: false, - }]); - - assert_eq!( - sources, - vec![ - SourceEntry::Ignored { - base: base.clone(), - pattern: "*".to_string(), - }, - SourceEntry::Ignored { - base: base.clone(), - pattern: "!/ef*/".to_string(), - }, - SourceEntry::Pattern { - base, - pattern: "/ef*/*.html".to_string(), - }, - ] - ); - } - - #[test] - fn unrestricted_sources_do_not_expand_patterns_for_the_same_base() { + fn sources_are_converted_in_order() { let dir = tempdir().unwrap(); fs::create_dir_all(dir.path().join("src")).unwrap(); let base = dunce::canonicalize(dir.path().join("src")).unwrap(); @@ -548,72 +373,6 @@ mod tests { pattern: "src/foo.html".to_string(), negated: false, }, - ]); - - assert_eq!( - sources, - vec![ - SourceEntry::Auto { base: base.clone() }, - SourceEntry::Pattern { - base, - pattern: "/foo.html".to_string(), - }, - ] - ); - } - - #[test] - fn restricted_parent_bases_do_not_open_unrelated_siblings() { - let dir = tempdir().unwrap(); - let project = dir.path().join("Users").join("robin").join("docus-test"); - fs::create_dir_all(&project).unwrap(); - - let users = dunce::canonicalize(dir.path().join("Users")).unwrap(); - let project = dunce::canonicalize(project).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![ - PublicSourceEntry { - base: project.to_string_lossy().to_string(), - pattern: "**/*".to_string(), - negated: false, - }, - PublicSourceEntry { - base: project.to_string_lossy().to_string(), - pattern: "../../app.config.ts".to_string(), - negated: false, - }, - ]); - - assert_eq!( - sources, - vec![ - SourceEntry::Auto { - base: project.clone(), - }, - SourceEntry::Ignored { - base: users.clone(), - pattern: "/*".to_string(), - }, - SourceEntry::Pattern { - base: users, - pattern: "/app.config.ts".to_string(), - }, - ] - ); - } - - #[test] - fn restricted_patterns_preserve_source_order() { - let dir = tempdir().unwrap(); - fs::create_dir_all(dir.path().join("src")).unwrap(); - let base = dunce::canonicalize(dir.path().join("src")).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![ - PublicSourceEntry { - base: dir.path().to_string_lossy().to_string(), - pattern: "src/foo.html".to_string(), - negated: false, - }, PublicSourceEntry { base: dir.path().to_string_lossy().to_string(), pattern: "src/foo.html".to_string(), @@ -624,10 +383,7 @@ mod tests { assert_eq!( sources, vec![ - SourceEntry::Ignored { - base: base.clone(), - pattern: "*".to_string(), - }, + SourceEntry::Auto { base: base.clone() }, SourceEntry::Pattern { base: base.clone(), pattern: "/foo.html".to_string(), @@ -660,7 +416,10 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::Auto { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { base } + ); } #[test] @@ -670,7 +429,10 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::External { base } + ); } #[test] @@ -684,7 +446,10 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::External { base } + ); } #[test] @@ -698,7 +463,54 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::External { base } + ); + } + + #[test] + fn folders_reincluded_by_a_deeper_gitignore_stay_auto_sources() { + let dir = tempdir().unwrap(); + fs::create_dir_all(dir.path().join(".git")).unwrap(); + fs::write(dir.path().join(".gitignore"), "generated/\n").unwrap(); + // The deepest `.gitignore` with a definitive answer wins: the re-include is reachable + // (no parent directory of `generated` is excluded), so `generated` is not ignored. + fs::create_dir_all(dir.path().join("packages").join("app")).unwrap(); + fs::write( + dir.path().join("packages").join("app").join(".gitignore"), + "!generated/\n", + ) + .unwrap(); + + let base = dir.path().join("packages").join("app").join("generated"); + fs::create_dir_all(&base).unwrap(); + let base = dunce::canonicalize(&base).unwrap(); + + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { base } + ); + } + + #[test] + fn folders_inside_excluded_directories_become_external_sources() { + let dir = tempdir().unwrap(); + fs::create_dir_all(dir.path().join(".git")).unwrap(); + fs::write(dir.path().join(".gitignore"), "parent/\n").unwrap(); + // This whitelist is unreachable: `parent` itself is excluded, so git never descends + // into it and the re-include of `child` has no effect. + fs::create_dir_all(dir.path().join("parent")).unwrap(); + fs::write(dir.path().join("parent").join(".gitignore"), "!child/\n").unwrap(); + + let base = dir.path().join("parent").join("child"); + fs::create_dir_all(&base).unwrap(); + let base = dunce::canonicalize(&base).unwrap(); + + assert_eq!( + auto_source_entry(&base), + SourceEntry::External { base } + ); } #[test] @@ -711,7 +523,10 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::Auto { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { base } + ); } } @@ -778,74 +593,79 @@ pub fn public_source_entries_to_private_source_entries( .map(|public_source| { let mut source: SourceEntry = public_source.into(); - // Promote auto-sources to external sources if they were gitignored + // Mark auto sources as external if their directory is gitignored if let SourceEntry::Auto { ref base } = source { let inside_git_repo = base.ancestors().any(|dir| dir.join(".git").exists()); - // Walk up from the folder, applying each `.gitignore` relative to the directory - // that contains it (matching git), and stop at the git repository root so - // `.gitignore` files outside of the repo are not considered. + // The chain of directories whose `.gitignore` files can apply: `base` itself and + // its ancestors, up to and including the git repository root so `.gitignore` + // files outside of the repo are not considered. + // + // Without a git repository there is no repository root to stop at. Stop once the + // directory contains the current working directory instead, so `.gitignore` + // files outside of the project (e.g. in the user's home directory) can never + // promote a source to an external source. Note that the file walker still + // applies those `.gitignore` files when deciding which files to scan. + let mut chain: Vec<&Path> = vec![]; for dir in base.ancestors() { - let gitignore = gitignores.entry(dir.to_path_buf()).or_insert_with(|| { - let path = dir.join(".gitignore"); + chain.push(dir); - // `Gitignore::new` roots the matcher at the directory containing the file, - // so patterns match relative to it. - path.is_file().then(|| Gitignore::new(&path).0) - }); - - // Only `.gitignore` files in ancestors of `base` can ignore `base` itself. - // Patterns in `base`'s own `.gitignore` only match paths _inside_ `base`, never - // `base` itself (the file walker still applies them to `base`'s contents). - // - // Skipping `base`'s own `.gitignore` also prevents a false positive for - // whitelist style `.gitignore` files, because relativizing `base` against - // itself yields the empty path, which incorrectly matches `/*`. - // - // E.g.: - // - // ```gitignore - // /* - // !/.gitignore - // !/app - // !/public - // ``` - // - // Everything inside `base` except `.gitignore`, `app` and `public` is ignored, - // but `base` itself is not. - if dir != base { - if let Some(gitignore) = gitignore { - if gitignore - .matched_path_or_any_parents(&base, true) - .is_ignore() - { - source = SourceEntry::External { base: base.into() }; - break; - } - } - } - - // Stop at the git repository root. if dir.join(".git").exists() { break; } - // Without a git repository there is no repository root to stop at. Stop once - // the directory contains the current working directory instead, so `.gitignore` - // files outside of the project (e.g. in the user's home directory) can never - // promote a source to an external source. Note that the file walker still - // applies those `.gitignore` files when deciding which files to scan. if !inside_git_repo && cwd.as_ref().is_some_and(|cwd| cwd.starts_with(dir)) { break; } } + + // Match git's semantics: a directory is ignored when the directory itself or any + // of its parent directories is excluded, and it is not possible to re-include a + // directory once a parent directory is excluded — git never descends into an + // excluded directory, so whitelist rules inside of it are unreachable. + // + // So walk the path from the top down (`chain` is ordered bottom-up: `base` at + // index 0, the boundary last) and decide for every directory along the way + // whether it is excluded. The first excluded directory settles it. For a single + // directory, only `.gitignore` files in its parent directories can match it (its + // own `.gitignore` only matches paths _inside_ of it), and the deepest + // `.gitignore` with a definitive answer wins, so a directory that is re-included + // by a deeper `!the-directory` pattern is not ignored, even when an ancestor + // `.gitignore` ignores it. + 'prefixes: for i in (0..chain.len().saturating_sub(1)).rev() { + let prefix = chain[i]; + + for dir in &chain[i + 1..] { + let gitignore = gitignores.entry(dir.to_path_buf()).or_insert_with(|| { + let path = dir.join(".gitignore"); + + // `Gitignore::new` roots the matcher at the directory containing the + // file, so patterns match relative to it. + path.is_file().then(|| Gitignore::new(&path).0) + }); + + let Some(gitignore) = gitignore else { + continue; + }; + + match gitignore.matched(prefix, true) { + ignore::Match::Ignore(_) => { + source = SourceEntry::External { base: base.into() }; + break 'prefixes; + } + // Re-included; this directory is reachable, move on to the next one. + ignore::Match::Whitelist(_) => continue 'prefixes, + ignore::Match::None => {} + } + } + } } source }) .collect::>(); - expand_restricted_patterns(sources) + sources } /// Convert a public source entry to a source entry @@ -858,8 +678,15 @@ impl From for SourceEntry { }; } - let auto = - value.pattern == "/**/*" || PathBuf::from(&value.base).join(&value.pattern).is_dir(); + // After a successful `optimize()` any trailing concrete directory has already been + // hoisted into the base, so a folder source always has the `/**/*` pattern. The + // `is_dir` check only matters when `optimize()` could not canonicalize the base and + // left the entry untouched. Note that the pinned leading `/` has to be stripped, since + // joining an absolute-looking path onto the base would discard the base entirely. + let auto = value.pattern == "/**/*" + || PathBuf::from(&value.base) + .join(value.pattern.trim_start_matches('/')) + .is_dir(); if !auto { return SourceEntry::Pattern { @@ -868,6 +695,8 @@ impl From for SourceEntry { }; } + // A directory inside e.g. `node_modules` is ignored by default, so listing it + // explicitly makes it an external source. let inside_ignored_content_dir = IGNORED_CONTENT_DIRS.iter().any(|dir| { value.base.contains(&format!( "{}{}{}", @@ -889,50 +718,3 @@ impl From for SourceEntry { } } } - -impl From for SourceEntry { - fn from(value: GlobEntry) -> Self { - SourceEntry::Pattern { - base: PathBuf::from(value.base), - pattern: value.pattern, - } - } -} - -impl From for GlobEntry { - fn from(value: SourceEntry) -> Self { - match value { - SourceEntry::Auto { base } | SourceEntry::External { base } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: "**/*".into(), - }, - SourceEntry::Pattern { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - SourceEntry::Ignored { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - } - } -} - -impl From<&SourceEntry> for GlobEntry { - fn from(value: &SourceEntry) -> Self { - match value { - SourceEntry::Auto { base } | SourceEntry::External { base } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: "**/*".into(), - }, - SourceEntry::Pattern { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - SourceEntry::Ignored { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - } - } -} diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 10a455d8b..42c08d9b3 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -1,5 +1,6 @@ #[cfg(test)] mod scanner { + use insta::assert_snapshot; use pretty_assertions::assert_eq; use std::path::{Path, PathBuf}; use std::process::Command; @@ -55,6 +56,162 @@ mod scanner { globs: Vec, normalized_sources: Vec, candidates: Vec, + tree: String, + } + + /// Renders the directory as a tree, annotating every file and folder with + /// an indicator that shows whether the scanner picked it up: + /// + /// - `✓` — scanned + /// - `✗` — ignored / skipped + /// + /// Symlinks are rendered as `link → target`, where the target is shown + /// relative to the folder containing the symlink. The contents of every + /// `.gitignore` file are printed right below the file itself. + /// + /// Folders that are a git repository root (they contain a `.git` folder) + /// are marked with `(git)`, because ignore rules behave differently + /// inside and outside of a repository. + fn fs_tree(root: &Path, scanned: &[String]) -> String { + /// Computes the relative path from `parent` to `target`, where both + /// are relative to the same root, e.g.: `b/c` seen from `b` is `c`, + /// and the root itself seen from `b` is `..`. + fn relative_to(target: &Path, parent: &Path) -> String { + let target = target.components().collect::>(); + let parent = parent.components().collect::>(); + + let common = target + .iter() + .zip(&parent) + .take_while(|(a, b)| **a == **b) + .count(); + + let mut parts = vec![".."; parent.len() - common]; + parts.extend( + target[common..] + .iter() + .map(|component| component.as_os_str().to_str().unwrap()), + ); + + if parts.is_empty() { + // The symlink points to its own parent folder + ".".into() + } else { + parts.join("/") + } + } + + fn walk( + dir: &Path, + root: &Path, + prefix: &str, + scanned: &[String], + visited: &mut Vec, + out: &mut String, + ) { + let mut entries = fs::read_dir(dir) + .unwrap() + .filter_map(Result::ok) + .filter(|entry| entry.file_name() != ".git") + .collect::>(); + entries.sort_by_key(|entry| entry.file_name()); + + for (i, entry) in entries.iter().enumerate() { + let last = i == entries.len() - 1; + let connector = if last { "└── " } else { "├── " }; + let child_prefix = if last { " " } else { "│ " }; + + let path = entry.path(); + let name = entry.file_name().to_string_lossy().to_string(); + let rel = path + .strip_prefix(root) + .unwrap() + .to_string_lossy() + .replace('\\', "/"); + + let file_type = entry.file_type().unwrap(); + + // A file is scanned when it shows up in the scanned files + // list. A folder is considered scanned when any scanned file + // lives inside of it. Paths that go through a symlink count + // towards the symlink, not towards the target directory. + let dir_prefix = format!("{rel}/"); + let is_scanned = scanned + .iter() + .any(|file| file == &rel || file.starts_with(&dir_prefix)); + let indicator = if is_scanned { "✓" } else { "✗" }; + + let mut display_name = name.clone(); + if path.join(".git").exists() { + display_name = format!("{display_name} (git)"); + } + if file_type.is_symlink() { + if let Ok(target) = fs::read_link(&path) { + // Show the target relative to the folder containing + // the symlink, or as-is when it points outside of the + // tree. + let target = match (target.strip_prefix(root), dir.strip_prefix(root)) { + (Ok(target), Ok(parent)) => relative_to(target, parent), + _ => target.to_string_lossy().replace('\\', "/"), + }; + display_name = format!("{name} → {target}"); + + // The symlink points to a target that doesn't exist + if path.metadata().is_err() { + display_name = format!("{display_name} (broken)"); + } + } + } + + out.push_str(&format!("{prefix}{connector}{indicator} {display_name}\n")); + + // Print the contents of `.gitignore` files below the file + // itself, so the ignore rules are visible in the same output. + if name == ".gitignore" { + if let Ok(contents) = fs::read_to_string(&path) { + for line in contents.lines() { + let line = format!("{prefix}{child_prefix} {line}"); + out.push_str(line.trim_end()); + out.push('\n'); + } + } + } + + // Descend into directories, including symlinked ones. The + // paths inside a symlinked directory are computed through the + // symlink, so the indicators show what the scanner saw via + // that route. A visited stack prevents symlink cycles from + // recursing forever. + if path.metadata().map(|meta| meta.is_dir()).unwrap_or(false) { + let Ok(canonical) = dunce::canonicalize(&path) else { + continue; + }; + if visited.contains(&canonical) { + continue; + } + + visited.push(canonical); + walk( + &path, + root, + &format!("{prefix}{child_prefix}"), + scanned, + visited, + out, + ); + visited.pop(); + } + } + } + + let mut visited = vec![dunce::canonicalize(root).unwrap()]; + let mut out = if root.join(".git").exists() { + String::from(". (git)\n") + } else { + String::from(".\n") + }; + walk(root, root, "", scanned, &mut visited, &mut out); + out } fn create_files_in(dir: &path::Path, paths: &[(&str, &str)]) { @@ -167,11 +324,14 @@ mod scanner { .collect::>(); normalized_sources.sort(); + let tree = fs_tree(&dir, &files); + ScanResult { files, globs, normalized_sources, candidates, + tree, } } @@ -185,6 +345,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -192,6 +353,14 @@ mod scanner { ("b.html", ""), ("c.html", ""), ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + └── ✓ index.html + "); assert_eq!(files, vec!["a.html", "b.html", "c.html", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -203,6 +372,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ (".gitignore", "b.html"), @@ -211,6 +381,17 @@ mod scanner { ("b.html", ""), ("c.html", ""), ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ b.html + ├── ✓ a.html + ├── ✗ b.html + ├── ✓ c.html + └── ✓ index.html + "); + assert_eq!(files, vec!["a.html", "c.html", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -222,6 +403,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -232,6 +414,20 @@ mod scanner { ("public/deeply/nested/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ public + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + ├── ✓ deeply + │ └── ✓ nested + │ └── ✓ c.html + └── ✓ nested + └── ✓ c.html + "); + assert_eq!( files, vec![ @@ -253,6 +449,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -266,6 +463,25 @@ mod scanner { ("public/very/deeply/nested/a.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ public + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + ├── ✓ nested + │ ├── ✓ a.html + │ ├── ✓ again + │ │ └── ✓ a.html + │ ├── ✓ b.html + │ └── ✓ c.html + └── ✓ very + └── ✓ deeply + └── ✓ nested + └── ✓ a.html + "); + assert_eq!( files, vec![ @@ -290,6 +506,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ (".gitignore", "public/b.html\na.html"), @@ -299,6 +516,18 @@ mod scanner { ("public/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ public/b.html + │ a.html + ├── ✓ index.html + └── ✓ public + ├── ✗ a.html + ├── ✗ b.html + └── ✓ c.html + "); + assert_eq!(files, vec!["index.html", "public/c.html",]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -310,6 +539,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -318,6 +548,15 @@ mod scanner { ("src/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ src + ├── ✓ a.html + ├── ✓ b.html + └── ✓ c.html + "); + assert_eq!( files, vec!["index.html", "src/a.html", "src/b.html", "src/c.html"] @@ -335,6 +574,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -343,6 +583,14 @@ mod scanner { ("c.lock", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ a.mp4 + ├── ✗ b.png + ├── ✗ c.lock + └── ✓ index.html + "); + assert_eq!(files, vec!["index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -355,6 +603,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ // Looks like `.pages` binary extension, but it's a folder @@ -368,6 +617,16 @@ mod scanner { ), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ other.pages + ├── ✗ other.pages + │ └── ✗ index.html + └── ✓ some.pages + └── ✓ index.html + "); + assert_eq!(files, vec!["some.pages/index.html"]); assert_eq!(globs, vec!["*", "some.pages/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -379,6 +638,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -387,6 +647,14 @@ mod scanner { ("c.less", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ a.css + ├── ✗ b.sass + ├── ✗ c.less + └── ✓ index.html + "); + assert_eq!(files, vec!["a.css", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -398,9 +666,16 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[("src/index.my-extension", "")]); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ index.my-extension + "); + assert_eq!(files, vec!["src/index.my-extension"]); assert_eq!(globs, vec!["*", "src/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,my-extension,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -412,6 +687,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -419,6 +695,13 @@ mod scanner { ("yarn.lock", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + ├── ✗ package-lock.json + └── ✗ yarn.lock + "); + assert_eq!(files, vec!["index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -430,6 +713,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ // Explicitly listed root files @@ -474,6 +758,58 @@ mod scanner { ("nested-d/very/deeply/nested/directory/again/foo.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✓ foo.html + ├── ✓ nested-a + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + ├── ✓ nested-b + │ └── ✓ deeply-nested + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + ├── ✓ nested-c + │ ├── ✗ .gitignore + │ │ ignored-folder/ + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ ├── ✓ foo.html + │ ├── ✗ ignored-folder + │ │ ├── ✗ bar.html + │ │ ├── ✗ baz.html + │ │ └── ✗ foo.html + │ └── ✓ sibling-folder + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + └── ✓ nested-d + ├── ✗ .gitignore + │ deep/ + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✓ foo.html + └── ✓ very + └── ✓ deeply + └── ✓ nested + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✗ deep + │ ├── ✗ bar.html + │ ├── ✗ baz.html + │ └── ✗ foo.html + ├── ✓ directory + │ ├── ✓ again + │ │ └── ✓ foo.html + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + └── ✓ foo.html + "); + assert_eq!( files, vec![ @@ -528,6 +864,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan(&[ // The gitignore file is used to filter out files but not scanned for candidates @@ -550,6 +887,21 @@ mod scanner { ("index4.svelte", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ # md:font-bold + │ foo.html + ├── ✗ foo.html + ├── ✗ foo.jpg + ├── ✓ index.angular.html + ├── ✓ index.html + ├── ✓ index.svelte + ├── ✓ index2.svelte + ├── ✓ index3.svelte + └── ✓ index4.svelte + "); + assert_eq!( candidates, vec![ @@ -573,12 +925,21 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[("foo/bar/baz/foo.html", "content-['foo.html']")], vec!["@source '**/*'", "@source './foo/bar/baz/..'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ foo + └── ✓ bar + └── ✓ baz + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['foo.html']"]); assert_eq!(normalized_sources, vec!["**/*", "foo/bar/**/*"]); } @@ -610,6 +971,19 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ project-a + │ └── ✗ src + │ └── ✗ index.css + └── ✓ project-b + ├── ✓ ignored + │ ├── ✓ except.html + │ └── ✗ ignored.html + └── ✓ keep + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['GOOD-1']", "content-['GOOD-2']"]); } @@ -619,12 +993,18 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[("my-file", "content-['my-file']")], vec!["@source '**/*'", "@source './my-file'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ my-file + "); + assert_eq!(candidates, vec!["content-['my-file']"]); assert_eq!(normalized_sources, vec!["**/*", "my-file"]); } @@ -635,6 +1015,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -654,6 +1035,14 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ my-folder.bin + │ └── ✓ foo.html + └── ✓ my-folder.templates + └── ✓ foo.html + "); + assert_eq!( candidates, vec![ @@ -672,6 +1061,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -682,6 +1072,11 @@ mod scanner { vec!["@source '**/*'", "@source '*.styl'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ foo.styl + "); + assert_eq!(candidates, vec!["content-['foo.styl']"]); assert_eq!(normalized_sources, vec!["**/*", "*.styl"]); } @@ -693,6 +1088,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ( @@ -707,6 +1104,18 @@ mod scanner { vec!["@source './blog/*/foo/bar/baz/**/*'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + └── ✓ foo + └── ✓ bar + ├── ✓ baz + │ └── ✓ index.html + └── ✗ qux + └── ✗ index.html + "); + assert_eq!( candidates, vec!["content-['blog/2024/foo/bar/baz/index.html']"] @@ -721,6 +1130,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -734,6 +1144,19 @@ mod scanner { vec!["@source '**/*'", "@source './**/*.{styl}'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ app + ├── ✓ (theme) + │ └── ✓ page.styl + ├── ✓ [...slug] + │ └── ✓ page.styl + ├── ✓ [[...slug]] + │ └── ✓ page.styl + └── ✓ [slug] + └── ✓ page.styl + "); + assert_eq!( candidates, vec![ @@ -775,6 +1198,14 @@ mod scanner { let mut scanner = Scanner::new(sources); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ └── ✓ index.html + └── ✓ project-b + └── ✓ index.html + "); + // We've done the initial scan and found the files assert_eq!( candidates, @@ -790,6 +1221,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -802,6 +1234,13 @@ mod scanner { vec!["@source '**/*'", "@source 'foo.styl'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ foo.styl + └── ✓ foo.styl + "); + assert_eq!(candidates, vec!["content-['foo.styl']"]); assert_eq!(normalized_sources, vec!["**/*", "foo.styl"]); } @@ -831,6 +1270,14 @@ mod scanner { let mut scanner = Scanner::new(sources); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ └── ✓ index.html + └── ✓ project-b + └── ✓ index.html + "); + // We've done the initial scan and found the files assert_eq!( candidates, @@ -935,6 +1382,24 @@ mod scanner { "content-['project-b/sub1/sub2/new.html']" ] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ ├── ✓ index.html + │ ├── ✓ new.html + │ └── ✓ sub1 + │ └── ✓ sub2 + │ ├── ✓ index.html + │ └── ✓ new.html + └── ✓ project-b + ├── ✓ index.html + ├── ✓ new.html + └── ✓ sub1 + └── ✓ sub2 + ├── ✓ index.html + └── ✓ new.html + "); } #[test] @@ -966,6 +1431,14 @@ mod scanner { vec!["src/index.html", "src/keep.html", "src/remove.html"] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + ├── ✓ keep.html + └── ✓ remove.html + "); + fs::remove_file(dir.join("src/remove.html")).unwrap(); scanner.scan(); @@ -973,6 +1446,13 @@ mod scanner { scanned_files(&mut scanner, &dir), vec!["src/index.html", "src/keep.html"] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✓ keep.html + "); } #[test] @@ -1013,6 +1493,18 @@ mod scanner { ] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + ├── ✓ keep + │ └── ✓ index.html + └── ✓ remove + ├── ✓ index.html + └── ✓ nested + └── ✓ index.html + "); + fs::remove_dir_all(dir.join("src/remove")).unwrap(); scanner.scan(); @@ -1020,6 +1512,14 @@ mod scanner { scanned_files(&mut scanner, &dir), vec!["src/index.html", "src/keep/index.html"] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✓ keep + └── ✓ index.html + "); } #[test] @@ -1046,6 +1546,13 @@ mod scanner { scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ index.html + └── ✓ src + └── ✓ index.html + "); + let globs = scanned_globs(&mut scanner, &dir); assert!(globs.iter().any(|glob| glob.starts_with("src/**/*"))); @@ -1053,6 +1560,11 @@ mod scanner { scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ index.html + "); + let globs = scanned_globs(&mut scanner, &dir); assert!(!globs.iter().any(|glob| glob.starts_with("src/**/*"))); } @@ -1111,6 +1623,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/index.ts", "content-['src/index.ts']"), @@ -1138,6 +1652,25 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ admin + │ └── ✓ foo + │ └── ✓ template.html + ├── ✗ colors + │ ├── ✗ blue.tsx + │ ├── ✗ green.tsx + │ └── ✗ red.jsx + ├── ✗ index.ts + ├── ✓ templates + │ └── ✓ index.html + └── ✗ utils + ├── ✗ date.ts + ├── ✗ file.ts + └── ✗ string.ts + "); + assert_eq!( candidates, vec![ @@ -1183,6 +1716,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( // Typically skipped &[ @@ -1194,6 +1729,15 @@ mod scanner { vec!["@source '**/*'", "@source 'src/**/*.{exe,bin}'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ out + │ └── ✗ out.exe + └── ✓ src + ├── ✓ index.bin + └── ✓ index.exe + "); + assert_eq!( candidates, vec!["content-['src/index.bin']", "content-['src/index.exe']",] @@ -1221,6 +1765,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("index.html", "content-['index.html']"), @@ -1245,6 +1791,21 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ index.html + └── ✓ src + ├── ✓ admin + │ ├── ✗ ignore.html + │ └── ✓ index.html + ├── ✓ dashboard + │ ├── ✗ ignore.html + │ └── ✓ index.html + ├── ✗ ignore.html + ├── ✗ index.html + └── ✗ lib.ts + "); + assert_eq!( candidates, vec![ @@ -1264,7 +1825,10 @@ mod scanner { #[test] fn it_should_restrict_explicit_file_sources_to_the_matching_file() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1273,6 +1837,13 @@ mod scanner { vec!["@source './src/foo.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar.html + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['src/foo.html']"]); assert_eq!(files, vec!["src/foo.html"]); } @@ -1280,7 +1851,10 @@ mod scanner { #[test] fn it_should_combine_multiple_restricted_sources_for_the_same_base() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1290,6 +1864,14 @@ mod scanner { vec!["@source './src/foo.html'", "@source './src/bar.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar.html + ├── ✗ baz.html + └── ✓ foo.html + "); + assert_eq!( candidates, vec!["content-['src/bar.html']", "content-['src/foo.html']"] @@ -1313,7 +1895,10 @@ mod scanner { ]; let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( paths_with_content, vec![ @@ -1322,6 +1907,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ component-sources.classes.txt + ├── ✗ ignore-me + │ └── ✗ component.html + ├── ✗ ignore-me.txt + └── ✓ nested + ├── ✓ component.html + └── ✗ ignore-me.html + "); + assert_eq!( candidates, vec![ @@ -1336,7 +1932,10 @@ mod scanner { // Same setup, but with the root-level source declared first let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( paths_with_content, vec![ @@ -1345,6 +1944,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ component-sources.classes.txt + ├── ✗ ignore-me + │ └── ✗ component.html + ├── ✗ ignore-me.txt + └── ✓ nested + ├── ✓ component.html + └── ✗ ignore-me.html + "); + assert_eq!( candidates, vec![ @@ -1361,12 +1971,21 @@ mod scanner { #[test] fn it_should_allow_later_ignores_to_override_restricted_sources() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[("src/foo.html", "content-['src/foo.html']")], vec!["@source './src/foo.html'", "@source not './src/foo.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✗ src + └── ✗ foo.html + "); + assert!(candidates.is_empty()); assert!(files.is_empty()); } @@ -1375,7 +1994,10 @@ mod scanner { fn it_should_handle_sources_with_parent_patterns() { { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1385,6 +2007,15 @@ mod scanner { vec!["@source './src/ba*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✓ ignore.html + │ └── ✓ index.html + └── ✗ foo.html + "); + assert_eq!( candidates, vec![ @@ -1397,7 +2028,10 @@ mod scanner { { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1410,13 +2044,25 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✗ ignore.html + │ └── ✓ index.html + └── ✗ foo.html + "); + assert_eq!(candidates, vec!["content-['src/bar/index.html']"]); assert_eq!(files, vec!["src/bar/index.html"]); } { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1426,13 +2072,25 @@ mod scanner { vec!["@source '**/*'", "@source not './src/ba*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar + │ ├── ✗ ignore.html + │ └── ✗ index.html + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['src/foo.html']"]); assert_eq!(files, vec!["src/foo.html"]); } { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1446,6 +2104,15 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✗ ignore.html + │ └── ✓ index.html + └── ✓ foo.html + "); + assert_eq!( candidates, vec!["content-['src/bar/index.html']", "content-['src/foo.html']"] @@ -1460,7 +2127,10 @@ mod scanner { // specific `@source` points at a single file inside a subdirectory. The restriction // added for the explicit file must not hide its siblings from the broad source. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/components/button.html", "content-['button']"), @@ -1469,6 +2139,14 @@ mod scanner { vec!["@source '**/*'", "@source './src/components/button.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ components + ├── ✓ button.html + └── ✓ card.html + "); + assert_eq!(candidates, vec!["content-['button']", "content-['card']"]); assert_eq!( files, @@ -1481,7 +2159,10 @@ mod scanner { // Same as above, but the broad source is an auto-detected directory (`@source "src"`) // and the explicit file lives in a nested subdirectory of it. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/components/button.html", "content-['button']"), @@ -1490,6 +2171,14 @@ mod scanner { vec!["@source 'src'", "@source './src/components/button.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ components + ├── ✓ button.html + └── ✓ card.html + "); + assert_eq!(candidates, vec!["content-['button']", "content-['card']"]); assert_eq!( files, @@ -1499,7 +2188,9 @@ mod scanner { #[test] fn root_file_source_should_not_suppress_sibling_source_roots() { - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ ("index.css", ""), ("src/index.html", "content-['src/index.html']"), @@ -1513,6 +2204,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.css + ├── ✓ pages + │ ├── ✓ foo.html + │ └── ✓ nested + │ └── ✓ foo.html + └── ✓ src + └── ✓ index.html + "); + assert_eq!( candidates, vec![ @@ -1530,6 +2232,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "ignore-1.html\nweb/ignore-2.html"), @@ -1541,6 +2245,19 @@ mod scanner { vec!["@source './src'", "@source './web'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ ignore-1.html + │ web/ignore-2.html + ├── ✓ src + │ └── ✓ index.html + └── ✓ web + ├── ✗ ignore-1.html + ├── ✗ ignore-2.html + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['src/index.html']", "content-['web/index.html']",] @@ -1558,6 +2275,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/logo.jpg", "content-['/src/logo.jpg']"), @@ -1566,6 +2285,13 @@ mod scanner { vec!["@source './src/logo.{jpg,png}'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ logo.jpg + └── ✓ logo.png + "); + assert_eq!( candidates, vec!["content-['/src/logo.jpg']", "content-['/src/logo.png']"] @@ -1582,6 +2308,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "ignore-1.html\n/web/ignore-2.html"), @@ -1591,6 +2319,17 @@ mod scanner { ], vec!["@source './web'", "@source './web/ignore-1.html'"], ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ ignore-1.html + │ /web/ignore-2.html + └── ✓ web + ├── ✓ ignore-1.html + ├── ✗ ignore-2.html + └── ✓ index.html + "); assert_eq!( candidates, vec![ @@ -1611,7 +2350,10 @@ mod scanner { // A `.gitignore` that ignores everything (`/*`) and then whitelists specific // directories and files using negated patterns. let ScanResult { - files, candidates, .. + files, + candidates, + tree, + .. } = scan(&[ ( ".gitignore", @@ -1629,6 +2371,28 @@ mod scanner { ), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /* + │ !/app + │ !/public + │ !/package.json + │ !/.gitignore + ├── ✓ app + │ └── ✓ index.html + ├── ✗ build + │ └── ✗ generated.html + ├── ✗ logs + │ └── ✗ dev.log + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + ├── ✓ package.json + └── ✓ public + └── ✓ index.html + "); + assert_eq!( files, vec!["app/index.html", "package.json", "public/index.html"] @@ -1654,7 +2418,10 @@ mod scanner { // !/foo/bar // ``` let ScanResult { - files, candidates, .. + files, + candidates, + tree, + .. } = scan(&[ ( ".gitignore", @@ -1671,6 +2438,25 @@ mod scanner { ("foo/baz/index.html", "content-['foo/baz/index.html']"), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ # exclude everything except directory foo/bar + │ /* + │ !/foo + │ /foo/* + │ !/foo/bar + ├── ✓ foo + │ ├── ✓ bar + │ │ ├── ✓ index.html + │ │ └── ✓ nested + │ │ └── ✓ index.html + │ ├── ✗ baz + │ │ └── ✗ index.html + │ └── ✗ index.html + └── ✗ index.html + "); + assert_eq!( files, vec!["foo/bar/index.html", "foo/bar/nested/index.html"] @@ -1691,7 +2477,10 @@ mod scanner { // contents. Only when the directory itself is ignored (by an ancestor // `.gitignore`, see the tests above) do we bypass the ignore rules. let ScanResult { - files, candidates, .. + files, + candidates, + tree, + .. } = scan_with_globs( &[ ("vendor/.gitignore", "ignored.html"), @@ -1701,6 +2490,15 @@ mod scanner { vec!["@source 'vendor'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ vendor + ├── ✗ .gitignore + │ ignored.html + ├── ✗ ignored.html + └── ✓ index.html + "); + assert_eq!(files, vec!["vendor/index.html"]); assert_eq!(candidates, vec!["content-['vendor/index.html']"]); } @@ -1711,6 +2509,7 @@ mod scanner { candidates, files, globs, + tree, .. } = scan_with_globs( &[ @@ -1722,6 +2521,18 @@ mod scanner { vec!["@source './src/ef*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ src/efgh/ + └── ✓ src + ├── ✗ abcd + │ └── ✗ index.html + └── ✓ efgh + ├── ✗ ignore.js + └── ✓ index.html + "); + assert_eq!(candidates, vec!["content-['src/efgh/index.html']"]); assert_eq!(files, vec!["src/efgh/index.html"]); assert_eq!(globs, vec!["src/ef*/*.html"]); @@ -1797,7 +2608,33 @@ mod scanner { ), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✗ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); // All ignore files are applied because there's no git repo assert_eq!( @@ -1815,7 +2652,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home (git) + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✗ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1834,7 +2697,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home/project")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project (git) + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1854,7 +2743,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home/project/apps")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps (git) + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✓ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1876,7 +2791,33 @@ mod scanner { .current_dir(dir.join("home/project/apps/web")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web (git) + ├── ✗ .gitignore + │ ignore-web.html + ├── ✓ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✓ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1910,7 +2851,15 @@ mod scanner { public_source_entry_from_pattern(dir.clone(), "@source not 'src/ignore-me.html'"), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + ├── ✗ ignore-me.html + └── ✓ keep-me.html + "); assert_eq!(candidates, vec!["content-['keep-me.html']"]); } @@ -1937,7 +2886,17 @@ mod scanner { ), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + └── ✓ app + └── ✓ [foo] + ├── ✗ ignore-me.html + └── ✓ keep-me.html + "); assert_eq!(candidates, vec!["content-['keep-me.html']"]); } @@ -1964,6 +2923,12 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/keep-me.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + └── ✓ keep-me.html + "); + // Create new files that should definitely be ignored create_files_in( &dir, @@ -1991,6 +2956,18 @@ mod scanner { let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ src/ignored-by-gitignore.html + └── ✓ src + ├── ✗ ignore-by-extension.bin + ├── ✓ ignored-by-gitignore.html + ├── ✗ ignored-by-source-not.html + ├── ✓ keep-me.html + └── ✓ new-file.html + "); + assert_eq!( candidates, vec![ @@ -2020,6 +2997,11 @@ mod scanner { let candidates = scanner.scan(); assert!(candidates.is_empty()); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✗ foo.styl + "); + // Explicitly allow `.styl` files let mut scanner = Scanner::new(vec![ public_source_entry_from_pattern(dir.clone(), "@source '**/*'"), @@ -2028,6 +3010,11 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['foo.styl']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ foo.styl + "); } #[test] @@ -2052,6 +3039,13 @@ mod scanner { let candidates = scanner.scan(); assert!(candidates.is_empty()); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ index.html + └── ✗ index.html + "); + let mut scanner = Scanner::new(vec![ public_source_entry_from_pattern(dir.clone(), "@source '**/*'"), public_source_entry_from_pattern(dir.clone(), "@source './*.html'"), @@ -2059,6 +3053,13 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['index.html']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ index.html + └── ✓ index.html + "); } #[test] @@ -2093,6 +3094,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + // Explicitly listing all `*.html` files, should not include `node_modules` because it's // ignored let sources = vec![public_source_entry_from_pattern( @@ -2104,6 +3116,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + // Explicitly listing all `*.html` files // Explicitly list the `node_modules/my-ui-lib` // @@ -2121,12 +3144,25 @@ mod scanner { "content-['src/index.html']" ] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✓ node_modules + │ └── ✓ my-ui-lib + │ └── ✓ index.html + └── ✓ src + └── ✓ index.html + "); } // https://github.com/tailwindlabs/tailwindcss/issues/19844 #[test] fn test_allow_explicit_sources_ignored_by_allow_list_gitignore() { - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ (".gitignore", "*\n!/app\n!/app/design\n!/app/design/**\n"), ( @@ -2141,6 +3177,27 @@ mod scanner { vec!["@source 'vendor/acme/theme'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ * + │ !/app + │ !/app/design + │ !/app/design/** + ├── ✗ app + │ └── ✗ design + │ └── ✗ frontend + │ └── ✗ theme + │ └── ✗ templates + │ └── ✗ component.phtml + └── ✓ vendor + └── ✓ acme + └── ✓ theme + └── ✓ module + └── ✓ templates + └── ✓ component.phtml + "); + assert_eq!( candidates, vec!["content-['vendor/acme/theme/module/templates/component.phtml']"] @@ -2154,6 +3211,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ( @@ -2172,6 +3231,17 @@ mod scanner { vec!["@source '**/*'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ node_modules + │ └── ✗ index.html + └── ✓ packages + └── ✓ web + ├── ✓ index.html + └── ✗ node_modules + └── ✗ index.html + "); + assert_eq!(candidates, vec!["content-['packages/web/index.html']"]); assert_eq!(files, vec!["packages/web/index.html",]); @@ -2186,6 +3256,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "node_modules\ndist"), @@ -2201,6 +3273,18 @@ mod scanner { vec!["@source 'node_modules/my-ui-lib'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + └── ✓ my-ui-lib + ├── ✓ dist + │ └── ✓ index.html + └── ✗ node.exe + "); + assert_eq!( candidates, vec!["content-['node_modules/my-ui-lib/dist/index.html']"] @@ -2242,6 +3326,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/components/button.tsx']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ *.jsx + │ generated/ + └── ✓ src + └── ✓ components + ├── ✗ button.jsx + └── ✓ button.tsx + "); + // Create 2 new files, one "good" and one "bad" file, and manually scan them. This should // only return the "good" file because the "bad" one is ignored by a `.gitignore` file. create_files_in( @@ -2302,6 +3397,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/💩.js", "content-['src/💩.js']"), @@ -2311,6 +3408,15 @@ mod scanner { vec!["@source '**/*'", "@source not 'src/🤦‍♂️'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ 💩.js + ├── ✗ 🤦‍♂️ + │ └── ✗ foo.tsx + └── ✓ 🤦‍♂️.tsx + "); + assert_eq!( candidates, vec!["content-['src/💩.js']", "content-['src/🤦‍♂️.tsx']"] @@ -2347,6 +3453,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✓ .gitkeep + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2358,6 +3481,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✗ .gitkeep + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2369,6 +3509,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✓ .gitkeep + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2386,11 +3543,14 @@ mod scanner { ], ); - // Create recursive symlinks - let _ = symlink(dir.join("a"), dir.join("b")); - let _ = symlink(dir.join("b/c"), dir.join("c")); - let _ = symlink(dir.join("b/root"), &dir); - let _ = symlink(dir.join("c"), dir.join("a")); + // Create recursive symlinks: + // + // - `a → b`, `b/c → c`, `c → a` form a cycle + // - `b/root → .` points back at the root directory + let _ = symlink(dir.join("b"), dir.join("a")); + let _ = symlink(dir.join("c"), dir.join("b/c")); + let _ = symlink(&dir, dir.join("b/root")); + let _ = symlink(dir.join("a"), dir.join("c")); let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( dir.clone(), @@ -2398,6 +3558,24 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ a → b + │ ├── ✗ c → ../c + │ ├── ✓ index.html + │ └── ✗ root → .. + ├── ✓ b + │ ├── ✗ c → ../c + │ ├── ✓ index.html + │ └── ✗ root → .. + ├── ✓ c → a + │ ├── ✗ c → . + │ ├── ✓ index.html + │ └── ✗ root → .. + └── ✓ z + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['b/index.html']", "content-['z/index.html']"] @@ -2422,6 +3600,14 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ abcd + │ └── ✓ xyz.html + └── ✓ efgh → abcd + └── ✓ xyz.html + "); + assert_eq!(candidates, vec!["content-['abcd/xyz.html']"]); // Partially referencing the symlinked folder with a glob, should find the file @@ -2462,6 +3648,18 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ actual-dir + │ └── ✗ ignore.html + ├── ✗ actual-file.html + ├── ✗ linked-dir → actual-dir + │ └── ✗ ignore.html + ├── ✗ linked-file.html → actual-file.html + └── ✓ src + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['src/keep.html']"]); let mut scanner = Scanner::new(vec![ @@ -2499,6 +3697,16 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ actual-dir + │ └── ✓ ignore.html + └── ✓ project + ├── ✓ keep.html + └── ✓ linked-dir → ../actual-dir + └── ✓ ignore.html + "); + assert_eq!( candidates, vec![ @@ -2513,6 +3721,16 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ actual-dir + │ └── ✗ ignore.html + └── ✓ project + ├── ✓ keep.html + └── ✗ linked-dir → ../actual-dir + └── ✗ ignore.html + "); + assert_eq!(candidates, vec!["content-['project/keep.html']"]); } @@ -2570,6 +3788,23 @@ mod scanner { ] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ linked.html → packages/other/index.html + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ ├── ✓ nested + │ │ └── ✓ deep.html + │ └── ✓ source.html + └── ✓ packages + ├── ✓ other + │ └── ✓ index.html + └── ✓ repro + ├── ✓ nested + │ └── ✓ deep.html + └── ✓ source.html + "); + // Both the symlinked paths and the canonical paths should be tracked, such that file // watchers watching the returned files also watch the real files on disk. let files = scanned_files(&mut scanner, &dir); @@ -2603,6 +3838,16 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['v1']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ └── ✓ source.html + └── ✓ packages + └── ✓ repro + └── ✓ source.html + "); + // Update the real file on disk. This is the path file watchers will report changes for. create_files_in(&dir, &[("packages/repro/source.html", "content-['v2']")]); @@ -2630,6 +3875,16 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['a']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ └── ✓ a.html + └── ✓ packages + └── ✓ repro + └── ✓ a.html + "); + // Create a new file in the real directory. File watchers watching the real directory // will report the new file with its canonical path. create_files_in(&dir, &[("packages/repro/b.html", "content-['b']")]); @@ -2639,12 +3894,26 @@ mod scanner { "html".into(), )]); assert_eq!(candidates, vec!["content-['b']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ ├── ✓ a.html + │ └── ✓ b.html + └── ✓ packages + └── ✓ repro + ├── ✓ a.html + └── ✗ b.html + "); } // https://github.com/tailwindlabs/tailwindcss/pull/20408 #[test] fn test_resolving_globs_does_not_traverse_gitignored_directories() { - let ScanResult { files, globs, .. } = scan_with_globs( + let ScanResult { + files, globs, tree, .. + } = scan_with_globs( &[ (".gitignore", "/vendor\n"), ("vendor/pkg/canary/index.html", ""), @@ -2653,6 +3922,18 @@ mod scanner { vec!["@source '**/*'", "@source './vendor/pkg/canary'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /vendor + ├── ✓ src + │ └── ✓ index.html + └── ✓ vendor + └── ✓ pkg + └── ✓ canary + └── ✓ index.html + "); + assert_eq!( files, vec!["src/index.html", "vendor/pkg/canary/index.html"] @@ -2666,6 +3947,974 @@ mod scanner { ]); } + // https://github.com/tailwindlabs/tailwindcss/issues/18870 + #[test] + fn glob_sources_can_descend_into_directories_ignored_from_within() { + // Laravel's `storage/` directories ship a `.gitignore` that ignores everything inside + // (`*` + `!.gitignore`). An explicit glob through such a directory should still find + // matching files, but nothing else. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ("storage/.gitignore", "*\n!.gitignore"), + ( + "storage/cms/section.html", + "content-['storage/cms/section.html']", + ), + ( + "storage/cms/nested/deep.html", + "content-['storage/cms/nested/deep.html']", + ), + ("storage/cms/data.json", "content-['storage/cms/data.json']"), + ("storage/other/other.html", "content-['storage/other.html']"), + ], + vec!["@source './storage/cms/**/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ storage + ├── ✗ .gitignore + │ * + │ !.gitignore + ├── ✓ cms + │ ├── ✗ data.json + │ ├── ✓ nested + │ │ └── ✓ deep.html + │ └── ✓ section.html + └── ✗ other + └── ✗ other.html + "); + + assert_eq!( + candidates, + vec![ + "content-['storage/cms/nested/deep.html']", + "content-['storage/cms/section.html']", + ] + ); + assert_eq!( + files, + vec!["storage/cms/nested/deep.html", "storage/cms/section.html"] + ); + } + + // https://github.com/tailwindlabs/tailwindcss/issues/18870 + #[test] + fn glob_sources_can_descend_into_directories_ignored_by_an_ancestor() { + // Same as above, but the ignore comes from an ancestor `.gitignore` rather than one + // inside the ignored directory itself. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "/storage"), + ( + "storage/cms/section.html", + "content-['storage/cms/section.html']", + ), + ( + "storage/cms/nested/deep.html", + "content-['storage/cms/nested/deep.html']", + ), + ("storage/cms/script.js", "content-['storage/cms/script.js']"), + ], + vec!["@source './storage/cms/**/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /storage + └── ✓ storage + └── ✓ cms + ├── ✓ nested + │ └── ✓ deep.html + ├── ✗ script.js + └── ✓ section.html + "); + + assert_eq!( + candidates, + vec![ + "content-['storage/cms/nested/deep.html']", + "content-['storage/cms/section.html']", + ] + ); + assert_eq!( + files, + vec!["storage/cms/nested/deep.html", "storage/cms/section.html"] + ); + } + + #[test] + fn glob_sources_inside_ignored_directories_do_not_include_other_extensions() { + // `@source "./foo/*.html"` where `foo` is git ignored: only the `.html` files were asked + // for. Other files in `foo` must not be scanned, even when a broader auto source exists. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "/foo"), + ("foo/index.html", "content-['foo/index.html']"), + ("foo/script.js", "content-['foo/script.js']"), + ( + "foo/nested/nested.html", + "content-['foo/nested/nested.html']", + ), + ("index.html", "content-['index.html']"), + ], + vec!["@source '**/*'", "@source './foo/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /foo + ├── ✓ foo + │ ├── ✓ index.html + │ ├── ✗ nested + │ │ └── ✗ nested.html + │ └── ✗ script.js + └── ✓ index.html + "); + + assert_eq!( + candidates, + vec!["content-['foo/index.html']", "content-['index.html']"] + ); + assert_eq!(files, vec!["foo/index.html", "index.html"]); + } + + // https://github.com/tailwindlabs/tailwindcss/pull/20406 + #[test] + fn concrete_file_sources_behind_node_modules_do_not_include_siblings() { + // A concrete file `@source` pointing into a default-ignored directory (`node_modules`) + // must only include that file, not its siblings, even when the project root is an auto + // source. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ( + "node_modules/.generated/ui/button.ts", + "content-['button.ts']", + ), + ("node_modules/.generated/ui/card.ts", "content-['card.ts']"), + ("index.html", "content-['index.html']"), + ], + vec![ + "@source '**/*'", + "@source './node_modules/.generated/ui/button.ts'", + ], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ node_modules + └── ✓ .generated + └── ✓ ui + ├── ✓ button.ts + └── ✗ card.ts + "); + + assert_eq!( + candidates, + vec!["content-['button.ts']", "content-['index.html']"] + ); + assert_eq!( + files, + vec!["index.html", "node_modules/.generated/ui/button.ts"] + ); + } + + // https://github.com/tailwindlabs/tailwindcss/pull/20406 + #[test] + fn concrete_file_sources_behind_gitignored_directories_do_not_include_siblings() { + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "/generated"), + ("generated/button.ts", "content-['button.ts']"), + ("generated/card.ts", "content-['card.ts']"), + ("index.html", "content-['index.html']"), + ], + vec!["@source '**/*'", "@source './generated/button.ts'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /generated + ├── ✓ generated + │ ├── ✓ button.ts + │ └── ✗ card.ts + └── ✓ index.html + "); + + assert_eq!( + candidates, + vec!["content-['button.ts']", "content-['index.html']"] + ); + assert_eq!(files, vec!["generated/button.ts", "index.html"]); + } + + #[test] + fn glob_sources_do_not_reopen_node_modules() { + // A glob source respects the default auto source detection rules: `node_modules` inside + // the globbed tree stays pruned unless it is targeted explicitly. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ("src/index.html", "content-['src/index.html']"), + ( + "src/node_modules/lib/index.html", + "content-['src/node_modules/lib/index.html']", + ), + ], + vec!["@source './src/**/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✗ node_modules + └── ✗ lib + └── ✗ index.html + "); + + assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_eq!(files, vec!["src/index.html"]); + } + + #[test] + fn wildcards_in_glob_sources_do_not_descend_into_gitignored_directories() { + // The static part of a glob source is the "explicit" part: it bypasses ignore rules. + // The wildcard part does not: `@source "./**/*.html"` does not descend into a git + // ignored directory — you weren't explicit about it. To opt in, name the directory: + // `@source "./dist/**/*.html"`. + // + // Files are different: a git ignored *file* that matches the glob is still included + // (you asked for all `.html` files). + let paths_with_content = &[ + (".gitignore", "dist/\nignored.html"), + ("src/index.html", "content-['src/index.html']"), + ("dist/index.html", "content-['dist/index.html']"), + ("ignored.html", "content-['ignored.html']"), + ]; + + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs(paths_with_content, vec!["@source './**/*.html'"]); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ dist/ + │ ignored.html + ├── ✗ dist + │ └── ✗ index.html + ├── ✓ ignored.html + └── ✓ src + └── ✓ index.html + "); + + assert_eq!( + candidates, + vec!["content-['ignored.html']", "content-['src/index.html']"] + ); + assert_eq!(files, vec!["ignored.html", "src/index.html"]); + + // Being explicit about the ignored directory bypasses the ignore rule. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + paths_with_content, + vec!["@source './**/*.html'", "@source './dist/**/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ dist/ + │ ignored.html + ├── ✓ dist + │ └── ✓ index.html + ├── ✓ ignored.html + └── ✓ src + └── ✓ index.html + "); + + assert_eq!( + candidates, + vec![ + "content-['dist/index.html']", + "content-['ignored.html']", + "content-['src/index.html']" + ] + ); + assert_eq!( + files, + vec!["dist/index.html", "ignored.html", "src/index.html"] + ); + } + + #[test] + fn glob_sources_only_rescue_matching_files_from_ignored_directories() { + // When a glob source forces the walker into a git ignored directory, files that do not + // match the glob must stay excluded, even though they are only ignored "via" their + // parent directory. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "gen/"), + ("gen/a.html", "content-['gen/a.html']"), + ("gen/b.js", "content-['gen/b.js']"), + ("index.html", "content-['index.html']"), + ], + vec!["@source '**/*'", "@source './gen/**/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ gen/ + ├── ✓ gen + │ ├── ✓ a.html + │ └── ✗ b.js + └── ✓ index.html + "); + + assert_eq!( + candidates, + vec!["content-['gen/a.html']", "content-['index.html']"] + ); + assert_eq!(files, vec!["gen/a.html", "index.html"]); + } + + #[test] + fn unpinned_extension_globs_apply_default_extension_rules() { + // A glob that doesn't pin an extension (`**/*`-style tail after a wildcard directory) + // still applies the default extension rules, so binary files are not included. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ("blog/2024/post/index.html", "content-['index.html']"), + ("blog/2024/post/image.png", "content-['image.png']"), + ("blog/2024/post/styles.scss", "content-['styles.scss']"), + ], + vec!["@source './blog/*/post/**/*'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + └── ✓ post + ├── ✗ image.png + ├── ✓ index.html + └── ✗ styles.scss + "); + + assert_eq!(candidates, vec!["content-['index.html']"]); + assert_eq!(files, vec!["blog/2024/post/index.html"]); + } + + #[test] + fn unpinned_globs_do_not_reinclude_gitignored_directories() { + // The whitelist of an unpinned glob like `./blog/*/**/*` could also match *directory* + // paths (unlike e.g. `./**/*.html`, which only ever matches files). It must not: the + // wildcard part of a glob is not explicit, so git ignored directories inside the + // walked subtree stay pruned, even when the whitelisted glob happens to match them. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ("blog/.gitignore", "drafts/"), + ("blog/2024/keep.html", "content-['blog/2024/keep.html']"), + ( + "blog/2024/drafts/secret.html", + "content-['blog/2024/drafts/secret.html']", + ), + ], + vec!["@source './blog/*/**/*'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + ├── ✗ .gitignore + │ drafts/ + └── ✓ 2024 + ├── ✗ drafts + │ └── ✗ secret.html + └── ✓ keep.html + "); + + assert_eq!(candidates, vec!["content-['blog/2024/keep.html']"]); + assert_eq!(files, vec!["blog/2024/keep.html"]); + } + + #[test] + fn external_sources_ignore_nested_default_ignored_directories() { + // An explicitly listed, git ignored directory (external source) still applies the + // default auto source detection rules inside of it: nested `node_modules` are not + // scanned. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "node_modules"), + ( + "node_modules/my-lib/dist/index.html", + "content-['node_modules/my-lib/dist/index.html']", + ), + ( + "node_modules/my-lib/node_modules/dep/index.html", + "content-['node_modules/my-lib/node_modules/dep/index.html']", + ), + ( + "node_modules/my-lib/logo.png", + "content-['node_modules/my-lib/logo.png']", + ), + ], + vec!["@source './node_modules/my-lib'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + └── ✓ node_modules + └── ✓ my-lib + ├── ✓ dist + │ └── ✓ index.html + ├── ✗ logo.png + └── ✗ node_modules + └── ✗ dep + └── ✗ index.html + "); + + assert_eq!( + candidates, + vec!["content-['node_modules/my-lib/dist/index.html']"] + ); + assert_eq!(files, vec!["node_modules/my-lib/dist/index.html"]); + } + + #[test] + fn external_sources_respect_gitignore_files_inside_of_them() { + // Listing an ignored directory explicitly bypasses the rules that made it ignored: + // `.gitignore` files above the base, and the base's own `.gitignore` — inside ignored + // trees that is typically the self-ignoring `*` file that generators drop into the + // directory. `.gitignore` files *deeper inside* the external tree were put there + // deliberately by whatever generates those files, so they still apply, just like they + // do for auto and pattern sources. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "node_modules"), + // The self-ignoring device at the base: bypassed + ("node_modules/.generated/.gitignore", "*"), + // A deliberate ignore deeper inside: applies + ( + "node_modules/.generated/ui/.gitignore", + "ignored.tsx\ncache/", + ), + ( + "node_modules/.generated/ui/button.tsx", + "content-['button.tsx']", + ), + ( + "node_modules/.generated/ui/input.tsx", + "content-['input.tsx']", + ), + ( + "node_modules/.generated/ui/ignored.tsx", + "content-['ignored.tsx']", + ), + ( + "node_modules/.generated/ui/cache/stale.tsx", + "content-['cache/stale.tsx']", + ), + ], + vec!["@source './node_modules/.generated'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + └── ✓ node_modules + └── ✓ .generated + ├── ✗ .gitignore + │ * + └── ✓ ui + ├── ✗ .gitignore + │ ignored.tsx + │ cache/ + ├── ✓ button.tsx + ├── ✗ cache + │ └── ✗ stale.tsx + ├── ✗ ignored.tsx + └── ✓ input.tsx + "); + + assert_eq!( + candidates, + vec!["content-['button.tsx']", "content-['input.tsx']"] + ); + assert_eq!( + files, + vec![ + "node_modules/.generated/ui/button.tsx", + "node_modules/.generated/ui/input.tsx" + ] + ); + } + + #[test] + fn double_star_sources_are_auto_sources() { + // `@source "./src/**"` means "everything underneath src", just like `./src/**/*` and + // `./src`: auto source detection applies, so `.gitignore` files and the default rules + // are respected instead of the glob rescuing every git ignored file. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "src/secret.html"), + ("src/index.html", "content-['src/index.html']"), + ("src/secret.html", "content-['src/secret.html']"), + ("src/logo.png", "content-['src/logo.png']"), + ], + vec!["@source './src/**'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ src/secret.html + └── ✓ src + ├── ✓ index.html + ├── ✗ logo.png + └── ✗ secret.html + "); + + assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_eq!(files, vec!["src/index.html"]); + } + + #[test] + fn gitignore_whitelists_do_not_reinclude_default_ignored_content() { + // A `.gitignore` whitelist re-includes files for git, but the default auto source + // detection rules are independent of git: `node_modules`, binary files, etc. stay + // ignored even when a `.gitignore` explicitly whitelists them. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "!node_modules/\n!*.png"), + ("index.html", "content-['index.html']"), + ("logo.png", "content-['logo.png']"), + ( + "node_modules/lib/index.html", + "content-['node_modules/lib/index.html']", + ), + ], + vec!["@source '**/*'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ !node_modules/ + │ !*.png + ├── ✓ index.html + ├── ✗ logo.png + └── ✗ node_modules + └── ✗ lib + └── ✗ index.html + "); + + assert_eq!(candidates, vec!["content-['index.html']"]); + assert_eq!(files, vec!["index.html"]); + } + + #[test] + fn dot_ignore_files_are_respected() { + // `.ignore` files (the gitignore-style files used by ripgrep and friends) work like + // `.gitignore` files and rank above them within the same directory. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "by-git.html\nby-both.html"), + (".ignore", "by-ignore.html\n!by-both.html"), + ("keep.html", "content-['keep.html']"), + ("by-git.html", "content-['by-git.html']"), + ("by-ignore.html", "content-['by-ignore.html']"), + // Ignored by the `.gitignore` file, re-included by the `.ignore` file + ("by-both.html", "content-['by-both.html']"), + ], + vec!["@source '**/*'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ by-git.html + │ by-both.html + ├── ✗ .ignore + ├── ✓ by-both.html + ├── ✗ by-git.html + ├── ✗ by-ignore.html + └── ✓ keep.html + "); + + assert_eq!( + candidates, + vec!["content-['by-both.html']", "content-['keep.html']"] + ); + assert_eq!(files, vec!["by-both.html", "keep.html"]); + } + + #[test] + fn git_info_exclude_is_respected() { + // `.git/info/exclude` works like the repository root's `.gitignore` file, ranking + // below it. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".git/info/exclude", "excluded.html\nreincluded.html"), + (".gitignore", "!reincluded.html"), + ("keep.html", "content-['keep.html']"), + ("excluded.html", "content-['excluded.html']"), + // Excluded by `.git/info/exclude`, re-included by the `.gitignore` file + ("reincluded.html", "content-['reincluded.html']"), + ], + vec!["@source '**/*'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ !reincluded.html + ├── ✗ excluded.html + ├── ✓ keep.html + └── ✓ reincluded.html + "); + + assert_eq!( + candidates, + vec!["content-['keep.html']", "content-['reincluded.html']"] + ); + assert_eq!(files, vec!["keep.html", "reincluded.html"]); + } + + #[test] + fn not_sources_apply_inside_external_sources() { + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "node_modules"), + ( + "node_modules/lib/src/index.html", + "content-['node_modules/lib/src/index.html']", + ), + ( + "node_modules/lib/dist/index.html", + "content-['node_modules/lib/dist/index.html']", + ), + ], + vec![ + "@source './node_modules/lib'", + "@source not './node_modules/lib/dist'", + ], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + └── ✓ node_modules + └── ✓ lib + ├── ✗ dist + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + + assert_eq!( + candidates, + vec!["content-['node_modules/lib/src/index.html']"] + ); + assert_eq!(files, vec!["node_modules/lib/src/index.html"]); + } + + #[test] + fn wildcards_in_glob_sources_respect_gitignores_deeper_in_the_subtree() { + // Like `wildcards_in_glob_sources_do_not_descend_into_gitignored_directories`, but the + // `.gitignore` sits in a directory between the glob's base and the ignored directory, + // not at the base itself. + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ("blog/2024/.gitignore", "drafts/"), + ("blog/2024/keep.html", "content-['blog/2024/keep.html']"), + ( + "blog/2024/drafts/secret.html", + "content-['blog/2024/drafts/secret.html']", + ), + ], + vec!["@source './blog/**/*.html'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + ├── ✗ .gitignore + │ drafts/ + ├── ✗ drafts + │ └── ✗ secret.html + └── ✓ keep.html + "); + + assert_eq!(candidates, vec!["content-['blog/2024/keep.html']"]); + assert_eq!(files, vec!["blog/2024/keep.html"]); + } + + #[test] + fn later_pattern_sources_override_earlier_not_sources() { + // Order matters: a later positive `@source` wins from an earlier `@source not`, and + // vice versa. + let ScanResult { + candidates, tree, .. + } = scan_with_globs( + &[ + ("src/keep.html", "content-['src/keep.html']"), + ("src/other.html", "content-['src/other.html']"), + ], + vec![ + "@source '**/*'", + "@source not './src'", + "@source './src/keep.html'", + ], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ keep.html + └── ✗ other.html + "); + + assert_eq!(candidates, vec!["content-['src/keep.html']"]); + + let ScanResult { + candidates, tree, .. + } = scan_with_globs( + &[ + ("src/keep.html", "content-['src/keep.html']"), + ("src/other.html", "content-['src/other.html']"), + ], + vec![ + "@source '**/*'", + "@source './src/keep.html'", + "@source not './src'", + ], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✗ src + ├── ✗ keep.html + └── ✗ other.html + "); + + assert!(candidates.is_empty()); + } + + #[test] + fn later_directory_sources_override_earlier_not_sources() { + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[("src/index.html", "content-['src/index.html']")], + vec!["@source '**/*'", "@source not './src'", "@source './src'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ index.html + "); + + assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_eq!(files, vec!["src/index.html"]); + } + + #[test] + fn wildcard_not_sources_exclude_matching_directories() { + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + ("src/bar/index.html", "content-['src/bar/index.html']"), + ("src/foo/index.html", "content-['src/foo/index.html']"), + ], + vec!["@source './src/**/*.html'", "@source not './src/ba*'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar + │ └── ✗ index.html + └── ✓ foo + └── ✓ index.html + "); + + assert_eq!(candidates, vec!["content-['src/foo/index.html']"]); + assert_eq!(files, vec!["src/foo/index.html"]); + } + + #[test] + fn explicit_file_sources_include_default_ignored_files_without_extensions() { + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".env", "content-['.env']"), + ("index.html", "content-['index.html']"), + ], + vec!["@source '.env'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ .env + └── ✗ index.html + "); + + assert_eq!(candidates, vec!["content-['.env']"]); + assert_eq!(files, vec![".env"]); + } + + #[test] + fn unreachable_whitelists_do_not_reinclude_explicit_source_directories() { + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + &[ + (".gitignore", "parent/"), + ("parent/.gitignore", "!child/"), + ("parent/child/.gitignore", "index.html"), + ( + "parent/child/index.html", + "content-['parent/child/index.html']", + ), + ], + vec!["@source './parent/child'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ parent/ + └── ✓ parent + ├── ✗ .gitignore + │ !child/ + └── ✓ child + ├── ✗ .gitignore + │ index.html + └── ✓ index.html + "); + + assert_eq!(candidates, vec!["content-['parent/child/index.html']"]); + assert_eq!(files, vec!["parent/child/index.html"]); + } + #[test] fn test_extract_used_css_variables_from_css() { let dir = tempdir().unwrap().into_path();