From a4d3f804cee635f5790ef6cb0c3ee2f89da7ef13 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Tue, 11 Aug 2026 20:23:43 +0200 Subject: [PATCH 01/23] rebase `ignore` crate from 0.4.24 to 0.4.33 --- Cargo.lock | 22 +- crates/ignore/Cargo.toml | 11 +- crates/ignore/examples/walk.rs | 4 +- crates/ignore/src/default_types.rs | 18 +- crates/ignore/src/dir.rs | 785 +++++++--- crates/ignore/src/gitignore.rs | 171 ++- crates/ignore/src/incremental.rs | 1286 +++++++++++++++++ crates/ignore/src/lib.rs | 93 +- crates/ignore/src/overrides.rs | 18 +- crates/ignore/src/pathutil.rs | 236 +-- crates/ignore/src/types.rs | 81 +- crates/ignore/src/walk.rs | 570 ++++++-- ...gnore_matched_path_or_any_parents_tests.rs | 61 +- 13 files changed, 2699 insertions(+), 657 deletions(-) create mode 100644 crates/ignore/src/incremental.rs diff --git a/Cargo.lock b/Cargo.lock index 657728c04..8a10094e3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -40,7 +40,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "531a9155a481e2ee699d4f98f43c0ca4ff8ee1bfd55c31e9e98fb29d2b176fe0" dependencies = [ "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "serde", ] @@ -241,14 +241,14 @@ dependencies = [ [[package]] name = "globset" -version = "0.4.17" +version = "0.4.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eab69130804d941f8075cfd713bf8848a2c3b3f201a9457a11e6f87e1ab62305" +checksum = "07c34a9410465b45bd9787443bc7370f37735bad04b0f0cd57ff1a3186c98988" dependencies = [ "aho-corasick", "bstr", "log", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "regex-syntax 0.8.5", ] @@ -273,7 +273,7 @@ dependencies = [ "globset", "log", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "same-file", "walkdir", "winapi-util", @@ -281,7 +281,7 @@ dependencies = [ [[package]] name = "ignore" -version = "0.4.24" +version = "0.4.33" dependencies = [ "bstr", "crossbeam-channel", @@ -290,7 +290,7 @@ dependencies = [ "globset", "log", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "same-file", "walkdir", "winapi-util", @@ -517,7 +517,7 @@ checksum = "b544ef1b4eac5dc2db33ea63606ae9ffcfac26c1416a2806ae0bf5f56b201191" dependencies = [ "aho-corasick", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "regex-syntax 0.8.5", ] @@ -532,9 +532,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.8" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "368758f23274712b504848e9d5a6f010445cc8b87a7cdb4d7cbee666c1288da3" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -646,7 +646,7 @@ dependencies = [ "dunce", "fast-glob", "globwalk", - "ignore 0.4.24", + "ignore 0.4.33", "log", "pretty_assertions", "rayon", diff --git a/crates/ignore/Cargo.toml b/crates/ignore/Cargo.toml index bd0352d15..d0af38ba9 100644 --- a/crates/ignore/Cargo.toml +++ b/crates/ignore/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "ignore" -version = "0.4.24" #:version +version = "0.4.33" #:version authors = ["Andrew Gallant "] description = """ A fast library for efficiently matching ignore files such as `.gitignore` @@ -12,7 +12,10 @@ repository = "https://github.com/BurntSushi/ripgrep/tree/master/crates/ignore" readme = "README.md" keywords = ["glob", "ignore", "gitignore", "pattern", "file"] license = "Unlicense OR MIT" +# CHANGED: Use an explicit edition instead of `edition.workspace = true` since this crate is +# vendored into the Tailwind CSS workspace. edition = "2024" +rust-version = "1.88" [lib] name = "ignore" @@ -20,15 +23,17 @@ bench = false [dependencies] crossbeam-deque = "0.8.3" -globset = "0.4.17" +# CHANGED: Use the published globset crate instead of a path dependency. +globset = "0.4.20" log = "0.4.20" memchr = "2.6.3" same-file = "1.0.6" walkdir = "2.4.0" +# CHANGED: Added `dunce` to canonicalize paths without UNC prefixes on Windows. dunce = "1.0.5" [dependencies.regex-automata] -version = "0.4.0" +version = "0.4.18" default-features = false features = ["std", "perf", "syntax", "meta", "nfa", "hybrid", "dfa-onepass"] diff --git a/crates/ignore/examples/walk.rs b/crates/ignore/examples/walk.rs index 9c627dc3e..c61d0515e 100644 --- a/crates/ignore/examples/walk.rs +++ b/crates/ignore/examples/walk.rs @@ -18,9 +18,7 @@ fn main() { let stdout_thread = std::thread::spawn(move || { let mut stdout = std::io::BufWriter::new(std::io::stdout()); for dent in rx { - stdout - .write_all(&Vec::from_path_lossy(dent.path())) - .unwrap(); + stdout.write_all(&Vec::from_path_lossy(dent.path())).unwrap(); stdout.write_all(b"\n").unwrap(); } }); diff --git a/crates/ignore/src/default_types.rs b/crates/ignore/src/default_types.rs index 4e060b76a..6b5bba0f0 100644 --- a/crates/ignore/src/default_types.rs +++ b/crates/ignore/src/default_types.rs @@ -47,6 +47,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["cml"], &["*.cml"]), (&["coffeescript"], &["*.coffee"]), (&["config"], &["*.cfg", "*.conf", "*.config", "*.ini"]), + (&["container"], &["*Containerfile*", "*Dockerfile*"]), (&["coq"], &["*.v"]), (&["cpp"], &[ "*.[ChH]", "*.cc", "*.[ch]pp", "*.[ch]xx", "*.hh", "*.inl", @@ -109,6 +110,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["hbs"], &["*.hbs"]), (&["hs"], &["*.hs", "*.lhs"]), (&["html"], &["*.htm", "*.html", "*.ejs"]), + (&["hurl"], &["*.hurl"]), (&["hy"], &["*.hy"]), (&["idris"], &["*.idr", "*.lidr"]), (&["janet"], &["*.janet"]), @@ -185,6 +187,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["mint"], &["*.mint"]), (&["mk"], &["mkfile"]), (&["ml"], &["*.ml"]), + (&["mojo"], &["*.mojo"]), (&["motoko"], &["*.mo"]), (&["msbuild"], &[ "*.csproj", "*.fsproj", "*.vcxproj", "*.proj", "*.props", "*.targets", @@ -206,11 +209,12 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ "*.php", "*.php3", "*.php4", "*.php5", "*.php7", "*.php8", "*.pht", "*.phtml" ]), + (&["pkgbuild"], &["PKGBUILD"]), (&["po"], &["*.po"]), (&["pod"], &["*.pod"]), (&["postscript"], &["*.eps", "*.ps"]), (&["prolog"], &["*.pl", "*.pro", "*.prolog", "*.P"]), - (&["protobuf"], &["*.proto"]), + (&["proto", "protobuf"], &["*.proto"]), (&["ps"], &["*.cdxml", "*.ps1", "*.ps1xml", "*.psd1", "*.psm1"]), (&["puppet"], &["*.epp", "*.erb", "*.pp", "*.rb"]), (&["purs"], &["*.purs"]), @@ -231,6 +235,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["red"], &["*.r", "*.red", "*.reds"]), (&["rescript"], &["*.res", "*.resi"]), (&["robot"], &["*.robot"]), + (&["rocq"], &["*.v"]), (&["rst"], &["*.rst"]), (&["ruby"], &[ // Idiomatic files @@ -274,6 +279,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["spark"], &["*.spark"]), (&["spec"], &["*.spec"]), (&["sql"], &["*.sql", "*.psql"]), + (&["ssa"], &["*.ssa"]), (&["stylus"], &["*.styl"]), (&["sv"], &["*.v", "*.vg", "*.sv", "*.svh", "*.h"]), (&["svelte"], &["*.svelte", "*.svelte.ts"]), @@ -359,4 +365,14 @@ mod tests { previous_name = name; } } + + #[test] + fn default_types_aliases_are_sorted() { + for (aliases, _) in DEFAULT_TYPES.iter() { + assert!( + aliases.is_sorted(), + "this alias list is not sorted: {aliases:?}", + ); + } + } } diff --git a/crates/ignore/src/dir.rs b/crates/ignore/src/dir.rs index 11b58f8ca..6bee724c8 100644 --- a/crates/ignore/src/dir.rs +++ b/crates/ignore/src/dir.rs @@ -16,7 +16,7 @@ use std::{ collections::HashMap, ffi::{OsStr, OsString}, - fs::{File, FileType}, + fs::{self, File, FileType}, io::{self, BufRead}, path::{Path, PathBuf}, sync::{Arc, RwLock, Weak}, @@ -25,7 +25,7 @@ use std::{ use crate::{ gitignore::{self, Gitignore, GitignoreBuilder}, overrides::{self, Override}, - pathutil::{is_hidden, strip_prefix}, + pathutil::{is_hidden_entry, strip_prefix}, types::{self, Types}, walk::DirEntry, {Error, Match, PartialErrorBuilder}, @@ -91,7 +91,25 @@ struct IgnoreOptions { /// Ignore is a matcher useful for recursively walking one or more directories. #[derive(Clone, Debug)] -pub(crate) struct Ignore(Arc); +pub(crate) struct Ignore { + inner: Arc, + // Parent matchers are cached independently of the path being walked, but + // matching them still needs the canonicalized path originally passed to + // `add_parents`. For example, when walking `/tmp/project/src`, parent + // matchers use `/tmp/project/src` to rewrite `/tmp/project/src/foo.py` + // before matching it against ignore files from `/tmp/project` and its + // ancestors. + // + // For ripgrep itself, this means that `rg pat src tests` must rewrite + // `src/foo` relative to `.../src`, and not whatever root was prepared + // first. + // + // See: https://github.com/BurntSushi/ripgrep/pull/3420 + // See: https://github.com/BurntSushi/ripgrep/issues/3376 + // See: https://github.com/BurntSushi/ripgrep/issues/3419 + // See: https://github.com/BurntSushi/ripgrep/issues/3320 + absolute_base: Option>, +} #[derive(Clone, Debug)] struct IgnoreInner { @@ -112,12 +130,9 @@ struct IgnoreInner { /// /// If this is the root directory or there are otherwise no more /// directories to match, then `parent` is `None`. - parent: Option, + parent: Option>, /// Whether this is an absolute parent matcher, as added by add_parent. is_absolute_parent: bool, - /// The absolute base path of this matcher. Populated only if parent - /// directories are added. - absolute_base: Option>, /// The directory that gitignores should be interpreted relative to. /// /// Usually this is the directory containing the gitignore file. But in @@ -152,34 +167,36 @@ struct IgnoreInner { impl Ignore { /// Return the directory path of this matcher. + #[cfg(test)] pub(crate) fn path(&self) -> &Path { - &self.0.dir + &self.inner.dir } /// Return true if this matcher has no parent. pub(crate) fn is_root(&self) -> bool { - self.0.parent.is_none() - } - - /// Returns true if this matcher was added via the `add_parents` method. - pub(crate) fn is_absolute_parent(&self) -> bool { - self.0.is_absolute_parent + self.inner.parent.is_none() } /// Return this matcher's parent, if one exists. pub(crate) fn parent(&self) -> Option { - self.0.parent.clone() + self.inner.parent.as_ref().map(|parent| Ignore { + inner: parent.clone(), + absolute_base: self.absolute_base.clone(), + }) } /// Create a new `Ignore` matcher with the parent directories of `dir`. /// /// Note that this can only be called on an `Ignore` matcher with no /// parents (i.e., `is_root` returns `true`). This will panic otherwise. - pub(crate) fn add_parents>(&self, path: P) -> (Ignore, Option) { - if !self.0.opts.parents - && !self.0.opts.git_ignore - && !self.0.opts.git_exclude - && !self.0.opts.git_global + pub(crate) fn add_parents>( + &self, + path: P, + ) -> (Ignore, Option) { + if !self.inner.opts.parents + && !self.inner.opts.git_ignore + && !self.inner.opts.git_exclude + && !self.inner.opts.git_global { // If we never need info from parent directories, then don't do // anything. @@ -209,25 +226,34 @@ impl Ignore { let mut errs = PartialErrorBuilder::default(); let mut ig = self.clone(); for parent in parents.into_iter().rev() { - let mut compiled = self.0.compiled.write().unwrap(); + let mut compiled = self.inner.compiled.write().unwrap(); if let Some(weak) = compiled.get(parent.as_os_str()) { if let Some(prebuilt) = weak.upgrade() { - ig = Ignore(prebuilt); + ig = Ignore { + inner: prebuilt, + absolute_base: Some(absolute_base.clone()), + }; continue; } } let (mut igtmp, err) = ig.add_child_path(parent); errs.maybe_push(err); igtmp.is_absolute_parent = true; - igtmp.absolute_base = Some(absolute_base.clone()); - igtmp.has_git = if self.0.opts.require_git && self.0.opts.git_ignore { - parent.join(".git").exists() || parent.join(".jj").exists() - } else { - false - }; + igtmp.has_git = + if self.inner.opts.require_git && self.inner.opts.git_ignore { + parent.join(".git").exists() || parent.join(".jj").exists() + } else { + false + }; let ig_arc = Arc::new(igtmp); - ig = Ignore(ig_arc.clone()); - compiled.insert(parent.as_os_str().to_os_string(), Arc::downgrade(&ig_arc)); + ig = Ignore { + inner: ig_arc.clone(), + absolute_base: Some(absolute_base.clone()), + }; + compiled.insert( + parent.as_os_str().to_os_string(), + Arc::downgrade(&ig_arc), + ); } (ig, errs.into_error_option()) } @@ -240,60 +266,161 @@ impl Ignore { /// returned if it exists. /// /// Note that all I/O errors are completely ignored. - pub(crate) fn add_child>(&self, dir: P) -> (Ignore, Option) { + pub(crate) fn add_child>( + &self, + dir: P, + ) -> (Ignore, Option) { let (ig, err) = self.add_child_path(dir.as_ref()); - (Ignore(Arc::new(ig)), err) + ( + Ignore { + inner: Arc::new(ig), + absolute_base: self.absolute_base.clone(), + }, + err, + ) + } + + /// Like add_child, but uses successful read_dir entries to reduce + /// probing when discovering ignore files. + pub(crate) fn add_child_with_entries>( + &self, + dir: P, + entries: &[fs::DirEntry], + ) -> (Ignore, Option) { + let files = self.collect_ignore_files(entries); + let (ig, err) = self.add_child_path_with_found_ignore_files( + dir.as_ref(), + Some(&files), + ); + ( + Ignore { + inner: Arc::new(ig), + absolute_base: self.absolute_base.clone(), + }, + err, + ) } /// Like add_child, but takes a full path and returns an IgnoreInner. fn add_child_path(&self, dir: &Path) -> (IgnoreInner, Option) { - let git_type = - if self.0.opts.require_git && (self.0.opts.git_ignore || self.0.opts.git_exclude) { - dir.join(".git").metadata().ok().map(|md| md.file_type()) - } else { - None - }; - let has_git = git_type.is_some() || dir.join(".jj").exists(); + self.add_child_path_with_found_ignore_files(dir, None) + } + + fn collect_ignore_files( + &self, + entries: &[fs::DirEntry], + ) -> IgnoreFilesFound { + let custom_ignore_filenames = &self.inner.custom_ignore_filenames; + let mut files = IgnoreFilesFound { + has_ignore: false, + has_git_ignore: false, + has_git_dir: false, + has_jj_dir: false, + custom_ignore_files: vec![false; custom_ignore_filenames.len()], + }; + for entry in entries { + let file_name = entry.file_name(); + if file_name == OsStr::new(".ignore") { + files.has_ignore = true; + } else if file_name == OsStr::new(".gitignore") { + files.has_git_ignore = true; + } else if file_name == OsStr::new(".git") { + files.has_git_dir = true; + } else if file_name == OsStr::new(".jj") { + files.has_jj_dir = true; + } + for (i, name) in custom_ignore_filenames.iter().enumerate() { + if file_name == name.as_os_str() { + files.custom_ignore_files[i] = true; + } + } + } + files + } + + fn add_child_path_with_found_ignore_files( + &self, + dir: &Path, + ignore_files_list: Option<&IgnoreFilesFound>, + ) -> (IgnoreInner, Option) { + let check_vcs_dir = self.inner.opts.require_git + && (self.inner.opts.git_ignore || self.inner.opts.git_exclude); + let git_type = if check_vcs_dir + && ignore_files_list.is_none_or(|i| i.has_git_dir) + { + dir.join(".git").metadata().ok().map(|md| md.file_type()) + } else { + None + }; + let has_jj = check_vcs_dir + && ignore_files_list.is_none_or(|i| i.has_jj_dir) + && dir.join(".jj").exists(); + let has_git = check_vcs_dir && (git_type.is_some() || has_jj); let mut errs = PartialErrorBuilder::default(); - let custom_ig_matcher = if self.0.custom_ignore_filenames.is_empty() { + let custom_ig_matcher = if self + .inner + .custom_ignore_filenames + .is_empty() + { Gitignore::empty() } else { - let (m, err) = create_gitignore( - &dir, - &dir, - &self.0.custom_ignore_filenames, - self.0.opts.ignore_case_insensitive, - ); - errs.maybe_push(err); - m + let custom_ignore_names: Vec<&OsString> = match ignore_files_list { + None => self.inner.custom_ignore_filenames.iter().collect(), + Some(m) => self + .inner + .custom_ignore_filenames + .iter() + .zip(m.custom_ignore_files.iter()) + .filter(|&(_, &matched)| matched) + .map(|(name, _)| name) + .collect(), + }; + if custom_ignore_names.is_empty() { + Gitignore::empty() + } else { + let (m, err) = create_gitignore( + &dir, + &dir, + &custom_ignore_names, + self.inner.opts.ignore_case_insensitive, + ); + errs.maybe_push(err); + m + } }; - let ig_matcher = if !self.0.opts.ignore { + let ig_matcher = if !self.inner.opts.ignore + || !ignore_files_list.is_none_or(|i| i.has_ignore) + { Gitignore::empty() } else { let (m, err) = create_gitignore( &dir, &dir, &[".ignore"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m }; - let gi_matcher = if !self.0.opts.git_ignore { + let gi_matcher = if !self.inner.opts.git_ignore + || !ignore_files_list.is_none_or(|i| i.has_git_ignore) + { Gitignore::empty() } else { let (m, err) = create_gitignore( &dir, &dir, &[".gitignore"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m }; - let gi_exclude_matcher = if !self.0.opts.git_exclude { + let gi_exclude_matcher = if !self.inner.opts.git_exclude + || !ignore_files_list.is_none_or(|i| i.has_git_dir) + { Gitignore::empty() } else { match resolve_git_commondir(dir, git_type) { @@ -302,7 +429,7 @@ impl Ignore { &dir, &git_dir, &["info/exclude"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m @@ -314,32 +441,38 @@ impl Ignore { } }; let ig = IgnoreInner { - compiled: self.0.compiled.clone(), + compiled: self.inner.compiled.clone(), dir: dir.to_path_buf(), - overrides: self.0.overrides.clone(), - types: self.0.types.clone(), - parent: Some(self.clone()), + overrides: self.inner.overrides.clone(), + types: self.inner.types.clone(), + parent: Some(self.inner.clone()), is_absolute_parent: false, - absolute_base: self.0.absolute_base.clone(), - global_gitignores_relative_to: self.0.global_gitignores_relative_to.clone(), - explicit_ignores: self.0.explicit_ignores.clone(), - custom_ignore_filenames: self.0.custom_ignore_filenames.clone(), + global_gitignores_relative_to: self + .inner + .global_gitignores_relative_to + .clone(), + explicit_ignores: self.inner.explicit_ignores.clone(), + custom_ignore_filenames: self + .inner + .custom_ignore_filenames + .clone(), custom_ignore_matcher: custom_ig_matcher, ignore_matcher: ig_matcher, - git_global_matcher: self.0.git_global_matcher.clone(), + git_global_matcher: self.inner.git_global_matcher.clone(), git_ignore_matcher: gi_matcher, git_exclude_matcher: gi_exclude_matcher, has_git, - opts: self.0.opts, + opts: self.inner.opts, }; (ig, errs.into_error_option()) } /// Returns true if at least one type of ignore rule should be matched. fn has_any_ignore_rules(&self) -> bool { - let opts = self.0.opts; - let has_custom_ignore_files = !self.0.custom_ignore_filenames.is_empty(); - let has_explicit_ignores = !self.0.explicit_ignores.is_empty(); + let opts = self.inner.opts; + let has_custom_ignore_files = + !self.inner.custom_ignore_filenames.is_empty(); + let has_explicit_ignores = !self.inner.explicit_ignores.is_empty(); opts.ignore || opts.git_global @@ -350,9 +483,12 @@ impl Ignore { } /// Like `matched`, but works with a directory entry instead. - pub(crate) fn matched_dir_entry<'a>(&'a self, dent: &DirEntry) -> Match> { + pub(crate) fn matched_dir_entry<'a>( + &'a self, + dent: &DirEntry, + ) -> Match> { let m = self.matched(dent.path(), dent.is_dir()); - if m.is_none() && self.0.opts.hidden && is_hidden(dent) { + if m.is_none() && self.inner.opts.hidden && is_hidden_entry(dent) { return Match::Ignore(IgnoreMatch::hidden()); } m @@ -362,7 +498,11 @@ impl Ignore { /// ignored or not. /// /// The match contains information about its origin. - fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub(crate) fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { // We need to be careful with our path. If it has a leading ./, then // strip it because it causes nothing but trouble. let mut path = path.as_ref(); @@ -373,9 +513,9 @@ impl Ignore { // regardless of whether it's whitelist/ignore, then we quit and // return that result immediately. Overrides have the highest // precedence. - if !self.0.overrides.is_empty() { + if !self.inner.overrides.is_empty() { let mat = self - .0 + .inner .overrides .matched(path, is_dir) .map(IgnoreMatch::overrides); @@ -392,8 +532,9 @@ impl Ignore { whitelisted = mat; } } - if !self.0.types.is_empty() { - let mat = self.0.types.matched(path, is_dir).map(IgnoreMatch::types); + if !self.inner.types.is_empty() { + let mat = + self.inner.types.matched(path, is_dir).map(IgnoreMatch::types); if mat.is_ignore() { return mat; } else if mat.is_whitelist() { @@ -405,98 +546,143 @@ impl Ignore { /// Performs matching only on the ignore files for this directory and /// all parent directories. - fn matched_ignore<'a>(&'a self, path: &Path, is_dir: bool) -> Match> { - let (mut m_custom_ignore, mut m_ignore, mut m_gi, mut m_gi_exclude, mut m_explicit) = ( - Match::None, - Match::None, - Match::None, - Match::None, - Match::None, - ); - let any_git = !self.0.opts.require_git || self.parents().any(|ig| ig.0.has_git); + pub(crate) fn matched_ignore<'a>( + &'a self, + path: &Path, + is_dir: bool, + ) -> Match> { + let ( + mut m_custom_ignore, + mut m_ignore, + mut m_gi, + mut m_gi_exclude, + mut m_explicit, + ) = (Match::None, Match::None, Match::None, Match::None, Match::None); + let any_git = !self.inner.opts.require_git + || self.parents().any(|ig| ig.inner.has_git); let mut saw_git = false; - for ig in self.parents().take_while(|ig| !ig.0.is_absolute_parent) { + for ig in self.parents().take_while(|ig| !ig.inner.is_absolute_parent) + { if m_custom_ignore.is_none() { - m_custom_ignore = - ig.0.custom_ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_custom_ignore = ig + .inner + .custom_ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if m_ignore.is_none() { - m_ignore = - ig.0.ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_ignore = ig + .inner + .ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi.is_none() { - m_gi = - ig.0.git_ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi = ig + .inner + .git_ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi_exclude.is_none() { - m_gi_exclude = - ig.0.git_exclude_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi_exclude = ig + .inner + .git_exclude_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } - saw_git = saw_git || ig.0.has_git; + saw_git = saw_git || ig.inner.has_git; } - if self.0.opts.parents { - if let Some(_) = self.absolute_base() { - // CHANGED: We removed a code path that rewrote the `path` to be relative to - // `self.absolute_base()` because it assumed that the every path is inside the base - // which is not the case for us as we use `WalkBuilder#add` to add roots outside of the - // base. - for ig in self.parents().skip_while(|ig| !ig.0.is_absolute_parent) { + if self.inner.opts.parents { + if let Some(abs_parent_path) = self.absolute_base() { + // What we want to do here is take the absolute base path of + // this directory and join it with the path we're searching. + // The main issue we want to avoid is accidentally duplicating + // directory components, so we try to strip any common prefix + // off of `path`. Overall, this seems a little ham-fisted, but + // it does fix a nasty bug. It should do fine until we overhaul + // this crate. + let path = abs_parent_path.join( + self.parents() + .take_while(|ig| !ig.inner.is_absolute_parent) + .last() + .map_or(path, |ig| { + // This is a weird special case when ripgrep users + // search with just a `.`, as some tools do + // automatically (like consult). In this case, if + // we don't bail out now, the code below will strip + // a leading `.` from `path`, which might mangle + // a hidden file name! + if ig.inner.dir.as_path() == Path::new(".") { + return path; + } + let without_dot_slash = strip_if_is_prefix( + "./", + ig.inner.dir.as_path(), + ); + let relative_base = + strip_if_is_prefix(without_dot_slash, path); + strip_if_is_prefix("/", relative_base) + }), + ); + + for ig in self + .parents() + .skip_while(|ig| !ig.inner.is_absolute_parent) + { if m_custom_ignore.is_none() { - m_custom_ignore = - ig.0.custom_ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_custom_ignore = ig + .inner + .custom_ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if m_ignore.is_none() { - m_ignore = - ig.0.ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_ignore = ig + .inner + .ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi.is_none() { - m_gi = - ig.0.git_ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi = ig + .inner + .git_ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi_exclude.is_none() { - m_gi_exclude = - ig.0.git_exclude_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi_exclude = ig + .inner + .git_exclude_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } - saw_git = saw_git || ig.0.has_git; + saw_git = saw_git || ig.inner.has_git; } } } - for gi in self.0.explicit_ignores.iter().rev() { - // CHANGED: We need to make sure that the explicit gitignore rules apply to the path - // - // path = Is the current file/folder we are traversing - // gi.path() = Is the path of the custom gitignore file - // - // E.g.: If we have a custom rule for `/src/utils` with `**/*`, and we are looking at - // just `/src`, then the `**/*` rules do not apply to this folder, so we can - // ignore the current custom gitignore file. - // - if !path.starts_with(gi.path()) { - continue; - } + for gi in self.inner.explicit_ignores.iter().rev() { if !m_explicit.is_none() { break; } + // CHANGED: We need to make sure that the explicit gitignore rules + // apply to the path + // + // path = Is the current file/folder we are traversing + // gi.path() = Is the path of the custom gitignore file + // + // E.g.: If we have a custom rule for `/src/utils` with `**/*`, and + // we are looking at just `/src`, then the `**/*` rules do + // not apply to this folder, so we can ignore the current + // custom gitignore file. + if !path.starts_with(gi.path()) { + continue; + } m_explicit = gi.matched(&path, is_dir).map(IgnoreMatch::gitignore); } let m_global = if any_git { - self.0 + self.inner .git_global_matcher .matched(&path, is_dir) .map(IgnoreMatch::gitignore) @@ -504,59 +690,90 @@ impl Ignore { Match::None }; - // CHANGED: We added logic to configure an order in which the ignore files are respected and - // allowed a whitelist in a later file to overrule a block on an earlier file. + // CHANGED: We added logic to configure an order in which the ignore + // files are respected. Explicitly added ignores (via + // `WalkBuilder::add_gitignore`) take precedence over all ignore files + // found on disk, and the first source with a definitive answer wins. let order = [ // Manually added ignores - &m_explicit, + m_explicit, // .custom-ignore - &m_custom_ignore, + m_custom_ignore, // .ignore - &m_ignore, + m_ignore, // .gitignore - &m_gi, + m_gi, // .git/info/exclude - &m_gi_exclude, + m_gi_exclude, // Global gitignore - &m_global, + m_global, ]; - for check in order.into_iter() { - if check.is_none() { - continue; + if !check.is_none() { + return check; } - - return check.clone(); } - - m_explicit + Match::None } /// Returns an iterator over parent ignore matchers, including this one. pub(crate) fn parents(&self) -> Parents<'_> { - Parents(Some(self)) + Parents(Some(IgnoreRef { inner: &self.inner })) } /// Returns the first absolute path of the first absolute parent, if /// one exists. fn absolute_base(&self) -> Option<&Path> { - self.0.absolute_base.as_ref().map(|p| &***p) + self.absolute_base.as_ref().map(|p| &***p) + } +} + +/// State for tracking what kinds of files ripgrep is interested in for a +/// given directory. +/// +/// This is computed over the entire set of files in a directory instead of +/// trying to stat each file individually. If a file is present, it's only then +/// that we stat it for more information, instead of relying on the stat to +/// determine its existence. +#[derive(Debug)] +struct IgnoreFilesFound { + has_ignore: bool, + has_git_ignore: bool, + has_git_dir: bool, + has_jj_dir: bool, + custom_ignore_files: Vec, +} + +#[derive(Clone, Copy)] +pub(crate) struct IgnoreRef<'a> { + inner: &'a IgnoreInner, +} + +impl IgnoreRef<'_> { + pub(crate) fn path(&self) -> &Path { + &self.inner.dir + } + + pub(crate) fn is_absolute_parent(&self) -> bool { + self.inner.is_absolute_parent } } /// An iterator over all parents of an ignore matcher, including itself. -/// -/// The lifetime `'a` refers to the lifetime of the initial `Ignore` matcher. -pub(crate) struct Parents<'a>(Option<&'a Ignore>); +pub(crate) struct Parents<'a>(Option>); impl<'a> Iterator for Parents<'a> { - type Item = &'a Ignore; + type Item = IgnoreRef<'a>; - fn next(&mut self) -> Option<&'a Ignore> { + fn next(&mut self) -> Option> { match self.0.take() { None => None, Some(ig) => { - self.0 = ig.0.parent.as_ref(); + self.0 = ig + .inner + .parent + .as_deref() + .map(|inner| IgnoreRef { inner }); Some(ig) } } @@ -645,33 +862,42 @@ impl IgnoreBuilder { } gi } else { - log::debug!("ignoring global gitignore file because CWD is not known"); + log::debug!( + "ignoring global gitignore file because CWD is not known" + ); Gitignore::empty() }; - Ignore(Arc::new(IgnoreInner { - compiled: Arc::new(RwLock::new(HashMap::new())), - dir: self.dir.clone(), - overrides: self.overrides.clone(), - types: self.types.clone(), - parent: None, - is_absolute_parent: true, + Ignore { + inner: Arc::new(IgnoreInner { + compiled: Arc::new(RwLock::new(HashMap::new())), + dir: self.dir.clone(), + overrides: self.overrides.clone(), + types: self.types.clone(), + parent: None, + is_absolute_parent: true, + global_gitignores_relative_to, + explicit_ignores: Arc::new(self.explicit_ignores.clone()), + custom_ignore_filenames: Arc::new( + self.custom_ignore_filenames.clone(), + ), + custom_ignore_matcher: Gitignore::empty(), + ignore_matcher: Gitignore::empty(), + git_global_matcher: Arc::new(git_global_matcher), + git_ignore_matcher: Gitignore::empty(), + git_exclude_matcher: Gitignore::empty(), + has_git: false, + opts: self.opts, + }), absolute_base: None, - global_gitignores_relative_to, - explicit_ignores: Arc::new(self.explicit_ignores.clone()), - custom_ignore_filenames: Arc::new(self.custom_ignore_filenames.clone()), - custom_ignore_matcher: Gitignore::empty(), - ignore_matcher: Gitignore::empty(), - git_global_matcher: Arc::new(git_global_matcher), - git_ignore_matcher: Gitignore::empty(), - git_exclude_matcher: Gitignore::empty(), - has_git: false, - opts: self.opts, - })) + } } /// Set the current directory used for matching global gitignores. - pub(crate) fn current_dir(&mut self, cwd: impl Into) -> &mut IgnoreBuilder { + pub(crate) fn current_dir( + &mut self, + cwd: impl Into, + ) -> &mut IgnoreBuilder { self.global_gitignores_relative_to = Some(cwd.into()); self } @@ -681,7 +907,10 @@ impl IgnoreBuilder { /// By default, no override matcher is used. /// /// This overrides any previous setting. - pub(crate) fn overrides(&mut self, overrides: Override) -> &mut IgnoreBuilder { + pub(crate) fn overrides( + &mut self, + overrides: Override, + ) -> &mut IgnoreBuilder { self.overrides = Arc::new(overrides); self } @@ -712,8 +941,7 @@ impl IgnoreBuilder { &mut self, file_name: S, ) -> &mut IgnoreBuilder { - self.custom_ignore_filenames - .push(file_name.as_ref().to_os_string()); + self.custom_ignore_filenames.push(file_name.as_ref().to_os_string()); self } @@ -725,6 +953,11 @@ impl IgnoreBuilder { self } + /// Whether ignoring hidden files is enabled or not. + pub(crate) fn is_hidden(&self) -> bool { + self.opts.hidden + } + /// Enables reading `.ignore` files. /// /// `.ignore` files have the same semantics as `gitignore` files and are @@ -795,7 +1028,10 @@ impl IgnoreBuilder { /// Process ignore files case insensitively /// /// This is disabled by default. - pub(crate) fn ignore_case_insensitive(&mut self, yes: bool) -> &mut IgnoreBuilder { + pub(crate) fn ignore_case_insensitive( + &mut self, + yes: bool, + ) -> &mut IgnoreBuilder { self.opts.ignore_case_insensitive = yes; self } @@ -854,7 +1090,10 @@ pub(crate) fn create_gitignore>( /// them when multiple repositories are searched. /// /// Some I/O errors are ignored. -fn resolve_git_commondir(dir: &Path, git_type: Option) -> Result> { +fn resolve_git_commondir( + dir: &Path, + git_type: Option, +) -> Result> { let git_dir_path = || dir.join(".git"); let git_dir = git_dir_path(); if !git_type.map_or(false, |ft| ft.is_file()) { @@ -899,15 +1138,20 @@ fn resolve_git_commondir(dir: &Path, git_type: Option) -> Result + ?Sized>(prefix: &'a P, path: &'a Path) -> &'a Path { +fn strip_if_is_prefix<'a, P: AsRef + ?Sized>( + prefix: &'a P, + path: &'a Path, +) -> &'a Path { strip_prefix(prefix, path).map_or(path, |p| p) } #[cfg(test)] mod tests { - use std::{io::Write, path::Path}; + use std::{io::Write, path::Path, sync::Arc}; - use crate::{Error, dir::IgnoreBuilder, gitignore::Gitignore, tests::TempDir}; + use crate::{ + Error, dir::IgnoreBuilder, gitignore::Gitignore, tests::TempDir, + }; fn wfile>(path: P, contents: &str) { let mut file = std::fs::File::create(path).unwrap(); @@ -936,11 +1180,11 @@ mod tests { let (gi, err) = Gitignore::new(td.path().join("not-an-ignore")); assert!(err.is_none()); - let (ig, err) = IgnoreBuilder::new() - .add_ignore(gi) - .build() - .add_child(td.path()); + let (ig, err) = + IgnoreBuilder::new().add_ignore(gi).build().add_child(td.path()); assert!(err.is_none()); + // CHANGED: Explicit ignores only apply to paths inside the directory + // of the ignore file, so we have to match against full paths. assert!(ig.matched(td.path().join("foo"), false).is_ignore()); assert!(ig.matched(td.path().join("bar"), false).is_whitelist()); assert!(ig.matched(td.path().join("baz"), false).is_none()); @@ -1211,15 +1455,152 @@ mod tests { let (ig2, err) = ig1.add_child("src"); assert!(err.is_none()); - // CHANGED: These test cases do not make sense for us as we never call the Ignore with - // relative paths. - assert!(ig1.matched("llvm", true).is_ignore()); - assert!(ig2.matched("llvm", true).is_ignore()); + assert!(ig1.matched("llvm", true).is_none()); + assert!(ig2.matched("llvm", true).is_none()); assert!(ig2.matched("src/llvm", true).is_none()); assert!(ig2.matched("foo", false).is_ignore()); assert!(ig2.matched("src/foo", false).is_ignore()); } + #[test] + fn absolute_parent_matchers_are_cached_across_roots() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("src/build")); + mkdirp(td.path().join("tests/build")); + wfile(td.path().join(".gitignore"), "tests/**/build/\n"); + + let ig0 = IgnoreBuilder::new().build(); + let (src_parents, err) = ig0.add_parents(td.path().join("src")); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(td.path().join("src")); + assert!(err.is_none()); + let (tests_parents, err) = ig0.add_parents(td.path().join("tests")); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(td.path().join("tests")); + assert!(err.is_none()); + + assert!(Arc::ptr_eq(&src_parents.inner, &tests_parents.inner)); + assert!(src.matched("build", true).is_none()); + assert!(tests.matched("build", true).is_ignore()); + } + + /// Parent matchers are shared across search roots, but path rewriting for + /// absolute parents must use each root's own base path. Otherwise a rule + /// like `src/invalid` is matched against the wrong absolute path when + /// `src` is searched before a sibling root (e.g. `tests`). + /// + /// Paths passed to `matched` use the same relative layout as `Walk` when + /// roots are given as relative directory names. + /// + /// Regression for: https://github.com/BurntSushi/ripgrep/issues/3376 + /// and https://github.com/BurntSushi/ripgrep/issues/3419 + #[test] + fn multi_root_gitignore_order_independent() { + let td = tmpdir(); + let cwd = std::env::current_dir().unwrap(); + // Use paths relative to CWD like the CLI walk does for `rg pat src tests`. + let root = td.path().strip_prefix(&cwd).unwrap_or(td.path()); + let src_root = root.join("src"); + let tests_root = root.join("tests"); + + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("src")); + mkdirp(td.path().join("tests")); + wfile(td.path().join(".gitignore"), "src/invalid\n"); + wfile(td.path().join("src/invalid"), "x"); + wfile(td.path().join("src/valid"), "x"); + wfile(td.path().join("tests/valid"), "x"); + + let ig0 = IgnoreBuilder::new().build(); + + // Historically buggy order: search `src` first, then `tests`. + let (src_parents, err) = ig0.add_parents(&src_root); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(&src_root); + assert!(err.is_none()); + let (tests_parents, err) = ig0.add_parents(&tests_root); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(&tests_root); + assert!(err.is_none()); + + assert!(Arc::ptr_eq(&src_parents.inner, &tests_parents.inner)); + // Each root must carry its own absolute_base even though inners are shared. + assert_ne!( + src.absolute_base.as_ref().unwrap().as_path(), + tests.absolute_base.as_ref().unwrap().as_path() + ); + assert!( + src.matched(src_root.join("invalid"), false).is_ignore(), + "parent .gitignore must apply for the src root even when \ + another root was prepared in the same process" + ); + assert!(src.matched(src_root.join("valid"), false).is_none()); + assert!(tests.matched(tests_root.join("valid"), false).is_none()); + + // Reverse order should behave the same way. + let ig0 = IgnoreBuilder::new().build(); + let (tests_parents, err) = ig0.add_parents(&tests_root); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(&tests_root); + assert!(err.is_none()); + let (src_parents, err) = ig0.add_parents(&src_root); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(&src_root); + assert!(err.is_none()); + + assert!(src.matched(src_root.join("invalid"), false).is_ignore()); + assert!(src.matched(src_root.join("valid"), false).is_none()); + assert!(tests.matched(tests_root.join("valid"), false).is_none()); + } + + /// Same multi-root / order issue for non-git ignore files (e.g. `.rgignore` + /// via custom ignore names). + /// + /// Regression for: https://github.com/BurntSushi/ripgrep/issues/3320 + #[test] + fn multi_root_custom_ignore_order_independent() { + let td = tmpdir(); + let cwd = std::env::current_dir().unwrap(); + let root = td.path().strip_prefix(&cwd).unwrap_or(td.path()); + let alpha_root = root.join("alpha"); + let beta_root = root.join("beta"); + + mkdirp(td.path().join("alpha")); + mkdirp(td.path().join("beta")); + wfile(td.path().join(".rgignore"), "beta/**/*.svg\n"); + wfile(td.path().join("alpha/a.txt"), "x"); + wfile(td.path().join("beta/x.svg"), "x"); + + let ig0 = IgnoreBuilder::new() + .add_custom_ignore_filename(".rgignore") + .ignore(false) + .git_ignore(false) + .git_global(false) + .git_exclude(false) + .build(); + + let (alpha_parents, err) = ig0.add_parents(&alpha_root); + assert!(err.is_none()); + let (alpha, err) = alpha_parents.add_child(&alpha_root); + assert!(err.is_none()); + let (beta_parents, err) = ig0.add_parents(&beta_root); + assert!(err.is_none()); + let (beta, err) = beta_parents.add_child(&beta_root); + assert!(err.is_none()); + + assert_ne!( + alpha.absolute_base.as_ref().unwrap().as_path(), + beta.absolute_base.as_ref().unwrap().as_path() + ); + assert!(alpha.matched(alpha_root.join("a.txt"), false).is_none()); + assert!( + beta.matched(beta_root.join("x.svg"), false).is_ignore(), + "parent .rgignore must apply for the beta root regardless of \ + which root was set up first" + ); + } + #[test] fn git_info_exclude_in_linked_worktree() { let td = tmpdir(); @@ -1227,16 +1608,14 @@ mod tests { mkdirp(git_dir.join("info")); wfile(git_dir.join("info/exclude"), "ignore_me"); mkdirp(git_dir.join("worktrees/linked-worktree")); - let commondir_path = || git_dir.join("worktrees/linked-worktree/commondir"); + let commondir_path = + || git_dir.join("worktrees/linked-worktree/commondir"); mkdirp(td.path().join("linked-worktree")); let worktree_git_dir_abs = format!( "gitdir: {}", git_dir.join("worktrees/linked-worktree").to_str().unwrap(), ); - wfile( - td.path().join("linked-worktree/.git"), - &worktree_git_dir_abs, - ); + wfile(td.path().join("linked-worktree/.git"), &worktree_git_dir_abs); // relative commondir wfile(commondir_path(), "../.."); diff --git a/crates/ignore/src/gitignore.rs b/crates/ignore/src/gitignore.rs index f822d8390..8824139b5 100644 --- a/crates/ignore/src/gitignore.rs +++ b/crates/ignore/src/gitignore.rs @@ -102,7 +102,9 @@ impl Gitignore { /// /// Note that I/O errors are ignored. For more granular control over /// errors, use `GitignoreBuilder`. - pub fn new>(gitignore_path: P) -> (Gitignore, Option) { + pub fn new>( + gitignore_path: P, + ) -> (Gitignore, Option) { let path = gitignore_path.as_ref(); let parent = path.parent().unwrap_or(Path::new("/")); let mut builder = GitignoreBuilder::new(parent); @@ -123,6 +125,17 @@ impl Gitignore { /// The global config file path is specified by git's `core.excludesFile` /// config option. /// + /// # Behavior + /// + /// This routine does its best to discover any global git exclude files. + /// This will try to parse out the `excludesFile` config option in your + /// global git configuration, if necessary. + /// + /// The specific things this routine tries (which are subject to change + /// based on how git behaves) are: + /// + /// + /// /// Git's config file location is `$HOME/.gitconfig`. If `$HOME/.gitconfig` /// does not exist or does not specify `core.excludesFile`, then /// `$XDG_CONFIG_HOME/git/ignore` is read. If `$XDG_CONFIG_HOME` is not @@ -145,7 +158,8 @@ impl Gitignore { num_ignores: 0, num_whitelists: 0, matches: None, - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: false, } } @@ -190,7 +204,11 @@ impl Gitignore { /// determined by a common suffix of the directory containing this /// gitignore) is stripped. If there is no common suffix/prefix overlap, /// then `path` is assumed to be relative to this matcher. - pub fn matched>(&self, path: P, is_dir: bool) -> Match<&Glob> { + pub fn matched>( + &self, + path: P, + is_dir: bool, + ) -> Match<&Glob> { if self.is_empty() { return Match::None; } @@ -243,11 +261,16 @@ impl Gitignore { } /// Like matched, but takes a path that has already been stripped. - fn matched_stripped>(&self, path: P, is_dir: bool) -> Match<&Glob> { + fn matched_stripped>( + &self, + path: P, + is_dir: bool, + ) -> Match<&Glob> { if self.is_empty() { return Match::None; } - // CHANGED: Rules marked as only_on_files can not match against directories. + // CHANGED: Rules marked as only_on_files can not match against + // directories. if self.only_on_files && is_dir { return Match::None; } @@ -270,7 +293,10 @@ impl Gitignore { /// Strips the given path such that it's suitable for matching with this /// gitignore matcher. - fn strip<'a, P: 'a + AsRef + ?Sized>(&'a self, path: &'a P) -> &'a Path { + fn strip<'a, P: 'a + AsRef + ?Sized>( + &'a self, + path: &'a P, + ) -> &'a Path { let mut path = path.as_ref(); // A leading ./ is completely superfluous. We also strip it from // our gitignore root path, so we need to strip it from our candidate @@ -326,7 +352,8 @@ impl GitignoreBuilder { globs: vec![], case_insensitive: false, allow_unclosed_class: true, - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: false, } } @@ -337,18 +364,21 @@ impl GitignoreBuilder { pub fn build(&self) -> Result { let nignore = self.globs.iter().filter(|g| !g.is_whitelist()).count(); let nwhite = self.globs.iter().filter(|g| g.is_whitelist()).count(); - let set = self.builder.build().map_err(|err| Error::Glob { - glob: None, - err: err.to_string(), - })?; + let set = self + .builder + .build() + .map_err(|err| Error::Glob { glob: None, err: err.to_string() })?; Ok(Gitignore { set, root: self.root.clone(), globs: self.globs.clone(), num_ignores: nignore as u64, num_whitelists: nwhite as u64, - matches: Some(Arc::new(Pool::new(|| vec![]))), - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + matches: Some(Arc::new( + Pool::with_available_parallelism_capacity(|| vec![]), + )), + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: self.only_on_files, }) } @@ -411,11 +441,8 @@ impl GitignoreBuilder { // Match Git's handling of .gitignore files that begin with the Unicode BOM const UTF8_BOM: &str = "\u{feff}"; - let line = if i == 0 { - line.trim_start_matches(UTF8_BOM) - } else { - &line - }; + let line = + if i == 0 { line.trim_start_matches(UTF8_BOM) } else { &line }; if let Err(err) = self.add_line(Some(path.to_path_buf()), &line) { errs.push(err.tagged(path, lineno)); @@ -537,7 +564,10 @@ impl GitignoreBuilder { /// affected. /// /// This is disabled by default. - pub fn case_insensitive(&mut self, yes: bool) -> Result<&mut GitignoreBuilder, Error> { + pub fn case_insensitive( + &mut self, + yes: bool, + ) -> Result<&mut GitignoreBuilder, Error> { // TODO: This should not return a `Result`. Fix this in the next semver // release. self.case_insensitive = yes; @@ -556,7 +586,10 @@ impl GitignoreBuilder { /// modes since the glob parser becomes more permissive. You might want to /// enable this when compatibility (e.g., with POSIX glob implementations) /// is more important than good error messages. - pub fn allow_unclosed_class(&mut self, yes: bool) -> &mut GitignoreBuilder { + pub fn allow_unclosed_class( + &mut self, + yes: bool, + ) -> &mut GitignoreBuilder { self.allow_unclosed_class = yes; self } @@ -576,32 +609,56 @@ impl GitignoreBuilder { /// /// Note that the file path returned may not exist. pub fn gitconfig_excludes_path() -> Option { - // git supports $HOME/.gitconfig and $XDG_CONFIG_HOME/git/config. Notably, - // both can be active at the same time, where $HOME/.gitconfig takes - // precedent. So if $HOME/.gitconfig defines a `core.excludesFile`, then - // we're done. - match gitconfig_home_contents().and_then(|x| parse_excludes_file(&x)) { - Some(path) => return Some(path), - None => {} + // When GIT_CONFIG_GLOBAL is set, it replaces both $HOME/.gitconfig and + // $XDG_CONFIG_HOME/git/config (per git 2.32+). Otherwise, git supports + // $HOME/.gitconfig and $XDG_CONFIG_HOME/git/config simultaneously, where + // $HOME/.gitconfig takes precedent. + gitconfig_global_env_contents() + .and_then(|x| parse_excludes_file(&x)) + .or_else(|| { + gitconfig_home_contents().and_then(|x| parse_excludes_file(&x)) + }) + .or_else(|| { + gitconfig_xdg_contents().and_then(|x| parse_excludes_file(&x)) + }) + // System-level config has the lowest priority for core.excludesFile. + // GIT_CONFIG_SYSTEM overrides the default /etc/gitconfig path. + .or_else(|| { + gitconfig_system_contents().and_then(|x| parse_excludes_file(&x)) + }) + .or_else(excludes_file_default) +} + +/// Returns the file contents of git's global config file from the path +/// specified by the `GIT_CONFIG_GLOBAL` environment variable. +fn gitconfig_global_env_contents() -> Option> { + let path = std::env::var_os("GIT_CONFIG_GLOBAL").map(PathBuf::from)?; + if path.as_os_str().is_empty() { + return None; } - match gitconfig_xdg_contents().and_then(|x| parse_excludes_file(&x)) { - Some(path) => return Some(path), - None => {} - } - excludes_file_default() + let mut file = BufReader::new(File::open(path).ok()?); + let mut contents = vec![]; + file.read_to_end(&mut contents).ok().map(|_| contents) +} + +/// Returns the file contents of git's system-level config file. +/// +/// Checks `GIT_CONFIG_SYSTEM` first, then falls back to `/etc/gitconfig`. +fn gitconfig_system_contents() -> Option> { + let path = std::env::var_os("GIT_CONFIG_SYSTEM") + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) + .unwrap_or_else(|| PathBuf::from("/etc/gitconfig")); + let mut file = BufReader::new(File::open(path).ok()?); + let mut contents = vec![]; + file.read_to_end(&mut contents).ok().map(|_| contents) } /// Returns the file contents of git's global config file, if one exists, in /// the user's home directory. fn gitconfig_home_contents() -> Option> { - let home = match home_dir() { - None => return None, - Some(home) => home, - }; - let mut file = match File::open(home.join(".gitconfig")) { - Err(_) => return None, - Ok(file) => BufReader::new(file), - }; + let home = home_dir()?; + let mut file = BufReader::new(File::open(home.join(".gitconfig")).ok()?); let mut contents = vec![]; file.read_to_end(&mut contents).ok().map(|_| contents) } @@ -610,19 +667,11 @@ fn gitconfig_home_contents() -> Option> { /// the user's XDG_CONFIG_HOME directory. fn gitconfig_xdg_contents() -> Option> { let path = std::env::var_os("XDG_CONFIG_HOME") - .and_then(|x| { - if x.is_empty() { - None - } else { - Some(PathBuf::from(x)) - } - }) + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) .or_else(|| home_dir().map(|p| p.join(".config"))) - .map(|x| x.join("git/config")); - let mut file = match path.and_then(|p| File::open(p).ok()) { - None => return None, - Some(file) => BufReader::new(file), - }; + .map(|x| x.join("git/config"))?; + let mut file = BufReader::new(File::open(path).ok()?); let mut contents = vec![]; file.read_to_end(&mut contents).ok().map(|_| contents) } @@ -632,13 +681,8 @@ fn gitconfig_xdg_contents() -> Option> { /// Specifically, this respects XDG_CONFIG_HOME. fn excludes_file_default() -> Option { std::env::var_os("XDG_CONFIG_HOME") - .and_then(|x| { - if x.is_empty() { - None - } else { - Some(PathBuf::from(x)) - } - }) + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) .or_else(|| home_dir().map(|p| p.join(".config"))) .map(|x| x.join("git/ignore")) } @@ -667,9 +711,7 @@ fn parse_excludes_file(data: &[u8]) -> Option { re.captures(data, &mut caps); let span = caps.get_group(1)?; let candidate = &data[span]; - std::str::from_utf8(candidate) - .ok() - .map(|s| PathBuf::from(expand_tilde(s))) + std::str::from_utf8(candidate).ok().map(|s| PathBuf::from(expand_tilde(s))) } /// Expands ~ in file paths to the value of $HOME. @@ -831,7 +873,10 @@ mod tests { fn parse_excludes_file4() { let data = bytes("[core]\nexcludesFile = \"~/foo/bar\""); let got = super::parse_excludes_file(&data); - assert_eq!(path_string(got.unwrap()), super::expand_tilde("~/foo/bar")); + assert_eq!( + path_string(got.unwrap()), + super::expand_tilde("~/foo/bar") + ); } #[test] diff --git a/crates/ignore/src/incremental.rs b/crates/ignore/src/incremental.rs new file mode 100644 index 000000000..5030b0d00 --- /dev/null +++ b/crates/ignore/src/incremental.rs @@ -0,0 +1,1286 @@ +use std::{ + collections::HashMap, + path::{Path, PathBuf}, + sync::OnceLock, +}; + +use crate::{ + Error, Match, PartialErrorBuilder, dir::Ignore, pathutil::is_hidden_path, +}; + +/// A cached matcher for checking paths against hierarchical ignore files. +/// +/// An `IncrementalIgnore` is built from a [`crate::WalkBuilder`]. Unlike a +/// recursive walk, it can check individual paths while still respecting the +/// ignore files in every relevant parent directory. Matchers for directories +/// are compiled on first use and then retained for later queries. +/// Each matcher corresponds to exactly one root configured on the builder, +/// and paths passed to it are interpreted relative to that root. +/// A matcher for the special `-` root representing standard input is inert +/// and always returns a non-match. +/// +/// The matcher checks path-based filters in the same precedence order as +/// a traversal. This includes glob overrides, `.ignore`, `.gitignore`, +/// `.git/info/exclude`, global and explicitly added ignore files, custom +/// ignore file names and file type selections. It does not apply filters that +/// require a directory entry or other traversal state, such as custom entry +/// predicates. Hidden-file detection, minimum and maximum depth limits and the +/// maximum file size are applied. +/// +/// A matcher is a snapshot at directory granularity. Once the ignore files in +/// a directory have been loaded, edits to those files are not observed. Build +/// a new matcher to reload them. +/// +/// # Warning +/// +/// The incremental path checking here necessarily needs to do a lot more work +/// per path matched. Callers should _not_ use this to run directory traversal. +/// This is intended to avoid the work of re-traversing an entire directory +/// tree when only a few changes are detected. (For example, in response to +/// file additions or deletions.) +/// +/// # Example +/// +/// ```rust,no_run +/// use ignore::WalkBuilder; +/// +/// let mut builder = WalkBuilder::new("."); +/// builder.add_custom_ignore_filename(".rgignore"); +/// let mut matchers = builder.build_matchers(); +/// let matcher = &mut matchers[0]; +/// +/// if matcher.matched("src/generated.rs", false).is_ignore() { +/// println!("ignored"); +/// } +/// ``` +#[derive(Clone, Debug)] +pub struct IncrementalIgnore { + /// The root exactly as it was given to `WalkBuilder`. + root: PathBuf, + /// The normalized root used only by the opt-in normalization routine. + normalized_root: OnceLock>, + /// The matcher for the configured root directory, loaded on first use. + ignore: RootIgnore, + /// Directory paths relative to `root`, excluding the root itself. + dirs: HashMap, + /// Options for additional filtering beyond gitignore. + options: IncrementalIgnoreOptions, +} + +/// The options for a matcher, mostly meant to duplicate as much as we can from +/// `WalkParallel`. +#[derive(Clone, Debug)] +pub(crate) struct IncrementalIgnoreOptions { + pub(crate) min_depth: Option, + pub(crate) max_depth: Option, + pub(crate) max_filesize: Option, + pub(crate) hidden: bool, + pub(crate) follow_links: bool, +} + +#[derive(Clone, Debug)] +enum RootIgnore { + Unloaded(Ignore), + Loaded(Ignore), + NotDirectory, + Stdin, +} + +/// Cached traversal state for a directory relative to the configured root. +/// +/// The presence of an entry means that the directory and every ancestor +/// between it and the root have already been checked. +#[derive(Clone, Debug)] +enum CachedDir { + /// The directory may be descended into. The matcher includes the ignore + /// rules loaded through this directory and is therefore the matcher to use + /// for its children. + Allowed(Ignore), + /// The directory may not be descended into because it was ignored by a + /// path rule or hidden-file filtering. Every descendant is consequently + /// ignored, and ignore files inside this directory are not loaded. + Ignored, +} + +impl IncrementalIgnore { + pub(crate) fn new( + root: PathBuf, + ignore: Ignore, + options: IncrementalIgnoreOptions, + ) -> IncrementalIgnore { + // File traversal special cases `-` to search stdin, so we recognize + // it here for completeness too. In particular, we really want + // `WalkBuilder::build_matchers` to return a matcher for every root, + // even when it's a simple file (handled automatically) or when it's + // stdin (necessarily special cased). + // + // If callers need to search a file or directory named `-`, then they + // can use `./-`. As is the case for file traversal too. + let ignore = if root == Path::new("-") { + RootIgnore::Stdin + } else { + RootIgnore::Unloaded(ignore) + }; + IncrementalIgnore { + root, + normalized_root: OnceLock::new(), + ignore, + dirs: HashMap::new(), + options, + } + } + + /// Return the root that paths matched by this matcher are relative to. + pub fn root(&self) -> &Path { + &self.root + } + + /// Normalize `path` and return it relative to this matcher's root. + /// + /// This returns `None` when `path` cannot be made absolute or + /// when it is known to be outside this matcher's root. Unlike + /// [`IncrementalIgnore::matched`], this performs absolute path conversion, + /// lexical normalization and allocation. It is intended as an opt-in + /// convenience for callers that do not already have root-relative paths. + /// + /// Note that `.` is interpreted relative to the process level current + /// working directory. It is _not_ interpreted relative to the root of + /// this matcher. + /// + /// Note also that this may reject paths that only differ in casing. For + /// example, if the root path for this matcher is `/FOO` but the provided + /// path is `/foo/bar`, then this may return `None`. Callers must ensure + /// casing is consistent between the path provided and the root path for + /// this matcher. + pub fn normalize>(&self, path: P) -> Option { + if matches!(self.ignore, RootIgnore::Stdin) { + return None; + } + let path = normalize_absolute(path.as_ref())?; + let root = self + .normalized_root + .get_or_init(|| normalize_absolute(&self.root)) + .as_ref()?; + path.strip_prefix(root).ok().map(Path::to_path_buf) + } + + /// Match a root-relative path against ignore files in its directory and + /// all relevant parent directories. + /// + /// `is_dir` should be true when `path` should be matched as a directory. + /// + /// For the return value, use [`IncrementalMatch::is_ignore`], + /// [`IncrementalMatch::is_whitelist`] or [`IncrementalMatch::is_none`] to + /// inspect it. + /// + /// Matchers for previously unseen directories are loaded and cached during + /// this call. Errors encountered while loading ignore files are logged. To + /// receive those errors, use [`IncrementalIgnore::matched_with_errors`]. + /// + /// `path` must be relative to this matcher's root and must not contain a + /// parent directory (`..`) component. Behavior is unspecified when these + /// preconditions are violated. Callers with an absolute path or with a + /// path containing `.` or `..` may use [`IncrementalIgnore::normalize`] + /// to get a path satisfying these preconditions. In all cases, a relative + /// path is *assumed* to be relative to the root of this ignore matcher. + /// + /// In general, it is intended that callers doing recursive directory + /// traversal on the root of this matcher may provide relative paths to + /// this routine *without* calling [`IncrementalIgnore::normalize`]. + /// + /// The empty path represents the explicitly configured root and also + /// returns non-match, consistent with recursive traversal where a root is + /// always treated as being at depth zero. + pub fn matched>( + &mut self, + path: P, + is_dir: bool, + ) -> IncrementalMatch { + let (matched, err) = self.matched_with_errors(path, is_dir); + if let Some(err) = err { + log::debug!("error while loading ignore files: {err}"); + } + matched + } + + /// Match a root-relative path and return errors encountered while loading + /// ignore files. + /// + /// This is equivalent to [`IncrementalIgnore::matched`], except that + /// it returns any errors from newly loaded ignore files. Loading can + /// partially succeed, so valid rules are always applied to the returned + /// match even when an error is present. + pub fn matched_with_errors>( + &mut self, + path: P, + is_dir: bool, + ) -> (IncrementalMatch, Option) { + let mut errs = PartialErrorBuilder::default(); + let matched = + self.matched_with_errors_impl(path.as_ref(), is_dir, &mut errs); + (matched, errs.into_error_option()) + } + + fn matched_with_errors_impl( + &mut self, + relative: &Path, + is_dir: bool, + errs: &mut PartialErrorBuilder, + ) -> IncrementalMatch { + // We short-circuit here when our matcher corresponds to `Stdin` in + // order to always return a non-match. This is somewhat redundant with + // `root_ignore()` which does this too, but we do it here so that it + // always happens, e.g., before depth filtering. + if relative.is_absolute() || matches!(self.ignore, RootIgnore::Stdin) { + return IncrementalMatch::none(is_dir); + } + + let (mut satisfies_min, mut edge_max) = (true, false); + if self.options.min_depth.is_some() || self.options.max_depth.is_some() + { + let depth = relative + .components() + .filter_map(|component| match component { + std::path::Component::CurDir => None, + component => Some(component), + }) + .count(); + satisfies_min = + self.options.min_depth.is_none_or(|min| depth >= min); + let satisfies_max = + self.options.max_depth.is_none_or(|max| depth <= max); + edge_max = self.options.max_depth.is_some_and(|max| depth == max); + // When we have a file that isn't past our min depth, we can give + // up right away. + if !is_dir && !satisfies_min { + return IncrementalMatch::ignore().not_within_depth(); + } + // Same for *anything* that exceeds our max depth. + if !satisfies_max { + return IncrementalMatch::ignore().not_within_depth(); + } + } + + // When the path is invalid in some way, we bail out early to avoid + // potentially doing a stat call below. + let (mut mat, valid) = self + .matched_with_errors_ignore(relative, is_dir, errs) + .map(|mat| (mat, true)) + .unwrap_or_else(|| (IncrementalMatch::none(is_dir), false)); + + if !is_dir + && !mat.is_ignore() + && valid + && let Some(max_filesize) = self.options.max_filesize + { + let path = self.root().join(relative); + let result = if self.options.follow_links { + path.metadata() + } else { + path.symlink_metadata() + }; + match result { + Ok(md) if md.len() > max_filesize => { + return IncrementalMatch::ignore(); + } + Ok(_) => {} + Err(err) => { + // Record the error but otherwise fall through + let err = Error::from(err); + errs.push(err.with_path(path)); + } + } + } + + // We still need to tag our `mat` if it doesn't pass the depth filter, + // or if it's a directory on the edge of a depth filter. This can only + // happen when the path is reported as a directory. Otherwise regular + // files are always handled above. + if is_dir { + if !satisfies_min { + mat = mat.not_within_depth(); + } + if edge_max { + mat = mat.no_descent(); + } + } + mat + } + + fn matched_with_errors_ignore( + &mut self, + relative: &Path, + is_dir: bool, + errs: &mut PartialErrorBuilder, + ) -> Option { + let mut components = relative + .components() + .filter_map(|component| match component { + std::path::Component::CurDir => None, + component => Some(component), + }) + .peekable(); + components.peek()?; + + // If the exact parent is cached, then all of its ancestors have + // already been checked. An allowed cache entry is the matcher to use + // for children of that directory, while an ignored cache entry is + // terminal for every descendant. In the usual case, this avoids both + // walking every component and allocating a relative directory path. + // + // Only try the fast path when the final component is a normal path + // component. Invalid paths are handled by the component walk below. + let has_normal_final_component = + relative.components().next_back().is_some_and(|component| { + matches!(component, std::path::Component::Normal(_)) + }); + if has_normal_final_component && let Some(parent) = relative.parent() { + match self.dirs.get(parent) { + Some(CachedDir::Allowed(ignore)) => { + return Some(self.match_path(ignore, relative, is_dir)); + } + Some(CachedDir::Ignored) => { + return Some(IncrementalMatch::ignore()); + } + None => {} + } + } + + let mut ignore = self.root_ignore(errs)?; + let mut dir = PathBuf::new(); + while let Some(component) = components.next() { + match component { + std::path::Component::ParentDir + | std::path::Component::RootDir + | std::path::Component::Prefix(_) => { + return None; + } + std::path::Component::CurDir => continue, + std::path::Component::Normal(_) => {} + } + if components.peek().is_none() { + break; + } + dir.push(component.as_os_str()); + match self.dirs.get(&dir) { + Some(CachedDir::Allowed(cached)) => { + ignore = cached.clone(); + continue; + } + Some(CachedDir::Ignored) => { + return Some(IncrementalMatch::ignore()); + } + None => {} + } + + let path = self.root.join(&dir); + let mat = ignore.matched(&path, true); + let is_hidden = + self.options.hidden && mat.is_none() && is_hidden_path(&path); + if mat.is_ignore() || is_hidden { + self.dirs.insert(dir.clone(), CachedDir::Ignored); + return Some(IncrementalMatch::ignore()); + } + let (child, err) = ignore.add_child(&path); + errs.maybe_push(err); + self.dirs.insert(dir.clone(), CachedDir::Allowed(child.clone())); + ignore = child; + } + + Some(self.match_path(&ignore, relative, is_dir)) + } + + fn match_path( + &self, + ignore: &Ignore, + relative: &Path, + is_dir: bool, + ) -> IncrementalMatch { + let path = self.root.join(relative); + let mut mat = IncrementalMatch::from_match( + ignore.matched(&path, is_dir).map(|_| ()), + is_dir, + ); + // Whether a file is hidden or not has low precedence in filtering. We + // only check it if we haven't matched anything above. This permits + // callers to whitelist hidden files or directories. + if self.options.hidden && mat.is_none() && is_hidden_path(&path) { + mat = IncrementalMatch::ignore(); + } + mat + } + + fn root_ignore( + &mut self, + errs: &mut PartialErrorBuilder, + ) -> Option { + let ignore = match self.ignore { + RootIgnore::Unloaded(ref ignore) => ignore.clone(), + RootIgnore::Loaded(ref ignore) => return Some(ignore.clone()), + RootIgnore::NotDirectory | RootIgnore::Stdin => return None, + }; + if !self.root.is_dir() { + self.ignore = RootIgnore::NotDirectory; + return None; + } + + let (parents, err) = ignore.add_parents(&self.root); + errs.maybe_push(err); + let (root, err) = parents.add_child(&self.root); + errs.maybe_push(err); + self.ignore = RootIgnore::Loaded(root.clone()); + Some(root) + } +} + +/// The result of an incremental match. +/// +/// This is similar to [`Match`] in that it reports whether a file path should +/// be ignored, whitelisted or didn't match anything at all. It also has extra +/// data, such as whether a directory matched but should not be descended into. +/// +/// Generally speaking, callers that only care about specific files can stick +/// to the `is_none()`, `is_ignore()` or `is_whitelist()` predicates. Directory +/// entries are more complicated because we sometimes want to yield directories +/// to descend into, but not actually visit (e.g., for the minimum depth +/// filter). Or, we may want to yield a directory to visit but not descend into +/// (e.g., for the maximum depth filter). +#[derive(Clone, Debug)] +pub struct IncrementalMatch { + mat: Match<()>, + should_descend: bool, + is_within_depth: bool, +} + +impl IncrementalMatch { + fn none(is_dir: bool) -> IncrementalMatch { + IncrementalMatch { + mat: Match::None, + should_descend: is_dir, + is_within_depth: true, + } + } + + fn ignore() -> IncrementalMatch { + IncrementalMatch { + mat: Match::Ignore(()), + should_descend: false, + is_within_depth: true, + } + } + + fn from_match(mat: Match<()>, is_dir: bool) -> IncrementalMatch { + let should_descend = is_dir && !mat.is_ignore(); + IncrementalMatch { mat, should_descend, is_within_depth: true } + } + + fn no_descent(self) -> IncrementalMatch { + IncrementalMatch { should_descend: false, ..self } + } + + fn not_within_depth(self) -> IncrementalMatch { + IncrementalMatch { is_within_depth: false, ..self } + } + + /// Returns true if the match result didn't match anything. + pub fn is_none(&self) -> bool { + self.mat.is_none() + } + + /// Returns true if the match result implies the path should be ignored. + pub fn is_ignore(&self) -> bool { + self.mat.is_ignore() + } + + /// Returns true if the match result implies the path should be + /// whitelisted. + pub fn is_whitelist(&self) -> bool { + self.mat.is_whitelist() + } + + /// Returns true only when this match corresponds to a directory *and* + /// whether the caller should look inside this directory for additional + /// results. + /// + /// It is possible for a match to report `false` for + /// [`IncrementalMatch::is_ignore`] _and_ `false` for + /// [`IncrementalMatch::should_descend`]. This can occur, for example, when + /// a maximum depth setting allows a directory through, but where none of + /// its children entries should be visited. + /// + /// This is always `false` for a file path that does *not* correspond to a + /// directory. + pub fn should_descend(&self) -> bool { + self.should_descend + } + + /// Returns true only when this result corresponds to an entry that is + /// within the depth filter. + /// + /// This is `false` when a path corresponds to a directory and is less than + /// the minimum depth. In this case, callers should continue looking inside + /// that directory. + /// + /// This is always `true` for a match corresponding to a file path that + /// isn't a directory. + pub fn is_within_depth(&self) -> bool { + self.is_within_depth + } + + /// Inverts the match so that `Ignore` becomes `Whitelist` and + /// `Whitelist` becomes `Ignore`. A non-match remains the same. + pub fn invert(self) -> IncrementalMatch { + IncrementalMatch { mat: self.mat.invert(), ..self } + } +} + +/// Return a lexically normalized absolute representation of `path`. +/// +/// This collapses `.` and `..`, but intentionally does not canonicalize or +/// resolve symlinks. Ignore rules apply to the lexical path, and resolving a +/// symlink below a root could move the result outside of that root. +fn normalize_absolute(path: &Path) -> Option { + let absolute = std::path::absolute(path).ok()?; + let mut normalized = PathBuf::new(); + for component in absolute.components() { + match component { + std::path::Component::CurDir => {} + std::path::Component::ParentDir => { + normalized.pop(); + } + _ => normalized.push(component.as_os_str()), + } + } + Some(normalized) +} + +#[cfg(test)] +mod tests { + use std::{ + fs::{self, File}, + io::Write, + path::{Path, PathBuf}, + }; + + use crate::{ + IncrementalIgnore, IncrementalMatch, WalkBuilder, + overrides::OverrideBuilder, tests::TempDir, types::TypesBuilder, + }; + + use super::CachedDir; + + fn wfile>(path: P, contents: &str) { + let mut file = File::create(path).unwrap(); + file.write_all(contents.as_bytes()).unwrap(); + } + + fn mkdirp>(path: P) { + fs::create_dir_all(path).unwrap(); + } + + fn tmpdir() -> TempDir { + TempDir::new().unwrap() + } + + fn builder>(path: P) -> WalkBuilder { + let mut builder = WalkBuilder::new(path); + builder.git_global(false); + builder + } + + fn one_matcher(builder: &WalkBuilder) -> IncrementalIgnore { + let mut matchers = builder.build_matchers(); + assert_eq!(matchers.len(), 1); + matchers.pop().unwrap() + } + + fn builders, P: AsRef>( + paths: I, + ) -> WalkBuilder { + let mut builder = WalkBuilder::from_iter(paths); + builder.git_global(false); + builder + } + + fn matchers(builder: &WalkBuilder) -> Vec { + builder.build_matchers() + } + + fn matchedf>( + matcher: &mut IncrementalIgnore, + path: P, + ) -> IncrementalMatch { + let (matched, err) = matcher.matched_with_errors(path, false); + assert!(err.is_none(), "unexpected matcher error: {err:?}"); + matched + } + + fn matchedd>( + matcher: &mut IncrementalIgnore, + path: P, + ) -> IncrementalMatch { + let (matched, err) = matcher.matched_with_errors(path, true); + assert!(err.is_none(), "unexpected matcher error: {err:?}"); + matched + } + + // Test that multiple parent ignore files, when nested, are respected. + #[test] + fn nested_parent_gitignores() { + let td = tmpdir(); + let root = td.path().join("project/work"); + mkdirp(td.path().join(".git")); + mkdirp(root.join("src")); + wfile(td.path().join(".gitignore"), "*.tmp\n"); + wfile(td.path().join("project/.gitignore"), "!keep.tmp\nnested.log\n"); + + let mut m = one_matcher(&builder(&root)); + assert_eq!(m.root(), root); + assert!(matchedf(&mut m, "src/drop.tmp").is_ignore()); + assert!(matchedf(&mut m, "src/keep.tmp").is_whitelist()); + assert!(matchedf(&mut m, "src/nested.log").is_ignore()); + assert!(matchedf(&mut m, "src/ok.rs").is_none()); + } + + // Test that an anchored rule in a child directory is matched relative to + // that directory, not relative to the configured root. + #[test] + fn anchored_child_rule_uses_child_root() { + let td = tmpdir(); + let root = td.path().join("root"); + mkdirp(root.join("a")); + wfile(root.join("a/.ignore"), "/foo\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "a/foo").is_ignore()); + assert!(matchedf(&mut m, "a/b/foo").is_none()); + } + + // Test that a leading `./` impacts how the rules are matched. + #[cfg(not(windows))] + #[test] + fn leading_dot_slash_impacts_matching() { + let td = tmpdir(); + let root = td.path().join("root"); + mkdirp(root.join("a")); + wfile(root.join(".ignore"), "/foo\n"); + wfile(root.join("a/.ignore"), "/foo\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "foo").is_ignore()); + assert!(matchedf(&mut m, "./foo").is_none()); + assert!(matchedf(&mut m, "a/allowed").is_none()); + assert!(matchedf(&mut m, "a/foo").is_ignore()); + assert!(matchedf(&mut m, "./a/foo").is_none()); + } + + // Test that custom ignore files are respected. + #[test] + fn parent_ignore_and_custom_ignore() { + let td = tmpdir(); + let root = td.path().join("project/work"); + mkdirp(root.join("src")); + wfile(td.path().join(".ignore"), "*.cache\n"); + wfile(td.path().join(".rgignore"), "*.svg\n"); + wfile(td.path().join("project/.ignore"), "!keep.cache\n"); + wfile(td.path().join("project/.rgignore"), "!keep.svg\n"); + + let mut builder = builder(&root); + builder.add_custom_ignore_filename(".rgignore"); + let mut m = one_matcher(&builder); + assert!(matchedf(&mut m, "src/drop.cache").is_ignore()); + assert!(matchedf(&mut m, "src/keep.cache").is_whitelist()); + assert!(matchedf(&mut m, "src/drop.svg").is_ignore()); + assert!(matchedf(&mut m, "src/keep.svg").is_whitelist()); + } + + // Test that glob overrides take precedence over ignore files. + #[test] + fn glob_overrides_are_applied() { + let td = tmpdir(); + wfile(td.path().join(".ignore"), "keep.rs\n!drop.rs\n"); + + let mut overrides = OverrideBuilder::new(td.path()); + overrides.add("keep.rs").unwrap(); + overrides.add("!drop.rs").unwrap(); + let mut b = builder(td.path()); + b.overrides(overrides.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "keep.rs").is_whitelist()); + assert!(matchedf(&mut m, "drop.rs").is_ignore()); + } + + // Test that an override-ignored directory prevents matching rules below + // it, just as it prevents a traversal from descending into the directory. + #[test] + fn glob_overrides_apply_to_ancestors() { + let td = tmpdir(); + mkdirp(td.path().join("blocked")); + wfile(td.path().join("blocked/.ignore"), "!keep.rs\n"); + + let mut overrides = OverrideBuilder::new(td.path()); + overrides.add("!blocked/").unwrap(); + let mut b = builder(td.path()); + b.overrides(overrides.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "blocked/keep.rs").is_ignore()); + } + + // Test that file type selections are applied to files, but not + // directories. + #[test] + fn file_types_are_applied() { + let td = tmpdir(); + mkdirp(td.path().join("src")); + + let mut types = TypesBuilder::new(); + types.add("rust", "*.rs").unwrap(); + types.select("rust"); + let mut b = builder(td.path()); + b.types(types.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedd(&mut m, "src").is_none()); + assert!(matchedf(&mut m, "src/lib.rs").is_whitelist()); + assert!(matchedf(&mut m, "README.md").is_ignore()); + } + + #[test] + fn directly_ignored_directory_is_not_descended() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("blocked")); + wfile(td.path().join(".gitignore"), "blocked/\n"); + + let mut m = one_matcher(&builder(td.path())); + let dir = matchedd(&mut m, "blocked"); + assert!(dir.is_ignore()); + assert!(!dir.should_descend()); + } + + // Test that when a directory is ignored, anything below it always ignored + // even when there are explicit whitelist rules. This matches directory + // traversal semantics. + #[test] + fn ignored_ancestor_wins() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("blocked")); + mkdirp(td.path().join("open")); + // This is the key line: since the entire `blocked` directory is + // ignored, a proper file traversal won't ever descend into it. So + // `blocked/keep.rs` should be ignored even if there are ignore rules + // "beneath" it that whitelist it. + wfile(td.path().join(".gitignore"), "blocked/\n!blocked/keep.rs\n"); + wfile(td.path().join("blocked/.gitignore"), "!keep.rs\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "blocked/keep.rs").is_ignore()); + assert!(matchedf(&mut m, "blocked/other.rs").is_ignore()); + assert!(matchedf(&mut m, "open/keep.rs").is_none()); + } + + // Test that we respect git boundaries. And that we don't respect git + // boundaries when not configured to do so. + #[test] + fn respects_git_repository_boundaries() { + let td = tmpdir(); + let root = td.path().join("repo/src"); + mkdirp(td.path().join("repo/.git")); + mkdirp(&root); + wfile(td.path().join(".gitignore"), "outside-rule\n"); + wfile(td.path().join(".ignore"), "tool-rule\n"); + wfile(td.path().join("repo/.gitignore"), "inside-rule\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "inside-rule").is_ignore()); + assert!(matchedf(&mut m, "outside-rule").is_none()); + assert!(matchedf(&mut m, "tool-rule").is_ignore()); + + let mut no_git_required = builder(&root); + no_git_required.require_git(false); + let mut m = one_matcher(&no_git_required); + assert!(matchedf(&mut m, "outside-rule").is_ignore()); + } + + // Test that when a gitignore matcher in a parent directory is created, we + // reuse that matcher from memory even if it's changed on disk. + #[test] + fn compiled_matchers_are_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a")); + wfile(td.path().join("a/.ignore"), "*.tmp\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/first.tmp").is_ignore()); + + // Below demonstrates that this new ignore file contents + // aren't actually picked up because it was already loaded. + wfile(td.path().join("a/.ignore"), "!*.tmp\n*.rs\n"); + assert!(matchedf(&mut m, "a/first.tmp").is_ignore()); + assert!(matchedf(&mut m, "a/second.tmp").is_ignore()); + assert!(matchedf(&mut m, "a/keep.rs").is_none()); + // To get the new ignore file, we need to rebuild the matcher. + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/first.tmp").is_whitelist()); + assert!(matchedf(&mut m, "a/second.tmp").is_whitelist()); + assert!(matchedf(&mut m, "a/keep.rs").is_ignore()); + } + + #[test] + fn cached_allowed_parent_matches_children() { + let td = tmpdir(); + mkdirp(td.path().join("a/b")); + wfile(td.path().join("a/b/.ignore"), "ignored\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/b/allowed").is_none()); + assert!(matches!( + m.dirs.get(Path::new("a/b")), + Some(CachedDir::Allowed(_)) + )); + assert!(matchedf(&mut m, "a/b/ignored").is_ignore()); + } + + #[test] + fn cached_ignored_parent_matches_children() { + let td = tmpdir(); + mkdirp(td.path().join("blocked")); + wfile(td.path().join(".ignore"), "blocked/\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "blocked/first").is_ignore()); + assert!(matches!( + m.dirs.get(Path::new("blocked")), + Some(CachedDir::Ignored) + )); + assert!(matchedf(&mut m, "blocked/second").is_ignore()); + } + + #[test] + fn cached_parent_matcher_is_used_for_directory() { + let td = tmpdir(); + mkdirp(td.path().join("a/b")); + wfile(td.path().join("a/b/.ignore"), "**\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/b/file").is_ignore()); + assert!(matches!( + m.dirs.get(Path::new("a/b")), + Some(CachedDir::Allowed(_)) + )); + let dir = matchedd(&mut m, "a/b"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + // Like `compiled_matchers_are_reused`, but with multiple roots. + #[test] + fn compiled_multi_matchers_are_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c")); + + wfile(td.path().join("a/.ignore"), "*.tmp\n"); + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + assert!(matchedf(&mut mats[0], "first.tmp").is_ignore()); + assert!(matchedf(&mut mats[0], "c/first.tmp").is_ignore()); + + // Even though we haven't used the second matcher, it should still + // reuse the `a/.ignore` above. If it didn't, then `a/b/c/first.tmp` + // below would be whitelisted. + wfile(td.path().join("a/.ignore"), "!*.tmp\n"); + assert!(matchedf(&mut mats[1], "first.tmp").is_ignore()); + } + + #[test] + fn compiled_multi_child_matchers_are_not_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c/d/e")); + + wfile(td.path().join("a/b/c/d/e/.ignore"), "*.tmp\n"); + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + // Some sanity checking first. + assert!(matchedf(&mut mats[0], "c/first.tmp").is_none()); + assert!(matchedf(&mut mats[0], "c/d/first.tmp").is_none()); + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_ignore()); + + // Now write a new ignore file at the same location as above + // and check that the other matcher still uses the "stale" data. + wfile(td.path().join("a/b/c/d/e/.ignore"), "!*.tmp\n"); + assert!(matchedf(&mut mats[1], "first.tmp").is_none()); + assert!(matchedf(&mut mats[1], "d/first.tmp").is_none()); + // This is the punch line: because we didn't load `mats[1]` before + // changing the ignore file, it loads it here and thus this path gets + // whitelisted. + assert!(matchedf(&mut mats[1], "d/e/first.tmp").is_whitelist()); + // ... but `mats[0]` still uses the stale gitignore matcher cached in + // memory! + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_ignore()); + + // If we rebuilder the matcher... then we force reloading and they're + // now consistent with one another. + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_whitelist()); + assert!(matchedf(&mut mats[1], "d/e/first.tmp").is_whitelist()); + } + + // Tests that even when there is an error with a glob pattern, we still + // respect other glob patterns that are valid. + #[test] + fn partial_errors_keep_valid_rules() { + let td = tmpdir(); + let root = td.path().join("work"); + mkdirp(&root); + wfile(td.path().join(".ignore"), "{bad\n*.tmp\n"); + + let mut builder = builder(&root); + builder.git_ignore(false).git_exclude(false); + let mut m = one_matcher(&builder); + let (matched, err) = m.matched_with_errors("drop.tmp", false); + assert!(err.is_some()); + assert!(matched.is_ignore()); + } + + // Tests that parent rules are not respected if the matcher is configured + // not to do so. + #[test] + fn parent_loading_can_be_disabled() { + let td = tmpdir(); + let root = td.path().join("work"); + mkdirp(&root); + wfile(td.path().join(".ignore"), "parent-rule\n"); + wfile(root.join(".ignore"), "root-rule\n"); + + let mut b = builder(&root); + b.parents(false); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "parent-rule").is_none()); + assert!(matchedf(&mut m, "root-rule").is_ignore()); + + // Sanity check that without `parents(false)`, the parent rule is + // respected. + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "parent-rule").is_ignore()); + assert!(matchedf(&mut m, "root-rule").is_ignore()); + } + + // Tests that we can normalize a file path that isn't already in "normal" + // relative form, and then use that to match on ignore files. + #[test] + fn paths_are_relative_to_the_root() { + let td = tmpdir(); + let root = td.path().join("root"); + let outside = td.path().join("outside"); + mkdirp(&root); + mkdirp(&outside); + wfile(root.join(".ignore"), "file\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedd(&mut m, "").is_none()); + assert!(matchedd(&mut m, ".").is_none()); + assert!(matchedf(&mut m, "file").is_ignore()); + // Doesn't work because it isn't relative to the root. It's absolute. + assert!(matchedf(&mut m, root.join("file")).is_none()); + // Also doesn't work because while it's relative, it contains `..`. + assert!(matchedf(&mut m, "dir/../file").is_none()); + + let norm = m.normalize(root.join("dir/../file")).unwrap(); + assert_eq!(norm, Path::new("file")); + assert!(matchedf(&mut m, "file").is_ignore()); + + // Doesn't normalize because its an absolute path outside of our root. + assert_eq!(m.normalize(outside.join("file")), None); + } + + // Tests that two matchers in two different directories correctly interpret + // the same parent gitignore file. + #[test] + fn multiple_roots_keep_their_own_context() { + let td = tmpdir(); + let root_a = td.path().join("a"); + let root_b = td.path().join("b"); + mkdirp(td.path().join(".git")); + mkdirp(&root_a); + mkdirp(&root_b); + wfile(td.path().join(".gitignore"), "/a/*.tmp\n/b/*.log\n"); + + let mut builder = + builders([root_a.as_path(), Path::new("-"), root_b.as_path()]); + builder.git_global(false); + let mut ms = matchers(&builder); + assert_eq!(ms.len(), 3); + assert_eq!(ms[0].root(), root_a); + assert_eq!(ms[1].root(), Path::new("-")); + assert_eq!(ms[2].root(), root_b); + assert!(matchedf(&mut ms[0], "drop.tmp").is_ignore()); + assert!(matchedf(&mut ms[0], "keep.log").is_none()); + assert!(matchedf(&mut ms[1], "anything").is_none()); + assert_eq!(ms[1].normalize("anything"), None); + assert!(matchedf(&mut ms[2], "keep.tmp").is_none()); + assert!(matchedf(&mut ms[2], "drop.log").is_ignore()); + } + + // Tests that only the exact `-` root represents standard input. + #[test] + fn dot_dash_root_is_not_stdin() { + let m = one_matcher(&builder("./-")); + assert_eq!(m.root(), Path::new("./-")); + assert_eq!(m.normalize("./-/file"), Some(PathBuf::from("file"))); + } + + #[test] + fn stdin_is_inert_with_depth_limits() { + let mut b = builder("-"); + b.min_depth(Some(2)); + let mut m = one_matcher(&b); + let mat = matchedf(&mut m, "file"); + assert!(mat.is_none()); + assert!(mat.is_within_depth()); + + let mut b = builder("-"); + b.max_depth(Some(0)); + let mut m = one_matcher(&b); + let mat = matchedd(&mut m, "dir"); + assert!(mat.is_none()); + assert!(mat.is_within_depth()); + assert!(mat.should_descend()); + } + + // Tests that ignore matching works lexically, and doesn't accidentally + // resolve symbolic links. + #[cfg(unix)] + #[test] + fn symlink_path_stays_under_lexical_root() { + use std::os::unix::fs::symlink; + + let td = tmpdir(); + let root = td.path().join("root"); + let outside = td.path().join("outside"); + mkdirp(&root); + mkdirp(&outside); + wfile(root.join(".ignore"), "link\n"); + wfile(outside.join("target"), ""); + symlink(outside.join("target"), root.join("link")).unwrap(); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "link").is_ignore()); + } + + #[test] + fn depth_limits() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c/d")); + + let mut b = builder(td.path()); + b.min_depth(Some(2)).max_depth(Some(3)); + let mut m = one_matcher(&b); + + let dir = matchedd(&mut m, "a"); + assert!(!dir.is_within_depth()); + assert!(dir.should_descend()); + assert!(matchedf(&mut m, "file").is_ignore()); + + let dir = matchedd(&mut m, "a/b"); + assert!(dir.is_within_depth()); + assert!(dir.should_descend()); + assert!(!matchedf(&mut m, "a/file").is_ignore()); + + let dir = matchedd(&mut m, "a/b/c"); + assert!(dir.is_within_depth()); + assert!(!dir.should_descend()); + assert!(!matchedf(&mut m, "a/b/file").is_ignore()); + assert!(matchedf(&mut m, "a/b/c/file").is_ignore()); + + let dir = matchedd(&mut m, "a/b/c/d"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + assert!(!dir.should_descend()); + assert!(matchedf(&mut m, "a/b/c/d/file").is_ignore()); + } + + #[test] + fn depth_limits_apply_to_root() { + let td = tmpdir(); + + let mut b = builder(td.path()); + b.min_depth(Some(1)); + let mut m = one_matcher(&b); + for path in ["", "."] { + let root = matchedd(&mut m, path); + assert!(root.is_none()); + assert!(!root.is_within_depth()); + assert!(root.should_descend()); + } + + let mut b = builder(td.path()); + b.max_depth(Some(0)); + let mut m = one_matcher(&b); + for path in ["", "."] { + let root = matchedd(&mut m, path); + assert!(root.is_none()); + assert!(root.is_within_depth()); + assert!(!root.should_descend()); + } + } + + #[test] + fn max_filesize() { + let td = tmpdir(); + mkdirp(td.path().join("dir")); + wfile(td.path().join("empty"), ""); + wfile(td.path().join("at-limit"), "12345"); + wfile(td.path().join("over-limit"), "123456"); + + let mut b = builder(td.path()); + b.max_filesize(Some(5)); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "empty").is_none()); + assert!(matchedf(&mut m, "at-limit").is_none()); + assert!(matchedf(&mut m, "over-limit").is_ignore()); + + let dir = matchedd(&mut m, "dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + #[test] + fn max_filesize_does_not_stat_ignored_file() { + let td = tmpdir(); + wfile(td.path().join(".ignore"), "ignored\n"); + + let mut b = builder(td.path()); + b.max_filesize(Some(0)); + let mut m = one_matcher(&b); + let (mat, err) = m.matched_with_errors("ignored", false); + assert!(mat.is_ignore()); + assert!(err.is_none(), "ignored missing file was statted: {err:?}"); + } + + #[cfg(unix)] + #[test] + fn max_filesize_respects_follow_links() { + use std::os::unix::fs::symlink; + + let td = tmpdir(); + wfile( + td.path().join("target"), + "target contents are much longer than the size limit", + ); + symlink("target", td.path().join("link")).unwrap(); + + let mut b = builder(td.path()); + b.max_filesize(Some(10)); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "link").is_none()); + + b.follow_links(true); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "link").is_ignore()); + } + + #[test] + fn hidden_files_and_directories() { + let td = tmpdir(); + mkdirp(td.path().join(".hidden-dir")); + mkdirp(td.path().join("visible-dir")); + wfile(td.path().join(".hidden-file"), ""); + wfile(td.path().join("visible-file"), ""); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden-file").is_ignore()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_ignore()); + assert!(!dir.should_descend()); + assert!(matchedf(&mut m, "visible-file").is_none()); + let dir = matchedd(&mut m, "visible-dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + + let mut b = builder(td.path()); + b.hidden(false); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, ".hidden-file").is_none()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + #[test] + fn gitignore_whitelist_overrides_hidden_filter() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden-dir")); + wfile(td.path().join(".hidden-file"), ""); + wfile(td.path().join(".gitignore"), "!.hidden-file\n!.hidden-dir/\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden-file").is_whitelist()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_whitelist()); + assert!(dir.should_descend()); + } + + #[test] + fn descendants_of_hidden_directories_are_ignored() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden/nested")); + mkdirp(td.path().join("visible/.hidden")); + wfile(td.path().join(".hidden/file"), ""); + wfile(td.path().join(".hidden/nested/file"), ""); + wfile(td.path().join("visible/.hidden/file"), ""); + wfile(td.path().join(".hidden/.gitignore"), "!file\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden/file").is_ignore()); + assert!(matchedf(&mut m, ".hidden/nested/file").is_ignore()); + assert!(matchedf(&mut m, "visible/.hidden/file").is_ignore()); + } + + #[test] + fn whitelisted_hidden_directory_allows_descendants() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden")); + wfile(td.path().join(".hidden/file"), ""); + wfile(td.path().join(".gitignore"), "!.hidden/\n"); + + let mut m = one_matcher(&builder(td.path())); + let dir = matchedd(&mut m, ".hidden"); + assert!(dir.is_whitelist()); + assert!(dir.should_descend()); + assert!(matchedf(&mut m, ".hidden/file").is_none()); + } + + #[test] + fn hidden_directories_outside_depth_limits() { + let td = tmpdir(); + mkdirp(td.path().join("a/.hidden")); + + let mut b = builder(td.path()); + b.min_depth(Some(3)); + let mut m = one_matcher(&b); + let dir = matchedd(&mut m, "a/.hidden"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + + let mut b = builder(td.path()); + b.max_depth(Some(1)); + let mut m = one_matcher(&b); + let dir = matchedd(&mut m, "a/.hidden"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + } +} diff --git a/crates/ignore/src/lib.rs b/crates/ignore/src/lib.rs index 609004c4e..b9c9a3ece 100644 --- a/crates/ignore/src/lib.rs +++ b/crates/ignore/src/lib.rs @@ -48,13 +48,16 @@ See the documentation for `WalkBuilder` for many other options. use std::path::{Path, PathBuf}; +pub use crate::incremental::{IncrementalIgnore, IncrementalMatch}; pub use crate::walk::{ - DirEntry, ParallelVisitor, ParallelVisitorBuilder, Walk, WalkBuilder, WalkParallel, WalkState, + DirEntry, ParallelVisitor, ParallelVisitorBuilder, Walk, WalkBuilder, + WalkParallel, WalkState, }; mod default_types; mod dir; pub mod gitignore; +mod incremental; pub mod overrides; mod pathutil; pub mod types; @@ -120,34 +123,31 @@ impl Clone for Error { fn clone(&self) -> Error { match *self { Error::Partial(ref errs) => Error::Partial(errs.clone()), - Error::WithLineNumber { line, ref err } => Error::WithLineNumber { - line, - err: err.clone(), - }, - Error::WithPath { ref path, ref err } => Error::WithPath { - path: path.clone(), - err: err.clone(), - }, - Error::WithDepth { depth, ref err } => Error::WithDepth { - depth, - err: err.clone(), - }, - Error::Loop { - ref ancestor, - ref child, - } => Error::Loop { + Error::WithLineNumber { line, ref err } => { + Error::WithLineNumber { line, err: err.clone() } + } + Error::WithPath { ref path, ref err } => { + Error::WithPath { path: path.clone(), err: err.clone() } + } + Error::WithDepth { depth, ref err } => { + Error::WithDepth { depth, err: err.clone() } + } + Error::Loop { ref ancestor, ref child } => Error::Loop { ancestor: ancestor.clone(), child: child.clone(), }, Error::Io(ref err) => match err.raw_os_error() { Some(e) => Error::Io(std::io::Error::from_raw_os_error(e)), - None => Error::Io(std::io::Error::new(err.kind(), err.to_string())), + None => { + Error::Io(std::io::Error::new(err.kind(), err.to_string())) + } }, - Error::Glob { ref glob, ref err } => Error::Glob { - glob: glob.clone(), - err: err.clone(), - }, - Error::UnrecognizedFileType(ref err) => Error::UnrecognizedFileType(err.clone()), + Error::Glob { ref glob, ref err } => { + Error::Glob { glob: glob.clone(), err: err.clone() } + } + Error::UnrecognizedFileType(ref err) => { + Error::UnrecognizedFileType(err.clone()) + } Error::InvalidDefinition => Error::InvalidDefinition, } } @@ -269,19 +269,14 @@ impl Error { /// Turn an error into a tagged error with the given depth. fn with_depth(self, depth: usize) -> Error { - Error::WithDepth { - depth, - err: Box::new(self), - } + Error::WithDepth { depth, err: Box::new(self) } } /// Turn an error into a tagged error with the given file path and line /// number. If path is empty, then it is omitted from the error. fn tagged>(self, path: P, lineno: u64) -> Error { - let errline = Error::WithLineNumber { - line: lineno, - err: Box::new(self), - }; + let errline = + Error::WithLineNumber { line: lineno, err: Box::new(self) }; if path.as_ref().as_os_str().is_empty() { return errline; } @@ -301,12 +296,12 @@ impl Error { }; } let path = err.path().map(|p| p.to_path_buf()); - let mut ig_err = Error::Io(std::io::Error::from(err)); + let mut ig_err = Error::WithDepth { + depth, + err: Box::new(Error::Io(std::io::Error::from(err))), + }; if let Some(path) = path { - ig_err = Error::WithPath { - path, - err: Box::new(ig_err), - }; + ig_err = Error::WithPath { path, err: Box::new(ig_err) }; } ig_err } @@ -333,7 +328,8 @@ impl std::fmt::Display for Error { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { match *self { Error::Partial(ref errs) => { - let msgs: Vec = errs.iter().map(|err| err.to_string()).collect(); + let msgs: Vec = + errs.iter().map(|err| err.to_string()).collect(); write!(f, "{}", msgs.join("\n")) } Error::WithLineNumber { line, ref err } => { @@ -343,10 +339,7 @@ impl std::fmt::Display for Error { write!(f, "{}: {}", path.display(), err) } Error::WithDepth { ref err, .. } => err.fmt(f), - Error::Loop { - ref ancestor, - ref child, - } => write!( + Error::Loop { ref ancestor, ref child } => write!( f, "File system loop found: \ {} points to an ancestor {}", @@ -354,14 +347,8 @@ impl std::fmt::Display for Error { ancestor.display() ), Error::Io(ref err) => err.fmt(f), - Error::Glob { - glob: None, - ref err, - } => write!(f, "{}", err), - Error::Glob { - glob: Some(ref glob), - ref err, - } => { + Error::Glob { glob: None, ref err } => write!(f, "{}", err), + Error::Glob { glob: Some(ref glob), ref err } => { write!(f, "error parsing glob '{}': {}", glob, err) } Error::UnrecognizedFileType(ref ty) => { @@ -507,7 +494,8 @@ mod tests { }; /// A convenient result type alias. - pub(crate) type Result = std::result::Result>; + pub(crate) type Result = + std::result::Result>; macro_rules! err { ($($tt:tt)*) => { @@ -545,8 +533,9 @@ mod tests { if path.is_dir() { continue; } - fs::create_dir_all(&path) - .map_err(|e| err!("failed to create {}: {}", path.display(), e))?; + fs::create_dir_all(&path).map_err(|e| { + err!("failed to create {}: {}", path.display(), e) + })?; return Ok(TempDir(path)); } Err(err!("failed to create temp dir after {} tries", TRIES)) diff --git a/crates/ignore/src/overrides.rs b/crates/ignore/src/overrides.rs index afb9f16ce..005cae8f2 100644 --- a/crates/ignore/src/overrides.rs +++ b/crates/ignore/src/overrides.rs @@ -94,7 +94,11 @@ impl Override { /// given) is stripped. If there is no common suffix/prefix overlap, then /// `path` is assumed to reside in the same directory as the root path for /// this set of overrides. - pub fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { if self.is_empty() { return Match::None; } @@ -146,7 +150,10 @@ impl OverrideBuilder { /// affected. /// /// This is disabled by default. - pub fn case_insensitive(&mut self, yes: bool) -> Result<&mut OverrideBuilder, Error> { + pub fn case_insensitive( + &mut self, + yes: bool, + ) -> Result<&mut OverrideBuilder, Error> { // TODO: This should not return a `Result`. Fix this in the next semver // release. self.builder.case_insensitive(yes)?; @@ -276,11 +283,8 @@ mod tests { #[test] fn default_case_sensitive() { - let ov = OverrideBuilder::new(ROOT) - .add("*.html") - .unwrap() - .build() - .unwrap(); + let ov = + OverrideBuilder::new(ROOT).add("*.html").unwrap().build().unwrap(); assert!(ov.matched("foo.html", false).is_whitelist()); assert!(ov.matched("foo.HTML", false).is_ignore()); assert!(ov.matched("foo.htm", false).is_ignore()); diff --git a/crates/ignore/src/pathutil.rs b/crates/ignore/src/pathutil.rs index 0ceb5a356..5de8c106c 100644 --- a/crates/ignore/src/pathutil.rs +++ b/crates/ignore/src/pathutil.rs @@ -2,55 +2,89 @@ use std::{ffi::OsStr, path::Path}; use crate::walk::DirEntry; -/// Returns true if and only if this entry is considered to be hidden. +/// Returns true if and only if this path is considered to be hidden. /// -/// This only returns true if the base name of the path starts with a `.`. +/// # Platform behavior /// -/// On Unix, this implements a more optimized check. -#[cfg(unix)] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - use std::os::unix::ffi::OsStrExt; - - if let Some(name) = file_name(dent.path()) { - name.as_bytes().get(0) == Some(&b'.') - } else { - false - } -} - -/// Returns true if and only if this entry is considered to be hidden. +/// ## Windows /// -/// On Windows, this returns true if one of the following is true: +/// This returns true if one of the following is true: /// /// * The base name of the path starts with a `.`. /// * The file attributes have the `HIDDEN` property set. -#[cfg(windows)] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - use std::os::windows::fs::MetadataExt; - use winapi_util::file; - - // This looks like we're doing an extra stat call, but on Windows, the - // directory traverser reuses the metadata retrieved from each directory - // entry and stores it on the DirEntry itself. So this is "free." - if let Ok(md) = dent.metadata() { - if file::is_hidden(md.file_attributes() as u64) { - return true; - } - } - if let Some(name) = file_name(dent.path()) { - name.to_str().map(|s| s.starts_with(".")).unwrap_or(false) - } else { - false - } -} - -/// Returns true if and only if this entry is considered to be hidden. +/// +/// ## All other platforms /// /// This only returns true if the base name of the path starts with a `.`. -#[cfg(not(any(unix, windows)))] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - if let Some(name) = file_name(dent.path()) { - name.to_str().map(|s| s.starts_with(".")).unwrap_or(false) +pub(crate) fn is_hidden_path(dent: &Path) -> bool { + #[cfg(not(windows))] + fn imp(path: &Path) -> bool { + is_hidden_path_only(path) + } + + #[cfg(windows)] + fn imp(path: &Path) -> bool { + use std::os::windows::fs::MetadataExt; + use winapi_util::file; + + if let Ok(md) = path.metadata() { + if file::is_hidden(md.file_attributes() as u64) { + return true; + } + } + is_hidden_path_only(path) + } + + imp(dent) +} + +/// Returns true if and only if this directory entry is considered to be +/// hidden. +/// +/// # Platform behavior +/// +/// ## Windows +/// +/// This returns true if one of the following is true: +/// +/// * The base name of the path starts with a `.`. +/// * The file attributes have the `HIDDEN` property set. +/// +/// ## All other platforms +/// +/// This only returns true if the base name of the path starts with a `.`. +pub(crate) fn is_hidden_entry(dent: &DirEntry) -> bool { + #[cfg(not(windows))] + fn imp(dent: &DirEntry) -> bool { + is_hidden_path_only(dent.path()) + } + + #[cfg(windows)] + fn imp(dent: &DirEntry) -> bool { + use std::os::windows::fs::MetadataExt; + use winapi_util::file; + + // This looks like we're doing an extra stat call, but on Windows, the + // directory traverser reuses the metadata retrieved from each directory + // entry and stores it on the DirEntry itself. So this is "free." + if let Ok(md) = dent.metadata() { + if file::is_hidden(md.file_attributes() as u64) { + return true; + } + } + is_hidden_path_only(dent.path()) + } + + imp(dent) +} + +/// Returns true if and only if this path is considered to be hidden from only +/// the path itself. +/// +/// This has the same behavior on all platforms. +fn is_hidden_path_only(path: &Path) -> bool { + if let Some(name) = file_name(path) { + name.as_encoded_bytes().starts_with(b".") } else { false } @@ -59,83 +93,79 @@ pub(crate) fn is_hidden(dent: &DirEntry) -> bool { /// Strip `prefix` from the `path` and return the remainder. /// /// If `path` doesn't have a prefix `prefix`, then return `None`. -#[cfg(unix)] pub(crate) fn strip_prefix<'a, P: AsRef + ?Sized>( prefix: &'a P, path: &'a Path, ) -> Option<&'a Path> { - use std::os::unix::ffi::OsStrExt; + #[cfg(unix)] + fn imp<'a>(prefix: &'a Path, path: &'a Path) -> Option<&'a Path> { + use std::os::unix::ffi::OsStrExt; - let prefix = prefix.as_ref().as_os_str().as_bytes(); - let path = path.as_os_str().as_bytes(); - if prefix.len() > path.len() || prefix != &path[0..prefix.len()] { - None - } else { - Some(&Path::new(OsStr::from_bytes(&path[prefix.len()..]))) + let prefix = prefix.as_os_str().as_bytes(); + let path = path.as_os_str().as_bytes(); + if prefix.len() > path.len() || prefix != &path[0..prefix.len()] { + None + } else { + Some(&Path::new(OsStr::from_bytes(&path[prefix.len()..]))) + } } -} -/// Strip `prefix` from the `path` and return the remainder. -/// -/// If `path` doesn't have a prefix `prefix`, then return `None`. -#[cfg(not(unix))] -pub(crate) fn strip_prefix<'a, P: AsRef + ?Sized>( - prefix: &'a P, - path: &'a Path, -) -> Option<&'a Path> { - path.strip_prefix(prefix).ok() + #[cfg(not(unix))] + fn imp<'a>(prefix: &'a Path, path: &'a Path) -> Option<&'a Path> { + path.strip_prefix(prefix).ok() + } + + imp(prefix.as_ref(), path) } /// Returns true if this file path is just a file name. i.e., Its parent is /// the empty string. -#[cfg(unix)] pub(crate) fn is_file_name>(path: P) -> bool { - use std::os::unix::ffi::OsStrExt; - - use memchr::memchr; - - let path = path.as_ref().as_os_str().as_bytes(); - memchr(b'/', path).is_none() -} - -/// Returns true if this file path is just a file name. i.e., Its parent is -/// the empty string. -#[cfg(not(unix))] -pub(crate) fn is_file_name>(path: P) -> bool { - path.as_ref() - .parent() - .map(|p| p.as_os_str().is_empty()) - .unwrap_or(false) -} - -/// The final component of the path, if it is a normal file. -/// -/// If the path terminates in ., .., or consists solely of a root of prefix, -/// file_name will return None. -#[cfg(unix)] -pub(crate) fn file_name<'a, P: AsRef + ?Sized>(path: &'a P) -> Option<&'a OsStr> { - use memchr::memrchr; - use std::os::unix::ffi::OsStrExt; - - let path = path.as_ref().as_os_str().as_bytes(); - if path.is_empty() { - return None; - } else if path.len() == 1 && path[0] == b'.' { - return None; - } else if path.last() == Some(&b'.') { - return None; - } else if path.len() >= 2 && &path[path.len() - 2..] == &b".."[..] { - return None; + #[cfg(unix)] + { + memchr::memchr(b'/', path.as_ref().as_os_str().as_encoded_bytes()) + .is_none() + } + #[cfg(not(unix))] + { + path.as_ref() + .parent() + .map(|p| p.as_os_str().is_empty()) + .unwrap_or(false) } - let last_slash = memrchr(b'/', path).map(|i| i + 1).unwrap_or(0); - Some(OsStr::from_bytes(&path[last_slash..])) } /// The final component of the path, if it is a normal file. /// -/// If the path terminates in ., .., or consists solely of a root of prefix, -/// file_name will return None. -#[cfg(not(unix))] -pub(crate) fn file_name<'a, P: AsRef + ?Sized>(path: &'a P) -> Option<&'a OsStr> { - path.as_ref().file_name() +/// If the path terminates in `.`, `..`, or consists solely of a root of +/// prefix, this will return `None`. +pub(crate) fn file_name<'a, P: AsRef + ?Sized>( + path: &'a P, +) -> Option<&'a OsStr> { + #[cfg(unix)] + fn imp(path: &Path) -> Option<&OsStr> { + use std::os::unix::ffi::OsStrExt; + + use memchr::memrchr; + + let path = path.as_os_str().as_bytes(); + if path.is_empty() { + return None; + } else if path.len() == 1 && path[0] == b'.' { + return None; + } else if path.last() == Some(&b'.') { + return None; + } else if path.len() >= 2 && &path[path.len() - 2..] == &b".."[..] { + return None; + } + let last_slash = memrchr(b'/', path).map(|i| i + 1).unwrap_or(0); + Some(OsStr::from_bytes(&path[last_slash..])) + } + + #[cfg(not(unix))] + fn imp(path: &Path) -> Option<&OsStr> { + path.file_name() + } + + imp(path.as_ref()) } diff --git a/crates/ignore/src/types.rs b/crates/ignore/src/types.rs index aa23999c0..313cf5c0e 100644 --- a/crates/ignore/src/types.rs +++ b/crates/ignore/src/types.rs @@ -204,8 +204,12 @@ impl Selection { fn map U>(self, f: F) -> Selection { match self { - Selection::Select(name, inner) => Selection::Select(name, f(inner)), - Selection::Negate(name, inner) => Selection::Negate(name, f(inner)), + Selection::Select(name, inner) => { + Selection::Select(name, f(inner)) + } + Selection::Negate(name, inner) => { + Selection::Negate(name, f(inner)) + } } } @@ -227,7 +231,9 @@ impl Types { has_selected: false, glob_to_selection: vec![], set: GlobSetBuilder::new().build().unwrap(), - matches: Arc::new(Pool::new(|| vec![])), + matches: Arc::new(Pool::with_available_parallelism_capacity( + || vec![], + )), } } @@ -254,7 +260,11 @@ impl Types { /// The path is considered ignored if it matches a negated file type. /// If at least one file type is selected and `path` doesn't match, then /// the path is also considered ignored. - pub fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { // File types don't apply to directories, and we can't do anything // if our glob set is empty. if is_dir || self.set.is_empty() { @@ -306,10 +316,7 @@ impl TypesBuilder { /// of default type definitions can be added with `add_defaults`, and /// additional type definitions can be added with `select` and `negate`. pub fn new() -> TypesBuilder { - TypesBuilder { - types: HashMap::new(), - selections: vec![], - } + TypesBuilder { types: HashMap::new(), selections: vec![] } } /// Build the current set of file type definitions *and* selections into @@ -343,17 +350,18 @@ impl TypesBuilder { } selections.push(selection.clone().map(move |_| def)); } - let set = build_set.build().map_err(|err| Error::Glob { - glob: None, - err: err.to_string(), - })?; + let set = build_set + .build() + .map_err(|err| Error::Glob { glob: None, err: err.to_string() })?; Ok(Types { defs, selections, has_selected, glob_to_selection, set, - matches: Arc::new(Pool::new(|| vec![])), + matches: Arc::new(Pool::with_available_parallelism_capacity( + || vec![], + )), }) } @@ -377,12 +385,10 @@ impl TypesBuilder { pub fn select(&mut self, name: &str) -> &mut TypesBuilder { if name == "all" { for name in self.types.keys() { - self.selections - .push(Selection::Select(name.to_string(), ())); + self.selections.push(Selection::Select(name.to_string(), ())); } } else { - self.selections - .push(Selection::Select(name.to_string(), ())); + self.selections.push(Selection::Select(name.to_string(), ())); } self } @@ -393,12 +399,10 @@ impl TypesBuilder { pub fn negate(&mut self, name: &str) -> &mut TypesBuilder { if name == "all" { for name in self.types.keys() { - self.selections - .push(Selection::Negate(name.to_string(), ())); + self.selections.push(Selection::Negate(name.to_string(), ())); } } else { - self.selections - .push(Selection::Negate(name.to_string(), ())); + self.selections.push(Selection::Negate(name.to_string(), ())); } self } @@ -453,7 +457,10 @@ impl TypesBuilder { 3 => { let name = parts[0]; let types_string = parts[2]; - if name.is_empty() || parts[1] != "include" || types_string.is_empty() { + if name.is_empty() + || parts[1] != "include" + || types_string.is_empty() + { return Err(Error::InvalidDefinition); } let types = types_string.split(','); @@ -463,7 +470,8 @@ impl TypesBuilder { return Err(Error::InvalidDefinition); } for type_name in types { - let globs = self.types.get(type_name).unwrap().globs.clone(); + let globs = + self.types.get(type_name).unwrap().globs.clone(); for glob in globs { self.add(name, &glob)?; } @@ -549,30 +557,9 @@ mod tests { matched!(not, matchnot1, types(), vec!["rust"], vec![], "index.html"); matched!(not, matchnot2, types(), vec![], vec!["rust"], "main.rs"); - matched!( - not, - matchnot3, - types(), - vec!["foo"], - vec!["rust"], - "main.rs" - ); - matched!( - not, - matchnot4, - types(), - vec!["rust"], - vec!["foo"], - "main.rs" - ); - matched!( - not, - matchnot5, - types(), - vec!["rust"], - vec!["foo"], - "main.foo" - ); + matched!(not, matchnot3, types(), vec!["foo"], vec!["rust"], "main.rs"); + matched!(not, matchnot4, types(), vec!["rust"], vec!["foo"], "main.rs"); + matched!(not, matchnot5, types(), vec!["rust"], vec!["foo"], "main.foo"); matched!(not, matchnot6, types(), vec!["combo"], vec![], "leftpad.js"); matched!(not, matchnot7, types(), vec!["py"], vec![], "index.html"); matched!(not, matchnot8, types(), vec!["python"], vec![], "doc.md"); diff --git a/crates/ignore/src/walk.rs b/crates/ignore/src/walk.rs index ebf20947a..2dff78852 100644 --- a/crates/ignore/src/walk.rs +++ b/crates/ignore/src/walk.rs @@ -17,7 +17,9 @@ use { use crate::{ Error, PartialErrorBuilder, dir::{Ignore, IgnoreBuilder}, + // CHANGED: Also import `Gitignore` for `WalkBuilder::add_gitignore`. gitignore::{Gitignore, GitignoreBuilder}, + incremental::{IncrementalIgnore, IncrementalIgnoreOptions}, overrides::Override, types::Types, }; @@ -104,24 +106,15 @@ impl DirEntry { } fn new_stdin() -> DirEntry { - DirEntry { - dent: DirEntryInner::Stdin, - err: None, - } + DirEntry { dent: DirEntryInner::Stdin, err: None } } fn new_walkdir(dent: walkdir::DirEntry, err: Option) -> DirEntry { - DirEntry { - dent: DirEntryInner::Walkdir(dent), - err, - } + DirEntry { dent: DirEntryInner::Walkdir(dent), err } } fn new_raw(dent: DirEntryRaw, err: Option) -> DirEntry { - DirEntry { - dent: DirEntryInner::Raw(dent), - err, - } + DirEntry { dent: DirEntryInner::Raw(dent), err } } } @@ -187,9 +180,11 @@ impl DirEntryInner { )); Err(err.with_path("")) } - Walkdir(ref x) => x - .metadata() - .map_err(|err| Error::Io(io::Error::from(err)).with_path(x.path())), + Walkdir(ref x) => x.metadata().map_err(|err| { + Error::Io(io::Error::from(err)) + .with_depth(x.depth()) + .with_path(x.path()) + }), Raw(ref x) => x.metadata(), } } @@ -308,7 +303,9 @@ impl DirEntryRaw { } else { fs::symlink_metadata(&self.path) } - .map_err(|err| Error::Io(io::Error::from(err)).with_path(&self.path)) + .map_err(|err| { + Error::Io(err).with_depth(self.depth).with_path(&self.path) + }) } fn file_type(&self) -> FileType { @@ -316,9 +313,7 @@ impl DirEntryRaw { } fn file_name(&self) -> &OsStr { - self.path - .file_name() - .unwrap_or_else(|| self.path.as_os_str()) + self.path.file_name().unwrap_or_else(|| self.path.as_os_str()) } fn depth(&self) -> usize { @@ -330,13 +325,13 @@ impl DirEntryRaw { self.ino } - fn from_entry(depth: usize, ent: &fs::DirEntry) -> Result { + fn from_entry( + depth: usize, + ent: &fs::DirEntry, + ) -> Result { let ty = ent.file_type().map_err(|err| { - let err = Error::Io(io::Error::from(err)).with_path(ent.path()); - Error::WithDepth { - depth, - err: Box::new(err), - } + let err = Error::Io(err).with_depth(depth).with_path(ent.path()); + Error::WithDepth { depth, err: Box::new(err) } })?; DirEntryRaw::from_entry_os(depth, ent, ty) } @@ -348,11 +343,8 @@ impl DirEntryRaw { ty: fs::FileType, ) -> Result { let md = ent.metadata().map_err(|err| { - let err = Error::Io(io::Error::from(err)).with_path(ent.path()); - Error::WithDepth { - depth, - err: Box::new(err), - } + let err = Error::Io(err).with_depth(depth).with_path(ent.path()); + Error::WithDepth { depth, err: Box::new(err) } })?; Ok(DirEntryRaw { path: ent.path(), @@ -395,8 +387,13 @@ impl DirEntryRaw { } #[cfg(windows)] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { - let md = fs::metadata(&pb).map_err(|err| Error::Io(err).with_path(&pb))?; + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { + let md = fs::metadata(&pb) + .map_err(|err| Error::Io(err).with_depth(depth).with_path(&pb))?; Ok(DirEntryRaw { path: pb, ty: md.file_type(), @@ -407,10 +404,15 @@ impl DirEntryRaw { } #[cfg(unix)] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { use std::os::unix::fs::MetadataExt; - let md = fs::metadata(&pb).map_err(|err| Error::Io(err).with_path(&pb))?; + let md = fs::metadata(&pb) + .map_err(|err| Error::Io(err).with_depth(depth).with_path(&pb))?; Ok(DirEntryRaw { path: pb, ty: md.file_type(), @@ -423,7 +425,11 @@ impl DirEntryRaw { // Placeholder implementation to allow compiling on non-standard platforms // (e.g. wasm32). #[cfg(not(any(windows, unix)))] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { Err(Error::Io(io::Error::new( io::ErrorKind::Other, "unsupported platform", @@ -502,7 +508,8 @@ pub struct WalkBuilder { /// /// When `None`, the CWD is fetched from `std::env::current_dir()`. If /// that fails, then global gitignores are ignored (an error is logged). - global_gitignores_relative_to: OnceLock>>, + global_gitignores_relative_to: + OnceLock>>, } #[derive(Clone)] @@ -544,8 +551,16 @@ impl WalkBuilder { /// is better to call `add` on this builder than to create multiple /// `Walk` values. pub fn new>(path: P) -> WalkBuilder { + WalkBuilder::from_iter([path]) + } + + /// Create an empty builder to which paths can be added. + /// + /// Note that if you call `build` on this instance before calling `add` + /// on it, it will return exactly zero items during iteration. + pub fn empty() -> WalkBuilder { WalkBuilder { - paths: vec![path.as_ref().to_path_buf()], + paths: vec![], ig_builder: IgnoreBuilder::new(), max_depth: None, min_depth: None, @@ -560,6 +575,21 @@ impl WalkBuilder { } } + /// Create a new builder for a recursive directory iterator from the + /// sequence of paths. + /// + /// Note that if the iterator is empty, this is the same as + /// `WalkBuilder::empty`. + pub fn from_iter, I: IntoIterator>( + paths: I, + ) -> WalkBuilder { + let mut builder = WalkBuilder::empty(); + for path in paths.into_iter() { + builder.add(path); + } + builder + } + /// Build a new `Walk` iterator. pub fn build(&self) -> Walk { let follow_links = self.follow_links; @@ -585,10 +615,14 @@ impl WalkBuilder { if let Some(ref sorter) = sorter { match sorter.clone() { Sorter::ByName(cmp) => { - wd = wd.sort_by(move |a, b| cmp(a.file_name(), b.file_name())); + wd = wd.sort_by(move |a, b| { + cmp(a.file_name(), b.file_name()) + }); } Sorter::ByPath(cmp) => { - wd = wd.sort_by(move |a, b| cmp(a.path(), b.path())); + wd = wd.sort_by(move |a, b| { + cmp(a.path(), b.path()) + }); } } } @@ -597,31 +631,76 @@ impl WalkBuilder { }) .collect::>() .into_iter(); - let ig_root = self - .get_or_set_current_dir() - .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) - .unwrap_or_else(|| self.ig_builder.build()); + let ig_root = self.build_ignore(); Walk { its, it: None, ig_root: ig_root.clone(), ig: ig_root.clone(), + max_depth: self.max_depth, max_filesize: self.max_filesize, skip: self.skip.clone(), filter: self.filter.clone(), } } + /// Build matchers for checking paths against ignore files without + /// recursively walking the configured roots. + /// + /// The returned matchers use the path-based filtering configuration + /// on this builder, including glob overrides, file type selections, + /// parent ignore files, `.ignore`, `.gitignore`, global Git + /// ignore files, explicitly added ignore files and custom ignore + /// file names. For example, ripgrep configures `.rgignore` via + /// [`WalkBuilder::add_custom_ignore_filename`]. Minimum and maximum depth + /// limits, maximum file size and hidden-file filtering are also applied. + /// Other options that only control traversal or require a directory entry, + /// such as custom entry predicates, are not applied. + /// + /// One matcher is returned for each configured path, in the same order as + /// the paths were added to this builder. Each matcher accepts paths + /// relative to its own [`IncrementalIgnore::root`]. The matcher for the + /// special `-` path representing standard input always returns a non-match + /// for all inputs. + /// + /// Ignore matchers are loaded lazily and cached by directory. + /// Thus, the first query may read ignore files from the root and + /// its parents, while later queries reuse the compiled matchers. + /// Errors encountered while loading ignore files are returned by + /// [`IncrementalIgnore::matched_with_errors`]. Once an ignore file has + /// been loaded, changes to it are not observed. Build new matchers to + /// reload changed ignore files. + /// + /// Matchers built together share the builder's base ignore configuration + /// and compiled parent matchers. + pub fn build_matchers(&self) -> Vec { + let ignore = self.build_ignore(); + let options = IncrementalIgnoreOptions { + min_depth: self.min_depth, + max_depth: self.max_depth, + max_filesize: self.max_filesize, + hidden: self.ig_builder.is_hidden(), + follow_links: self.follow_links, + }; + self.paths + .iter() + .map(move |path| { + IncrementalIgnore::new( + path.clone(), + ignore.clone(), + options.clone(), + ) + }) + .collect() + } + /// Build a new `WalkParallel` iterator. /// /// Note that this *doesn't* return something that implements `Iterator`. /// Instead, the returned value must be run with a closure. e.g., /// `builder.build_parallel().run(|| |path| { println!("{path:?}"); WalkState::Continue })`. pub fn build_parallel(&self) -> WalkParallel { - let ig_root = self - .get_or_set_current_dir() - .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) - .unwrap_or_else(|| self.ig_builder.build()); + let ig_root = self.build_ignore(); WalkParallel { paths: self.paths.clone().into_iter(), ig_root, @@ -651,7 +730,10 @@ impl WalkBuilder { /// The default, `None`, imposes no depth restriction. pub fn max_depth(&mut self, depth: Option) -> &mut WalkBuilder { self.max_depth = depth; - if self.min_depth.is_some() && self.max_depth.is_some() && self.max_depth < self.min_depth { + if self.min_depth.is_some() + && self.max_depth.is_some() + && self.max_depth < self.min_depth + { self.max_depth = self.min_depth; } self @@ -662,7 +744,10 @@ impl WalkBuilder { /// The default, `None`, imposes no minimum depth restriction. pub fn min_depth(&mut self, depth: Option) -> &mut WalkBuilder { self.min_depth = depth; - if self.max_depth.is_some() && self.min_depth.is_some() && self.min_depth > self.max_depth { + if self.max_depth.is_some() + && self.min_depth.is_some() + && self.min_depth > self.max_depth + { self.min_depth = self.max_depth; } self @@ -705,7 +790,12 @@ impl WalkBuilder { /// An error will also occur if this walker could not get the current /// working directory (and `WalkBuilder::current_dir` isn't set). pub fn add_ignore>(&mut self, path: P) -> Option { - // CHANGED: Dropped this code + // CHANGED: Root the ignore file at `""` instead of the current working + // directory. Explicit ignores are scoped to the directory of the + // ignore file (see `matched_ignore`), and a root of `""` makes the + // rules apply to every walked path regardless of the walk root. This + // also avoids depending on the current working directory entirely. + // // let path = path.as_ref(); // let Some(cwd) = self.get_or_set_current_dir() else { // let err = std::io::Error::other(format!( @@ -729,7 +819,11 @@ impl WalkBuilder { errs.into_error_option() } - /// CHANGED: Add a Gitignore to the builder. + /// CHANGED: Add a prebuilt Gitignore to the builder. + /// + /// Like the ignore file added via `add_ignore`, these rules are matched + /// against the full path of each walked entry, scoped to the `Gitignore`'s + /// root path. pub fn add_gitignore(&mut self, gi: Gitignore) { self.ig_builder.add_ignore(gi); } @@ -982,7 +1076,10 @@ impl WalkBuilder { /// /// Global gitignore files come from things like a user's git configuration /// or from gitignore files added via [`WalkBuilder::add_ignore`]. - pub fn current_dir(&mut self, cwd: impl Into) -> &mut WalkBuilder { + pub fn current_dir( + &mut self, + cwd: impl Into, + ) -> &mut WalkBuilder { let cwd = cwd.into(); self.ig_builder.current_dir(cwd.clone()); if let Err(cwd) = self.global_gitignores_relative_to.set(Ok(cwd)) { @@ -1002,7 +1099,10 @@ impl WalkBuilder { let result = std::env::current_dir().map_err(Arc::new); match result { Ok(ref path) => { - log::trace!("automatically discovered CWD: {}", path.display()); + log::trace!( + "automatically discovered CWD: {}", + path.display() + ); } Err(ref err) => { log::debug!( @@ -1016,6 +1116,13 @@ impl WalkBuilder { }); result.as_ref().ok().map(|path| &**path) } + + /// Build the root ignore matcher shared by all consumers of this builder. + fn build_ignore(&self) -> Ignore { + self.get_or_set_current_dir() + .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) + .unwrap_or_else(|| self.ig_builder.build()) + } } /// Walk is a recursive directory iterator over file paths in one or more @@ -1029,6 +1136,7 @@ pub struct Walk { it: Option, ig_root: Ignore, ig: Ignore, + max_depth: Option, max_filesize: Option, skip: Option>, filter: Option, @@ -1044,6 +1152,17 @@ impl Walk { WalkBuilder::new(path).build() } + /// Create a new recursive directory iterator from the sequence of paths + /// given. + /// + /// Note that if the provided iterator is empty, then `Walk` is guaranteed + /// to yield zero entries. + pub fn from_iter, I: IntoIterator>( + paths: I, + ) -> Walk { + WalkBuilder::from_iter(paths).build() + } + fn skip_entry(&self, ent: &DirEntry) -> Result { if ent.depth() == 0 { return Ok(false); @@ -1128,12 +1247,17 @@ impl Iterator for Walk { self.it.as_mut().unwrap().it.skip_current_dir(); // Still need to push this on the stack because // we'll get a WalkEvent::Exit event for this dir. - // We don't care if it errors though. - let (igtmp, _) = self.ig.add_child(ent.path()); + // Its ignore files cannot apply to any visited entry. + let (igtmp, _) = + self.ig.add_child_with_entries(ent.path(), &[]); self.ig = igtmp; continue; } - let (igtmp, err) = self.ig.add_child(ent.path()); + let (igtmp, err) = if self.max_depth == Some(ent.depth()) { + self.ig.add_child_with_entries(ent.path(), &[]) + } else { + self.ig.add_child(ent.path()) + }; self.ig = igtmp; ent.err = err; return Some(Ok(ent)); @@ -1175,11 +1299,7 @@ enum WalkEvent { impl From for WalkEventIter { fn from(it: WalkDir) -> WalkEventIter { - WalkEventIter { - depth: 0, - it: it.into_iter(), - next: None, - } + WalkEventIter { depth: 0, it: it.into_iter(), next: None } } } @@ -1252,7 +1372,9 @@ pub trait ParallelVisitorBuilder<'s> { fn build(&mut self) -> Box; } -impl<'a, 's, P: ParallelVisitorBuilder<'s>> ParallelVisitorBuilder<'s> for &'a mut P { +impl<'a, 's, P: ParallelVisitorBuilder<'s>> ParallelVisitorBuilder<'s> + for &'a mut P +{ fn build(&mut self) -> Box { (**self).build() } @@ -1273,14 +1395,17 @@ struct FnBuilder { builder: F, } -impl<'s, F: FnMut() -> FnVisitor<'s>> ParallelVisitorBuilder<'s> for FnBuilder { +impl<'s, F: FnMut() -> FnVisitor<'s>> ParallelVisitorBuilder<'s> + for FnBuilder +{ fn build(&mut self) -> Box { let visitor = (self.builder)(); Box::new(FnVisitorImp { visitor }) } } -type FnVisitor<'s> = Box) -> WalkState + Send + 's>; +type FnVisitor<'s> = + Box) -> WalkState + Send + 's>; struct FnVisitorImp<'s> { visitor: FnVisitor<'s>, @@ -1370,7 +1495,9 @@ impl WalkParallel { } }; match DirEntryRaw::from_path(0, path, false) { - Ok(dent) => (DirEntry::new_raw(dent, None), root_device), + Ok(dent) => { + (DirEntry::new_raw(dent, None), root_device) + } Err(err) => { if visitor.visit(Err(err)).is_quit() { return; @@ -1394,21 +1521,28 @@ impl WalkParallel { let quit_now = Arc::new(AtomicBool::new(false)); let active_workers = Arc::new(AtomicUsize::new(threads)); let stacks = Stack::new_for_each_thread(threads, stack); + // Collect all of the workers first. In the case that + // `builder.build()` panics, we want that to happen and + // propagate before we actually start to run any of the + // workers. + let workers: Vec<_> = stacks + .into_iter() + .map(|stack| Worker { + visitor: builder.build(), + stack, + quit_now: quit_now.clone(), + active_workers: active_workers.clone(), + max_depth: self.max_depth, + min_depth: self.min_depth, + max_filesize: self.max_filesize, + follow_links: self.follow_links, + skip: self.skip.clone(), + filter: self.filter.clone(), + }) + .collect(); std::thread::scope(|s| { - let handles: Vec<_> = stacks + let handles: Vec<_> = workers .into_iter() - .map(|stack| Worker { - visitor: builder.build(), - stack, - quit_now: quit_now.clone(), - active_workers: active_workers.clone(), - max_depth: self.max_depth, - min_depth: self.min_depth, - max_filesize: self.max_filesize, - follow_links: self.follow_links, - skip: self.skip.clone(), - filter: self.filter.clone(), - }) .map(|worker| s.spawn(|| worker.run())) .collect(); for handle in handles { @@ -1419,9 +1553,7 @@ impl WalkParallel { fn threads(&self) -> usize { if self.threads == 0 { - std::thread::available_parallelism() - .map_or(1, |n| n.get()) - .min(12) + std::thread::available_parallelism().map_or(1, |n| n.get()).min(12) } else { self.threads } @@ -1452,6 +1584,12 @@ struct Work { root_device: Option, } +#[derive(Default)] +struct ReadDirResult { + entries: Vec, + errors: Vec, +} + impl Work { /// Returns true if and only if this work item is a directory. fn is_dir(&self) -> bool { @@ -1478,6 +1616,13 @@ impl Work { err } + /// Adds ignore rules for this directory without reading its contents. + fn add_ignore(&mut self) { + let (ig, err) = self.ignore.add_child(self.dent.path()); + self.ignore = ig; + self.dent.err = err; + } + /// Reads the directory contents of this work item and adds ignore /// rules for this directory. /// @@ -1485,7 +1630,7 @@ impl Work { /// an error is returned. If there was a problem reading the ignore /// rules for this directory, then the error is attached to this /// work item's directory entry. - fn read_dir(&mut self) -> Result { + fn read_dir(&mut self) -> Result { let readdir = match fs::read_dir(self.dent.path()) { Ok(readdir) => readdir, Err(err) => { @@ -1495,10 +1640,24 @@ impl Work { return Err(err); } }; - let (ig, err) = self.ignore.add_child(self.dent.path()); + // Actually descend into the directory and read its contents + let mut result = ReadDirResult::default(); + for entry in readdir { + match entry { + Ok(entry) => result.entries.push(entry), + Err(err) => result.errors.push( + Error::from(err) + .with_path(self.dent.path()) + .with_depth(self.dent.depth() + 1), + ), + } + } + let (ig, err) = self + .ignore + .add_child_with_entries(self.dent.path(), &result.entries); self.ignore = ig; self.dent.err = err; - Ok(readdir) + Ok(result) } } @@ -1522,11 +1681,11 @@ impl Stack { // breadth-first. We do depth-first because a breadth first traversal // on wide directories with a lot of gitignores is disastrous (for // example, searching a directory tree containing all of crates.io). - let deques: Vec> = std::iter::repeat_with(Deque::new_lifo) - .take(threads) - .collect(); - let stealers = - Arc::<[Stealer]>::from(deques.iter().map(Deque::stealer).collect::>()); + let deques: Vec> = + std::iter::repeat_with(Deque::new_lifo).take(threads).collect(); + let stealers = Arc::<[Stealer]>::from( + deques.iter().map(Deque::stealer).collect::>(), + ); let stacks: Vec = deques .into_iter() .enumerate() @@ -1668,8 +1827,13 @@ impl<'s> Worker<'s> { // have sufficient read permissions to list the directory. // In that case we still want to provide the closure with a valid // entry before passing the error value. - let readdir = work.read_dir(); let depth = work.dent.depth(); + let readdir = if descend && self.max_depth.is_none_or(|m| depth < m) { + Some(work.read_dir()) + } else { + work.add_ignore(); + None + }; if should_visit { let state = self.visitor.visit(Ok(work.dent)); if !state.is_continue() { @@ -1680,6 +1844,10 @@ impl<'s> Worker<'s> { return WalkState::Skip; } + let readdir = match readdir { + Some(readdir) => readdir, + None => return WalkState::Skip, + }; let readdir = match readdir { Ok(readdir) => readdir, Err(err) => { @@ -1687,11 +1855,19 @@ impl<'s> Worker<'s> { } }; - if self.max_depth.map_or(false, |max| depth >= max) { - return WalkState::Skip; + for result in readdir.entries { + let state = self.generate_work( + &work.ignore, + depth + 1, + work.root_device, + result, + ); + if state.is_quit() { + return state; + } } - for result in readdir { - let state = self.generate_work(&work.ignore, depth + 1, work.root_device, result); + for err in readdir.errors { + let state = self.visitor.visit(Err(err)); if state.is_quit() { return state; } @@ -1717,14 +1893,8 @@ impl<'s> Worker<'s> { ig: &Ignore, depth: usize, root_device: Option, - result: Result, + fs_dent: fs::DirEntry, ) -> WalkState { - let fs_dent = match result { - Ok(fs_dent) => fs_dent, - Err(err) => { - return self.visitor.visit(Err(Error::from(err).with_depth(depth))); - } - }; let mut dent = match DirEntryRaw::from_entry(depth, &fs_dent) { Ok(dent) => DirEntry::new_raw(dent, None), Err(err) => { @@ -1760,26 +1930,24 @@ impl<'s> Worker<'s> { return WalkState::Continue; } } - let should_skip_filesize = if self.max_filesize.is_some() && !dent.is_dir() { - skip_filesize( - self.max_filesize.unwrap(), - dent.path(), - &dent.metadata().ok(), - ) - } else { - false - }; - let should_skip_filtered = if let Some(Filter(predicate)) = &self.filter { - !predicate(&dent) - } else { - false - }; + let should_skip_filesize = + if self.max_filesize.is_some() && !dent.is_dir() { + skip_filesize( + self.max_filesize.unwrap(), + dent.path(), + &dent.metadata().ok(), + ) + } else { + false + }; + let should_skip_filtered = + if let Some(Filter(predicate)) = &self.filter { + !predicate(&dent) + } else { + false + }; if !should_skip_filesize && !should_skip_filtered { - self.send(Work { - dent, - ignore: ig.clone(), - root_device, - }); + self.send(Work { dent, ignore: ig.clone(), root_device }); } WalkState::Continue } @@ -1820,6 +1988,9 @@ impl<'s> Worker<'s> { } // Wait for next `Work` or `Quit` message. loop { + if self.is_quit_now() { + return None; + } if let Some(v) = self.recv() { self.activate_worker(); value = Some(v); @@ -1873,24 +2044,25 @@ impl<'s> Worker<'s> { } } +impl<'s> Drop for Worker<'s> { + fn drop(&mut self) { + if std::thread::panicking() { + self.quit_now(); + } + } +} + fn check_symlink_loop( ig_parent: &Ignore, child_path: &Path, child_depth: usize, ) -> Result<(), Error> { let hchild = Handle::from_path(child_path).map_err(|err| { - Error::from(err) - .with_path(child_path) - .with_depth(child_depth) + Error::from(err).with_path(child_path).with_depth(child_depth) })?; - for ig in ig_parent - .parents() - .take_while(|ig| !ig.is_absolute_parent()) - { + for ig in ig_parent.parents().take_while(|ig| !ig.is_absolute_parent()) { let h = Handle::from_path(ig.path()).map_err(|err| { - Error::from(err) - .with_path(child_path) - .with_depth(child_depth) + Error::from(err).with_path(child_path).with_depth(child_depth) })?; if hchild == h { return Err(Error::Loop { @@ -1905,7 +2077,11 @@ fn check_symlink_loop( // Before calling this function, make sure that you ensure that is really // necessary as the arguments imply a file stat. -fn skip_filesize(max_filesize: u64, path: &Path, ent: &Option) -> bool { +fn skip_filesize( + max_filesize: u64, + path: &Path, + ent: &Option, +) -> bool { let filesize = match *ent { Some(ref md) => Some(md.len()), None => None, @@ -1977,9 +2153,9 @@ fn path_equals(dent: &DirEntry, handle: &Handle) -> Result { if dent.is_stdin() || never_equal(dent, handle) { return Ok(false); } - Handle::from_path(dent.path()) - .map(|h| &h == handle) - .map_err(|err| Error::Io(err).with_path(dent.path())) + Handle::from_path(dent.path()).map(|h| &h == handle).map_err(|err| { + Error::Io(err).with_depth(dent.depth()).with_path(dent.path()) + }) } /// Returns true if the given walkdir entry corresponds to a directory. @@ -1997,16 +2173,14 @@ fn walkdir_is_dir(dent: &walkdir::DirEntry) -> bool { if !dent.file_type().is_symlink() || dent.depth() > 0 { return false; } - dent.path() - .metadata() - .ok() - .map_or(false, |md| md.file_type().is_dir()) + dent.path().metadata().ok().map_or(false, |md| md.file_type().is_dir()) } /// Returns true if and only if the given path is on the same device as the /// given root device. fn is_same_file_system(root_device: u64, path: &Path) -> Result { - let dent_device = device_num(path).map_err(|err| Error::Io(err).with_path(path))?; + let dent_device = + device_num(path).map_err(|err| Error::Io(err).with_path(path))?; Ok(root_device == dent_device) } @@ -2065,11 +2239,7 @@ mod tests { } fn normal_path(unix: &str) -> String { - if cfg!(windows) { - unix.replace("\\", "/") - } else { - unix.to_string() - } + if cfg!(windows) { unix.replace("\\", "/") } else { unix.to_string() } } fn walk_collect(prefix: &Path, builder: &WalkBuilder) -> Vec { @@ -2089,7 +2259,10 @@ mod tests { paths } - fn walk_collect_parallel(prefix: &Path, builder: &WalkBuilder) -> Vec { + fn walk_collect_parallel( + prefix: &Path, + builder: &WalkBuilder, + ) -> Vec { let mut paths = vec![]; for dent in walk_collect_entries_parallel(builder) { let path = dent.path().strip_prefix(prefix).unwrap(); @@ -2278,6 +2451,27 @@ mod tests { ); } + #[test] + fn max_depth_does_not_load_unreachable_ignore_files() { + let td = tmpdir(); + let leaf = td.path().join("leaf"); + mkdirp(&leaf); + wfile(leaf.join(".ignore"), "{invalid\n"); + + let mut builder = WalkBuilder::new(td.path()); + builder.max_depth(Some(1)); + let entry = builder + .build() + .find_map(|result| { + let entry = result.unwrap(); + (entry.path() == leaf).then_some(entry) + }) + .unwrap(); + + assert!(entry.error().is_none()); + assert_paths(td.path(), &builder, &["leaf"]); + } + #[test] fn min_depth() { let td = tmpdir(); @@ -2388,7 +2582,9 @@ mod tests { assert_eq!(1, dents.len()); assert!(!dents[0].path_is_symlink()); - let dents = walk_collect_entries_parallel(&WalkBuilder::new(td.path().join("foo"))); + let dents = walk_collect_entries_parallel(&WalkBuilder::new( + td.path().join("foo"), + )); assert_eq!(1, dents.len()); assert!(!dents[0].path_is_symlink()); } @@ -2474,8 +2670,88 @@ mod tests { assert_paths( td.path(), - &WalkBuilder::new(td.path()).filter_entry(|entry| entry.file_name() != OsStr::new("a")), + &WalkBuilder::new(td.path()) + .filter_entry(|entry| entry.file_name() != OsStr::new("a")), &["x", "x/y", "x/y/foo"], ); } + + #[test] + fn empty() { + let td = tmpdir(); + assert_paths(td.path(), &WalkBuilder::empty(), &[]); + + let empty_paths: Vec<&OsStr> = Vec::new(); + assert_paths(td.path(), &WalkBuilder::from_iter(empty_paths), &[]); + } + + #[test] + fn from_iter() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c")); + mkdirp(td.path().join("d/e/f")); + mkdirp(td.path().join("x/y")); + wfile(td.path().join("a/b/foo"), ""); + wfile(td.path().join("d/e/f/foo"), ""); + wfile(td.path().join("x/y/foo"), ""); + + let paths = vec![ + td.path().join("a"), + td.path().join("d"), + td.path().join("x"), + ]; + + assert_paths( + td.path(), + &WalkBuilder::from_iter(paths), + &[ + "x", + "x/y", + "x/y/foo", + "d", + "d/e", + "d/e/f", + "d/e/f/foo", + "a", + "a/b", + "a/b/foo", + "a/b/c", + ], + ); + } + + // This should always panic and never hang. + // + // Ref: https://github.com/BurntSushi/ripgrep/issues/3009 + #[test] + #[should_panic] + fn panic_in_parallel() { + let td = tmpdir(); + wfile(td.path().join("foo.txt"), ""); + + WalkBuilder::new(td.path()) + .threads(40) + .build_parallel() + .run(|| Box::new(|_| panic!("oops!"))); + } + + // This should always panic and never hang. The first call to the visitor + // builder is used while processing the root paths. Previously, a panic on + // the third call occurred after the first worker had already been spawned, + // leaving it waiting indefinitely for workers that were never created. + #[test] + #[should_panic(expected = "builder panic")] + fn panic_in_parallel_builder() { + let td = tmpdir(); + wfile(td.path().join("foo.txt"), ""); + + let mut builds = 0; + WalkBuilder::new(td.path()).threads(2).build_parallel().run(|| { + builds += 1; + if builds == 3 { + panic!("builder panic"); + } + Box::new(|_| WalkState::Continue) + }); + } } diff --git a/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs b/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs index b7b7c6f95..ecb7b47e3 100644 --- a/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs +++ b/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs @@ -2,7 +2,8 @@ use std::path::Path; use ignore::gitignore::{Gitignore, GitignoreBuilder}; -const IGNORE_FILE: &'static str = "tests/gitignore_matched_path_or_any_parents_tests.gitignore"; +const IGNORE_FILE: &'static str = + "tests/gitignore_matched_path_or_any_parents_tests.gitignore"; fn get_gitignore() -> Gitignore { let mut builder = GitignoreBuilder::new("ROOT"); @@ -23,7 +24,9 @@ fn test_path_should_be_under_root() { #[test] fn test_files_in_root() { let gitignore = get_gitignore(); - let m = |path: &str| gitignore.matched_path_or_any_parents(Path::new(path), false); + let m = |path: &str| { + gitignore.matched_path_or_any_parents(Path::new(path), false) + }; // 0x assert!(m("ROOT/file_root_00").is_ignore()); @@ -53,7 +56,9 @@ fn test_files_in_root() { #[test] fn test_files_in_deep() { let gitignore = get_gitignore(); - let m = |path: &str| gitignore.matched_path_or_any_parents(Path::new(path), false); + let m = |path: &str| { + gitignore.matched_path_or_any_parents(Path::new(path), false) + }; // 0x assert!(m("ROOT/parent_dir/file_deep_00").is_ignore()); @@ -83,8 +88,9 @@ fn test_files_in_deep() { #[test] fn test_dirs_in_root() { let gitignore = get_gitignore(); - let m = - |path: &str, is_dir: bool| gitignore.matched_path_or_any_parents(Path::new(path), is_dir); + let m = |path: &str, is_dir: bool| { + gitignore.matched_path_or_any_parents(Path::new(path), is_dir) + }; // 00 assert!(m("ROOT/dir_root_00", true).is_ignore()); @@ -186,20 +192,25 @@ fn test_dirs_in_root() { #[test] fn test_dirs_in_deep() { let gitignore = get_gitignore(); - let m = - |path: &str, is_dir: bool| gitignore.matched_path_or_any_parents(Path::new(path), is_dir); + let m = |path: &str, is_dir: bool| { + gitignore.matched_path_or_any_parents(Path::new(path), is_dir) + }; // 00 assert!(m("ROOT/parent_dir/dir_deep_00", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_00/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_00/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_00/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_00/child_dir/file", false).is_ignore() + ); // 01 assert!(m("ROOT/parent_dir/dir_deep_01", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_01/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_01/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_01/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_01/child_dir/file", false).is_ignore() + ); // 02 assert!(m("ROOT/parent_dir/dir_deep_02", true).is_none()); @@ -241,51 +252,67 @@ fn test_dirs_in_deep() { assert!(m("ROOT/parent_dir/dir_deep_20", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_20/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_20/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_20/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_20/child_dir/file", false).is_ignore() + ); // 21 assert!(m("ROOT/parent_dir/dir_deep_21", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_21/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_21/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_21/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_21/child_dir/file", false).is_ignore() + ); // 22 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_22", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_22/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_22/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_22/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_22/child_dir/file", false).is_ignore() + ); // 23 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_23", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_23/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_23/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_23/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_23/child_dir/file", false).is_ignore() + ); // 30 assert!(m("ROOT/parent_dir/dir_deep_30", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_30/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_30/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_30/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_30/child_dir/file", false).is_ignore() + ); // 31 assert!(m("ROOT/parent_dir/dir_deep_31", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_31/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_31/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_31/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_31/child_dir/file", false).is_ignore() + ); // 32 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_32", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_32/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_32/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_32/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_32/child_dir/file", false).is_ignore() + ); // 33 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_33", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_33/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_33/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_33/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_33/child_dir/file", false).is_ignore() + ); } From 44765c4167e52189330f3efa623a9a146c222dfb Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Thu, 13 Aug 2026 01:44:59 +0200 Subject: [PATCH 02/23] visualize scanned files --- Cargo.lock | 49 +- crates/oxide/Cargo.toml | 1 + crates/oxide/tests/scanner.rs | 1282 ++++++++++++++++++++++++++++++++- 3 files changed, 1306 insertions(+), 26 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 8a10094e3..73fd9e995 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -59,6 +59,17 @@ dependencies = [ "syn", ] +[[package]] +name = "console" +version = "0.16.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4fe5f465a4f6fee88fad41b85d990f84c835335e85b5d9e6e63e0d06d28cba7c" +dependencies = [ + "encode_unicode", + "libc", + "windows-sys 0.61.2", +] + [[package]] name = "convert_case" version = "0.11.0" @@ -126,6 +137,12 @@ version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7fcaabb2fef8c910e7f4c7ce9f67a1283a1715879a7c230ca9d6d1ae31f16d91" +[[package]] +name = "encode_unicode" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" + [[package]] name = "errno" version = "0.3.9" @@ -296,6 +313,18 @@ dependencies = [ "winapi-util", ] +[[package]] +name = "insta" +version = "1.48.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "86f0f8fee8c926415c58d6ae43a08523a26faccb2323f5e6b644fe7dd4ef6b82" +dependencies = [ + "console", + "once_cell", + "similar", + "tempfile", +] + [[package]] name = "itertools" version = "0.11.0" @@ -445,9 +474,9 @@ dependencies = [ [[package]] name = "once_cell" -version = "1.19.0" +version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fdb12b2476b595f9358c5161aa467c2438859caa136dec86c26fdd2efe17b92" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "overload" @@ -602,6 +631,12 @@ dependencies = [ "lazy_static", ] +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + [[package]] name = "slab" version = "0.4.12" @@ -647,6 +682,7 @@ dependencies = [ "fast-glob", "globwalk", "ignore 0.4.33", + "insta", "log", "pretty_assertions", "rayon", @@ -832,6 +868,15 @@ dependencies = [ "windows-targets", ] +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + [[package]] name = "windows-targets" version = "0.52.6" diff --git a/crates/oxide/Cargo.toml b/crates/oxide/Cargo.toml index 0965320c9..f8a5f6137 100644 --- a/crates/oxide/Cargo.toml +++ b/crates/oxide/Cargo.toml @@ -20,6 +20,7 @@ ignore = { path = "../ignore" } regex = "1.11.1" [dev-dependencies] +insta = "1.48.0" tempfile = "3.13.0" pretty_assertions = "1.4.1" unicode-width = "0.2.0" diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 10a455d8b..54d485cf8 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -1,5 +1,6 @@ #[cfg(test)] mod scanner { + use insta::assert_snapshot; use pretty_assertions::assert_eq; use std::path::{Path, PathBuf}; use std::process::Command; @@ -55,6 +56,126 @@ mod scanner { globs: Vec, normalized_sources: Vec, candidates: Vec, + tree: String, + } + + /// Renders the directory as a tree, annotating every file and folder with + /// an indicator that shows whether the scanner picked it up: + /// + /// - `✓` — scanned + /// - `✗` — ignored / skipped + /// + /// Symlinks are rendered as `link → target`, and the contents of every + /// `.gitignore` file are printed right below the file itself. + /// + /// Folders that are a git repository root (they contain a `.git` folder) + /// are marked with `(git)`, because ignore rules behave differently + /// inside and outside of a repository. + fn fs_tree(root: &Path, scanned: &[String]) -> String { + fn walk( + dir: &Path, + root: &Path, + prefix: &str, + scanned: &[String], + visited: &mut Vec, + out: &mut String, + ) { + let mut entries = fs::read_dir(dir) + .unwrap() + .filter_map(Result::ok) + .filter(|entry| entry.file_name() != ".git") + .collect::>(); + entries.sort_by_key(|entry| entry.file_name()); + + for (i, entry) in entries.iter().enumerate() { + let last = i == entries.len() - 1; + let connector = if last { "└── " } else { "├── " }; + let child_prefix = if last { " " } else { "│ " }; + + let path = entry.path(); + let name = entry.file_name().to_string_lossy().to_string(); + let rel = path + .strip_prefix(root) + .unwrap() + .to_string_lossy() + .replace('\\', "/"); + + let file_type = entry.file_type().unwrap(); + + // A file is scanned when it shows up in the scanned files + // list. A folder is considered scanned when any scanned file + // lives inside of it. Paths that go through a symlink count + // towards the symlink, not towards the target directory. + let dir_prefix = format!("{rel}/"); + let is_scanned = scanned + .iter() + .any(|file| file == &rel || file.starts_with(&dir_prefix)); + let indicator = if is_scanned { "✓" } else { "✗" }; + + let mut display_name = name.clone(); + if path.join(".git").exists() { + display_name = format!("{display_name} (git)"); + } + if file_type.is_symlink() { + if let Ok(target) = fs::read_link(&path) { + let target = target + .strip_prefix(root) + .unwrap_or(&target) + .to_string_lossy() + .replace('\\', "/"); + display_name = format!("{name} → {target}"); + } + } + + out.push_str(&format!("{prefix}{connector}{indicator} {display_name}\n")); + + // Print the contents of `.gitignore` files below the file + // itself, so the ignore rules are visible in the same output. + if name == ".gitignore" { + if let Ok(contents) = fs::read_to_string(&path) { + for line in contents.lines() { + let line = format!("{prefix}{child_prefix} {line}"); + out.push_str(line.trim_end()); + out.push('\n'); + } + } + } + + // Descend into directories, including symlinked ones. The + // paths inside a symlinked directory are computed through the + // symlink, so the indicators show what the scanner saw via + // that route. A visited stack prevents symlink cycles from + // recursing forever. + if path.metadata().map(|meta| meta.is_dir()).unwrap_or(false) { + let Ok(canonical) = dunce::canonicalize(&path) else { + continue; + }; + if visited.contains(&canonical) { + continue; + } + + visited.push(canonical); + walk( + &path, + root, + &format!("{prefix}{child_prefix}"), + scanned, + visited, + out, + ); + visited.pop(); + } + } + } + + let mut visited = vec![dunce::canonicalize(root).unwrap()]; + let mut out = if root.join(".git").exists() { + String::from(". (git)\n") + } else { + String::from(".\n") + }; + walk(root, root, "", scanned, &mut visited, &mut out); + out } fn create_files_in(dir: &path::Path, paths: &[(&str, &str)]) { @@ -167,11 +288,14 @@ mod scanner { .collect::>(); normalized_sources.sort(); + let tree = fs_tree(&dir, &files); + ScanResult { files, globs, normalized_sources, candidates, + tree, } } @@ -185,6 +309,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -192,6 +317,14 @@ mod scanner { ("b.html", ""), ("c.html", ""), ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + └── ✓ index.html + "); assert_eq!(files, vec!["a.html", "b.html", "c.html", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -203,6 +336,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ (".gitignore", "b.html"), @@ -211,6 +345,17 @@ mod scanner { ("b.html", ""), ("c.html", ""), ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ b.html + ├── ✓ a.html + ├── ✗ b.html + ├── ✓ c.html + └── ✓ index.html + "); + assert_eq!(files, vec!["a.html", "c.html", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -222,6 +367,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -232,6 +378,20 @@ mod scanner { ("public/deeply/nested/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ public + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + ├── ✓ deeply + │ └── ✓ nested + │ └── ✓ c.html + └── ✓ nested + └── ✓ c.html + "); + assert_eq!( files, vec![ @@ -253,6 +413,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -266,6 +427,25 @@ mod scanner { ("public/very/deeply/nested/a.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ public + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + ├── ✓ nested + │ ├── ✓ a.html + │ ├── ✓ again + │ │ └── ✓ a.html + │ ├── ✓ b.html + │ └── ✓ c.html + └── ✓ very + └── ✓ deeply + └── ✓ nested + └── ✓ a.html + "); + assert_eq!( files, vec![ @@ -290,6 +470,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ (".gitignore", "public/b.html\na.html"), @@ -299,6 +480,18 @@ mod scanner { ("public/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ public/b.html + │ a.html + ├── ✓ index.html + └── ✓ public + ├── ✗ a.html + ├── ✗ b.html + └── ✓ c.html + "); + assert_eq!(files, vec!["index.html", "public/c.html",]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -310,6 +503,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -318,6 +512,15 @@ mod scanner { ("src/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ src + ├── ✓ a.html + ├── ✓ b.html + └── ✓ c.html + "); + assert_eq!( files, vec!["index.html", "src/a.html", "src/b.html", "src/c.html"] @@ -335,6 +538,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -343,6 +547,14 @@ mod scanner { ("c.lock", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ a.mp4 + ├── ✗ b.png + ├── ✗ c.lock + └── ✓ index.html + "); + assert_eq!(files, vec!["index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -355,6 +567,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ // Looks like `.pages` binary extension, but it's a folder @@ -368,6 +581,16 @@ mod scanner { ), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ other.pages + ├── ✗ other.pages + │ └── ✗ index.html + └── ✓ some.pages + └── ✓ index.html + "); + assert_eq!(files, vec!["some.pages/index.html"]); assert_eq!(globs, vec!["*", "some.pages/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -379,6 +602,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -387,6 +611,14 @@ mod scanner { ("c.less", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ a.css + ├── ✗ b.sass + ├── ✗ c.less + └── ✓ index.html + "); + assert_eq!(files, vec!["a.css", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -398,9 +630,16 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[("src/index.my-extension", "")]); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ index.my-extension + "); + assert_eq!(files, vec!["src/index.my-extension"]); assert_eq!(globs, vec!["*", "src/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,my-extension,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -412,6 +651,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -419,6 +659,13 @@ mod scanner { ("yarn.lock", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + ├── ✗ package-lock.json + └── ✗ yarn.lock + "); + assert_eq!(files, vec!["index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -430,6 +677,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ // Explicitly listed root files @@ -474,6 +722,58 @@ mod scanner { ("nested-d/very/deeply/nested/directory/again/foo.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✓ foo.html + ├── ✓ nested-a + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + ├── ✓ nested-b + │ └── ✓ deeply-nested + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + ├── ✓ nested-c + │ ├── ✗ .gitignore + │ │ ignored-folder/ + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ ├── ✓ foo.html + │ ├── ✗ ignored-folder + │ │ ├── ✗ bar.html + │ │ ├── ✗ baz.html + │ │ └── ✗ foo.html + │ └── ✓ sibling-folder + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + └── ✓ nested-d + ├── ✗ .gitignore + │ deep/ + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✓ foo.html + └── ✓ very + └── ✓ deeply + └── ✓ nested + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✗ deep + │ ├── ✗ bar.html + │ ├── ✗ baz.html + │ └── ✗ foo.html + ├── ✓ directory + │ ├── ✓ again + │ │ └── ✓ foo.html + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + └── ✓ foo.html + "); + assert_eq!( files, vec![ @@ -528,6 +828,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan(&[ // The gitignore file is used to filter out files but not scanned for candidates @@ -550,6 +851,21 @@ mod scanner { ("index4.svelte", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ # md:font-bold + │ foo.html + ├── ✗ foo.html + ├── ✗ foo.jpg + ├── ✓ index.angular.html + ├── ✓ index.html + ├── ✓ index.svelte + ├── ✓ index2.svelte + ├── ✓ index3.svelte + └── ✓ index4.svelte + "); + assert_eq!( candidates, vec![ @@ -573,12 +889,21 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[("foo/bar/baz/foo.html", "content-['foo.html']")], vec!["@source '**/*'", "@source './foo/bar/baz/..'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ foo + └── ✓ bar + └── ✓ baz + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['foo.html']"]); assert_eq!(normalized_sources, vec!["**/*", "foo/bar/**/*"]); } @@ -610,6 +935,19 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ project-a + │ └── ✗ src + │ └── ✗ index.css + └── ✓ project-b + ├── ✓ ignored + │ ├── ✓ except.html + │ └── ✗ ignored.html + └── ✓ keep + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['GOOD-1']", "content-['GOOD-2']"]); } @@ -619,12 +957,18 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[("my-file", "content-['my-file']")], vec!["@source '**/*'", "@source './my-file'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ my-file + "); + assert_eq!(candidates, vec!["content-['my-file']"]); assert_eq!(normalized_sources, vec!["**/*", "my-file"]); } @@ -635,6 +979,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -654,6 +999,14 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ my-folder.bin + │ └── ✓ foo.html + └── ✓ my-folder.templates + └── ✓ foo.html + "); + assert_eq!( candidates, vec![ @@ -672,6 +1025,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -682,6 +1036,11 @@ mod scanner { vec!["@source '**/*'", "@source '*.styl'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ foo.styl + "); + assert_eq!(candidates, vec!["content-['foo.styl']"]); assert_eq!(normalized_sources, vec!["**/*", "*.styl"]); } @@ -693,6 +1052,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ( @@ -707,6 +1068,18 @@ mod scanner { vec!["@source './blog/*/foo/bar/baz/**/*'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + └── ✓ foo + └── ✓ bar + ├── ✓ baz + │ └── ✓ index.html + └── ✗ qux + └── ✗ index.html + "); + assert_eq!( candidates, vec!["content-['blog/2024/foo/bar/baz/index.html']"] @@ -721,6 +1094,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -734,6 +1108,19 @@ mod scanner { vec!["@source '**/*'", "@source './**/*.{styl}'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ app + ├── ✓ (theme) + │ └── ✓ page.styl + ├── ✓ [...slug] + │ └── ✓ page.styl + ├── ✓ [[...slug]] + │ └── ✓ page.styl + └── ✓ [slug] + └── ✓ page.styl + "); + assert_eq!( candidates, vec![ @@ -775,6 +1162,14 @@ mod scanner { let mut scanner = Scanner::new(sources); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ └── ✓ index.html + └── ✓ project-b + └── ✓ index.html + "); + // We've done the initial scan and found the files assert_eq!( candidates, @@ -790,6 +1185,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -802,6 +1198,13 @@ mod scanner { vec!["@source '**/*'", "@source 'foo.styl'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ foo.styl + └── ✓ foo.styl + "); + assert_eq!(candidates, vec!["content-['foo.styl']"]); assert_eq!(normalized_sources, vec!["**/*", "foo.styl"]); } @@ -831,6 +1234,14 @@ mod scanner { let mut scanner = Scanner::new(sources); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ └── ✓ index.html + └── ✓ project-b + └── ✓ index.html + "); + // We've done the initial scan and found the files assert_eq!( candidates, @@ -935,6 +1346,24 @@ mod scanner { "content-['project-b/sub1/sub2/new.html']" ] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ ├── ✓ index.html + │ ├── ✓ new.html + │ └── ✓ sub1 + │ └── ✓ sub2 + │ ├── ✓ index.html + │ └── ✓ new.html + └── ✓ project-b + ├── ✓ index.html + ├── ✓ new.html + └── ✓ sub1 + └── ✓ sub2 + ├── ✓ index.html + └── ✓ new.html + "); } #[test] @@ -966,6 +1395,14 @@ mod scanner { vec!["src/index.html", "src/keep.html", "src/remove.html"] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + ├── ✓ keep.html + └── ✓ remove.html + "); + fs::remove_file(dir.join("src/remove.html")).unwrap(); scanner.scan(); @@ -973,6 +1410,13 @@ mod scanner { scanned_files(&mut scanner, &dir), vec!["src/index.html", "src/keep.html"] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✓ keep.html + "); } #[test] @@ -1013,6 +1457,18 @@ mod scanner { ] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + ├── ✓ keep + │ └── ✓ index.html + └── ✓ remove + ├── ✓ index.html + └── ✓ nested + └── ✓ index.html + "); + fs::remove_dir_all(dir.join("src/remove")).unwrap(); scanner.scan(); @@ -1020,6 +1476,14 @@ mod scanner { scanned_files(&mut scanner, &dir), vec!["src/index.html", "src/keep/index.html"] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✓ keep + └── ✓ index.html + "); } #[test] @@ -1046,6 +1510,13 @@ mod scanner { scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ index.html + └── ✓ src + └── ✓ index.html + "); + let globs = scanned_globs(&mut scanner, &dir); assert!(globs.iter().any(|glob| glob.starts_with("src/**/*"))); @@ -1053,6 +1524,11 @@ mod scanner { scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ index.html + "); + let globs = scanned_globs(&mut scanner, &dir); assert!(!globs.iter().any(|glob| glob.starts_with("src/**/*"))); } @@ -1111,6 +1587,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/index.ts", "content-['src/index.ts']"), @@ -1138,6 +1616,25 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ admin + │ └── ✓ foo + │ └── ✓ template.html + ├── ✗ colors + │ ├── ✗ blue.tsx + │ ├── ✗ green.tsx + │ └── ✗ red.jsx + ├── ✗ index.ts + ├── ✓ templates + │ └── ✓ index.html + └── ✗ utils + ├── ✗ date.ts + ├── ✗ file.ts + └── ✗ string.ts + "); + assert_eq!( candidates, vec![ @@ -1183,6 +1680,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( // Typically skipped &[ @@ -1194,6 +1693,15 @@ mod scanner { vec!["@source '**/*'", "@source 'src/**/*.{exe,bin}'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ out + │ └── ✗ out.exe + └── ✓ src + ├── ✓ index.bin + └── ✓ index.exe + "); + assert_eq!( candidates, vec!["content-['src/index.bin']", "content-['src/index.exe']",] @@ -1221,6 +1729,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("index.html", "content-['index.html']"), @@ -1245,6 +1755,21 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ index.html + └── ✓ src + ├── ✓ admin + │ ├── ✗ ignore.html + │ └── ✓ index.html + ├── ✓ dashboard + │ ├── ✗ ignore.html + │ └── ✓ index.html + ├── ✗ ignore.html + ├── ✗ index.html + └── ✗ lib.ts + "); + assert_eq!( candidates, vec![ @@ -1264,7 +1789,10 @@ mod scanner { #[test] fn it_should_restrict_explicit_file_sources_to_the_matching_file() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1273,6 +1801,13 @@ mod scanner { vec!["@source './src/foo.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar.html + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['src/foo.html']"]); assert_eq!(files, vec!["src/foo.html"]); } @@ -1280,7 +1815,10 @@ mod scanner { #[test] fn it_should_combine_multiple_restricted_sources_for_the_same_base() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1290,6 +1828,14 @@ mod scanner { vec!["@source './src/foo.html'", "@source './src/bar.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar.html + ├── ✗ baz.html + └── ✓ foo.html + "); + assert_eq!( candidates, vec!["content-['src/bar.html']", "content-['src/foo.html']"] @@ -1313,7 +1859,10 @@ mod scanner { ]; let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( paths_with_content, vec![ @@ -1322,6 +1871,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ component-sources.classes.txt + ├── ✗ ignore-me + │ └── ✗ component.html + ├── ✗ ignore-me.txt + └── ✓ nested + ├── ✓ component.html + └── ✗ ignore-me.html + "); + assert_eq!( candidates, vec![ @@ -1336,7 +1896,10 @@ mod scanner { // Same setup, but with the root-level source declared first let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( paths_with_content, vec![ @@ -1345,6 +1908,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ component-sources.classes.txt + ├── ✗ ignore-me + │ └── ✗ component.html + ├── ✗ ignore-me.txt + └── ✓ nested + ├── ✓ component.html + └── ✗ ignore-me.html + "); + assert_eq!( candidates, vec![ @@ -1361,12 +1935,21 @@ mod scanner { #[test] fn it_should_allow_later_ignores_to_override_restricted_sources() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[("src/foo.html", "content-['src/foo.html']")], vec!["@source './src/foo.html'", "@source not './src/foo.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✗ src + └── ✗ foo.html + "); + assert!(candidates.is_empty()); assert!(files.is_empty()); } @@ -1375,7 +1958,10 @@ mod scanner { fn it_should_handle_sources_with_parent_patterns() { { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1385,6 +1971,15 @@ mod scanner { vec!["@source './src/ba*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✓ ignore.html + │ └── ✓ index.html + └── ✗ foo.html + "); + assert_eq!( candidates, vec![ @@ -1397,7 +1992,10 @@ mod scanner { { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1410,13 +2008,25 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✗ ignore.html + │ └── ✓ index.html + └── ✗ foo.html + "); + assert_eq!(candidates, vec!["content-['src/bar/index.html']"]); assert_eq!(files, vec!["src/bar/index.html"]); } { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1426,13 +2036,25 @@ mod scanner { vec!["@source '**/*'", "@source not './src/ba*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar + │ ├── ✗ ignore.html + │ └── ✗ index.html + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['src/foo.html']"]); assert_eq!(files, vec!["src/foo.html"]); } { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1446,6 +2068,15 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✗ ignore.html + │ └── ✓ index.html + └── ✓ foo.html + "); + assert_eq!( candidates, vec!["content-['src/bar/index.html']", "content-['src/foo.html']"] @@ -1460,7 +2091,10 @@ mod scanner { // specific `@source` points at a single file inside a subdirectory. The restriction // added for the explicit file must not hide its siblings from the broad source. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/components/button.html", "content-['button']"), @@ -1469,6 +2103,14 @@ mod scanner { vec!["@source '**/*'", "@source './src/components/button.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ components + ├── ✓ button.html + └── ✓ card.html + "); + assert_eq!(candidates, vec!["content-['button']", "content-['card']"]); assert_eq!( files, @@ -1481,7 +2123,10 @@ mod scanner { // Same as above, but the broad source is an auto-detected directory (`@source "src"`) // and the explicit file lives in a nested subdirectory of it. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/components/button.html", "content-['button']"), @@ -1490,6 +2135,14 @@ mod scanner { vec!["@source 'src'", "@source './src/components/button.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ components + ├── ✓ button.html + └── ✓ card.html + "); + assert_eq!(candidates, vec!["content-['button']", "content-['card']"]); assert_eq!( files, @@ -1499,7 +2152,9 @@ mod scanner { #[test] fn root_file_source_should_not_suppress_sibling_source_roots() { - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ ("index.css", ""), ("src/index.html", "content-['src/index.html']"), @@ -1513,6 +2168,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.css + ├── ✓ pages + │ ├── ✓ foo.html + │ └── ✓ nested + │ └── ✓ foo.html + └── ✓ src + └── ✓ index.html + "); + assert_eq!( candidates, vec![ @@ -1530,6 +2196,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "ignore-1.html\nweb/ignore-2.html"), @@ -1541,6 +2209,19 @@ mod scanner { vec!["@source './src'", "@source './web'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ ignore-1.html + │ web/ignore-2.html + ├── ✓ src + │ └── ✓ index.html + └── ✓ web + ├── ✗ ignore-1.html + ├── ✗ ignore-2.html + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['src/index.html']", "content-['web/index.html']",] @@ -1558,6 +2239,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/logo.jpg", "content-['/src/logo.jpg']"), @@ -1566,6 +2249,13 @@ mod scanner { vec!["@source './src/logo.{jpg,png}'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ logo.jpg + └── ✓ logo.png + "); + assert_eq!( candidates, vec!["content-['/src/logo.jpg']", "content-['/src/logo.png']"] @@ -1582,6 +2272,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "ignore-1.html\n/web/ignore-2.html"), @@ -1591,6 +2283,17 @@ mod scanner { ], vec!["@source './web'", "@source './web/ignore-1.html'"], ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ ignore-1.html + │ /web/ignore-2.html + └── ✓ web + ├── ✓ ignore-1.html + ├── ✗ ignore-2.html + └── ✓ index.html + "); assert_eq!( candidates, vec![ @@ -1611,7 +2314,10 @@ mod scanner { // A `.gitignore` that ignores everything (`/*`) and then whitelists specific // directories and files using negated patterns. let ScanResult { - files, candidates, .. + files, + candidates, + tree, + .. } = scan(&[ ( ".gitignore", @@ -1629,6 +2335,28 @@ mod scanner { ), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /* + │ !/app + │ !/public + │ !/package.json + │ !/.gitignore + ├── ✓ app + │ └── ✓ index.html + ├── ✗ build + │ └── ✗ generated.html + ├── ✗ logs + │ └── ✗ dev.log + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + ├── ✓ package.json + └── ✓ public + └── ✓ index.html + "); + assert_eq!( files, vec!["app/index.html", "package.json", "public/index.html"] @@ -1654,7 +2382,10 @@ mod scanner { // !/foo/bar // ``` let ScanResult { - files, candidates, .. + files, + candidates, + tree, + .. } = scan(&[ ( ".gitignore", @@ -1671,6 +2402,25 @@ mod scanner { ("foo/baz/index.html", "content-['foo/baz/index.html']"), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ # exclude everything except directory foo/bar + │ /* + │ !/foo + │ /foo/* + │ !/foo/bar + ├── ✓ foo + │ ├── ✓ bar + │ │ ├── ✓ index.html + │ │ └── ✓ nested + │ │ └── ✓ index.html + │ ├── ✗ baz + │ │ └── ✗ index.html + │ └── ✗ index.html + └── ✗ index.html + "); + assert_eq!( files, vec!["foo/bar/index.html", "foo/bar/nested/index.html"] @@ -1691,7 +2441,10 @@ mod scanner { // contents. Only when the directory itself is ignored (by an ancestor // `.gitignore`, see the tests above) do we bypass the ignore rules. let ScanResult { - files, candidates, .. + files, + candidates, + tree, + .. } = scan_with_globs( &[ ("vendor/.gitignore", "ignored.html"), @@ -1701,6 +2454,15 @@ mod scanner { vec!["@source 'vendor'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ vendor + ├── ✗ .gitignore + │ ignored.html + ├── ✗ ignored.html + └── ✓ index.html + "); + assert_eq!(files, vec!["vendor/index.html"]); assert_eq!(candidates, vec!["content-['vendor/index.html']"]); } @@ -1711,6 +2473,7 @@ mod scanner { candidates, files, globs, + tree, .. } = scan_with_globs( &[ @@ -1722,6 +2485,18 @@ mod scanner { vec!["@source './src/ef*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ src/efgh/ + └── ✓ src + ├── ✗ abcd + │ └── ✗ index.html + └── ✓ efgh + ├── ✗ ignore.js + └── ✓ index.html + "); + assert_eq!(candidates, vec!["content-['src/efgh/index.html']"]); assert_eq!(files, vec!["src/efgh/index.html"]); assert_eq!(globs, vec!["src/ef*/*.html"]); @@ -1797,7 +2572,33 @@ mod scanner { ), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✗ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); // All ignore files are applied because there's no git repo assert_eq!( @@ -1815,7 +2616,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home (git) + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✗ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1834,7 +2661,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home/project")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project (git) + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1854,7 +2707,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home/project/apps")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps (git) + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✓ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1876,7 +2755,33 @@ mod scanner { .current_dir(dir.join("home/project/apps/web")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web (git) + ├── ✗ .gitignore + │ ignore-web.html + ├── ✓ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✓ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1910,7 +2815,15 @@ mod scanner { public_source_entry_from_pattern(dir.clone(), "@source not 'src/ignore-me.html'"), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + ├── ✗ ignore-me.html + └── ✓ keep-me.html + "); assert_eq!(candidates, vec!["content-['keep-me.html']"]); } @@ -1937,7 +2850,17 @@ mod scanner { ), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + └── ✓ app + └── ✓ [foo] + ├── ✗ ignore-me.html + └── ✓ keep-me.html + "); assert_eq!(candidates, vec!["content-['keep-me.html']"]); } @@ -1964,6 +2887,12 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/keep-me.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + └── ✓ keep-me.html + "); + // Create new files that should definitely be ignored create_files_in( &dir, @@ -1991,6 +2920,18 @@ mod scanner { let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ src/ignored-by-gitignore.html + └── ✓ src + ├── ✗ ignore-by-extension.bin + ├── ✓ ignored-by-gitignore.html + ├── ✗ ignored-by-source-not.html + ├── ✓ keep-me.html + └── ✓ new-file.html + "); + assert_eq!( candidates, vec![ @@ -2020,6 +2961,11 @@ mod scanner { let candidates = scanner.scan(); assert!(candidates.is_empty()); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✗ foo.styl + "); + // Explicitly allow `.styl` files let mut scanner = Scanner::new(vec![ public_source_entry_from_pattern(dir.clone(), "@source '**/*'"), @@ -2028,6 +2974,11 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['foo.styl']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ foo.styl + "); } #[test] @@ -2052,6 +3003,13 @@ mod scanner { let candidates = scanner.scan(); assert!(candidates.is_empty()); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ index.html + └── ✗ index.html + "); + let mut scanner = Scanner::new(vec![ public_source_entry_from_pattern(dir.clone(), "@source '**/*'"), public_source_entry_from_pattern(dir.clone(), "@source './*.html'"), @@ -2059,6 +3017,13 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['index.html']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ index.html + └── ✓ index.html + "); } #[test] @@ -2093,6 +3058,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + // Explicitly listing all `*.html` files, should not include `node_modules` because it's // ignored let sources = vec![public_source_entry_from_pattern( @@ -2104,6 +3080,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + // Explicitly listing all `*.html` files // Explicitly list the `node_modules/my-ui-lib` // @@ -2121,12 +3108,25 @@ mod scanner { "content-['src/index.html']" ] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✓ node_modules + │ └── ✓ my-ui-lib + │ └── ✓ index.html + └── ✓ src + └── ✓ index.html + "); } // https://github.com/tailwindlabs/tailwindcss/issues/19844 #[test] fn test_allow_explicit_sources_ignored_by_allow_list_gitignore() { - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ (".gitignore", "*\n!/app\n!/app/design\n!/app/design/**\n"), ( @@ -2141,6 +3141,27 @@ mod scanner { vec!["@source 'vendor/acme/theme'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ * + │ !/app + │ !/app/design + │ !/app/design/** + ├── ✗ app + │ └── ✗ design + │ └── ✗ frontend + │ └── ✗ theme + │ └── ✗ templates + │ └── ✗ component.phtml + └── ✓ vendor + └── ✓ acme + └── ✓ theme + └── ✓ module + └── ✓ templates + └── ✓ component.phtml + "); + assert_eq!( candidates, vec!["content-['vendor/acme/theme/module/templates/component.phtml']"] @@ -2154,6 +3175,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ( @@ -2172,6 +3195,17 @@ mod scanner { vec!["@source '**/*'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ node_modules + │ └── ✗ index.html + └── ✓ packages + └── ✓ web + ├── ✓ index.html + └── ✗ node_modules + └── ✗ index.html + "); + assert_eq!(candidates, vec!["content-['packages/web/index.html']"]); assert_eq!(files, vec!["packages/web/index.html",]); @@ -2186,6 +3220,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "node_modules\ndist"), @@ -2201,6 +3237,18 @@ mod scanner { vec!["@source 'node_modules/my-ui-lib'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + └── ✓ my-ui-lib + ├── ✓ dist + │ └── ✓ index.html + └── ✗ node.exe + "); + assert_eq!( candidates, vec!["content-['node_modules/my-ui-lib/dist/index.html']"] @@ -2242,6 +3290,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/components/button.tsx']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ *.jsx + │ generated/ + └── ✓ src + └── ✓ components + ├── ✗ button.jsx + └── ✓ button.tsx + "); + // Create 2 new files, one "good" and one "bad" file, and manually scan them. This should // only return the "good" file because the "bad" one is ignored by a `.gitignore` file. create_files_in( @@ -2302,6 +3361,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/💩.js", "content-['src/💩.js']"), @@ -2311,6 +3372,15 @@ mod scanner { vec!["@source '**/*'", "@source not 'src/🤦‍♂️'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ 💩.js + ├── ✗ 🤦‍♂️ + │ └── ✗ foo.tsx + └── ✓ 🤦‍♂️.tsx + "); + assert_eq!( candidates, vec!["content-['src/💩.js']", "content-['src/🤦‍♂️.tsx']"] @@ -2347,6 +3417,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✓ .gitkeep + └── ✓ my-ui-library → node_modules/.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2358,6 +3445,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✗ .gitkeep + └── ✓ my-ui-library → node_modules/.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2369,6 +3473,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✓ .gitkeep + └── ✓ my-ui-library → node_modules/.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2398,6 +3519,16 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ a → c + ├── ✓ b + │ └── ✓ index.html + ├── ✗ c → b/c + └── ✓ z + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['b/index.html']", "content-['z/index.html']"] @@ -2422,6 +3553,14 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ abcd + │ └── ✓ xyz.html + └── ✓ efgh → abcd + └── ✓ xyz.html + "); + assert_eq!(candidates, vec!["content-['abcd/xyz.html']"]); // Partially referencing the symlinked folder with a glob, should find the file @@ -2462,6 +3601,18 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ actual-dir + │ └── ✗ ignore.html + ├── ✗ actual-file.html + ├── ✗ linked-dir → actual-dir + │ └── ✗ ignore.html + ├── ✗ linked-file.html → actual-file.html + └── ✓ src + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['src/keep.html']"]); let mut scanner = Scanner::new(vec![ @@ -2499,6 +3650,16 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ actual-dir + │ └── ✓ ignore.html + └── ✓ project + ├── ✓ keep.html + └── ✓ linked-dir → actual-dir + └── ✓ ignore.html + "); + assert_eq!( candidates, vec![ @@ -2513,6 +3674,16 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ actual-dir + │ └── ✗ ignore.html + └── ✓ project + ├── ✓ keep.html + └── ✗ linked-dir → actual-dir + └── ✗ ignore.html + "); + assert_eq!(candidates, vec!["content-['project/keep.html']"]); } @@ -2570,6 +3741,23 @@ mod scanner { ] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ linked.html → packages/other/index.html + ├── ✓ node_modules + │ └── ✓ repro → packages/repro + │ ├── ✓ nested + │ │ └── ✓ deep.html + │ └── ✓ source.html + └── ✓ packages + ├── ✓ other + │ └── ✓ index.html + └── ✓ repro + ├── ✓ nested + │ └── ✓ deep.html + └── ✓ source.html + "); + // Both the symlinked paths and the canonical paths should be tracked, such that file // watchers watching the returned files also watch the real files on disk. let files = scanned_files(&mut scanner, &dir); @@ -2603,6 +3791,16 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['v1']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → packages/repro + │ └── ✓ source.html + └── ✓ packages + └── ✓ repro + └── ✓ source.html + "); + // Update the real file on disk. This is the path file watchers will report changes for. create_files_in(&dir, &[("packages/repro/source.html", "content-['v2']")]); @@ -2630,6 +3828,16 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['a']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → packages/repro + │ └── ✓ a.html + └── ✓ packages + └── ✓ repro + └── ✓ a.html + "); + // Create a new file in the real directory. File watchers watching the real directory // will report the new file with its canonical path. create_files_in(&dir, &[("packages/repro/b.html", "content-['b']")]); @@ -2639,12 +3847,26 @@ mod scanner { "html".into(), )]); assert_eq!(candidates, vec!["content-['b']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → packages/repro + │ ├── ✓ a.html + │ └── ✓ b.html + └── ✓ packages + └── ✓ repro + ├── ✓ a.html + └── ✗ b.html + "); } // https://github.com/tailwindlabs/tailwindcss/pull/20408 #[test] fn test_resolving_globs_does_not_traverse_gitignored_directories() { - let ScanResult { files, globs, .. } = scan_with_globs( + let ScanResult { + files, globs, tree, .. + } = scan_with_globs( &[ (".gitignore", "/vendor\n"), ("vendor/pkg/canary/index.html", ""), @@ -2653,6 +3875,18 @@ mod scanner { vec!["@source '**/*'", "@source './vendor/pkg/canary'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /vendor + ├── ✓ src + │ └── ✓ index.html + └── ✓ vendor + └── ✓ pkg + └── ✓ canary + └── ✓ index.html + "); + assert_eq!( files, vec!["src/index.html", "vendor/pkg/canary/index.html"] From e0d271bb62815dc42f4b58c7f4bfc50eb29131f7 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Thu, 13 Aug 2026 01:45:42 +0200 Subject: [PATCH 03/23] render broken paths (symlinks) --- crates/oxide/tests/scanner.rs | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 54d485cf8..8efe813f9 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -124,6 +124,11 @@ mod scanner { .to_string_lossy() .replace('\\', "/"); display_name = format!("{name} → {target}"); + + // The symlink points to a target that doesn't exist + if path.metadata().is_err() { + display_name = format!("{display_name} (broken)"); + } } } @@ -3521,10 +3526,10 @@ mod scanner { assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" . - ├── ✗ a → c + ├── ✗ a → c (broken) ├── ✓ b │ └── ✓ index.html - ├── ✗ c → b/c + ├── ✗ c → b/c (broken) └── ✓ z └── ✓ index.html "); From 0308dd4adb5f8ec8e85a65964450022fdcf01c18 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Thu, 13 Aug 2026 01:49:35 +0200 Subject: [PATCH 04/23] fix broken symlink tests --- crates/oxide/tests/scanner.rs | 33 ++++++++++++++++++++++++--------- 1 file changed, 24 insertions(+), 9 deletions(-) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 8efe813f9..67979d420 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -118,11 +118,15 @@ mod scanner { } if file_type.is_symlink() { if let Ok(target) = fs::read_link(&path) { - let target = target + let mut target = target .strip_prefix(root) .unwrap_or(&target) .to_string_lossy() .replace('\\', "/"); + if target.is_empty() { + // The symlink points to the root itself + target = ".".into(); + } display_name = format!("{name} → {target}"); // The symlink points to a target that doesn't exist @@ -3512,11 +3516,14 @@ mod scanner { ], ); - // Create recursive symlinks - let _ = symlink(dir.join("a"), dir.join("b")); - let _ = symlink(dir.join("b/c"), dir.join("c")); - let _ = symlink(dir.join("b/root"), &dir); - let _ = symlink(dir.join("c"), dir.join("a")); + // Create recursive symlinks: + // + // - `a → b`, `b/c → c`, `c → a` form a cycle + // - `b/root → .` points back at the root directory + let _ = symlink(dir.join("b"), dir.join("a")); + let _ = symlink(dir.join("c"), dir.join("b/c")); + let _ = symlink(&dir, dir.join("b/root")); + let _ = symlink(dir.join("a"), dir.join("c")); let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( dir.clone(), @@ -3526,10 +3533,18 @@ mod scanner { assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" . - ├── ✗ a → c (broken) + ├── ✓ a → b + │ ├── ✗ c → c + │ ├── ✓ index.html + │ └── ✗ root → . ├── ✓ b - │ └── ✓ index.html - ├── ✗ c → b/c (broken) + │ ├── ✗ c → c + │ ├── ✓ index.html + │ └── ✗ root → . + ├── ✓ c → a + │ ├── ✗ c → c + │ ├── ✓ index.html + │ └── ✗ root → . └── ✓ z └── ✓ index.html "); From 54e2c1ab155e725b359cbf080a91c87fa86c65af Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Thu, 13 Aug 2026 01:53:02 +0200 Subject: [PATCH 05/23] render symlinks relatively --- crates/oxide/tests/scanner.rs | 77 +++++++++++++++++++++++------------ 1 file changed, 52 insertions(+), 25 deletions(-) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 67979d420..c30327bdd 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -65,13 +65,42 @@ mod scanner { /// - `✓` — scanned /// - `✗` — ignored / skipped /// - /// Symlinks are rendered as `link → target`, and the contents of every + /// Symlinks are rendered as `link → target`, where the target is shown + /// relative to the folder containing the symlink. The contents of every /// `.gitignore` file are printed right below the file itself. /// /// Folders that are a git repository root (they contain a `.git` folder) /// are marked with `(git)`, because ignore rules behave differently /// inside and outside of a repository. fn fs_tree(root: &Path, scanned: &[String]) -> String { + /// Computes the relative path from `parent` to `target`, where both + /// are relative to the same root, e.g.: `b/c` seen from `b` is `c`, + /// and the root itself seen from `b` is `..`. + fn relative_to(target: &Path, parent: &Path) -> String { + let target = target.components().collect::>(); + let parent = parent.components().collect::>(); + + let common = target + .iter() + .zip(&parent) + .take_while(|(a, b)| **a == **b) + .count(); + + let mut parts = vec![".."; parent.len() - common]; + parts.extend( + target[common..] + .iter() + .map(|component| component.as_os_str().to_str().unwrap()), + ); + + if parts.is_empty() { + // The symlink points to its own parent folder + ".".into() + } else { + parts.join("/") + } + } + fn walk( dir: &Path, root: &Path, @@ -118,15 +147,13 @@ mod scanner { } if file_type.is_symlink() { if let Ok(target) = fs::read_link(&path) { - let mut target = target - .strip_prefix(root) - .unwrap_or(&target) - .to_string_lossy() - .replace('\\', "/"); - if target.is_empty() { - // The symlink points to the root itself - target = ".".into(); - } + // Show the target relative to the folder containing + // the symlink, or as-is when it points outside of the + // tree. + let target = match (target.strip_prefix(root), dir.strip_prefix(root)) { + (Ok(target), Ok(parent)) => relative_to(target, parent), + _ => target.to_string_lossy().replace('\\', "/"), + }; display_name = format!("{name} → {target}"); // The symlink points to a target that doesn't exist @@ -3438,7 +3465,7 @@ mod scanner { │ └── ✓ index.ts └── ✓ @org ├── ✓ .gitkeep - └── ✓ my-ui-library → node_modules/.pnpm/@org+my-ui-library + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library └── ✓ dist └── ✓ index.ts "); @@ -3466,7 +3493,7 @@ mod scanner { │ └── ✓ index.ts └── ✓ @org ├── ✗ .gitkeep - └── ✓ my-ui-library → node_modules/.pnpm/@org+my-ui-library + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library └── ✓ dist └── ✓ index.ts "); @@ -3494,7 +3521,7 @@ mod scanner { │ └── ✓ index.ts └── ✓ @org ├── ✓ .gitkeep - └── ✓ my-ui-library → node_modules/.pnpm/@org+my-ui-library + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library └── ✓ dist └── ✓ index.ts "); @@ -3534,17 +3561,17 @@ mod scanner { assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" . ├── ✓ a → b - │ ├── ✗ c → c + │ ├── ✗ c → ../c │ ├── ✓ index.html - │ └── ✗ root → . + │ └── ✗ root → .. ├── ✓ b - │ ├── ✗ c → c + │ ├── ✗ c → ../c │ ├── ✓ index.html - │ └── ✗ root → . + │ └── ✗ root → .. ├── ✓ c → a - │ ├── ✗ c → c + │ ├── ✗ c → . │ ├── ✓ index.html - │ └── ✗ root → . + │ └── ✗ root → .. └── ✓ z └── ✓ index.html "); @@ -3676,7 +3703,7 @@ mod scanner { │ └── ✓ ignore.html └── ✓ project ├── ✓ keep.html - └── ✓ linked-dir → actual-dir + └── ✓ linked-dir → ../actual-dir └── ✓ ignore.html "); @@ -3700,7 +3727,7 @@ mod scanner { │ └── ✗ ignore.html └── ✓ project ├── ✓ keep.html - └── ✗ linked-dir → actual-dir + └── ✗ linked-dir → ../actual-dir └── ✗ ignore.html "); @@ -3765,7 +3792,7 @@ mod scanner { . ├── ✓ linked.html → packages/other/index.html ├── ✓ node_modules - │ └── ✓ repro → packages/repro + │ └── ✓ repro → ../packages/repro │ ├── ✓ nested │ │ └── ✓ deep.html │ └── ✓ source.html @@ -3814,7 +3841,7 @@ mod scanner { assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" . ├── ✓ node_modules - │ └── ✓ repro → packages/repro + │ └── ✓ repro → ../packages/repro │ └── ✓ source.html └── ✓ packages └── ✓ repro @@ -3851,7 +3878,7 @@ mod scanner { assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" . ├── ✓ node_modules - │ └── ✓ repro → packages/repro + │ └── ✓ repro → ../packages/repro │ └── ✓ a.html └── ✓ packages └── ✓ repro @@ -3871,7 +3898,7 @@ mod scanner { assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" . ├── ✓ node_modules - │ └── ✓ repro → packages/repro + │ └── ✓ repro → ../packages/repro │ ├── ✓ a.html │ └── ✓ b.html └── ✓ packages From 5fcd42dd9f9216adf2b3b2891915c4f8b54b5808 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Tue, 11 Aug 2026 20:24:17 +0200 Subject: [PATCH 06/23] don't promote re-included directories to external sources When deciding whether an explicitly listed directory is git ignored (and should therefore bypass the ignore rules as an "external" source), follow git's precedence: walk up from the directory and let the nearest `.gitignore` with a definitive answer win. A directory that a deeper `.gitignore` re-includes via a `!dir` pattern is not ignored, even when an ancestor `.gitignore` ignores it, so it keeps its regular auto source detection behavior instead of bypassing the gitignore rules inside of it. --- crates/oxide/src/scanner/sources.rs | 40 ++++++++++++++++++++++++----- 1 file changed, 34 insertions(+), 6 deletions(-) diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 53c09387f..99a4a5ec2 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -701,6 +701,29 @@ mod tests { assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); } + #[test] + fn folders_reincluded_by_a_deeper_gitignore_stay_auto_sources() { + let dir = tempdir().unwrap(); + fs::create_dir_all(dir.path().join(".git")).unwrap(); + fs::write(dir.path().join(".gitignore"), "generated/\n").unwrap(); + // The nearest `.gitignore` with a definitive answer wins: `packages/app/.gitignore` + // re-includes `generated`, overruling the ignore rule in the root `.gitignore`, so the + // directory keeps its regular auto source detection behavior (which respects the + // `.gitignore` rules inside of it) instead of bypassing them as an external source. + fs::create_dir_all(dir.path().join("packages").join("app")).unwrap(); + fs::write( + dir.path().join("packages").join("app").join(".gitignore"), + "!generated/\n", + ) + .unwrap(); + + let base = dir.path().join("packages").join("app").join("generated"); + fs::create_dir_all(&base).unwrap(); + let base = dunce::canonicalize(&base).unwrap(); + + assert_eq!(auto_source_entry(&base), SourceEntry::Auto { base }); + } + #[test] fn folders_not_ignored_by_gitignore_stay_auto_sources() { let dir = tempdir().unwrap(); @@ -815,12 +838,17 @@ pub fn public_source_entries_to_private_source_entries( // but `base` itself is not. if dir != base { if let Some(gitignore) = gitignore { - if gitignore - .matched_path_or_any_parents(&base, true) - .is_ignore() - { - source = SourceEntry::External { base: base.into() }; - break; + // Match git's precedence: the nearest `.gitignore` with a definitive + // answer wins. A directory that is re-included by a deeper + // `!the-directory` pattern is not ignored, even when an ancestor + // `.gitignore` ignores it. + match gitignore.matched_path_or_any_parents(&base, true) { + ignore::Match::Ignore(_) => { + source = SourceEntry::External { base: base.into() }; + break; + } + ignore::Match::Whitelist(_) => break, + ignore::Match::None => {} } } } From 6d9d3b65e4175b19e1e6b1ed9a7314acb425efcf Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Tue, 11 Aug 2026 20:24:32 +0200 Subject: [PATCH 07/23] rewrite source scanning around one walker per source kind Previously all `@source` directives were compiled into a single file walker: directives became layered gitignore rules, `expand_restricted_patterns` injected `*` / `/*` / `!/dir/` counter-rules to restrict concrete patterns, and a global `filter_entry` re-checked files against every source. Because all rules were global to the one walker, sources contaminated each other: pattern-driven re-includes leaked into auto source results, and auto source bases legitimized files that only a pattern's walk root reached. This caused a family of bugs: - #18870: globs through directories that are ignored from within (Laravel's `storage/` layout) found nothing - a concrete file `@source` pointing behind `node_modules` or a git ignored directory scanned all of its siblings - `@source "./foo/*.html"` on an ignored folder also scanned other extensions in `foo` - external sources (explicitly listed ignored directories) scanned nested `node_modules` Sources are now split over multiple walkers: - One gitignore-aware walker for all auto/external roots. Default rules and `@source not` rules are registered as explicit in-memory gitignores in directive order (later directives win); external roots get a `!/**/*` whitelist plus a re-statement of the default rules so that e.g. nested `node_modules` stay ignored. - One walker per pattern base. The static prefix of a glob is the explicit part: it is used as the walk root and `.gitignore` files above it never apply, even when the base is hidden inside an ignored directory. The wildcard part is not explicit: `.gitignore` files inside the subtree and the default-ignored directories still prune (`@source "./**/*.html"` does not descend into `node_modules` or a git ignored `dist/"; `@source "./dist/**/*.html"` does). Matching files always win over file-level ignores, and globs that don't pin an extension re-apply the default extension rules. A filter closure makes the exact per-file decision, honoring the relative order of `@source not` directives, and `dir_could_contain_matches` keeps the walk pruned to directories that can actually contribute files. The `@source` semantics are now documented as a spec at the top of `scanner/mod.rs` and pinned by eleven new tests (six of which failed on the old implementation). Four unit tests asserting the internals of the removed `expand_restricted_patterns` mechanism were deleted; the user-visible behavior they guarded is covered by the scanner-level tests. --- .../src/scanner/auto_source_detection.rs | 6 +- crates/oxide/src/scanner/mod.rs | 536 +++++++++++++----- crates/oxide/src/scanner/sources.rs | 295 +--------- crates/oxide/tests/scanner.rs | 334 +++++++++++ 4 files changed, 722 insertions(+), 449 deletions(-) diff --git a/crates/oxide/src/scanner/auto_source_detection.rs b/crates/oxide/src/scanner/auto_source_detection.rs index 0d723a1ab..c53756456 100644 --- a/crates/oxide/src/scanner/auto_source_detection.rs +++ b/crates/oxide/src/scanner/auto_source_detection.rs @@ -34,10 +34,10 @@ pub static IGNORED_CONTENT_DIRS: sync::LazyLock> = sync::LazyL .collect() }); -static IGNORED_CONTENT_DIRS_GLOB: sync::LazyLock = +pub static IGNORED_CONTENT_DIRS_GLOB: sync::LazyLock = sync::LazyLock::new(|| format!("{{{}}}/", IGNORED_CONTENT_DIRS.join(","))); -static IGNORED_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { +pub static IGNORED_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { format!( "*.{{{}}}", include_str!("fixtures/ignored-extensions.txt") @@ -59,7 +59,7 @@ pub static BINARY_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new( ) }); -static IGNORED_FILES_GLOB: sync::LazyLock = sync::LazyLock::new(|| { +pub static IGNORED_FILES_GLOB: sync::LazyLock = sync::LazyLock::new(|| { format!( "{{{}}}", include_str!("fixtures/ignored-files.txt") diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index ab61b11b3..66696e7ec 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -22,16 +22,58 @@ use std::sync::{Arc, Mutex}; use std::time::SystemTime; use tracing::event; -// @source "some/folder"; // This is auto source detection -// @source "some/folder/**/*"; // This is auto source detection -// @source "some/folder/*.html"; // This is just a glob, but new files matching this should be included -// @source "node_modules/my-ui-lib"; // Auto source detection but since node_modules is explicit we allow it -// // Maybe could be considered `external(…)` automatically if: -// // 1. It's git ignored but listed explicitly -// // 2. It exists outside of the current working directory (do we know that?) +// # `@source` semantics // -// @source "do-include-me.bin"; // `.bin` is typically ignored, but now it's explicit so should be included -// @source "git-ignored.html"; // A git ignored file that is listed explicitly, should be scanned +// Every `@source` directive is classified as one of: +// +// - `Auto`: `@source "some/folder"` or `@source "some/folder/**/*"` — auto source detection. +// The folder is scanned recursively while respecting `.gitignore` files and the default rules +// (skip `node_modules`/`.git`/…, skip binary and irrelevant extensions, skip lock files, …). +// +// - `External`: an `Auto` source whose folder is itself ignored (by a `.gitignore` or because +// it's a default-ignored directory like `node_modules`), e.g. +// `@source "node_modules/my-ui-lib"`. Since the folder was listed explicitly, its ignoredness +// is bypassed: everything inside is scanned as if it were an `Auto` source, except that +// `.gitignore` files no longer apply inside (git ignores the whole tree anyway). The default +// rules still apply inside: nested `node_modules`, binary extensions, etc. stay ignored. +// +// - `Pattern`: `@source "some/folder/*.html"` — an explicit glob. Only files matching the glob +// are scanned. The *static* prefix of the glob (`some/folder`) is the explicit part: it is +// reached even when it is git ignored or hidden behind a default-ignored directory +// (`@source "node_modules/lib/dist/*.html"` works). The *wildcard* part is not explicit: +// while expanding it we still respect `.gitignore` files inside the walked subtree and the +// default-ignored directories (`@source "./**/*.html"` does not descend into `node_modules` +// or a git ignored `dist/`; `@source "./dist/**/*.html"` does descend into `dist/`). +// Individual *files* matching the glob are always included, even when git ignored — you were +// explicit about wanting files of that shape (`@source "git-ignored.html"` and +// `@source "*.styl"` work). Extensions that are ignored by default are only included when the +// glob pins an extension (`@source "logo.{jpg,png}"`, `@source "do-include-me.bin"`), an +// extension-less glob like `@source "some/folder/**/*"`… is `Auto`, and e.g. +// `@source "blog/*/post/**/*"` still applies the default extension rules. +// +// - `Ignored`: `@source not "…"` — excludes matching files/folders, even when a `.gitignore` +// or another `@source` allows them. +// +// Later directives win over earlier ones on conflict: `@source not "./x"` followed by +// `@source "./x/keep.html"` scans `keep.html`, and vice versa excludes it. +// +// # Implementation +// +// Scanning uses the vendored `ignore` crate for gitignore-aware, pruned directory walking (we +// never descend into directories that cannot contribute files). Sources are split over multiple +// walkers: +// +// - One walker for all `Auto` and `External` roots. Default rules and `@source not` rules are +// registered as explicit in-memory gitignores (they rank above `.gitignore` files found on +// disk, later directives above earlier ones). `External` roots additionally get a `!/**/*` +// whitelist (bypassing gitignore rules) plus a re-statement of the default rules. +// +// - One walker per `Pattern` base. It never looks at `.gitignore` files *above* its base +// (the static prefix is explicit), while `.gitignore` files inside the subtree still prune +// directories. Matching files are whitelisted (a glob match beats file-level ignores), and a +// filter closure makes the final per-file decision: the file must match a glob, honoring the +// relative order of `@source not` directives, and non-extension-pinning globs re-apply the +// default file rules. #[derive(Debug, Clone)] pub enum ChangedContent { @@ -60,8 +102,9 @@ pub struct Scanner { /// Content sources sources: Sources, - /// The walker to detect all files that we have to scan - walker: Option, + /// The walkers to detect all files that we have to scan: one for all auto/external source + /// roots, and one per pattern source base + walkers: Vec, /// All found extensions extensions: FxHashSet, @@ -111,11 +154,11 @@ impl Scanner { } } - let walker = create_walker(&sources); + let walkers = create_walkers(&sources); Self { sources, - walker, + walkers, ..Default::default() } } @@ -174,7 +217,7 @@ impl Scanner { // Figure out if the new unknown files are allowed to be scanned if !new_unknown_files.is_empty() { - if let Some(walk_builder) = &mut self.walker { + 'outer: for walk_builder in self.walkers.iter_mut() { for entry in walk_builder.build().filter_map(Result::ok) { let path = entry.path(); if !path.is_file() { @@ -219,7 +262,7 @@ impl Scanner { // We can stop walking the file system if all files we are interested in have // been found. if new_unknown_files.is_empty() { - break; + break 'outer; } } } @@ -379,17 +422,20 @@ impl Scanner { } self.sources_scanned = true; - let Some(walker) = &mut self.walker else { + if self.walkers.is_empty() { return (vec![], vec![], vec![]); - }; + } - // Use synchronous walk for the initial build (lower overhead) and parallel - // walk for subsequent calls (watch mode) where the overhead is amortised. - let all_entries = if self.has_scanned_once { - walk_parallel(walker) - } else { - walk_synchronous(walker) - }; + // Use synchronous walks for the initial build (lower overhead) and parallel + // walks for subsequent calls (watch mode) where the overhead is amortised. + let mut all_entries = vec![]; + for walker in self.walkers.iter_mut() { + if self.has_scanned_once { + all_entries.extend(walk_parallel(walker)); + } else { + all_entries.extend(walk_synchronous(walker)); + } + } let mut css_files: Vec = vec![]; let mut content_paths: Vec<(PathBuf, String)> = vec![]; @@ -707,75 +753,43 @@ fn walk_parallel(walker: &mut WalkBuilder) -> Vec { Arc::try_unwrap(collected).unwrap().into_inner().unwrap() } -/// Sets up a WalkBuilder with all source roots, gitignore rules, and source pattern matching. +/// Sets up all walkers for the given sources: one gitignore-aware walker for all `Auto` and +/// `External` roots, and one walker per `Pattern` base. +fn create_walkers(sources: &Sources) -> Vec { + let mut walkers = vec![]; + walkers.extend(create_auto_walker(sources)); + walkers.extend(create_pattern_walkers(sources)); + walkers +} + +/// A single walker for all `Auto` and `External` roots. /// -/// This is the common setup shared between the full walker (with mtime tracking for re-scans) -/// and the parallel walker (without mtime tracking for the initial scan). -fn create_walker(sources: &Sources) -> Option { - let mut other_roots: FxHashSet<&PathBuf> = FxHashSet::default(); - let mut first_root: Option<&PathBuf> = None; +/// The walker respects `.gitignore` files (also from parent directories, up to the git +/// repository root) plus a stack of explicit in-memory gitignores. Explicit gitignores rank +/// above the ignore files found on disk, and later-registered ones above earlier ones (see the +/// `CHANGED:` annotations in the vendored `ignore` crate). From low to high precedence: +/// +/// 1. The default auto source detection rules +/// 2. Per `@source` directive, in order: +/// - `External`: a `!/**/*` whitelist at its base (bypassing all gitignore rules), plus a +/// re-statement of the default rules so that e.g. nested `node_modules` stay ignored +/// - `Ignored` (`@source not`): its pattern at its base +fn create_auto_walker(sources: &Sources) -> Option { + let mut roots = sources.iter().filter_map(|source| match source { + SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base), + _ => None, + }); - let mut ignores: Vec<(&PathBuf, Vec)> = Default::default(); - let mut emit = |base, pattern| match ignores.last_mut() { - Some((prev_base, patterns)) if *prev_base == base => { - patterns.push(pattern); - } - _ => { - ignores.push((base, vec![pattern])); - } - }; + let first_root = roots.next()?; - for source in sources.iter() { - match source { - SourceEntry::Auto { base } => { - if first_root.is_none() { - first_root = Some(base); - } else { - other_roots.insert(base); - } - } - SourceEntry::Pattern { base, pattern } => { - let pattern = pattern.to_owned(); - - if first_root.is_none() { - first_root = Some(base); - } else { - other_roots.insert(base); - } - - if !pattern.contains("**") { - // Specific patterns should take precedence even over git-ignored files: - emit(base, format!("!{}", pattern)); - } else { - // Assumption: the pattern we receive will already be brace expanded. So - // `*.{html,jsx}` will result in two separate patterns: `*.html` and `*.jsx`. - if let Some(extension) = Path::new(&pattern).extension() { - // Extend auto source detection to include the extension - emit(base, format!("!*.{}", extension.to_string_lossy())); - } - } - } - SourceEntry::Ignored { base, pattern } => { - emit(base, pattern.to_owned()); - } - SourceEntry::External { base } => { - if first_root.is_none() { - first_root = Some(base); - } else { - other_roots.insert(base); - } - - // External sources should take precedence even over git-ignored files: - emit(base, "!/**/*".to_owned()); - - // External sources should still disallow binary extensions: - emit(base, BINARY_EXTENSIONS_GLOB.clone()); - } + let mut builder = WalkBuilder::new(first_root); + let mut seen_roots = FxHashSet::from_iter([first_root]); + for root in roots { + if seen_roots.insert(root) { + builder.add(root); } } - let mut builder = WalkBuilder::new(first_root?); - // We have to follow symlinks builder.follow_links(true); @@ -820,92 +834,302 @@ fn create_walker(sources: &Sources) -> Option { // - my-project/apps/.gitignore // // Setting the require_git(true) flag conditionally allows us to do this. - for parent in first_root?.ancestors() { + for parent in first_root.ancestors() { if parent.join(".git").exists() { builder.require_git(true); break; } } - for root in other_roots { - builder.add(root); - } - // Setup auto source detection rules for ignore in auto_source_detection::RULES.iter() { builder.add_gitignore(ignore.clone()); } - // Setup ignores based on `@source` definitions - for (base, patterns) in ignores { - let mut ignore_builder = GitignoreBuilder::new(base); - for pattern in patterns { - ignore_builder.add_line(None, &pattern).unwrap(); + // Setup ignores based on `@source` definitions, in directive order so later directives win + for source in sources.iter() { + match source { + SourceEntry::External { base } => { + // External sources bypass all gitignore rules (the directory was explicitly + // listed even though it is ignored)… + let mut ignore_builder = GitignoreBuilder::new(base); + ignore_builder.add_line(None, "!/**/*").unwrap(); + + // …but the default auto source detection rules still apply inside of them, so + // nested `node_modules`, ignored extensions, and lock files stay ignored. + ignore_builder + .add_line(None, &auto_source_detection::IGNORED_CONTENT_DIRS_GLOB) + .unwrap(); + ignore_builder + .add_line(None, &auto_source_detection::IGNORED_EXTENSIONS_GLOB) + .unwrap(); + ignore_builder + .add_line(None, &auto_source_detection::IGNORED_FILES_GLOB) + .unwrap(); + builder.add_gitignore(ignore_builder.build().unwrap()); + + // Binary extensions are ignored as well, but only for files, so that a folder + // named e.g. `some.pages` is still scanned. + let mut ignore_builder = GitignoreBuilder::new(base); + ignore_builder + .only_on_files(true) + .add_line(None, &BINARY_EXTENSIONS_GLOB) + .unwrap(); + builder.add_gitignore(ignore_builder.build().unwrap()); + } + SourceEntry::Ignored { base, pattern } => { + let mut ignore_builder = GitignoreBuilder::new(base); + ignore_builder.add_line(None, pattern).unwrap(); + builder.add_gitignore(ignore_builder.build().unwrap()); + } + _ => {} } - let ignore = ignore_builder.build().unwrap(); - builder.add_gitignore(ignore); } - // Pre-compute source matching data to avoid allocations in the hot filter_entry path - let auto_bases: Vec = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base.clone()), - _ => None, - }) - .collect(); - - let pattern_sources: Vec<(PathBuf, String)> = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Pattern { base, pattern } => Some((base.into(), pattern.into())), - _ => None, - }) - .collect(); - - // Source pattern matching filter (lock-free, safe for parallel walking) - builder.filter_entry(move |entry| { - let path = entry.path(); - - // Ensure the entries are matching any of the provided source patterns (this is - // necessary for manual-patterns that can filter the file extension) - if path.is_file() { - let mut matches = false; - - for base in &auto_bases { - if path.starts_with(base) { - matches = true; - break; - } - } - - if !matches { - for (base, pattern) in &pattern_sources { - let remainder = path.strip_prefix(base); - if remainder.is_ok_and(|remainder| { - let mut path_str = remainder.to_string_lossy().to_string(); - if !path_str.starts_with("/") { - path_str = format!("/{path_str}"); - } - glob_match(pattern, path_str.as_bytes()) - }) { - matches = true; - break; - } - } - } - - if !matches { - return false; - } - } - - true - }); - Some(builder) } +/// A glob pattern of a `Pattern` source, together with the position of its `@source` directive. +#[derive(Debug, Clone)] +struct PatternRule { + /// Position of the `@source` directive, used to resolve conflicts with `@source not` + /// directives: the later directive wins. + idx: usize, + + /// The glob pattern, relative to the walker's base, e.g. `/ba*/*.html` + pattern: String, + + /// Whether the pattern pins a specific extension (e.g. `*.html` or `logo.png`). Patterns + /// that don't (e.g. `blog/*/**/*`) re-apply the default extension rules. + pins_extension: bool, +} + +/// An `@source not` directive, together with its position. +#[derive(Debug, Clone)] +struct NotRule { + /// Position of the `@source not` directive + idx: usize, + + base: PathBuf, + + /// The glob pattern, relative to `base`, e.g. `/ignored/**/*` + pattern: String, +} + +impl NotRule { + /// Whether this directive excludes the given file. + fn matches_file(&self, path: &Path) -> bool { + let Ok(remainder) = path.strip_prefix(&self.base) else { + return false; + }; + glob_match(&self.pattern, rooted_posix(remainder).as_bytes()) + } + + /// Whether this directive excludes the entire directory (and everything inside). + fn covers_dir(&self, path: &Path) -> bool { + // Directory-shaped `@source not "./some/dir"` directives are normalized to a `/**/*` + // pattern with the directory as its base. + self.pattern == "/**/*" && path.starts_with(&self.base) + } +} + +/// Serialize a path relative to some base as a `/`-rooted posix style string, e.g. +/// `/ba*/index.html`, matching how source patterns are stored. +fn rooted_posix(path: &Path) -> String { + let posix = crate::scanner::sources::path_to_posix_string(path); + if posix.starts_with('/') { + posix + } else { + format!("/{posix}") + } +} + +/// Whether a directory (relative to the pattern's base) can contain files matching the pattern. +/// Used to prune directories that can never contribute, e.g. for `/ba*/*.html` only `ba*` +/// directories are entered. +fn dir_could_contain_matches(pattern: &str, dir: &Path) -> bool { + let pattern_components: Vec<&str> = pattern + .trim_start_matches('/') + .split('/') + .filter(|c| !c.is_empty()) + .collect(); + + for (i, component) in dir.components().enumerate() { + let component = component.as_os_str().to_string_lossy(); + + // Once we see a `**` everything nested can contain matches + match pattern_components.get(i) { + Some(&"**") => return true, + // The last pattern component matches files, not directories. A directory nested + // deeper than the pattern's directory part can never contain matches. + Some(_) if i + 1 >= pattern_components.len() => return false, + Some(pattern_component) => { + if !glob_match(pattern_component, component.as_bytes()) { + return false; + } + } + None => return false, + } + } + + true +} + +/// One walker per `Pattern` base. +/// +/// The static base of a glob is the explicit part: it is used as the walk root, so `.gitignore` +/// files *above* it never apply (`parents(false)`), even when the base is hidden inside an +/// ignored directory. The wildcard part is not explicit: +/// +/// - `.gitignore` files inside the subtree still prune directories +/// - the default rules still prune directories (`node_modules` etc., unless the base itself +/// points inside of one) +/// +/// Files matching a glob are whitelisted explicitly (a glob match beats file-level gitignore +/// and default rules). The filter closure then makes the exact per-file decision. +fn create_pattern_walkers(sources: &Sources) -> Vec { + // Collect all `@source not` directives with their positions + let nots: Vec = sources + .iter() + .enumerate() + .filter_map(|(idx, source)| match source { + SourceEntry::Ignored { base, pattern } => Some(NotRule { + idx, + base: base.clone(), + pattern: pattern.clone(), + }), + _ => None, + }) + .collect(); + + // Group patterns by base, preserving directive order + let mut bases: Vec = vec![]; + let mut patterns_by_base: FxHashMap> = FxHashMap::default(); + for (idx, source) in sources.iter().enumerate() { + let SourceEntry::Pattern { base, pattern } = source else { + continue; + }; + + patterns_by_base + .entry(base.clone()) + .or_insert_with(|| { + bases.push(base.clone()); + vec![] + }) + .push(PatternRule { + idx, + pattern: pattern.clone(), + pins_extension: pattern_pins_extension(pattern), + }); + } + + bases + .into_iter() + .map(|base| { + let patterns = patterns_by_base.remove(&base).unwrap(); + + let mut builder = WalkBuilder::new(&base); + + // We have to follow symlinks + builder.follow_links(true); + + // Scan hidden files / directories + builder.hidden(false); + + // Don't respect global gitignore files + builder.git_global(false); + + // The static base is explicit: `.gitignore` files above it do not apply + builder.parents(false); + + // Apply `.gitignore` files inside the subtree regardless of whether a `.git` + // directory is present + builder.require_git(false); + + // The default rules prune directories (`node_modules`, …). Their file-level rules + // are rescued by the whitelists below when the glob matches. + for ignore in auto_source_detection::RULES.iter() { + builder.add_gitignore(ignore.clone()); + } + + // Whitelist the patterns themselves so that matching files win from file-level + // gitignore rules and the default rules. Restricted to files: the wildcard part of + // a pattern must not re-include ignored directories. + let mut ignore_builder = GitignoreBuilder::new(&base); + ignore_builder.only_on_files(true); + for pattern in &patterns { + ignore_builder + .add_line(None, &format!("!{}", pattern.pattern)) + .unwrap(); + } + builder.add_gitignore(ignore_builder.build().unwrap()); + + // The exact per-file decision (lock-free, safe for parallel walking) + let filter_base = base.clone(); + let filter_nots = nots.clone(); + builder.filter_entry(move |entry| { + // Always keep the walk root itself + if entry.depth() == 0 { + return true; + } + + let path = entry.path(); + let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); + + let Ok(remainder) = path.strip_prefix(&filter_base) else { + return false; + }; + + if is_dir { + // Only descend when some pattern can match files inside this directory… + let relevant = patterns + .iter() + .filter(|p| dir_could_contain_matches(&p.pattern, remainder)) + .collect::>(); + if relevant.is_empty() { + return false; + } + + // …and no later `@source not` directive excludes the directory for all of + // those patterns. When a pattern comes after the `not`, keep walking; the + // file-level check below resolves the conflict exactly. + return !filter_nots.iter().any(|not| { + not.covers_dir(path) && relevant.iter().all(|p| p.idx < not.idx) + }); + } + + let rel = rooted_posix(remainder); + patterns.iter().any(|p| { + glob_match(&p.pattern, rel.as_bytes()) + && !filter_nots + .iter() + .any(|not| not.idx > p.idx && not.matches_file(path)) + && (p.pins_extension || !is_ignored_by_default_file_rules(path)) + }) + }); + + builder + }) + .collect() +} + +/// Whether a pattern pins a specific extension, e.g. `/*.html` or `/logo.png`. Patterns that +/// don't (e.g. `/blog/*/**/*`) keep the default extension rules applied. +fn pattern_pins_extension(pattern: &str) -> bool { + match Path::new(pattern).extension().and_then(|ext| ext.to_str()) { + Some(ext) => !ext.contains(['*', '?', '[']), + None => false, + } +} + +/// Whether a file is ignored by the default file-level rules (binary extensions, ignored +/// extensions, lock files, …). +fn is_ignored_by_default_file_rules(path: &Path) -> bool { + auto_source_detection::RULES + .iter() + .any(|ignore| ignore.matched(path, false).is_ignore()) +} + #[cfg(test)] mod tests { use super::{ChangedContent, Scanner}; diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 99a4a5ec2..49619d41f 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -1,6 +1,6 @@ use crate::GlobEntry; use bexpand::Expression; -use fxhash::{FxHashMap, FxHashSet}; +use fxhash::FxHashMap; use ignore::gitignore::Gitignore; use std::path::{Component, Path, PathBuf}; use tracing::{event, Level}; @@ -77,164 +77,6 @@ impl Sources { } } -/// When dealing with a pattern, then it could be that we end up with: -/// -/// ```json -/// { base: '/some/folder', pattern: 'foo.ts' } -/// ``` -/// -/// If we just emit `!foo.ts` for the `/some/folder` path, then _everything_ else in that folder -/// would still be walked (but the result will be ignored). -/// -/// Instead, we have to ensure that we ignore everything in that folder _except_ for the `foo.ts` -/// pattern. -/// -/// This should be equivalent to: -/// ```gitignore -/// * -/// !foo.ts -/// ``` -/// -/// However, we have to be careful that we don't start ignoring files/folders that already exist. -/// ```css -/// @source "./some/folder/foo.ts"; -/// @source "./some/folder/bar.ts"; -/// ``` -/// Would result in: -/// ```json -/// { base: '/some/folder', pattern: 'foo.ts' } -/// { base: '/some/folder', pattern: 'bar.ts' } -/// ``` -/// -/// If we were to blindly emit `*` for each pattern, then the `.gitignore` equivalent would look like -/// this: -/// ```gitignore -/// * -/// !foo.ts -/// * -/// !bar.ts -/// ``` -/// -/// This would result in ignoring the `foo.ts` file as well. Therefore we only want to insert -/// this `*` pattern when nothing else exists yet. -/// -/// There is another problem that we need to solve. Let's say you have a pattern that contains a `*` -/// in the pattern: -/// ```css -/// @source './src/ba*/*.html'; -/// ``` -/// -/// This would result in -/// ```json -/// { base: '/src', pattern: '/ba*/*.html' } -/// ``` -/// -/// If we now inject the `*` pattern for the `/src` folder, then we wouldn't scan any `ba*` folders -/// (e.g. `bar` or `baz`). For this, we have to make sure that we add inverse patterns for these -/// folders. This would essentially result in: -/// ```gitignore -/// * ← ignore everything -/// !/ba*/ ← except for the `ba*/` pattern, so we scan these folders -/// !/ba*/*.html ← then ensure we scan the `*.html` files in it as well -/// ``` -/// -fn expand_restricted_patterns(sources: Vec) -> Vec { - let unrestricted_roots = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base.clone()), - SourceEntry::Pattern { base, pattern } if pattern.contains("**") => Some(base.clone()), - _ => None, - }) - .collect::>(); - - // Bases of restricted patterns. Each of these becomes its own walk root with its own - // `*` + `!` rules, so an ancestor base must not ignore them recursively. - let pattern_roots = sources - .iter() - .filter_map(|source| match source { - SourceEntry::Pattern { base, .. } => Some(base.clone()), - _ => None, - }) - .collect::>(); - - let mut restricted_roots: FxHashSet = FxHashSet::default(); - let mut expanded = vec![]; - - for source in sources { - let SourceEntry::Pattern { base, pattern } = &source else { - expanded.push(source); - continue; - }; - - // `base` is already included by another `@source` that we know should be walked. This - // includes the case where `base` is _nested_ inside such a root, because everything under - // an unrestricted root is walked already. Restricting it would incorrectly hide siblings - // that the broader source is supposed to pick up. - if unrestricted_roots.iter().any(|root| base.starts_with(root)) { - expanded.push(source); - continue; - } - - // Ignore everything in the directory. We will later add the specific patterns we are - // interested in. - if restricted_roots.insert(base.clone()) { - // When another source root is nested inside this base — an unrestricted root, or the - // base of another restricted pattern (which is walked from its own root with its own - // rules) — only ignore direct children so the nested root can still be walked. - let has_nested_root = unrestricted_roots.iter().any(|root| root.starts_with(base)) - || pattern_roots - .iter() - .any(|root| root != base && root.starts_with(base)); - - let pattern = if has_nested_root { "/*" } else { "*" }; - - expanded.push(SourceEntry::Ignored { - base: base.clone(), - pattern: pattern.to_owned(), - }); - } - - // Ensure to _include_ parent paths, otherwise the `*` from above would block walking the - // folders that need to be walked. - // - // ```css - // @source './src/ba*/*.html'; - // ``` - // - // ```gitignore - // * ← added by the above rule - // !/ba*/ ← this is what we're focusing on in this block - // !/ba*/*.html ← this is added later - // ``` - { - let mut dir = PathBuf::new(); - let mut components = Path::new(pattern).components().peekable(); - - while let Some(component) = components.next() { - if components.peek().is_none() { - break; - } - - match component { - Component::Prefix(_) | Component::RootDir | Component::CurDir => continue, - Component::ParentDir | Component::Normal(_) => dir.push(component), - } - - expanded.push(SourceEntry::Ignored { - base: base.clone(), - pattern: format!("!/{}/", path_to_posix_string(&dir).trim_start_matches('/')), - }); - } - } - - // Track the original source - expanded.push(source); - } - - expanded -} - impl PublicSourceEntry { /// Optimize the PublicSourceEntry by trying to move all the static parts of the pattern to the /// base of the PublicSourceEntry. @@ -343,7 +185,7 @@ impl PublicSourceEntry { } } -fn path_to_posix_string(path: &Path) -> String { +pub(crate) fn path_to_posix_string(path: &Path) -> String { let mut parts = Vec::new(); let mut is_rooted = false; @@ -474,65 +316,7 @@ mod tests { } #[test] - fn concrete_patterns_are_expanded_to_restrict_their_base() { - let dir = tempdir().unwrap(); - fs::create_dir_all(dir.path().join("src")).unwrap(); - let base = dunce::canonicalize(dir.path().join("src")).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![PublicSourceEntry { - base: dir.path().to_string_lossy().to_string(), - pattern: "src/foo.html".to_string(), - negated: false, - }]); - - assert_eq!( - sources, - vec![ - SourceEntry::Ignored { - base: base.clone(), - pattern: "*".to_string(), - }, - SourceEntry::Pattern { - base, - pattern: "/foo.html".to_string(), - }, - ] - ); - } - - #[test] - fn restricted_patterns_include_parent_directory_allow_rules() { - let dir = tempdir().unwrap(); - fs::create_dir_all(dir.path().join("src")).unwrap(); - let base = dunce::canonicalize(dir.path().join("src")).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![PublicSourceEntry { - base: dir.path().to_string_lossy().to_string(), - pattern: "src/ef*/*.html".to_string(), - negated: false, - }]); - - assert_eq!( - sources, - vec![ - SourceEntry::Ignored { - base: base.clone(), - pattern: "*".to_string(), - }, - SourceEntry::Ignored { - base: base.clone(), - pattern: "!/ef*/".to_string(), - }, - SourceEntry::Pattern { - base, - pattern: "/ef*/*.html".to_string(), - }, - ] - ); - } - - #[test] - fn unrestricted_sources_do_not_expand_patterns_for_the_same_base() { + fn sources_are_converted_in_order() { let dir = tempdir().unwrap(); fs::create_dir_all(dir.path().join("src")).unwrap(); let base = dunce::canonicalize(dir.path().join("src")).unwrap(); @@ -548,72 +332,6 @@ mod tests { pattern: "src/foo.html".to_string(), negated: false, }, - ]); - - assert_eq!( - sources, - vec![ - SourceEntry::Auto { base: base.clone() }, - SourceEntry::Pattern { - base, - pattern: "/foo.html".to_string(), - }, - ] - ); - } - - #[test] - fn restricted_parent_bases_do_not_open_unrelated_siblings() { - let dir = tempdir().unwrap(); - let project = dir.path().join("Users").join("robin").join("docus-test"); - fs::create_dir_all(&project).unwrap(); - - let users = dunce::canonicalize(dir.path().join("Users")).unwrap(); - let project = dunce::canonicalize(project).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![ - PublicSourceEntry { - base: project.to_string_lossy().to_string(), - pattern: "**/*".to_string(), - negated: false, - }, - PublicSourceEntry { - base: project.to_string_lossy().to_string(), - pattern: "../../app.config.ts".to_string(), - negated: false, - }, - ]); - - assert_eq!( - sources, - vec![ - SourceEntry::Auto { - base: project.clone(), - }, - SourceEntry::Ignored { - base: users.clone(), - pattern: "/*".to_string(), - }, - SourceEntry::Pattern { - base: users, - pattern: "/app.config.ts".to_string(), - }, - ] - ); - } - - #[test] - fn restricted_patterns_preserve_source_order() { - let dir = tempdir().unwrap(); - fs::create_dir_all(dir.path().join("src")).unwrap(); - let base = dunce::canonicalize(dir.path().join("src")).unwrap(); - - let sources = public_source_entries_to_private_source_entries(vec![ - PublicSourceEntry { - base: dir.path().to_string_lossy().to_string(), - pattern: "src/foo.html".to_string(), - negated: false, - }, PublicSourceEntry { base: dir.path().to_string_lossy().to_string(), pattern: "src/foo.html".to_string(), @@ -624,10 +342,7 @@ mod tests { assert_eq!( sources, vec![ - SourceEntry::Ignored { - base: base.clone(), - pattern: "*".to_string(), - }, + SourceEntry::Auto { base: base.clone() }, SourceEntry::Pattern { base: base.clone(), pattern: "/foo.html".to_string(), @@ -873,7 +588,7 @@ pub fn public_source_entries_to_private_source_entries( }) .collect::>(); - expand_restricted_patterns(sources) + sources } /// Convert a public source entry to a source entry diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index c30327bdd..764ad5a12 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -3947,6 +3947,340 @@ mod scanner { ]); } + // https://github.com/tailwindlabs/tailwindcss/issues/18870 + #[test] + fn glob_sources_can_descend_into_directories_ignored_from_within() { + // Laravel's `storage/` directories ship a `.gitignore` that ignores everything inside + // (`*` + `!.gitignore`). An explicit glob through such a directory should still find + // matching files, but nothing else. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ("storage/.gitignore", "*\n!.gitignore"), + ( + "storage/cms/section.html", + "content-['storage/cms/section.html']", + ), + ( + "storage/cms/nested/deep.html", + "content-['storage/cms/nested/deep.html']", + ), + ("storage/cms/data.json", "content-['storage/cms/data.json']"), + ("storage/other/other.html", "content-['storage/other.html']"), + ], + vec!["@source './storage/cms/**/*.html'"], + ); + + assert_eq!( + candidates, + vec![ + "content-['storage/cms/nested/deep.html']", + "content-['storage/cms/section.html']", + ] + ); + assert_eq!( + files, + vec!["storage/cms/nested/deep.html", "storage/cms/section.html"] + ); + } + + // https://github.com/tailwindlabs/tailwindcss/issues/18870 + #[test] + fn glob_sources_can_descend_into_directories_ignored_by_an_ancestor() { + // Same as above, but the ignore comes from an ancestor `.gitignore` rather than one + // inside the ignored directory itself. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "/storage"), + ( + "storage/cms/section.html", + "content-['storage/cms/section.html']", + ), + ( + "storage/cms/nested/deep.html", + "content-['storage/cms/nested/deep.html']", + ), + ("storage/cms/script.js", "content-['storage/cms/script.js']"), + ], + vec!["@source './storage/cms/**/*.html'"], + ); + + assert_eq!( + candidates, + vec![ + "content-['storage/cms/nested/deep.html']", + "content-['storage/cms/section.html']", + ] + ); + assert_eq!( + files, + vec!["storage/cms/nested/deep.html", "storage/cms/section.html"] + ); + } + + #[test] + fn glob_sources_inside_ignored_directories_do_not_include_other_extensions() { + // `@source "./foo/*.html"` where `foo` is git ignored: only the `.html` files were asked + // for. Other files in `foo` must not be scanned, even when a broader auto source exists. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "/foo"), + ("foo/index.html", "content-['foo/index.html']"), + ("foo/script.js", "content-['foo/script.js']"), + ( + "foo/nested/nested.html", + "content-['foo/nested/nested.html']", + ), + ("index.html", "content-['index.html']"), + ], + vec!["@source '**/*'", "@source './foo/*.html'"], + ); + + assert_eq!( + candidates, + vec!["content-['foo/index.html']", "content-['index.html']"] + ); + assert_eq!(files, vec!["foo/index.html", "index.html"]); + } + + // https://github.com/tailwindlabs/tailwindcss/pull/20406 + #[test] + fn concrete_file_sources_behind_node_modules_do_not_include_siblings() { + // A concrete file `@source` pointing into a default-ignored directory (`node_modules`) + // must only include that file, not its siblings, even when the project root is an auto + // source. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ( + "node_modules/.generated/ui/button.ts", + "content-['button.ts']", + ), + ("node_modules/.generated/ui/card.ts", "content-['card.ts']"), + ("index.html", "content-['index.html']"), + ], + vec![ + "@source '**/*'", + "@source './node_modules/.generated/ui/button.ts'", + ], + ); + + assert_eq!( + candidates, + vec!["content-['button.ts']", "content-['index.html']"] + ); + assert_eq!( + files, + vec!["index.html", "node_modules/.generated/ui/button.ts"] + ); + } + + // https://github.com/tailwindlabs/tailwindcss/pull/20406 + #[test] + fn concrete_file_sources_behind_gitignored_directories_do_not_include_siblings() { + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "/generated"), + ("generated/button.ts", "content-['button.ts']"), + ("generated/card.ts", "content-['card.ts']"), + ("index.html", "content-['index.html']"), + ], + vec!["@source '**/*'", "@source './generated/button.ts'"], + ); + + assert_eq!( + candidates, + vec!["content-['button.ts']", "content-['index.html']"] + ); + assert_eq!(files, vec!["generated/button.ts", "index.html"]); + } + + #[test] + fn glob_sources_do_not_reopen_node_modules() { + // A glob source respects the default auto source detection rules: `node_modules` inside + // the globbed tree stays pruned unless it is targeted explicitly. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ("src/index.html", "content-['src/index.html']"), + ( + "src/node_modules/lib/index.html", + "content-['src/node_modules/lib/index.html']", + ), + ], + vec!["@source './src/**/*.html'"], + ); + + assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_eq!(files, vec!["src/index.html"]); + } + + #[test] + fn wildcards_in_glob_sources_do_not_descend_into_gitignored_directories() { + // The static part of a glob source is the "explicit" part: it bypasses ignore rules. + // The wildcard part does not: `@source "./**/*.html"` does not descend into a git + // ignored directory — you weren't explicit about it. To opt in, name the directory: + // `@source "./dist/**/*.html"`. + // + // Files are different: a git ignored *file* that matches the glob is still included + // (you asked for all `.html` files). + let paths_with_content = &[ + (".gitignore", "dist/\nignored.html"), + ("src/index.html", "content-['src/index.html']"), + ("dist/index.html", "content-['dist/index.html']"), + ("ignored.html", "content-['ignored.html']"), + ]; + + let ScanResult { + candidates, files, .. + } = scan_with_globs(paths_with_content, vec!["@source './**/*.html'"]); + + assert_eq!( + candidates, + vec!["content-['ignored.html']", "content-['src/index.html']"] + ); + assert_eq!(files, vec!["ignored.html", "src/index.html"]); + + // Being explicit about the ignored directory bypasses the ignore rule. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + paths_with_content, + vec!["@source './**/*.html'", "@source './dist/**/*.html'"], + ); + + assert_eq!( + candidates, + vec![ + "content-['dist/index.html']", + "content-['ignored.html']", + "content-['src/index.html']" + ] + ); + assert_eq!( + files, + vec!["dist/index.html", "ignored.html", "src/index.html"] + ); + } + + #[test] + fn glob_sources_only_rescue_matching_files_from_ignored_directories() { + // When a glob source forces the walker into a git ignored directory, files that do not + // match the glob must stay excluded, even though they are only ignored "via" their + // parent directory. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "gen/"), + ("gen/a.html", "content-['gen/a.html']"), + ("gen/b.js", "content-['gen/b.js']"), + ("index.html", "content-['index.html']"), + ], + vec!["@source '**/*'", "@source './gen/**/*.html'"], + ); + + assert_eq!( + candidates, + vec!["content-['gen/a.html']", "content-['index.html']"] + ); + assert_eq!(files, vec!["gen/a.html", "index.html"]); + } + + #[test] + fn unpinned_extension_globs_apply_default_extension_rules() { + // A glob that doesn't pin an extension (`**/*`-style tail after a wildcard directory) + // still applies the default extension rules, so binary files are not included. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ("blog/2024/post/index.html", "content-['index.html']"), + ("blog/2024/post/image.png", "content-['image.png']"), + ("blog/2024/post/styles.scss", "content-['styles.scss']"), + ], + vec!["@source './blog/*/post/**/*'"], + ); + + assert_eq!(candidates, vec!["content-['index.html']"]); + assert_eq!(files, vec!["blog/2024/post/index.html"]); + } + + #[test] + fn external_sources_ignore_nested_default_ignored_directories() { + // An explicitly listed, git ignored directory (external source) still applies the + // default auto source detection rules inside of it: nested `node_modules` are not + // scanned. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "node_modules"), + ( + "node_modules/my-lib/dist/index.html", + "content-['node_modules/my-lib/dist/index.html']", + ), + ( + "node_modules/my-lib/node_modules/dep/index.html", + "content-['node_modules/my-lib/node_modules/dep/index.html']", + ), + ( + "node_modules/my-lib/logo.png", + "content-['node_modules/my-lib/logo.png']", + ), + ], + vec!["@source './node_modules/my-lib'"], + ); + + assert_eq!( + candidates, + vec!["content-['node_modules/my-lib/dist/index.html']"] + ); + assert_eq!(files, vec!["node_modules/my-lib/dist/index.html"]); + } + + #[test] + fn later_pattern_sources_override_earlier_not_sources() { + // Order matters: a later positive `@source` wins from an earlier `@source not`, and + // vice versa. + let ScanResult { candidates, .. } = scan_with_globs( + &[ + ("src/keep.html", "content-['src/keep.html']"), + ("src/other.html", "content-['src/other.html']"), + ], + vec![ + "@source '**/*'", + "@source not './src'", + "@source './src/keep.html'", + ], + ); + + assert_eq!(candidates, vec!["content-['src/keep.html']"]); + + let ScanResult { candidates, .. } = scan_with_globs( + &[ + ("src/keep.html", "content-['src/keep.html']"), + ("src/other.html", "content-['src/other.html']"), + ], + vec![ + "@source '**/*'", + "@source './src/keep.html'", + "@source not './src'", + ], + ); + + assert!(candidates.is_empty()); + } + #[test] fn test_extract_used_css_variables_from_css() { let dir = tempdir().unwrap().into_path(); From cdcbf98744fee85f0c912e4aeda623e357dbbe79 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 18:55:05 +0200 Subject: [PATCH 08/23] questionable test --- crates/oxide/tests/scanner.rs | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 764ad5a12..e318ba69d 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4215,6 +4215,30 @@ mod scanner { assert_eq!(files, vec!["blog/2024/post/index.html"]); } + #[test] + fn unpinned_globs_do_not_reinclude_gitignored_directories() { + // The whitelist of an unpinned glob like `./blog/*/**/*` could also match *directory* + // paths (unlike e.g. `./**/*.html`, which only ever matches files). It must not: the + // wildcard part of a glob is not explicit, so git ignored directories inside the + // walked subtree stay pruned, even when the whitelisted glob happens to match them. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ("blog/.gitignore", "drafts/"), + ("blog/2024/keep.html", "content-['blog/2024/keep.html']"), + ( + "blog/2024/drafts/secret.html", + "content-['blog/2024/drafts/secret.html']", + ), + ], + vec!["@source './blog/*/**/*'"], + ); + + assert_eq!(candidates, vec!["content-['blog/2024/keep.html']"]); + assert_eq!(files, vec!["blog/2024/keep.html"]); + } + #[test] fn external_sources_ignore_nested_default_ignored_directories() { // An explicitly listed, git ignored directory (external source) still applies the From 35d2ea025dc2a532db8c37849cd56bafa0a569f8 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Tue, 11 Aug 2026 20:28:02 +0200 Subject: [PATCH 09/23] simplify `SourceEntry` MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Fold `External { base }` into `Auto { base, external: bool }`. External sources were never a distinct kind of source: they are auto sources whose directory is itself ignored but was explicitly listed (set either directly for `node_modules`-style directories, or by the gitignore promotion walk). Every consumer except the walker-layer emission matched `Auto { base } | External { base }`; those pairings collapse to a single arm, and the promotion becomes setting a field instead of swapping variants. - Delete the unused `SourceEntry` ⇄ `GlobEntry` `From` conversions; the glob accessors build `GlobEntry` values inline and nothing ever converted through these impls. - Document that directory-shaped `@source not` directives are normalized to a `/**/*` pattern (semantically identical), which `NotRule::covers_dir` relies on. - Make the folder check in `From` honest: joining the "/"-pinned pattern onto the base discarded the base entirely (an absolute path replaces it), so the check accidentally tested a nonsense path like `/index.html` against the filesystem root. Strip the leading slash and document that this arm only matters when `optimize()` could not canonicalize the base. --- crates/oxide/src/scanner/mod.rs | 11 +- crates/oxide/src/scanner/sources.rs | 159 ++++++++++++++-------------- 2 files changed, 88 insertions(+), 82 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index 66696e7ec..bd3f24fa5 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -330,7 +330,7 @@ impl Scanner { let mut globs = vec![]; for source in self.sources.iter() { match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => { + SourceEntry::Auto { base, .. } => { globs.extend(resolve_globs( base.to_path_buf(), &self.dirs, @@ -361,7 +361,7 @@ impl Scanner { self.sources .iter() .filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(GlobEntry { + SourceEntry::Auto { base, .. } => Some(GlobEntry { base: base.to_string_lossy().to_string(), pattern: "**/*".to_string(), }), @@ -776,7 +776,7 @@ fn create_walkers(sources: &Sources) -> Vec { /// - `Ignored` (`@source not`): its pattern at its base fn create_auto_walker(sources: &Sources) -> Option { let mut roots = sources.iter().filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base), + SourceEntry::Auto { base, .. } => Some(base), _ => None, }); @@ -849,7 +849,10 @@ fn create_auto_walker(sources: &Sources) -> Option { // Setup ignores based on `@source` definitions, in directive order so later directives win for source in sources.iter() { match source { - SourceEntry::External { base } => { + SourceEntry::Auto { + base, + external: true, + } => { // External sources bypass all gitignore rules (the directory was explicitly // listed even though it is ignored)… let mut ignore_builder = GitignoreBuilder::new(base); diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 49619d41f..14455acd2 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -1,4 +1,3 @@ -use crate::GlobEntry; use bexpand::Expression; use fxhash::FxHashMap; use ignore::gitignore::Gitignore; @@ -29,7 +28,20 @@ pub enum SourceEntry { /// @source "src";` /// @source "src/**/*";` /// ``` - Auto { base: PathBuf }, + /// + /// `external` is set when the directory itself is ignored (by the default rules, e.g. + /// `node_modules`, or by a `.gitignore`) but was explicitly listed anyway: + /// + /// ```css + /// @source "../node_modules/my-lib";` + /// @source "../node_modules/my-lib/**/*";` + /// ``` + /// + /// Being explicit bypasses the ignoredness of the directory: everything inside is scanned + /// as if it were a regular auto source, except that `.gitignore` files no longer apply + /// inside (git ignores the whole tree anyway). The default rules still apply inside, so + /// e.g. nested `node_modules` stay ignored. + Auto { base: PathBuf, external: bool }, /// Explicit source pattern regardless of any auto source detection rules /// @@ -48,18 +60,11 @@ pub enum SourceEntry { /// @source not "src";` /// @source not "src/**/*.html";` /// ``` + /// + /// Note that directory-shaped directives (`@source not "src"`) are normalized to + /// `base: "src", pattern: "/**/*"`, which is semantically identical: everything under the + /// directory is ignored. Ignored { base: PathBuf, pattern: String }, - - /// External sources are directories that are ignored (by us or .gitignore rules), but should be - /// included bypassing the default ignore rules. - /// - /// Represented by: - /// - /// ```css - /// @source "../node_modules/my-lib";` - /// @source "../node_modules/my-lib/**/*";` - /// ``` - External { base: PathBuf }, } #[derive(Debug, Clone, Default)] @@ -342,7 +347,10 @@ mod tests { assert_eq!( sources, vec![ - SourceEntry::Auto { base: base.clone() }, + SourceEntry::Auto { + base: base.clone(), + external: false, + }, SourceEntry::Pattern { base: base.clone(), pattern: "/foo.html".to_string(), @@ -375,7 +383,13 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::Auto { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: false, + } + ); } #[test] @@ -385,7 +399,13 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: true, + } + ); } #[test] @@ -399,7 +419,13 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: true, + } + ); } #[test] @@ -413,7 +439,13 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::External { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: true, + } + ); } #[test] @@ -449,7 +481,13 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::Auto { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: false, + } + ); } } @@ -516,8 +554,12 @@ pub fn public_source_entries_to_private_source_entries( .map(|public_source| { let mut source: SourceEntry = public_source.into(); - // Promote auto-sources to external sources if they were gitignored - if let SourceEntry::Auto { ref base } = source { + // Mark auto sources as external if their directory is gitignored + if let SourceEntry::Auto { + ref base, + external: false, + } = source + { let inside_git_repo = base.ancestors().any(|dir| dir.join(".git").exists()); // Walk up from the folder, applying each `.gitignore` relative to the directory @@ -559,7 +601,10 @@ pub fn public_source_entries_to_private_source_entries( // `.gitignore` ignores it. match gitignore.matched_path_or_any_parents(&base, true) { ignore::Match::Ignore(_) => { - source = SourceEntry::External { base: base.into() }; + source = SourceEntry::Auto { + base: base.into(), + external: true, + }; break; } ignore::Match::Whitelist(_) => break, @@ -601,8 +646,15 @@ impl From for SourceEntry { }; } - let auto = - value.pattern == "/**/*" || PathBuf::from(&value.base).join(&value.pattern).is_dir(); + // After a successful `optimize()` any trailing concrete directory has already been + // hoisted into the base, so a folder source always has the `/**/*` pattern. The + // `is_dir` check only matters when `optimize()` could not canonicalize the base and + // left the entry untouched. Note that the pinned leading `/` has to be stripped, since + // joining an absolute-looking path onto the base would discard the base entirely. + let auto = value.pattern == "/**/*" + || PathBuf::from(&value.base) + .join(value.pattern.trim_start_matches('/')) + .is_dir(); if !auto { return SourceEntry::Pattern { @@ -611,6 +663,8 @@ impl From for SourceEntry { }; } + // A directory inside e.g. `node_modules` is ignored by default, so listing it + // explicitly makes it an external source. let inside_ignored_content_dir = IGNORED_CONTENT_DIRS.iter().any(|dir| { value.base.contains(&format!( "{}{}{}", @@ -622,60 +676,9 @@ impl From for SourceEntry { .ends_with(&format!("{}{}", std::path::MAIN_SEPARATOR, dir)) }); - match inside_ignored_content_dir { - false => SourceEntry::Auto { - base: value.base.into(), - }, - true => SourceEntry::External { - base: value.base.into(), - }, - } - } -} - -impl From for SourceEntry { - fn from(value: GlobEntry) -> Self { - SourceEntry::Pattern { - base: PathBuf::from(value.base), - pattern: value.pattern, - } - } -} - -impl From for GlobEntry { - fn from(value: SourceEntry) -> Self { - match value { - SourceEntry::Auto { base } | SourceEntry::External { base } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: "**/*".into(), - }, - SourceEntry::Pattern { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - SourceEntry::Ignored { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - } - } -} - -impl From<&SourceEntry> for GlobEntry { - fn from(value: &SourceEntry) -> Self { - match value { - SourceEntry::Auto { base } | SourceEntry::External { base } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: "**/*".into(), - }, - SourceEntry::Pattern { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, - SourceEntry::Ignored { base, pattern } => GlobEntry { - base: base.to_string_lossy().into(), - pattern: pattern.clone(), - }, + SourceEntry::Auto { + base: value.base.into(), + external: inside_ignored_content_dir, } } } From fb43708997d0209822f6d0cbcf372dd8f77c76db Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:31:51 +0200 Subject: [PATCH 10/23] add test for wildcard directory source exclusions --- crates/oxide/tests/scanner.rs | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index e318ba69d..66c334d24 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4305,6 +4305,22 @@ mod scanner { assert!(candidates.is_empty()); } + #[test] + fn wildcard_not_sources_exclude_matching_directories() { + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ("src/bar/index.html", "content-['src/bar/index.html']"), + ("src/foo/index.html", "content-['src/foo/index.html']"), + ], + vec!["@source './src/**/*.html'", "@source not './src/ba*'"], + ); + + assert_eq!(candidates, vec!["content-['src/foo/index.html']"]); + assert_eq!(files, vec!["src/foo/index.html"]); + } + #[test] fn test_extract_used_css_variables_from_css() { let dir = tempdir().unwrap().into_path(); From d78a138ca332b8362c1a0e87d0b8719a876ba3bf Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:37:58 +0200 Subject: [PATCH 11/23] exclude whole directories when a `@source not` glob matches them `@source not "./src/ba*"` is normalized to base `src` + pattern `/ba*` (the wildcard can't be hoisted into the base), but the pattern-walker's not-directive check only glob-matched the file path itself: `/ba*` never matches `/bar/index.html` since `*` doesn't cross `/`, and the directory pruning only recognized the normalized `/**/*` shape. So wildcard directory exclusions were silently ignored by pattern sources. Follow gitignore semantics instead: a pattern that matches a directory excludes the whole subtree. `NotRule::matches` now tests the path itself and every ancestor directory up to the directive's base, and replaces both the file-level check and the directory pruning check (the `/**/*` sentinel special case falls out naturally). The auto/external walker already behaved correctly because on-disk gitignore semantics apply there natively. --- crates/oxide/src/scanner/mod.rs | 31 ++++++++++++++++++------------- 1 file changed, 18 insertions(+), 13 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index bd3f24fa5..2d9f58f77 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -920,19 +920,24 @@ struct NotRule { } impl NotRule { - /// Whether this directive excludes the given file. - fn matches_file(&self, path: &Path) -> bool { + /// Whether this directive excludes the given path: a file, or a directory and thereby + /// everything inside of it. + /// + /// Like a gitignore rule, the pattern excludes a whole subtree when it matches a + /// directory, so besides the path itself every ancestor directory (up to the directive's + /// base) is tested as well. E.g. `@source not "./src/ba*"` excludes `src/bar/index.html` + /// because `/ba*` matches the `src/bar` directory. Note that directory-shaped directives + /// (`@source not "./some/dir"`) are normalized to a `/**/*` pattern with the directory as + /// its base, which matches everything inside the directory directly. + fn matches(&self, path: &Path) -> bool { let Ok(remainder) = path.strip_prefix(&self.base) else { return false; }; - glob_match(&self.pattern, rooted_posix(remainder).as_bytes()) - } - /// Whether this directive excludes the entire directory (and everything inside). - fn covers_dir(&self, path: &Path) -> bool { - // Directory-shaped `@source not "./some/dir"` directives are normalized to a `/**/*` - // pattern with the directory as its base. - self.pattern == "/**/*" && path.starts_with(&self.base) + remainder.ancestors().any(|prefix| { + !prefix.as_os_str().is_empty() + && glob_match(&self.pattern, rooted_posix(prefix).as_bytes()) + }) } } @@ -1096,9 +1101,9 @@ fn create_pattern_walkers(sources: &Sources) -> Vec { // …and no later `@source not` directive excludes the directory for all of // those patterns. When a pattern comes after the `not`, keep walking; the // file-level check below resolves the conflict exactly. - return !filter_nots.iter().any(|not| { - not.covers_dir(path) && relevant.iter().all(|p| p.idx < not.idx) - }); + return !filter_nots + .iter() + .any(|not| not.matches(path) && relevant.iter().all(|p| p.idx < not.idx)); } let rel = rooted_posix(remainder); @@ -1106,7 +1111,7 @@ fn create_pattern_walkers(sources: &Sources) -> Vec { glob_match(&p.pattern, rel.as_bytes()) && !filter_nots .iter() - .any(|not| not.idx > p.idx && not.matches_file(path)) + .any(|not| not.idx > p.idx && not.matches(path)) && (p.pins_extension || !is_ignored_by_default_file_rules(path)) }) }); From 79b2ce7af8d2c6c49dc715175e242a032283d1fc Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:32:35 +0200 Subject: [PATCH 12/23] add test for explicit extensionless file sources --- crates/oxide/tests/scanner.rs | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 66c334d24..b1bf7f857 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4321,6 +4321,19 @@ mod scanner { assert_eq!(files, vec!["src/foo/index.html"]); } + #[test] + fn explicit_file_sources_include_default_ignored_files_without_extensions() { + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[(".env", "content-['.env']"), ("index.html", "content-['index.html']")], + vec!["@source '.env'"], + ); + + assert_eq!(candidates, vec!["content-['.env']"]); + assert_eq!(files, vec![".env"]); + } + #[test] fn test_extract_used_css_variables_from_css() { let dir = tempdir().unwrap().into_path(); From ae45aca2958ce60199e045d63881acf1594e7eeb Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:43:11 +0200 Subject: [PATCH 13/23] always include concrete file sources, even default-ignored ones MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `@source ".env"` scanned nothing: the pattern walker only bypasses the default file rules when the glob pins an extension, and Rust's `Path::extension()` returns `None` for leading-dot names like `.env`, so the file fell through to the default rules — which ignore `.env` — and was dropped. A pattern without any wildcards names a concrete file and is the most explicit source there is, so it now always bypasses the default file rules, extension or not. Extension-pinning globs behave as before, and globs that pin nothing (e.g. `blog/*/**/*`) still apply the default rules. --- crates/oxide/src/scanner/mod.rs | 36 ++++++++++++++++++++++----------- 1 file changed, 24 insertions(+), 12 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index 2d9f58f77..3f87bc0a1 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -46,10 +46,10 @@ use tracing::event; // or a git ignored `dist/`; `@source "./dist/**/*.html"` does descend into `dist/`). // Individual *files* matching the glob are always included, even when git ignored — you were // explicit about wanting files of that shape (`@source "git-ignored.html"` and -// `@source "*.styl"` work). Extensions that are ignored by default are only included when the -// glob pins an extension (`@source "logo.{jpg,png}"`, `@source "do-include-me.bin"`), an -// extension-less glob like `@source "some/folder/**/*"`… is `Auto`, and e.g. -// `@source "blog/*/post/**/*"` still applies the default extension rules. +// `@source "*.styl"` work). Files that are ignored by default are only included when the +// pattern is explicit about them: it names a concrete file (`@source "do-include-me.bin"`, +// `@source ".env"`) or pins an extension (`@source "logo.{jpg,png}"`). A glob that does +// neither (e.g. `@source "blog/*/post/**/*"`) still applies the default file rules. // // - `Ignored`: `@source not "…"` — excludes matching files/folders, even when a `.gitignore` // or another `@source` allows them. @@ -902,9 +902,11 @@ struct PatternRule { /// The glob pattern, relative to the walker's base, e.g. `/ba*/*.html` pattern: String, - /// Whether the pattern pins a specific extension (e.g. `*.html` or `logo.png`). Patterns - /// that don't (e.g. `blog/*/**/*`) re-apply the default extension rules. - pins_extension: bool, + /// Whether the pattern is explicit enough to bypass the default file rules: it names a + /// concrete file (e.g. `.env` or `do-include-me.bin`) or pins a specific extension (e.g. + /// `*.html` or `logo.png`). Patterns that don't (e.g. `blog/*/**/*`) re-apply the default + /// file rules. + bypasses_default_file_rules: bool, } /// An `@source not` directive, together with its position. @@ -1027,7 +1029,7 @@ fn create_pattern_walkers(sources: &Sources) -> Vec { .push(PatternRule { idx, pattern: pattern.clone(), - pins_extension: pattern_pins_extension(pattern), + bypasses_default_file_rules: pattern_bypasses_default_file_rules(pattern), }); } @@ -1112,7 +1114,7 @@ fn create_pattern_walkers(sources: &Sources) -> Vec { && !filter_nots .iter() .any(|not| not.idx > p.idx && not.matches(path)) - && (p.pins_extension || !is_ignored_by_default_file_rules(path)) + && (p.bypasses_default_file_rules || !is_ignored_by_default_file_rules(path)) }) }); @@ -1121,9 +1123,19 @@ fn create_pattern_walkers(sources: &Sources) -> Vec { .collect() } -/// Whether a pattern pins a specific extension, e.g. `/*.html` or `/logo.png`. Patterns that -/// don't (e.g. `/blog/*/**/*`) keep the default extension rules applied. -fn pattern_pins_extension(pattern: &str) -> bool { +/// Whether a pattern is explicit enough to bypass the default file rules. +/// +/// A pattern without any wildcards names a concrete file, e.g. `/.env` or +/// `/do-include-me.bin` — you asked for exactly this file, so the default rules never apply. +/// A pattern that pins a specific extension, e.g. `/*.html` or `/**/*.bin`, bypasses them as +/// well. Patterns that do neither (e.g. `/blog/*/**/*`) keep the default file rules applied. +fn pattern_bypasses_default_file_rules(pattern: &str) -> bool { + // Concrete file, no wildcards (braces have already been expanded away) + if !pattern.contains(['*', '?', '[']) { + return true; + } + + // Pinned extension match Path::new(pattern).extension().and_then(|ext| ext.to_str()) { Some(ext) => !ext.contains(['*', '?', '[']), None => false, From 7de8663522f5c41f79e6025f5db86058619a208a Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:33:12 +0200 Subject: [PATCH 14/23] add test for unreachable gitignore whitelists --- crates/oxide/tests/scanner.rs | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index b1bf7f857..3653d6473 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4334,6 +4334,27 @@ mod scanner { assert_eq!(files, vec![".env"]); } + #[test] + fn unreachable_whitelists_do_not_reinclude_explicit_source_directories() { + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "parent/"), + ("parent/.gitignore", "!child/"), + ("parent/child/.gitignore", "index.html"), + ( + "parent/child/index.html", + "content-['parent/child/index.html']", + ), + ], + vec!["@source './parent/child'"], + ); + + assert_eq!(candidates, vec!["content-['parent/child/index.html']"]); + assert_eq!(files, vec!["parent/child/index.html"]); + } + #[test] fn test_extract_used_css_variables_from_css() { let dir = tempdir().unwrap().into_path(); From 367b59371ffe3334cdee64d20ec3362fa8b01cf2 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:48:44 +0200 Subject: [PATCH 15/23] ignore unreachable gitignore whitelists when promoting sources MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Whether an explicitly listed directory is git ignored (and should bypass the ignore rules as an external source) was decided by walking up from the directory and letting the nearest `.gitignore` with a definitive answer win. That's only half of git's precedence: a whitelist is unreachable when a parent directory is excluded — git never descends into an excluded directory, so re-include rules inside of it have no effect. With .gitignore parent/ parent/.gitignore !child/ `@source "./parent/child"` stayed a regular auto source (the `!child/` whitelist answered first), so the `.gitignore` files inside it kept applying, even though git considers the whole tree ignored. Decide exclusion the way git does instead: walk the path from the top down, settling for every directory along the way whether it is excluded — the first excluded directory makes everything below it ignored. Within a single directory's decision the deepest `.gitignore` still wins, so directories re-included by a deeper, reachable `!dir` pattern stay regular auto sources. Both directions are now pinned by unit tests next to the other promotion tests. --- crates/oxide/src/scanner/sources.rs | 147 ++++++++++++++++------------ 1 file changed, 87 insertions(+), 60 deletions(-) diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 14455acd2..139e4ef24 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -453,10 +453,8 @@ mod tests { let dir = tempdir().unwrap(); fs::create_dir_all(dir.path().join(".git")).unwrap(); fs::write(dir.path().join(".gitignore"), "generated/\n").unwrap(); - // The nearest `.gitignore` with a definitive answer wins: `packages/app/.gitignore` - // re-includes `generated`, overruling the ignore rule in the root `.gitignore`, so the - // directory keeps its regular auto source detection behavior (which respects the - // `.gitignore` rules inside of it) instead of bypassing them as an external source. + // The deepest `.gitignore` with a definitive answer wins: the re-include is reachable + // (no parent directory of `generated` is excluded), so `generated` is not ignored. fs::create_dir_all(dir.path().join("packages").join("app")).unwrap(); fs::write( dir.path().join("packages").join("app").join(".gitignore"), @@ -468,7 +466,36 @@ mod tests { fs::create_dir_all(&base).unwrap(); let base = dunce::canonicalize(&base).unwrap(); - assert_eq!(auto_source_entry(&base), SourceEntry::Auto { base }); + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: false, + } + ); + } + + #[test] + fn folders_inside_excluded_directories_become_external_sources() { + let dir = tempdir().unwrap(); + fs::create_dir_all(dir.path().join(".git")).unwrap(); + fs::write(dir.path().join(".gitignore"), "parent/\n").unwrap(); + // This whitelist is unreachable: `parent` itself is excluded, so git never descends + // into it and the re-include of `child` has no effect. + fs::create_dir_all(dir.path().join("parent")).unwrap(); + fs::write(dir.path().join("parent").join(".gitignore"), "!child/\n").unwrap(); + + let base = dir.path().join("parent").join("child"); + fs::create_dir_all(&base).unwrap(); + let base = dunce::canonicalize(&base).unwrap(); + + assert_eq!( + auto_source_entry(&base), + SourceEntry::Auto { + base, + external: true, + } + ); } #[test] @@ -562,71 +589,71 @@ pub fn public_source_entries_to_private_source_entries( { let inside_git_repo = base.ancestors().any(|dir| dir.join(".git").exists()); - // Walk up from the folder, applying each `.gitignore` relative to the directory - // that contains it (matching git), and stop at the git repository root so - // `.gitignore` files outside of the repo are not considered. + // The chain of directories whose `.gitignore` files can apply: `base` itself and + // its ancestors, up to and including the git repository root so `.gitignore` + // files outside of the repo are not considered. + // + // Without a git repository there is no repository root to stop at. Stop once the + // directory contains the current working directory instead, so `.gitignore` + // files outside of the project (e.g. in the user's home directory) can never + // promote a source to an external source. Note that the file walker still + // applies those `.gitignore` files when deciding which files to scan. + let mut chain: Vec<&Path> = vec![]; for dir in base.ancestors() { - let gitignore = gitignores.entry(dir.to_path_buf()).or_insert_with(|| { - let path = dir.join(".gitignore"); + chain.push(dir); - // `Gitignore::new` roots the matcher at the directory containing the file, - // so patterns match relative to it. - path.is_file().then(|| Gitignore::new(&path).0) - }); - - // Only `.gitignore` files in ancestors of `base` can ignore `base` itself. - // Patterns in `base`'s own `.gitignore` only match paths _inside_ `base`, never - // `base` itself (the file walker still applies them to `base`'s contents). - // - // Skipping `base`'s own `.gitignore` also prevents a false positive for - // whitelist style `.gitignore` files, because relativizing `base` against - // itself yields the empty path, which incorrectly matches `/*`. - // - // E.g.: - // - // ```gitignore - // /* - // !/.gitignore - // !/app - // !/public - // ``` - // - // Everything inside `base` except `.gitignore`, `app` and `public` is ignored, - // but `base` itself is not. - if dir != base { - if let Some(gitignore) = gitignore { - // Match git's precedence: the nearest `.gitignore` with a definitive - // answer wins. A directory that is re-included by a deeper - // `!the-directory` pattern is not ignored, even when an ancestor - // `.gitignore` ignores it. - match gitignore.matched_path_or_any_parents(&base, true) { - ignore::Match::Ignore(_) => { - source = SourceEntry::Auto { - base: base.into(), - external: true, - }; - break; - } - ignore::Match::Whitelist(_) => break, - ignore::Match::None => {} - } - } - } - - // Stop at the git repository root. if dir.join(".git").exists() { break; } - // Without a git repository there is no repository root to stop at. Stop once - // the directory contains the current working directory instead, so `.gitignore` - // files outside of the project (e.g. in the user's home directory) can never - // promote a source to an external source. Note that the file walker still - // applies those `.gitignore` files when deciding which files to scan. if !inside_git_repo && cwd.as_ref().is_some_and(|cwd| cwd.starts_with(dir)) { break; } } + + // Match git's semantics: a directory is ignored when the directory itself or any + // of its parent directories is excluded, and it is not possible to re-include a + // directory once a parent directory is excluded — git never descends into an + // excluded directory, so whitelist rules inside of it are unreachable. + // + // So walk the path from the top down (`chain` is ordered bottom-up: `base` at + // index 0, the boundary last) and decide for every directory along the way + // whether it is excluded. The first excluded directory settles it. For a single + // directory, only `.gitignore` files in its parent directories can match it (its + // own `.gitignore` only matches paths _inside_ of it), and the deepest + // `.gitignore` with a definitive answer wins, so a directory that is re-included + // by a deeper `!the-directory` pattern is not ignored, even when an ancestor + // `.gitignore` ignores it. + 'prefixes: for i in (0..chain.len().saturating_sub(1)).rev() { + let prefix = chain[i]; + + for dir in &chain[i + 1..] { + let gitignore = gitignores.entry(dir.to_path_buf()).or_insert_with(|| { + let path = dir.join(".gitignore"); + + // `Gitignore::new` roots the matcher at the directory containing the + // file, so patterns match relative to it. + path.is_file().then(|| Gitignore::new(&path).0) + }); + + let Some(gitignore) = gitignore else { + continue; + }; + + match gitignore.matched(prefix, true) { + ignore::Match::Ignore(_) => { + source = SourceEntry::Auto { + base: base.into(), + external: true, + }; + break 'prefixes; + } + // Re-included; this directory is reachable, move on to the next one. + ignore::Match::Whitelist(_) => continue 'prefixes, + ignore::Match::None => {} + } + } + } } source From 7536feb8010aceeb10876c552e8459be7a15f5a3 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:33:48 +0200 Subject: [PATCH 16/23] add test for directory source ordering --- crates/oxide/tests/scanner.rs | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 3653d6473..d06d2b95a 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4305,6 +4305,23 @@ mod scanner { assert!(candidates.is_empty()); } + #[test] + fn later_directory_sources_override_earlier_not_sources() { + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[("src/index.html", "content-['src/index.html']")], + vec![ + "@source '**/*'", + "@source not './src'", + "@source './src'", + ], + ); + + assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_eq!(files, vec!["src/index.html"]); + } + #[test] fn wildcard_not_sources_exclude_matching_directories() { let ScanResult { From c0ab06d5151c617a3a4df8efc424d7aa172ea7a7 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 15:57:03 +0200 Subject: [PATCH 17/23] let later directory sources re-include earlier `@source not` exclusions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `@source not "./src"` followed by `@source "./src"` scanned nothing: in the auto/external walker, `@source not` directives were registered as explicit gitignore layers, but plain directory sources emit no layer at all, so there was nothing a later directive could win with — the exclusion applied regardless of order. (A counter-whitelist layer would be wrong too: it would rank above the `.gitignore` files on disk and bypass them inside the re-included directory.) Handle `@source not` in the auto/external walker with a filter closure instead, mirroring how the pattern walkers already resolve ordering: the last directive that covers a path wins. When that is a `@source not` the entry is excluded; when it is a later auto/external source the entry falls through to the normal gitignore + default rules handling, so re-included directories keep their regular auto source semantics. Excluded directories can still be pruned safely, because every auto/external base is its own walk root and stays reachable even when it is nested inside an excluded directory. While moving the exclusion out of the gitignore layers, directory-shaped directives now also have to exclude their base directory itself (the normalized `/**/*` pattern only matches the directory's contents), so the directory is pruned and doesn't widen the generated watch globs. --- crates/oxide/src/scanner/mod.rs | 101 ++++++++++++++++++++++++-------- 1 file changed, 78 insertions(+), 23 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index 3f87bc0a1..93cd7a84e 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -55,7 +55,9 @@ use tracing::event; // or another `@source` allows them. // // Later directives win over earlier ones on conflict: `@source not "./x"` followed by -// `@source "./x/keep.html"` scans `keep.html`, and vice versa excludes it. +// `@source "./x/keep.html"` scans `keep.html`, and vice versa excludes it. The same holds for +// directory sources: `@source not "./x"` followed by `@source "./x"` re-includes `./x` (with +// the normal auto source detection rules applied inside). // // # Implementation // @@ -63,10 +65,11 @@ use tracing::event; // never descend into directories that cannot contribute files). Sources are split over multiple // walkers: // -// - One walker for all `Auto` and `External` roots. Default rules and `@source not` rules are -// registered as explicit in-memory gitignores (they rank above `.gitignore` files found on -// disk, later directives above earlier ones). `External` roots additionally get a `!/**/*` -// whitelist (bypassing gitignore rules) plus a re-statement of the default rules. +// - One walker for all `Auto` and `External` roots. Default rules are registered as explicit +// in-memory gitignores (they rank above `.gitignore` files found on disk). `External` roots +// additionally get a `!/**/*` whitelist (bypassing gitignore rules) plus a re-statement of +// the default rules. `@source not` rules are handled by a filter closure that honors the +// directive order: the last directive covering a path wins. // // - One walker per `Pattern` base. It never looks at `.gitignore` files *above* its base // (the static prefix is explicit), while `.gitignore` files inside the subtree still prune @@ -880,15 +883,57 @@ fn create_auto_walker(sources: &Sources) -> Option { .unwrap(); builder.add_gitignore(ignore_builder.build().unwrap()); } - SourceEntry::Ignored { base, pattern } => { - let mut ignore_builder = GitignoreBuilder::new(base); - ignore_builder.add_line(None, pattern).unwrap(); - builder.add_gitignore(ignore_builder.build().unwrap()); - } _ => {} } } + // `@source not` directives are handled in a filter closure instead of a gitignore layer, + // because their effect depends on the directive order: the last directive that covers a + // path wins. When that is a `@source not`, the entry is excluded; when it is an + // auto/external source, the entry falls through to the normal gitignore + default rules + // handling. E.g.: + // + // ```css + // @source not "./src"; + // @source "./src"; /* re-includes ./src, .gitignore files still apply inside */ + // ``` + // + // Directories excluded here can be pruned safely: every auto/external base is its own walk + // root, so a source root nested inside an excluded directory is still walked. + let includes: Vec<(usize, PathBuf)> = sources + .iter() + .enumerate() + .filter_map(|(idx, source)| match source { + SourceEntry::Auto { base, .. } => Some((idx, base.clone())), + _ => None, + }) + .collect(); + let nots = collect_not_rules(sources); + + builder.filter_entry(move |entry| { + // Always keep the walk roots themselves + if entry.depth() == 0 { + return true; + } + + let path = entry.path(); + + // The last `@source not` directive covering the path… + let Some(not_idx) = nots + .iter() + .filter(|not| not.matches(path)) + .map(|not| not.idx) + .max() + else { + return true; + }; + + // …loses when a later auto/external source covers it as well + includes + .iter() + .any(|(idx, base)| *idx > not_idx && path.starts_with(base)) + }); + Some(builder) } @@ -936,6 +981,12 @@ impl NotRule { return false; }; + // A directory-shaped directive (normalized to a `/**/*` pattern) also excludes the + // base directory itself, not just its contents, so the directory can be pruned. + if remainder.as_os_str().is_empty() { + return self.pattern == "/**/*"; + } + remainder.ancestors().any(|prefix| { !prefix.as_os_str().is_empty() && glob_match(&self.pattern, rooted_posix(prefix).as_bytes()) @@ -943,6 +994,22 @@ impl NotRule { } } +/// Collect all `@source not` directives with their positions. +fn collect_not_rules(sources: &Sources) -> Vec { + sources + .iter() + .enumerate() + .filter_map(|(idx, source)| match source { + SourceEntry::Ignored { base, pattern } => Some(NotRule { + idx, + base: base.clone(), + pattern: pattern.clone(), + }), + _ => None, + }) + .collect() +} + /// Serialize a path relative to some base as a `/`-rooted posix style string, e.g. /// `/ba*/index.html`, matching how source patterns are stored. fn rooted_posix(path: &Path) -> String { @@ -998,19 +1065,7 @@ fn dir_could_contain_matches(pattern: &str, dir: &Path) -> bool { /// Files matching a glob are whitelisted explicitly (a glob match beats file-level gitignore /// and default rules). The filter closure then makes the exact per-file decision. fn create_pattern_walkers(sources: &Sources) -> Vec { - // Collect all `@source not` directives with their positions - let nots: Vec = sources - .iter() - .enumerate() - .filter_map(|(idx, source)| match source { - SourceEntry::Ignored { base, pattern } => Some(NotRule { - idx, - base: base.clone(), - pattern: pattern.clone(), - }), - _ => None, - }) - .collect(); + let nots = collect_not_rules(sources); // Group patterns by base, preserving directive order let mut bases: Vec = vec![]; From 97b7d342f889cc04252134bf523951f1a04032be Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 21:02:54 +0200 Subject: [PATCH 18/23] re-add `External` --- crates/oxide/src/scanner/mod.rs | 15 +++---- crates/oxide/src/scanner/sources.rs | 70 ++++++++++------------------- 2 files changed, 30 insertions(+), 55 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index 93cd7a84e..de1cceca8 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -333,7 +333,7 @@ impl Scanner { let mut globs = vec![]; for source in self.sources.iter() { match source { - SourceEntry::Auto { base, .. } => { + SourceEntry::Auto { base } | SourceEntry::External { base } => { globs.extend(resolve_globs( base.to_path_buf(), &self.dirs, @@ -364,7 +364,7 @@ impl Scanner { self.sources .iter() .filter_map(|source| match source { - SourceEntry::Auto { base, .. } => Some(GlobEntry { + SourceEntry::Auto { base } | SourceEntry::External { base } => Some(GlobEntry { base: base.to_string_lossy().to_string(), pattern: "**/*".to_string(), }), @@ -779,7 +779,7 @@ fn create_walkers(sources: &Sources) -> Vec { /// - `Ignored` (`@source not`): its pattern at its base fn create_auto_walker(sources: &Sources) -> Option { let mut roots = sources.iter().filter_map(|source| match source { - SourceEntry::Auto { base, .. } => Some(base), + SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base), _ => None, }); @@ -852,10 +852,7 @@ fn create_auto_walker(sources: &Sources) -> Option { // Setup ignores based on `@source` definitions, in directive order so later directives win for source in sources.iter() { match source { - SourceEntry::Auto { - base, - external: true, - } => { + SourceEntry::External { base } => { // External sources bypass all gitignore rules (the directory was explicitly // listed even though it is ignored)… let mut ignore_builder = GitignoreBuilder::new(base); @@ -904,7 +901,9 @@ fn create_auto_walker(sources: &Sources) -> Option { .iter() .enumerate() .filter_map(|(idx, source)| match source { - SourceEntry::Auto { base, .. } => Some((idx, base.clone())), + SourceEntry::Auto { base } | SourceEntry::External { base } => { + Some((idx, base.clone())) + } _ => None, }) .collect(); diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 139e4ef24..ef717483d 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -28,9 +28,12 @@ pub enum SourceEntry { /// @source "src";` /// @source "src/**/*";` /// ``` + Auto { base: PathBuf }, + + /// An `Auto` source whose directory is itself ignored (by the default rules, e.g. + /// `node_modules`, or by a `.gitignore`) but was explicitly listed anyway. /// - /// `external` is set when the directory itself is ignored (by the default rules, e.g. - /// `node_modules`, or by a `.gitignore`) but was explicitly listed anyway: + /// Represented by: /// /// ```css /// @source "../node_modules/my-lib";` @@ -41,7 +44,7 @@ pub enum SourceEntry { /// as if it were a regular auto source, except that `.gitignore` files no longer apply /// inside (git ignores the whole tree anyway). The default rules still apply inside, so /// e.g. nested `node_modules` stay ignored. - Auto { base: PathBuf, external: bool }, + External { base: PathBuf }, /// Explicit source pattern regardless of any auto source detection rules /// @@ -347,10 +350,7 @@ mod tests { assert_eq!( sources, vec![ - SourceEntry::Auto { - base: base.clone(), - external: false, - }, + SourceEntry::Auto { base: base.clone() }, SourceEntry::Pattern { base: base.clone(), pattern: "/foo.html".to_string(), @@ -385,10 +385,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: false, - } + SourceEntry::Auto { base } ); } @@ -401,10 +398,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: true, - } + SourceEntry::External { base } ); } @@ -421,10 +415,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: true, - } + SourceEntry::External { base } ); } @@ -441,10 +432,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: true, - } + SourceEntry::External { base } ); } @@ -468,10 +456,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: false, - } + SourceEntry::Auto { base } ); } @@ -491,10 +476,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: true, - } + SourceEntry::External { base } ); } @@ -510,10 +492,7 @@ mod tests { assert_eq!( auto_source_entry(&base), - SourceEntry::Auto { - base, - external: false, - } + SourceEntry::Auto { base } ); } } @@ -582,11 +561,7 @@ pub fn public_source_entries_to_private_source_entries( let mut source: SourceEntry = public_source.into(); // Mark auto sources as external if their directory is gitignored - if let SourceEntry::Auto { - ref base, - external: false, - } = source - { + if let SourceEntry::Auto { ref base } = source { let inside_git_repo = base.ancestors().any(|dir| dir.join(".git").exists()); // The chain of directories whose `.gitignore` files can apply: `base` itself and @@ -642,10 +617,7 @@ pub fn public_source_entries_to_private_source_entries( match gitignore.matched(prefix, true) { ignore::Match::Ignore(_) => { - source = SourceEntry::Auto { - base: base.into(), - external: true, - }; + source = SourceEntry::External { base: base.into() }; break 'prefixes; } // Re-included; this directory is reachable, move on to the next one. @@ -703,9 +675,13 @@ impl From for SourceEntry { .ends_with(&format!("{}{}", std::path::MAIN_SEPARATOR, dir)) }); - SourceEntry::Auto { - base: value.base.into(), - external: inside_ignored_content_dir, + match inside_ignored_content_dir { + false => SourceEntry::Auto { + base: value.base.into(), + }, + true => SourceEntry::External { + base: value.base.into(), + }, } } } From 5360b0d40834267bee39fbd5f8d61aff88fdb532 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 21:31:12 +0200 Subject: [PATCH 19/23] refactor --- .../src/scanner/auto_source_detection.rs | 8 +- crates/oxide/src/scanner/mod.rs | 807 +++++++++++------- 2 files changed, 505 insertions(+), 310 deletions(-) diff --git a/crates/oxide/src/scanner/auto_source_detection.rs b/crates/oxide/src/scanner/auto_source_detection.rs index c53756456..7c0ac0b98 100644 --- a/crates/oxide/src/scanner/auto_source_detection.rs +++ b/crates/oxide/src/scanner/auto_source_detection.rs @@ -34,10 +34,10 @@ pub static IGNORED_CONTENT_DIRS: sync::LazyLock> = sync::LazyL .collect() }); -pub static IGNORED_CONTENT_DIRS_GLOB: sync::LazyLock = +static IGNORED_CONTENT_DIRS_GLOB: sync::LazyLock = sync::LazyLock::new(|| format!("{{{}}}/", IGNORED_CONTENT_DIRS.join(","))); -pub static IGNORED_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { +static IGNORED_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { format!( "*.{{{}}}", include_str!("fixtures/ignored-extensions.txt") @@ -48,7 +48,7 @@ pub static IGNORED_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new ) }); -pub static BINARY_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { +static BINARY_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new(|| { format!( "*.{{{}}}", include_str!("fixtures/binary-extensions.txt") @@ -59,7 +59,7 @@ pub static BINARY_EXTENSIONS_GLOB: sync::LazyLock = sync::LazyLock::new( ) }); -pub static IGNORED_FILES_GLOB: sync::LazyLock = sync::LazyLock::new(|| { +static IGNORED_FILES_GLOB: sync::LazyLock = sync::LazyLock::new(|| { format!( "{{{}}}", include_str!("fixtures/ignored-files.txt") diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index de1cceca8..0350a03f9 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -10,11 +10,13 @@ use crate::scanner::sources::{ public_source_entries_to_private_source_entries, PublicSourceEntry, SourceEntry, Sources, }; use crate::GlobEntry; -use auto_source_detection::BINARY_EXTENSIONS_GLOB; use bstr::ByteSlice; use fast_glob::glob_match; use fxhash::{FxHashMap, FxHashSet}; -use ignore::{gitignore::GitignoreBuilder, WalkBuilder}; +use ignore::{ + gitignore::{Gitignore, GitignoreBuilder}, + WalkBuilder, +}; use init_tracing::{init_tracing, SHOULD_TRACE}; use rayon::prelude::*; use std::path::{Path, PathBuf}; @@ -61,22 +63,30 @@ use tracing::event; // // # Implementation // -// Scanning uses the vendored `ignore` crate for gitignore-aware, pruned directory walking (we -// never descend into directories that cannot contribute files). Sources are split over multiple -// walkers: +// All sources are scanned in a single file system walk. The vendored `ignore` crate only +// provides the traversal itself (parallel walking, symlink loop handling); its built-in +// gitignore handling is disabled, because it computes one global verdict per path while the +// `@source` semantics are per source: the same directory can be pruned for an auto source but +// walkable for a pattern source, and a file can be gitignored for an auto source but rescued +// by a glob. // -// - One walker for all `Auto` and `External` roots. Default rules are registered as explicit -// in-memory gitignores (they rank above `.gitignore` files found on disk). `External` roots -// additionally get a `!/**/*` whitelist (bypassing gitignore rules) plus a re-statement of -// the default rules. `@source not` rules are handled by a filter closure that honors the -// directive order: the last directive covering a path wins. +// Instead, the [`Resolver`] implements the semantics in the walker's `filter_entry` callback, +// backed by its own lazily-loaded cache of the on-disk ignore files (`.gitignore`, `.ignore`, +// and the repository's `.git/info/exclude`, applied up to the git repository root): // -// - One walker per `Pattern` base. It never looks at `.gitignore` files *above* its base -// (the static prefix is explicit), while `.gitignore` files inside the subtree still prune -// directories. Matching files are whitelisted (a glob match beats file-level ignores), and a -// filter closure makes the final per-file decision: the file must match a glob, honoring the -// relative order of `@source not` directives, and non-extension-pinning globs re-apply the -// default file rules. +// - The walk roots are the "maximal" source bases; nested bases are reached by walking, and +// the resolver keeps the static path towards an explicitly listed base open even through +// ignored directories. +// +// - A directory is entered when at least one source can contribute files inside of it, where +// each source kind applies its own rules: auto sources check the default rules and the full +// gitignore chain, external sources only the default rules, and pattern sources check the +// default rules, the `.gitignore` files at or below their base, and whether the glob can +// match anything inside the directory. Directories that cannot contribute anything are +// never descended into. +// +// - A file is kept when at least one source includes it, honoring the directive order of +// `@source not` (the later directive wins). #[derive(Debug, Clone)] pub enum ChangedContent { @@ -105,9 +115,11 @@ pub struct Scanner { /// Content sources sources: Sources, - /// The walkers to detect all files that we have to scan: one for all auto/external source - /// roots, and one per pattern source base - walkers: Vec, + /// The walker to detect all files that we have to scan + walker: Option, + + /// The resolver implementing the `@source` semantics for the walker + resolver: Option>, /// All found extensions extensions: FxHashSet, @@ -157,11 +169,13 @@ impl Scanner { } } - let walkers = create_walkers(&sources); + let resolver = Arc::new(Resolver::new(&sources)); + let walker = create_walker(resolver.clone()); Self { sources, - walkers, + walker, + resolver: Some(resolver), ..Default::default() } } @@ -220,7 +234,7 @@ impl Scanner { // Figure out if the new unknown files are allowed to be scanned if !new_unknown_files.is_empty() { - 'outer: for walk_builder in self.walkers.iter_mut() { + if let Some(walk_builder) = &mut self.walker { for entry in walk_builder.build().filter_map(Result::ok) { let path = entry.path(); if !path.is_file() { @@ -265,7 +279,7 @@ impl Scanner { // We can stop walking the file system if all files we are interested in have // been found. if new_unknown_files.is_empty() { - break 'outer; + break; } } } @@ -425,20 +439,17 @@ impl Scanner { } self.sources_scanned = true; - if self.walkers.is_empty() { + let Some(walker) = &mut self.walker else { return (vec![], vec![], vec![]); - } + }; - // Use synchronous walks for the initial build (lower overhead) and parallel - // walks for subsequent calls (watch mode) where the overhead is amortised. - let mut all_entries = vec![]; - for walker in self.walkers.iter_mut() { - if self.has_scanned_once { - all_entries.extend(walk_parallel(walker)); - } else { - all_entries.extend(walk_synchronous(walker)); - } - } + // Use synchronous walk for the initial build (lower overhead) and parallel + // walk for subsequent calls (watch mode) where the overhead is amortised. + let all_entries = if self.has_scanned_once { + walk_parallel(walker) + } else { + walk_synchronous(walker) + }; let mut css_files: Vec = vec![]; let mut content_paths: Vec<(PathBuf, String)> = vec![]; @@ -457,7 +468,16 @@ impl Scanner { for entry in all_entries { match entry { WalkEntry::Dir(path) => { - self.dirs.insert(path); + // Directories that are only walked to reach an explicitly listed base are + // not part of any source's content: they must not widen the generated + // file watcher globs. + let contributes = self + .resolver + .as_ref() + .is_some_and(|resolver| resolver.contributes_dir(&path)); + if contributes { + self.dirs.insert(path); + } } WalkEntry::File { path, @@ -756,184 +776,485 @@ fn walk_parallel(walker: &mut WalkBuilder) -> Vec { Arc::try_unwrap(collected).unwrap().into_inner().unwrap() } -/// Sets up all walkers for the given sources: one gitignore-aware walker for all `Auto` and -/// `External` roots, and one walker per `Pattern` base. -fn create_walkers(sources: &Sources) -> Vec { - let mut walkers = vec![]; - walkers.extend(create_auto_walker(sources)); - walkers.extend(create_pattern_walkers(sources)); - walkers -} - -/// A single walker for all `Auto` and `External` roots. +/// Sets up the single walker for all sources. /// -/// The walker respects `.gitignore` files (also from parent directories, up to the git -/// repository root) plus a stack of explicit in-memory gitignores. Explicit gitignores rank -/// above the ignore files found on disk, and later-registered ones above earlier ones (see the -/// `CHANGED:` annotations in the vendored `ignore` crate). From low to high precedence: +/// The walker is only used for the (parallel, symlink aware) traversal itself: all ignore +/// semantics are implemented by the [`Resolver`] in the `filter_entry` callback. The crate's +/// built-in gitignore handling can't be used because it computes one global verdict per path, +/// while the `@source` semantics are per source (see the spec at the top of this file): the +/// same directory can be pruned for an auto source but walkable for a pattern source, and a +/// file can be gitignored for an auto source but rescued by a glob. /// -/// 1. The default auto source detection rules -/// 2. Per `@source` directive, in order: -/// - `External`: a `!/**/*` whitelist at its base (bypassing all gitignore rules), plus a -/// re-statement of the default rules so that e.g. nested `node_modules` stay ignored -/// - `Ignored` (`@source not`): its pattern at its base -fn create_auto_walker(sources: &Sources) -> Option { - let mut roots = sources.iter().filter_map(|source| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => Some(base), - _ => None, - }); - +/// The walk roots are the "maximal" source bases: a base contained in another base is reached +/// by walking, which the resolver allows even through ignored directories (the static path to +/// an explicit base bypasses ignore rules). +fn create_walker(resolver: Arc) -> Option { + let mut roots = resolver.walk_roots().into_iter(); let first_root = roots.next()?; let mut builder = WalkBuilder::new(first_root); - let mut seen_roots = FxHashSet::from_iter([first_root]); for root in roots { - if seen_roots.insert(root) { - builder.add(root); - } + builder.add(root); } // We have to follow symlinks builder.follow_links(true); - // Scan hidden files / directories - builder.hidden(false); + // Disable all of the built-in filtering (hidden files, .gitignore files, parent + // directories, global gitignore files, …): the resolver implements the ignore semantics. + builder.standard_filters(false); - // Don't respect global gitignore files - builder.git_global(false); + builder.filter_entry(move |entry| resolver.keep(entry)); - // By default, allow .gitignore files to be used regardless of whether or not - // a .git directory is present. This is an optimization for when projects - // are first created and may not be in a git repo yet. - builder.require_git(false); + Some(builder) +} - // If we are in a git repo then require it to ensure that only rules within - // the repo are used. For example, we don't want to consider a .gitignore file - // in the user's home folder if we're in a git repo. - // - // The alternative is using a call like `.parents(false)` but that will - // prevent looking at parent directories for .gitignore files from within - // the repo and that's not what we want. - // - // For example, in a project with this structure: - // - // home - // .gitignore - // my-project - // .gitignore - // apps - // .gitignore - // web - // {root} - // - // We do want to consider all .gitignore files listed: - // - home/.gitignore - // - my-project/.gitignore - // - my-project/apps/.gitignore - // - // However, if a repo is initialized inside my-project then only the following - // make sense for consideration: - // - my-project/.gitignore - // - my-project/apps/.gitignore - // - // Setting the require_git(true) flag conditionally allows us to do this. - for parent in first_root.ancestors() { - if parent.join(".git").exists() { - builder.require_git(true); - break; +/// Implements the `@source` semantics for a single file system walk. +/// +/// For every walked entry, the resolver computes the union of the per-source verdicts: +/// +/// - a directory is entered when at least one source can contribute files inside of it +/// - a file is kept when at least one source includes it, honoring the directive order of +/// `@source not` (the later directive wins) +/// +/// The per-source verdicts require gitignore decisions relative to different anchors (an auto +/// source respects the full `.gitignore` chain, a pattern source only the `.gitignore` files +/// at or below its base, an external source none at all), so the resolver maintains its own +/// lazily-loaded cache of ignore files instead of using the walker's built-in handling. +#[derive(Debug)] +struct Resolver { + /// `Auto` sources: directive position and base + autos: Vec<(usize, PathBuf)>, + + /// `External` sources: directive position and base + externals: Vec<(usize, PathBuf)>, + + /// `Pattern` sources: base and the pattern with its directive position + patterns: Vec<(PathBuf, PatternRule)>, + + /// `@source not` directives + nots: Vec, + + /// Lazily loaded ignore files (`.gitignore`, `.ignore`, `.git/info/exclude`) per directory + ignore_files: IgnoreFiles, + + /// Memoized "is this directory reachable for an auto source": every directory on the path + /// from an auto base down to it passes the gitignore chain and the default rules + auto_reachable: Mutex>, + + /// Memoized "is this directory reachable for an external source": like `auto_reachable`, + /// but without the gitignore chain (git ignores the whole tree anyway) + external_reachable: Mutex>, + + /// Memoized "is this directory reachable for a pattern source with this base": like + /// `auto_reachable`, but only `.gitignore` files at or below the base apply (the static + /// base is explicit, everything above it is bypassed) + pattern_reachable: Mutex>, +} + +impl Resolver { + fn new(sources: &Sources) -> Self { + let mut autos = vec![]; + let mut externals = vec![]; + let mut patterns = vec![]; + + for (idx, source) in sources.iter().enumerate() { + match source { + SourceEntry::Auto { base } => autos.push((idx, base.clone())), + SourceEntry::External { base } => externals.push((idx, base.clone())), + SourceEntry::Pattern { base, pattern } => patterns.push(( + base.clone(), + PatternRule { + idx, + pattern: pattern.clone(), + bypasses_default_file_rules: pattern_bypasses_default_file_rules( + pattern, + ), + }, + )), + SourceEntry::Ignored { .. } => {} + } + } + + Self { + autos, + externals, + patterns, + nots: collect_not_rules(sources), + ignore_files: IgnoreFiles::default(), + auto_reachable: Default::default(), + external_reachable: Default::default(), + pattern_reachable: Default::default(), } } - // Setup auto source detection rules - for ignore in auto_source_detection::RULES.iter() { - builder.add_gitignore(ignore.clone()); + /// All source bases + fn bases(&self) -> impl Iterator { + self.autos + .iter() + .map(|(_, base)| base) + .chain(self.externals.iter().map(|(_, base)| base)) + .chain(self.patterns.iter().map(|(base, _)| base)) } - // Setup ignores based on `@source` definitions, in directive order so later directives win - for source in sources.iter() { - match source { - SourceEntry::External { base } => { - // External sources bypass all gitignore rules (the directory was explicitly - // listed even though it is ignored)… - let mut ignore_builder = GitignoreBuilder::new(base); - ignore_builder.add_line(None, "!/**/*").unwrap(); - - // …but the default auto source detection rules still apply inside of them, so - // nested `node_modules`, ignored extensions, and lock files stay ignored. - ignore_builder - .add_line(None, &auto_source_detection::IGNORED_CONTENT_DIRS_GLOB) - .unwrap(); - ignore_builder - .add_line(None, &auto_source_detection::IGNORED_EXTENSIONS_GLOB) - .unwrap(); - ignore_builder - .add_line(None, &auto_source_detection::IGNORED_FILES_GLOB) - .unwrap(); - builder.add_gitignore(ignore_builder.build().unwrap()); - - // Binary extensions are ignored as well, but only for files, so that a folder - // named e.g. `some.pages` is still scanned. - let mut ignore_builder = GitignoreBuilder::new(base); - ignore_builder - .only_on_files(true) - .add_line(None, &BINARY_EXTENSIONS_GLOB) - .unwrap(); - builder.add_gitignore(ignore_builder.build().unwrap()); + /// The walk roots: all bases that are not contained in another base. Nested bases are + /// reached by walking (the resolver keeps the path to an explicit base open). + fn walk_roots(&self) -> Vec { + let mut roots: Vec = vec![]; + for base in self.bases() { + if self + .bases() + .any(|other| other != base && base.starts_with(other)) + { + continue; + } + if !roots.contains(base) { + roots.push(base.clone()); } - _ => {} } + roots } - // `@source not` directives are handled in a filter closure instead of a gitignore layer, - // because their effect depends on the directive order: the last directive that covers a - // path wins. When that is a `@source not`, the entry is excluded; when it is an - // auto/external source, the entry falls through to the normal gitignore + default rules - // handling. E.g.: - // - // ```css - // @source not "./src"; - // @source "./src"; /* re-includes ./src, .gitignore files still apply inside */ - // ``` - // - // Directories excluded here can be pruned safely: every auto/external base is its own walk - // root, so a source root nested inside an excluded directory is still walked. - let includes: Vec<(usize, PathBuf)> = sources - .iter() - .enumerate() - .filter_map(|(idx, source)| match source { - SourceEntry::Auto { base } | SourceEntry::External { base } => { - Some((idx, base.clone())) - } - _ => None, - }) - .collect(); - let nots = collect_not_rules(sources); - - builder.filter_entry(move |entry| { - // Always keep the walk roots themselves + /// Whether to keep the given walk entry + fn keep(&self, entry: &ignore::DirEntry) -> bool { + // Always keep the walk roots themselves; they are explicitly listed bases if entry.depth() == 0 { return true; } let path = entry.path(); + let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); - // The last `@source not` directive covering the path… - let Some(not_idx) = nots - .iter() - .filter(|not| not.matches(path)) - .map(|not| not.idx) - .max() - else { + if is_dir { + self.keep_dir(path) + } else { + self.keep_file(path) + } + } + + /// Whether to keep walking the given directory: either a source can contribute files + /// inside of it, or it is on the static path towards an explicitly listed base. + fn keep_dir(&self, dir: &Path) -> bool { + // A directory on the static path to an explicit base can always be entered (the base + // is explicit, so its ignoredness is bypassed), unless everything below it is excluded + // again by a later `@source not` directive. + let leads_to_base = |idx: usize, base: &PathBuf| { + base != dir + && base.starts_with(dir) + && !self + .nots + .iter() + .any(|not| not.idx > idx && not.matches(base)) + }; + if self.autos.iter().any(|(idx, base)| leads_to_base(*idx, base)) + || self + .externals + .iter() + .any(|(idx, base)| leads_to_base(*idx, base)) + || self + .patterns + .iter() + .any(|(base, pattern)| leads_to_base(pattern.idx, base)) + { return true; + } + + self.contributes_dir(dir) + } + + /// Whether at least one source can contribute files inside the given directory. Unlike + /// [`Resolver::keep_dir`], directories that are only walked to reach an explicitly listed + /// base don't count: they are not part of any source's content, e.g. for the purpose of + /// generating file watcher globs. + fn contributes_dir(&self, dir: &Path) -> bool { + // Some source must be able to contribute files inside the directory, and not be + // overridden by a later `@source not` directive. + let not_after = |idx: usize| { + self.nots + .iter() + .any(|not| not.idx > idx && not.matches(dir)) }; - // …loses when a later auto/external source covers it as well - includes + if self + .autos .iter() - .any(|(idx, base)| *idx > not_idx && path.starts_with(base)) - }); + .any(|(idx, _)| self.auto_reachable(dir) && !not_after(*idx)) + { + return true; + } - Some(builder) + if self + .externals + .iter() + .any(|(idx, _)| self.external_reachable(dir) && !not_after(*idx)) + { + return true; + } + + self.patterns.iter().any(|(base, pattern)| { + dir.strip_prefix(base).is_ok_and(|remainder| { + dir_could_contain_matches(&pattern.pattern, remainder) + && self.pattern_reachable(base, dir) + && !not_after(pattern.idx) + }) + }) + } + + /// Whether at least one source includes the given file + fn keep_file(&self, file: &Path) -> bool { + let Some(parent) = file.parent() else { + return false; + }; + + let not_after = |idx: usize| { + self.nots + .iter() + .any(|not| not.idx > idx && not.matches(file)) + }; + + // Auto sources: the file must pass the default rules and the gitignore chain + if self.autos.iter().any(|(idx, base)| { + file.starts_with(base) + && self.auto_reachable(parent) + && !is_ignored_by_default_file_rules(file) + && self.ignore_files.matched(file, false, parent, None) != Some(Match::Ignore) + && !not_after(*idx) + }) { + return true; + } + + // External sources: like auto sources, but the gitignore chain doesn't apply + if self.externals.iter().any(|(idx, base)| { + file.starts_with(base) + && self.external_reachable(parent) + && !is_ignored_by_default_file_rules(file) + && !not_after(*idx) + }) { + return true; + } + + // Pattern sources: the file must match the glob. A match beats file-level gitignore + // rules — you were explicit about wanting files of that shape — and the default file + // rules only apply when the pattern isn't explicit about the file's shape. + self.patterns.iter().any(|(base, pattern)| { + file.strip_prefix(base).is_ok_and(|remainder| { + glob_match(&pattern.pattern, rooted_posix(remainder).as_bytes()) + && self.pattern_reachable(base, parent) + && (pattern.bypasses_default_file_rules + || !is_ignored_by_default_file_rules(file)) + && !not_after(pattern.idx) + }) + }) + } + + /// Whether the given directory is reachable for an auto source: every directory on the + /// path from the auto base down to it passes the default rules and the gitignore chain. + /// Auto bases themselves are always reachable — a base that is itself ignored would have + /// been promoted to an external source. + fn auto_reachable(&self, dir: &Path) -> bool { + if let Some(reachable) = self.auto_reachable.lock().unwrap().get(dir) { + return *reachable; + } + + let reachable = if self.autos.iter().any(|(_, base)| base == dir) { + true + } else if !self.autos.iter().any(|(_, base)| dir.starts_with(base)) { + false + } else { + dir.parent().is_some_and(|parent| { + self.auto_reachable(parent) + && !is_ignored_by_default_dir_rules(dir) + && self.ignore_files.matched(dir, true, parent, None) != Some(Match::Ignore) + }) + }; + + self.auto_reachable + .lock() + .unwrap() + .insert(dir.to_path_buf(), reachable); + reachable + } + + /// Like [`Resolver::auto_reachable`], but for external sources: the gitignore chain + /// doesn't apply inside of them (git ignores the whole tree anyway), only the default + /// rules do. + fn external_reachable(&self, dir: &Path) -> bool { + if let Some(reachable) = self.external_reachable.lock().unwrap().get(dir) { + return *reachable; + } + + let reachable = if self.externals.iter().any(|(_, base)| base == dir) { + true + } else if !self.externals.iter().any(|(_, base)| dir.starts_with(base)) { + false + } else { + dir.parent().is_some_and(|parent| { + self.external_reachable(parent) && !is_ignored_by_default_dir_rules(dir) + }) + }; + + self.external_reachable + .lock() + .unwrap() + .insert(dir.to_path_buf(), reachable); + reachable + } + + /// Like [`Resolver::auto_reachable`], but for a pattern source with the given base: only + /// `.gitignore` files at or below the base apply — the static base is explicit, so + /// everything above it is bypassed. + fn pattern_reachable(&self, base: &Path, dir: &Path) -> bool { + if let Some(reachable) = self + .pattern_reachable + .lock() + .unwrap() + .get(&(base.to_path_buf(), dir.to_path_buf())) + { + return *reachable; + } + + let reachable = if base == dir { + true + } else if !dir.starts_with(base) { + false + } else { + dir.parent().is_some_and(|parent| { + self.pattern_reachable(base, parent) + && !is_ignored_by_default_dir_rules(dir) + && self.ignore_files.matched(dir, true, parent, Some(base)) + != Some(Match::Ignore) + }) + }; + + self.pattern_reachable + .lock() + .unwrap() + .insert((base.to_path_buf(), dir.to_path_buf()), reachable); + reachable + } +} + +/// The verdict of the on-disk ignore files for a path +#[derive(Debug, Clone, Copy, PartialEq)] +enum Match { + Ignore, + Whitelist, +} + +/// A lazily-loaded cache of the on-disk ignore files (`.gitignore`, `.ignore` and the +/// repository's `.git/info/exclude`). +#[derive(Debug, Default)] +struct IgnoreFiles { + /// Compiled ignore rules per directory (`None` when the directory has no ignore files) + matchers: Mutex>>>, + + /// The matcher chains per directory: all matchers that apply to paths inside the + /// directory, deepest first, from the directory itself up to the repository root (or the + /// file system root outside of a git repository, matching `git init`-less projects where + /// all ancestor `.gitignore` files apply) + chains: Mutex>>>>, +} + +impl IgnoreFiles { + /// The verdict of the ignore files for the given path. `dir` is the directory containing + /// the path (matchers are looked up for that directory), and `below` optionally restricts + /// the chain to ignore files at or below the given directory. + /// + /// The deepest ignore file with a definitive answer wins, matching git's precedence. + fn matched(&self, path: &Path, is_dir: bool, dir: &Path, below: Option<&Path>) -> Option { + for matcher in self.chain(dir).iter() { + if below.is_some_and(|below| !matcher.path().starts_with(below)) { + // Chains are ordered deepest first, so nothing below the boundary can follow + break; + } + + match matcher.matched(path, is_dir) { + ignore::Match::Ignore(_) => return Some(Match::Ignore), + ignore::Match::Whitelist(_) => return Some(Match::Whitelist), + ignore::Match::None => {} + } + } + + None + } + + /// The matcher chain for paths inside the given directory: the directory's own matcher + /// first, then its parents' matchers, up to and including the git repository root. + fn chain(&self, dir: &Path) -> Arc>> { + if let Some(chain) = self.chains.lock().unwrap().get(dir) { + return chain.clone(); + } + + let is_repo_root = dir.join(".git").exists(); + + let mut chain = vec![]; + if let Some(matcher) = self.matcher(dir, is_repo_root) { + chain.push(matcher); + } + + // Stop at the git repository root so that ignore files outside of the repository are + // not considered. Without a repository, all ancestor ignore files apply. + if !is_repo_root { + if let Some(parent) = dir.parent() { + chain.extend(self.chain(parent).iter().cloned()); + } + } + + let chain = Arc::new(chain); + self.chains + .lock() + .unwrap() + .insert(dir.to_path_buf(), chain.clone()); + chain + } + + /// The compiled ignore rules of the given directory, combining (from low to high + /// precedence) the repository's `.git/info/exclude`, the `.gitignore` file, and the + /// `.ignore` file. + fn matcher(&self, dir: &Path, is_repo_root: bool) -> Option> { + if let Some(matcher) = self.matchers.lock().unwrap().get(dir) { + return matcher.clone(); + } + + let mut builder = GitignoreBuilder::new(dir); + let mut any = false; + + let mut add = |file: PathBuf| { + if file.is_file() { + // I/O errors and partially invalid ignore files are ignored, matching the + // walker's behavior. + let _ = builder.add(file); + any = true; + } + }; + + if is_repo_root { + add(dir.join(".git").join("info").join("exclude")); + } + add(dir.join(".gitignore")); + add(dir.join(".ignore")); + + let matcher = if any { + builder.build().ok().map(Arc::new) + } else { + None + }; + + self.matchers + .lock() + .unwrap() + .insert(dir.to_path_buf(), matcher.clone()); + matcher + } +} + +/// Whether a directory is ignored by the default rules (e.g. `node_modules`). Only the +/// directory's own name is checked; the path towards it is checked by the reachability +/// helpers one directory at a time. +fn is_ignored_by_default_dir_rules(dir: &Path) -> bool { + auto_source_detection::RULES + .iter() + .any(|ignore| ignore.matched(dir, true).is_ignore()) } /// A glob pattern of a `Pattern` source, together with the position of its `@source` directive. @@ -1051,132 +1372,6 @@ fn dir_could_contain_matches(pattern: &str, dir: &Path) -> bool { true } -/// One walker per `Pattern` base. -/// -/// The static base of a glob is the explicit part: it is used as the walk root, so `.gitignore` -/// files *above* it never apply (`parents(false)`), even when the base is hidden inside an -/// ignored directory. The wildcard part is not explicit: -/// -/// - `.gitignore` files inside the subtree still prune directories -/// - the default rules still prune directories (`node_modules` etc., unless the base itself -/// points inside of one) -/// -/// Files matching a glob are whitelisted explicitly (a glob match beats file-level gitignore -/// and default rules). The filter closure then makes the exact per-file decision. -fn create_pattern_walkers(sources: &Sources) -> Vec { - let nots = collect_not_rules(sources); - - // Group patterns by base, preserving directive order - let mut bases: Vec = vec![]; - let mut patterns_by_base: FxHashMap> = FxHashMap::default(); - for (idx, source) in sources.iter().enumerate() { - let SourceEntry::Pattern { base, pattern } = source else { - continue; - }; - - patterns_by_base - .entry(base.clone()) - .or_insert_with(|| { - bases.push(base.clone()); - vec![] - }) - .push(PatternRule { - idx, - pattern: pattern.clone(), - bypasses_default_file_rules: pattern_bypasses_default_file_rules(pattern), - }); - } - - bases - .into_iter() - .map(|base| { - let patterns = patterns_by_base.remove(&base).unwrap(); - - let mut builder = WalkBuilder::new(&base); - - // We have to follow symlinks - builder.follow_links(true); - - // Scan hidden files / directories - builder.hidden(false); - - // Don't respect global gitignore files - builder.git_global(false); - - // The static base is explicit: `.gitignore` files above it do not apply - builder.parents(false); - - // Apply `.gitignore` files inside the subtree regardless of whether a `.git` - // directory is present - builder.require_git(false); - - // The default rules prune directories (`node_modules`, …). Their file-level rules - // are rescued by the whitelists below when the glob matches. - for ignore in auto_source_detection::RULES.iter() { - builder.add_gitignore(ignore.clone()); - } - - // Whitelist the patterns themselves so that matching files win from file-level - // gitignore rules and the default rules. Restricted to files: the wildcard part of - // a pattern must not re-include ignored directories. - let mut ignore_builder = GitignoreBuilder::new(&base); - ignore_builder.only_on_files(true); - for pattern in &patterns { - ignore_builder - .add_line(None, &format!("!{}", pattern.pattern)) - .unwrap(); - } - builder.add_gitignore(ignore_builder.build().unwrap()); - - // The exact per-file decision (lock-free, safe for parallel walking) - let filter_base = base.clone(); - let filter_nots = nots.clone(); - builder.filter_entry(move |entry| { - // Always keep the walk root itself - if entry.depth() == 0 { - return true; - } - - let path = entry.path(); - let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); - - let Ok(remainder) = path.strip_prefix(&filter_base) else { - return false; - }; - - if is_dir { - // Only descend when some pattern can match files inside this directory… - let relevant = patterns - .iter() - .filter(|p| dir_could_contain_matches(&p.pattern, remainder)) - .collect::>(); - if relevant.is_empty() { - return false; - } - - // …and no later `@source not` directive excludes the directory for all of - // those patterns. When a pattern comes after the `not`, keep walking; the - // file-level check below resolves the conflict exactly. - return !filter_nots - .iter() - .any(|not| not.matches(path) && relevant.iter().all(|p| p.idx < not.idx)); - } - - let rel = rooted_posix(remainder); - patterns.iter().any(|p| { - glob_match(&p.pattern, rel.as_bytes()) - && !filter_nots - .iter() - .any(|not| not.idx > p.idx && not.matches(path)) - && (p.bypasses_default_file_rules || !is_ignored_by_default_file_rules(path)) - }) - }); - - builder - }) - .collect() -} - /// Whether a pattern is explicit enough to bypass the default file rules. /// /// A pattern without any wildcards names a concrete file, e.g. `/.env` or From a547b31500da2ba7dbfd416352fc8e62859d7c81 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 21:44:41 +0200 Subject: [PATCH 20/23] refactor --- crates/oxide/src/scanner/mod.rs | 376 ++++++++++++++------------------ 1 file changed, 167 insertions(+), 209 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index 0350a03f9..a522f8c98 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -829,8 +829,8 @@ struct Resolver { /// `External` sources: directive position and base externals: Vec<(usize, PathBuf)>, - /// `Pattern` sources: base and the pattern with its directive position - patterns: Vec<(PathBuf, PatternRule)>, + /// `Pattern` sources, grouped by base + patterns: Vec, /// `@source not` directives nots: Vec, @@ -845,34 +845,51 @@ struct Resolver { /// Memoized "is this directory reachable for an external source": like `auto_reachable`, /// but without the gitignore chain (git ignores the whole tree anyway) external_reachable: Mutex>, +} - /// Memoized "is this directory reachable for a pattern source with this base": like - /// `auto_reachable`, but only `.gitignore` files at or below the base apply (the static - /// base is explicit, everything above it is bypassed) - pattern_reachable: Mutex>, +/// All `Pattern` sources sharing a base, e.g. `@source "src/*.{html,jsx}"` produces the +/// patterns `/*.html` and `/*.jsx` for the base `src`. Patterns carry the position of their +/// `@source` directive to resolve conflicts with `@source not` directives: the later +/// directive wins. +#[derive(Debug)] +struct PatternGroup { + base: PathBuf, + + /// The glob patterns, relative to `base`, with their directive positions + patterns: Vec<(usize, String)>, + + /// Memoized "is this directory reachable for this group": like + /// `Resolver::auto_reachable`, but only `.gitignore` files at or below the base apply — + /// the static base is explicit, so everything above it is bypassed + reachable: Mutex>, } impl Resolver { fn new(sources: &Sources) -> Self { let mut autos = vec![]; let mut externals = vec![]; - let mut patterns = vec![]; + let mut patterns: Vec = vec![]; + let mut nots = vec![]; for (idx, source) in sources.iter().enumerate() { match source { SourceEntry::Auto { base } => autos.push((idx, base.clone())), SourceEntry::External { base } => externals.push((idx, base.clone())), - SourceEntry::Pattern { base, pattern } => patterns.push(( - base.clone(), - PatternRule { - idx, - pattern: pattern.clone(), - bypasses_default_file_rules: pattern_bypasses_default_file_rules( - pattern, - ), - }, - )), - SourceEntry::Ignored { .. } => {} + SourceEntry::Pattern { base, pattern } => { + match patterns.iter_mut().find(|group| &group.base == base) { + Some(group) => group.patterns.push((idx, pattern.clone())), + None => patterns.push(PatternGroup { + base: base.clone(), + patterns: vec![(idx, pattern.clone())], + reachable: Default::default(), + }), + } + } + SourceEntry::Ignored { base, pattern } => nots.push(NotRule { + idx, + base: base.clone(), + pattern: pattern.clone(), + }), } } @@ -880,11 +897,10 @@ impl Resolver { autos, externals, patterns, - nots: collect_not_rules(sources), + nots, ignore_files: IgnoreFiles::default(), auto_reachable: Default::default(), external_reachable: Default::default(), - pattern_reachable: Default::default(), } } @@ -894,7 +910,7 @@ impl Resolver { .iter() .map(|(_, base)| base) .chain(self.externals.iter().map(|(_, base)| base)) - .chain(self.patterns.iter().map(|(base, _)| base)) + .chain(self.patterns.iter().map(|group| &group.base)) } /// The walk roots: all bases that are not contained in another base. Nested bases are @@ -951,10 +967,12 @@ impl Resolver { .externals .iter() .any(|(idx, base)| leads_to_base(*idx, base)) - || self - .patterns - .iter() - .any(|(base, pattern)| leads_to_base(pattern.idx, base)) + || self.patterns.iter().any(|group| { + group + .patterns + .iter() + .any(|(idx, _)| leads_to_base(*idx, &group.base)) + }) { return true; } @@ -991,11 +1009,11 @@ impl Resolver { return true; } - self.patterns.iter().any(|(base, pattern)| { - dir.strip_prefix(base).is_ok_and(|remainder| { - dir_could_contain_matches(&pattern.pattern, remainder) - && self.pattern_reachable(base, dir) - && !not_after(pattern.idx) + self.patterns.iter().any(|group| { + dir.strip_prefix(&group.base).is_ok_and(|remainder| { + group.patterns.iter().any(|(idx, pattern)| { + dir_could_contain_matches(pattern, remainder) && !not_after(*idx) + }) && self.pattern_reachable(group, dir) }) }) } @@ -1016,8 +1034,8 @@ impl Resolver { if self.autos.iter().any(|(idx, base)| { file.starts_with(base) && self.auto_reachable(parent) - && !is_ignored_by_default_file_rules(file) - && self.ignore_files.matched(file, false, parent, None) != Some(Match::Ignore) + && !is_ignored_by_default_rules(file, false) + && !self.ignore_files.is_ignored(file, false, parent, None) && !not_after(*idx) }) { return true; @@ -1027,127 +1045,117 @@ impl Resolver { if self.externals.iter().any(|(idx, base)| { file.starts_with(base) && self.external_reachable(parent) - && !is_ignored_by_default_file_rules(file) + && !is_ignored_by_default_rules(file, false) && !not_after(*idx) }) { return true; } - // Pattern sources: the file must match the glob. A match beats file-level gitignore + // Pattern sources: the file must match a glob. A match beats file-level gitignore // rules — you were explicit about wanting files of that shape — and the default file // rules only apply when the pattern isn't explicit about the file's shape. - self.patterns.iter().any(|(base, pattern)| { - file.strip_prefix(base).is_ok_and(|remainder| { - glob_match(&pattern.pattern, rooted_posix(remainder).as_bytes()) - && self.pattern_reachable(base, parent) - && (pattern.bypasses_default_file_rules - || !is_ignored_by_default_file_rules(file)) - && !not_after(pattern.idx) + self.patterns.iter().any(|group| { + file.strip_prefix(&group.base).is_ok_and(|remainder| { + let remainder = rooted_posix(remainder); + group.patterns.iter().any(|(idx, pattern)| { + glob_match(pattern, remainder.as_bytes()) + && (pattern_bypasses_default_file_rules(pattern) + || !is_ignored_by_default_rules(file, false)) + && !not_after(*idx) + }) && self.pattern_reachable(group, parent) }) }) } /// Whether the given directory is reachable for an auto source: every directory on the /// path from the auto base down to it passes the default rules and the gitignore chain. - /// Auto bases themselves are always reachable — a base that is itself ignored would have - /// been promoted to an external source. fn auto_reachable(&self, dir: &Path) -> bool { - if let Some(reachable) = self.auto_reachable.lock().unwrap().get(dir) { - return *reachable; - } - - let reachable = if self.autos.iter().any(|(_, base)| base == dir) { - true - } else if !self.autos.iter().any(|(_, base)| dir.starts_with(base)) { - false - } else { - dir.parent().is_some_and(|parent| { - self.auto_reachable(parent) - && !is_ignored_by_default_dir_rules(dir) - && self.ignore_files.matched(dir, true, parent, None) != Some(Match::Ignore) - }) - }; - - self.auto_reachable - .lock() - .unwrap() - .insert(dir.to_path_buf(), reachable); - reachable + reachable( + &self.auto_reachable, + dir, + |dir| self.autos.iter().any(|(_, base)| base == dir), + |dir| { + !is_ignored_by_default_rules(dir, true) + && dir + .parent() + .is_some_and(|parent| !self.ignore_files.is_ignored(dir, true, parent, None)) + }, + ) } /// Like [`Resolver::auto_reachable`], but for external sources: the gitignore chain /// doesn't apply inside of them (git ignores the whole tree anyway), only the default /// rules do. fn external_reachable(&self, dir: &Path) -> bool { - if let Some(reachable) = self.external_reachable.lock().unwrap().get(dir) { - return *reachable; - } - - let reachable = if self.externals.iter().any(|(_, base)| base == dir) { - true - } else if !self.externals.iter().any(|(_, base)| dir.starts_with(base)) { - false - } else { - dir.parent().is_some_and(|parent| { - self.external_reachable(parent) && !is_ignored_by_default_dir_rules(dir) - }) - }; - - self.external_reachable - .lock() - .unwrap() - .insert(dir.to_path_buf(), reachable); - reachable + reachable( + &self.external_reachable, + dir, + |dir| self.externals.iter().any(|(_, base)| base == dir), + |dir| !is_ignored_by_default_rules(dir, true), + ) } - /// Like [`Resolver::auto_reachable`], but for a pattern source with the given base: only - /// `.gitignore` files at or below the base apply — the static base is explicit, so - /// everything above it is bypassed. - fn pattern_reachable(&self, base: &Path, dir: &Path) -> bool { - if let Some(reachable) = self - .pattern_reachable - .lock() - .unwrap() - .get(&(base.to_path_buf(), dir.to_path_buf())) - { - return *reachable; - } - - let reachable = if base == dir { - true - } else if !dir.starts_with(base) { - false - } else { - dir.parent().is_some_and(|parent| { - self.pattern_reachable(base, parent) - && !is_ignored_by_default_dir_rules(dir) - && self.ignore_files.matched(dir, true, parent, Some(base)) - != Some(Match::Ignore) - }) - }; - - self.pattern_reachable - .lock() - .unwrap() - .insert((base.to_path_buf(), dir.to_path_buf()), reachable); - reachable + /// Like [`Resolver::auto_reachable`], but for a pattern group: only `.gitignore` files at + /// or below the base apply — the static base is explicit, so everything above it is + /// bypassed. + fn pattern_reachable(&self, group: &PatternGroup, dir: &Path) -> bool { + reachable( + &group.reachable, + dir, + |dir| dir == group.base, + |dir| { + !is_ignored_by_default_rules(dir, true) + && dir.parent().is_some_and(|parent| { + !self + .ignore_files + .is_ignored(dir, true, parent, Some(&group.base)) + }) + }, + ) } } -/// The verdict of the on-disk ignore files for a path -#[derive(Debug, Clone, Copy, PartialEq)] -enum Match { - Ignore, - Whitelist, +/// Whether every directory on the path from a source base down to `dir` passes the source's +/// `enter` rule. Bases themselves are always reachable: they are explicitly listed (and an +/// auto base that is itself ignored would have been promoted to an external source). Verdicts +/// are memoized per directory. +fn reachable( + memo: &Mutex>, + dir: &Path, + is_base: impl Fn(&Path) -> bool, + enter: impl Fn(&Path) -> bool, +) -> bool { + // Walk up to the nearest base or directory with a memoized verdict… + let mut pending = vec![]; + let mut current = dir; + let mut reachable = loop { + if is_base(current) { + break true; + } + if let Some(reachable) = memo.lock().unwrap().get(current) { + break *reachable; + } + pending.push(current.to_path_buf()); + match current.parent() { + Some(parent) => current = parent, + // Reached the file system root without finding a base + None => break false, + } + }; + + // …then fill in the verdicts back down towards `dir` + for dir in pending.into_iter().rev() { + reachable = reachable && enter(&dir); + memo.lock().unwrap().insert(dir, reachable); + } + + reachable } /// A lazily-loaded cache of the on-disk ignore files (`.gitignore`, `.ignore` and the /// repository's `.git/info/exclude`). #[derive(Debug, Default)] struct IgnoreFiles { - /// Compiled ignore rules per directory (`None` when the directory has no ignore files) - matchers: Mutex>>>, - /// The matcher chains per directory: all matchers that apply to paths inside the /// directory, deepest first, from the directory itself up to the repository root (or the /// file system root outside of a git repository, matching `git init`-less projects where @@ -1156,12 +1164,13 @@ struct IgnoreFiles { } impl IgnoreFiles { - /// The verdict of the ignore files for the given path. `dir` is the directory containing - /// the path (matchers are looked up for that directory), and `below` optionally restricts - /// the chain to ignore files at or below the given directory. + /// Whether the ignore files definitively ignore the given path. `dir` is the directory + /// containing the path, and `below` optionally restricts the chain to ignore files at or + /// below the given directory. /// - /// The deepest ignore file with a definitive answer wins, matching git's precedence. - fn matched(&self, path: &Path, is_dir: bool, dir: &Path, below: Option<&Path>) -> Option { + /// The deepest ignore file with a definitive answer wins, matching git's precedence, so a + /// path that a deeper ignore file re-includes via a `!` pattern is not ignored. + fn is_ignored(&self, path: &Path, is_dir: bool, dir: &Path, below: Option<&Path>) -> bool { for matcher in self.chain(dir).iter() { if below.is_some_and(|below| !matcher.path().starts_with(below)) { // Chains are ordered deepest first, so nothing below the boundary can follow @@ -1169,13 +1178,13 @@ impl IgnoreFiles { } match matcher.matched(path, is_dir) { - ignore::Match::Ignore(_) => return Some(Match::Ignore), - ignore::Match::Whitelist(_) => return Some(Match::Whitelist), + ignore::Match::Ignore(_) => return true, + ignore::Match::Whitelist(_) => return false, ignore::Match::None => {} } } - None + false } /// The matcher chain for paths inside the given directory: the directory's own matcher @@ -1188,7 +1197,7 @@ impl IgnoreFiles { let is_repo_root = dir.join(".git").exists(); let mut chain = vec![]; - if let Some(matcher) = self.matcher(dir, is_repo_root) { + if let Some(matcher) = load_ignore_files(dir, is_repo_root) { chain.push(matcher); } @@ -1207,71 +1216,44 @@ impl IgnoreFiles { .insert(dir.to_path_buf(), chain.clone()); chain } +} - /// The compiled ignore rules of the given directory, combining (from low to high - /// precedence) the repository's `.git/info/exclude`, the `.gitignore` file, and the - /// `.ignore` file. - fn matcher(&self, dir: &Path, is_repo_root: bool) -> Option> { - if let Some(matcher) = self.matchers.lock().unwrap().get(dir) { - return matcher.clone(); +/// Compile the ignore rules of the given directory, combining (from low to high precedence) +/// the repository's `.git/info/exclude`, the `.gitignore` file, and the `.ignore` file. +fn load_ignore_files(dir: &Path, is_repo_root: bool) -> Option> { + let mut builder = GitignoreBuilder::new(dir); + let mut any = false; + + let mut add = |file: PathBuf| { + if file.is_file() { + // I/O errors and partially invalid ignore files are ignored, matching the + // walker's behavior. + let _ = builder.add(file); + any = true; } + }; - let mut builder = GitignoreBuilder::new(dir); - let mut any = false; + if is_repo_root { + add(dir.join(".git").join("info").join("exclude")); + } + add(dir.join(".gitignore")); + add(dir.join(".ignore")); - let mut add = |file: PathBuf| { - if file.is_file() { - // I/O errors and partially invalid ignore files are ignored, matching the - // walker's behavior. - let _ = builder.add(file); - any = true; - } - }; - - if is_repo_root { - add(dir.join(".git").join("info").join("exclude")); - } - add(dir.join(".gitignore")); - add(dir.join(".ignore")); - - let matcher = if any { - builder.build().ok().map(Arc::new) - } else { - None - }; - - self.matchers - .lock() - .unwrap() - .insert(dir.to_path_buf(), matcher.clone()); - matcher + if any { + builder.build().ok().map(Arc::new) + } else { + None } } -/// Whether a directory is ignored by the default rules (e.g. `node_modules`). Only the +/// Whether a path is ignored by the default auto source detection rules: directories like +/// `node_modules`, binary and irrelevant extensions, lock files, … For directories only the /// directory's own name is checked; the path towards it is checked by the reachability /// helpers one directory at a time. -fn is_ignored_by_default_dir_rules(dir: &Path) -> bool { +fn is_ignored_by_default_rules(path: &Path, is_dir: bool) -> bool { auto_source_detection::RULES .iter() - .any(|ignore| ignore.matched(dir, true).is_ignore()) -} - -/// A glob pattern of a `Pattern` source, together with the position of its `@source` directive. -#[derive(Debug, Clone)] -struct PatternRule { - /// Position of the `@source` directive, used to resolve conflicts with `@source not` - /// directives: the later directive wins. - idx: usize, - - /// The glob pattern, relative to the walker's base, e.g. `/ba*/*.html` - pattern: String, - - /// Whether the pattern is explicit enough to bypass the default file rules: it names a - /// concrete file (e.g. `.env` or `do-include-me.bin`) or pins a specific extension (e.g. - /// `*.html` or `logo.png`). Patterns that don't (e.g. `blog/*/**/*`) re-apply the default - /// file rules. - bypasses_default_file_rules: bool, + .any(|ignore| ignore.matched(path, is_dir).is_ignore()) } /// An `@source not` directive, together with its position. @@ -1314,22 +1296,6 @@ impl NotRule { } } -/// Collect all `@source not` directives with their positions. -fn collect_not_rules(sources: &Sources) -> Vec { - sources - .iter() - .enumerate() - .filter_map(|(idx, source)| match source { - SourceEntry::Ignored { base, pattern } => Some(NotRule { - idx, - base: base.clone(), - pattern: pattern.clone(), - }), - _ => None, - }) - .collect() -} - /// Serialize a path relative to some base as a `/`-rooted posix style string, e.g. /// `/ba*/index.html`, matching how source patterns are stored. fn rooted_posix(path: &Path) -> String { @@ -1391,14 +1357,6 @@ fn pattern_bypasses_default_file_rules(pattern: &str) -> bool { } } -/// Whether a file is ignored by the default file-level rules (binary extensions, ignored -/// extensions, lock files, …). -fn is_ignored_by_default_file_rules(path: &Path) -> bool { - auto_source_detection::RULES - .iter() - .any(|ignore| ignore.matched(path, false).is_ignore()) -} - #[cfg(test)] mod tests { use super::{ChangedContent, Scanner}; From 6ecef9db768fe830d1e2f5acef7c6ac8e282522e Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Wed, 12 Aug 2026 23:12:00 +0200 Subject: [PATCH 21/23] fix more stuff --- crates/oxide/src/scanner/mod.rs | 96 ++++++++++++++++++++++------- crates/oxide/src/scanner/sources.rs | 8 ++- crates/oxide/tests/scanner.rs | 50 +++++++++++++++ 3 files changed, 130 insertions(+), 24 deletions(-) diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index a522f8c98..2abae99f5 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -36,8 +36,10 @@ use tracing::event; // it's a default-ignored directory like `node_modules`), e.g. // `@source "node_modules/my-ui-lib"`. Since the folder was listed explicitly, its ignoredness // is bypassed: everything inside is scanned as if it were an `Auto` source, except that -// `.gitignore` files no longer apply inside (git ignores the whole tree anyway). The default -// rules still apply inside: nested `node_modules`, binary extensions, etc. stay ignored. +// `.gitignore` files from at or above the folder no longer apply — they (including the +// self-ignoring `*` file that generators typically place inside such folders) are what made +// it ignored in the first place. `.gitignore` files *deeper inside* the folder still apply, +// and so do the default rules: nested `node_modules`, binary extensions, etc. stay ignored. // // - `Pattern`: `@source "some/folder/*.html"` — an explicit glob. Only files matching the glob // are scanned. The *static* prefix of the glob (`some/folder`) is the explicit part: it is @@ -817,10 +819,10 @@ fn create_walker(resolver: Arc) -> Option { /// - a file is kept when at least one source includes it, honoring the directive order of /// `@source not` (the later directive wins) /// -/// The per-source verdicts require gitignore decisions relative to different anchors (an auto -/// source respects the full `.gitignore` chain, a pattern source only the `.gitignore` files -/// at or below its base, an external source none at all), so the resolver maintains its own -/// lazily-loaded cache of ignore files instead of using the walker's built-in handling. +/// The per-source verdicts require gitignore decisions relative to different anchors (see +/// [`Boundary`]: each source kind respects a different part of the `.gitignore` chain), so the +/// resolver maintains its own lazily-loaded cache of ignore files instead of using the +/// walker's built-in handling. #[derive(Debug)] struct Resolver { /// `Auto` sources: directive position and base @@ -843,7 +845,7 @@ struct Resolver { auto_reachable: Mutex>, /// Memoized "is this directory reachable for an external source": like `auto_reachable`, - /// but without the gitignore chain (git ignores the whole tree anyway) + /// but only `.gitignore` files strictly below the external base apply external_reachable: Mutex>, } @@ -1035,17 +1037,25 @@ impl Resolver { file.starts_with(base) && self.auto_reachable(parent) && !is_ignored_by_default_rules(file, false) - && !self.ignore_files.is_ignored(file, false, parent, None) + && !self + .ignore_files + .is_ignored(file, false, parent, Boundary::None) && !not_after(*idx) }) { return true; } - // External sources: like auto sources, but the gitignore chain doesn't apply + // External sources: like auto sources, but only `.gitignore` files strictly below the + // base apply. Rules from at or above the base are bypassed — they are what made the + // directory ignored, and it was listed explicitly anyway — while `.gitignore` files + // deeper inside the external tree still apply. if self.externals.iter().any(|(idx, base)| { file.starts_with(base) && self.external_reachable(parent) && !is_ignored_by_default_rules(file, false) + && !self + .ignore_files + .is_ignored(file, false, parent, Boundary::Inside(base)) && !not_after(*idx) }) { return true; @@ -1076,22 +1086,40 @@ impl Resolver { |dir| self.autos.iter().any(|(_, base)| base == dir), |dir| { !is_ignored_by_default_rules(dir, true) - && dir - .parent() - .is_some_and(|parent| !self.ignore_files.is_ignored(dir, true, parent, None)) + && dir.parent().is_some_and(|parent| { + !self.ignore_files.is_ignored(dir, true, parent, Boundary::None) + }) }, ) } - /// Like [`Resolver::auto_reachable`], but for external sources: the gitignore chain - /// doesn't apply inside of them (git ignores the whole tree anyway), only the default - /// rules do. + /// Like [`Resolver::auto_reachable`], but for external sources: only `.gitignore` files + /// strictly below the external base apply (see [`Boundary`]). Rules from at or above the + /// base are bypassed — they are what made the directory ignored, and it was listed + /// explicitly anyway. + /// + /// When external bases are nested, the deepest base containing the directory bounds the + /// chain: reachability from an outer base implies reachability from a nested base (the + /// outer chain checks a superset of the ignore files), so this computes the union of the + /// per-base verdicts. fn external_reachable(&self, dir: &Path) -> bool { reachable( &self.external_reachable, dir, |dir| self.externals.iter().any(|(_, base)| base == dir), - |dir| !is_ignored_by_default_rules(dir, true), + |dir| { + !is_ignored_by_default_rules(dir, true) + && dir.parent().is_some_and(|parent| { + let boundary = self + .externals + .iter() + .map(|(_, base)| base) + .filter(|base| dir.starts_with(base)) + .max_by_key(|base| base.components().count()) + .map_or(Boundary::None, |base| Boundary::Inside(base)); + !self.ignore_files.is_ignored(dir, true, parent, boundary) + }) + }, ) } @@ -1108,7 +1136,7 @@ impl Resolver { && dir.parent().is_some_and(|parent| { !self .ignore_files - .is_ignored(dir, true, parent, Some(&group.base)) + .is_ignored(dir, true, parent, Boundary::At(&group.base)) }) }, ) @@ -1163,16 +1191,42 @@ struct IgnoreFiles { chains: Mutex>>>>, } +/// Which part of the ignore file chain applies to a source, anchored at its base: +/// +/// - `Auto` sources respect the full chain, up to the git repository root +/// - `Pattern` sources respect ignore files at or below their base (the static base is +/// explicit, everything above it is bypassed) +/// - `External` sources respect ignore files strictly below their base: the base's own +/// ignore file is part of its bypassed ignoredness — inside ignored trees it is typically +/// the self-ignoring `*` file that generators drop into the directory — while deeper ignore +/// files are deliberate signals about specific contents and still apply +#[derive(Debug, Clone, Copy)] +enum Boundary<'a> { + None, + At(&'a Path), + Inside(&'a Path), +} + +impl Boundary<'_> { + /// Whether an ignore file rooted at the given directory applies + fn applies_to(&self, dir: &Path) -> bool { + match self { + Boundary::None => true, + Boundary::At(base) => dir.starts_with(base), + Boundary::Inside(base) => dir != *base && dir.starts_with(base), + } + } +} + impl IgnoreFiles { /// Whether the ignore files definitively ignore the given path. `dir` is the directory - /// containing the path, and `below` optionally restricts the chain to ignore files at or - /// below the given directory. + /// containing the path, and `boundary` restricts which ignore files of the chain apply. /// /// The deepest ignore file with a definitive answer wins, matching git's precedence, so a /// path that a deeper ignore file re-includes via a `!` pattern is not ignored. - fn is_ignored(&self, path: &Path, is_dir: bool, dir: &Path, below: Option<&Path>) -> bool { + fn is_ignored(&self, path: &Path, is_dir: bool, dir: &Path, boundary: Boundary) -> bool { for matcher in self.chain(dir).iter() { - if below.is_some_and(|below| !matcher.path().starts_with(below)) { + if !boundary.applies_to(matcher.path()) { // Chains are ordered deepest first, so nothing below the boundary can follow break; } diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index ef717483d..25b1a7598 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -41,9 +41,11 @@ pub enum SourceEntry { /// ``` /// /// Being explicit bypasses the ignoredness of the directory: everything inside is scanned - /// as if it were a regular auto source, except that `.gitignore` files no longer apply - /// inside (git ignores the whole tree anyway). The default rules still apply inside, so - /// e.g. nested `node_modules` stay ignored. + /// as if it were a regular auto source, except that `.gitignore` files from at or above + /// the directory no longer apply — they (including the self-ignoring `*` file that + /// generators typically place inside such directories) are what made it ignored in the + /// first place. `.gitignore` files *deeper inside* the directory still apply, and so do + /// the default rules, so e.g. nested `node_modules` stay ignored. External { base: PathBuf }, /// Explicit source pattern regardless of any auto source detection rules diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index d06d2b95a..ed42a9174 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4272,6 +4272,56 @@ mod scanner { assert_eq!(files, vec!["node_modules/my-lib/dist/index.html"]); } + #[test] + fn external_sources_respect_gitignore_files_inside_of_them() { + // Listing an ignored directory explicitly bypasses the rules that made it ignored: + // `.gitignore` files above the base, and the base's own `.gitignore` — inside ignored + // trees that is typically the self-ignoring `*` file that generators drop into the + // directory. `.gitignore` files *deeper inside* the external tree were put there + // deliberately by whatever generates those files, so they still apply, just like they + // do for auto and pattern sources. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "node_modules"), + // The self-ignoring device at the base: bypassed + ("node_modules/.generated/.gitignore", "*"), + // A deliberate ignore deeper inside: applies + ("node_modules/.generated/ui/.gitignore", "ignored.tsx\ncache/"), + ( + "node_modules/.generated/ui/button.tsx", + "content-['button.tsx']", + ), + ( + "node_modules/.generated/ui/input.tsx", + "content-['input.tsx']", + ), + ( + "node_modules/.generated/ui/ignored.tsx", + "content-['ignored.tsx']", + ), + ( + "node_modules/.generated/ui/cache/stale.tsx", + "content-['cache/stale.tsx']", + ), + ], + vec!["@source './node_modules/.generated'"], + ); + + assert_eq!( + candidates, + vec!["content-['button.tsx']", "content-['input.tsx']"] + ); + assert_eq!( + files, + vec![ + "node_modules/.generated/ui/button.tsx", + "node_modules/.generated/ui/input.tsx" + ] + ); + } + #[test] fn later_pattern_sources_override_earlier_not_sources() { // Order matters: a later positive `@source` wins from an earlier `@source not`, and From 7c89b317d6db1e8cf406bac62f65e24b0944b3cb Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Thu, 13 Aug 2026 00:53:33 +0200 Subject: [PATCH 22/23] more cases --- .../src/scanner/fixtures/ignored-files.txt | 1 + crates/oxide/src/scanner/sources.rs | 31 ++++ crates/oxide/tests/scanner.rs | 148 ++++++++++++++++++ 3 files changed, 180 insertions(+) diff --git a/crates/oxide/src/scanner/fixtures/ignored-files.txt b/crates/oxide/src/scanner/fixtures/ignored-files.txt index 1e8f1029f..fe9aeb1bf 100644 --- a/crates/oxide/src/scanner/fixtures/ignored-files.txt +++ b/crates/oxide/src/scanner/fixtures/ignored-files.txt @@ -2,5 +2,6 @@ package-lock.json pnpm-lock.yaml bun.lockb .gitignore +.ignore .env .env.* diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 25b1a7598..15ca9ae54 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -192,6 +192,12 @@ impl PublicSourceEntry { else if !self.pattern.starts_with("/") { self.pattern = format!("/{}", self.pattern); } + + // `src/**` means everything underneath `src`, just like `src/**/*` and `src` do. + // Normalize it so all three are classified as auto source detection. + if self.pattern == "/**" { + self.pattern = "/**/*".to_owned(); + } } } @@ -325,6 +331,31 @@ mod tests { assert_eq!(source.pattern, "/**/*.html"); } + #[test] + fn optimize_normalizes_double_star_to_auto_source_detection() { + let dir = tempdir().unwrap(); + fs::create_dir_all(dir.path().join("src")).unwrap(); + + let mut source = PublicSourceEntry { + base: dir.path().to_string_lossy().to_string(), + pattern: "src/**".to_string(), + negated: false, + }; + + source.optimize(); + + assert_eq!(source.pattern, "/**/*"); + + // …and therefore `src/**` is classified as an auto source, like `src/**/*` and `src` + let base = dunce::canonicalize(dir.path().join("src")).unwrap(); + let sources = public_source_entries_to_private_source_entries(vec![PublicSourceEntry { + base: dir.path().to_string_lossy().to_string(), + pattern: "src/**".to_string(), + negated: false, + }]); + assert_eq!(sources, vec![SourceEntry::Auto { base }]); + } + #[test] fn sources_are_converted_in_order() { let dir = tempdir().unwrap(); diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index ed42a9174..f21e4d642 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -4322,6 +4322,154 @@ mod scanner { ); } + #[test] + fn double_star_sources_are_auto_sources() { + // `@source "./src/**"` means "everything underneath src", just like `./src/**/*` and + // `./src`: auto source detection applies, so `.gitignore` files and the default rules + // are respected instead of the glob rescuing every git ignored file. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "src/secret.html"), + ("src/index.html", "content-['src/index.html']"), + ("src/secret.html", "content-['src/secret.html']"), + ("src/logo.png", "content-['src/logo.png']"), + ], + vec!["@source './src/**'"], + ); + + assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_eq!(files, vec!["src/index.html"]); + } + + #[test] + fn gitignore_whitelists_do_not_reinclude_default_ignored_content() { + // A `.gitignore` whitelist re-includes files for git, but the default auto source + // detection rules are independent of git: `node_modules`, binary files, etc. stay + // ignored even when a `.gitignore` explicitly whitelists them. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "!node_modules/\n!*.png"), + ("index.html", "content-['index.html']"), + ("logo.png", "content-['logo.png']"), + ( + "node_modules/lib/index.html", + "content-['node_modules/lib/index.html']", + ), + ], + vec!["@source '**/*'"], + ); + + assert_eq!(candidates, vec!["content-['index.html']"]); + assert_eq!(files, vec!["index.html"]); + } + + #[test] + fn dot_ignore_files_are_respected() { + // `.ignore` files (the gitignore-style files used by ripgrep and friends) work like + // `.gitignore` files and rank above them within the same directory. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "by-git.html\nby-both.html"), + (".ignore", "by-ignore.html\n!by-both.html"), + ("keep.html", "content-['keep.html']"), + ("by-git.html", "content-['by-git.html']"), + ("by-ignore.html", "content-['by-ignore.html']"), + // Ignored by the `.gitignore` file, re-included by the `.ignore` file + ("by-both.html", "content-['by-both.html']"), + ], + vec!["@source '**/*'"], + ); + + assert_eq!( + candidates, + vec!["content-['by-both.html']", "content-['keep.html']"] + ); + assert_eq!(files, vec!["by-both.html", "keep.html"]); + } + + #[test] + fn git_info_exclude_is_respected() { + // `.git/info/exclude` works like the repository root's `.gitignore` file, ranking + // below it. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".git/info/exclude", "excluded.html\nreincluded.html"), + (".gitignore", "!reincluded.html"), + ("keep.html", "content-['keep.html']"), + ("excluded.html", "content-['excluded.html']"), + // Excluded by `.git/info/exclude`, re-included by the `.gitignore` file + ("reincluded.html", "content-['reincluded.html']"), + ], + vec!["@source '**/*'"], + ); + + assert_eq!( + candidates, + vec!["content-['keep.html']", "content-['reincluded.html']"] + ); + assert_eq!(files, vec!["keep.html", "reincluded.html"]); + } + + #[test] + fn not_sources_apply_inside_external_sources() { + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + (".gitignore", "node_modules"), + ( + "node_modules/lib/src/index.html", + "content-['node_modules/lib/src/index.html']", + ), + ( + "node_modules/lib/dist/index.html", + "content-['node_modules/lib/dist/index.html']", + ), + ], + vec![ + "@source './node_modules/lib'", + "@source not './node_modules/lib/dist'", + ], + ); + + assert_eq!( + candidates, + vec!["content-['node_modules/lib/src/index.html']"] + ); + assert_eq!(files, vec!["node_modules/lib/src/index.html"]); + } + + #[test] + fn wildcards_in_glob_sources_respect_gitignores_deeper_in_the_subtree() { + // Like `wildcards_in_glob_sources_do_not_descend_into_gitignored_directories`, but the + // `.gitignore` sits in a directory between the glob's base and the ignored directory, + // not at the base itself. + let ScanResult { + candidates, files, .. + } = scan_with_globs( + &[ + ("blog/2024/.gitignore", "drafts/"), + ("blog/2024/keep.html", "content-['blog/2024/keep.html']"), + ( + "blog/2024/drafts/secret.html", + "content-['blog/2024/drafts/secret.html']", + ), + ], + vec!["@source './blog/**/*.html'"], + ); + + assert_eq!(candidates, vec!["content-['blog/2024/keep.html']"]); + assert_eq!(files, vec!["blog/2024/keep.html"]); + } + #[test] fn later_pattern_sources_override_earlier_not_sources() { // Order matters: a later positive `@source` wins from an earlier `@source not`, and From ea25394532f5d11656f45166bc702a3fddac2118 Mon Sep 17 00:00:00 2001 From: Robin Malfait Date: Thu, 13 Aug 2026 11:23:32 +0200 Subject: [PATCH 23/23] visualize new walker tests --- crates/oxide/tests/scanner.rs | 409 +++++++++++++++++++++++++++++++--- 1 file changed, 377 insertions(+), 32 deletions(-) diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index f21e4d642..42c08d9b3 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -3954,7 +3954,10 @@ mod scanner { // (`*` + `!.gitignore`). An explicit glob through such a directory should still find // matching files, but nothing else. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("storage/.gitignore", "*\n!.gitignore"), @@ -3972,6 +3975,21 @@ mod scanner { vec!["@source './storage/cms/**/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ storage + ├── ✗ .gitignore + │ * + │ !.gitignore + ├── ✓ cms + │ ├── ✗ data.json + │ ├── ✓ nested + │ │ └── ✓ deep.html + │ └── ✓ section.html + └── ✗ other + └── ✗ other.html + "); + assert_eq!( candidates, vec![ @@ -3991,7 +4009,10 @@ mod scanner { // Same as above, but the ignore comes from an ancestor `.gitignore` rather than one // inside the ignored directory itself. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "/storage"), @@ -4008,6 +4029,18 @@ mod scanner { vec!["@source './storage/cms/**/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /storage + └── ✓ storage + └── ✓ cms + ├── ✓ nested + │ └── ✓ deep.html + ├── ✗ script.js + └── ✓ section.html + "); + assert_eq!( candidates, vec![ @@ -4026,7 +4059,10 @@ mod scanner { // `@source "./foo/*.html"` where `foo` is git ignored: only the `.html` files were asked // for. Other files in `foo` must not be scanned, even when a broader auto source exists. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "/foo"), @@ -4041,6 +4077,18 @@ mod scanner { vec!["@source '**/*'", "@source './foo/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /foo + ├── ✓ foo + │ ├── ✓ index.html + │ ├── ✗ nested + │ │ └── ✗ nested.html + │ └── ✗ script.js + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['foo/index.html']", "content-['index.html']"] @@ -4055,7 +4103,10 @@ mod scanner { // must only include that file, not its siblings, even when the project root is an auto // source. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ( @@ -4071,6 +4122,16 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ node_modules + └── ✓ .generated + └── ✓ ui + ├── ✓ button.ts + └── ✗ card.ts + "); + assert_eq!( candidates, vec!["content-['button.ts']", "content-['index.html']"] @@ -4085,7 +4146,10 @@ mod scanner { #[test] fn concrete_file_sources_behind_gitignored_directories_do_not_include_siblings() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "/generated"), @@ -4096,6 +4160,16 @@ mod scanner { vec!["@source '**/*'", "@source './generated/button.ts'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /generated + ├── ✓ generated + │ ├── ✓ button.ts + │ └── ✗ card.ts + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['button.ts']", "content-['index.html']"] @@ -4108,7 +4182,10 @@ mod scanner { // A glob source respects the default auto source detection rules: `node_modules` inside // the globbed tree stays pruned unless it is targeted explicitly. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/index.html", "content-['src/index.html']"), @@ -4120,6 +4197,15 @@ mod scanner { vec!["@source './src/**/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✗ node_modules + └── ✗ lib + └── ✗ index.html + "); + assert_eq!(candidates, vec!["content-['src/index.html']"]); assert_eq!(files, vec!["src/index.html"]); } @@ -4141,9 +4227,24 @@ mod scanner { ]; let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs(paths_with_content, vec!["@source './**/*.html'"]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ dist/ + │ ignored.html + ├── ✗ dist + │ └── ✗ index.html + ├── ✓ ignored.html + └── ✓ src + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['ignored.html']", "content-['src/index.html']"] @@ -4152,12 +4253,27 @@ mod scanner { // Being explicit about the ignored directory bypasses the ignore rule. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( paths_with_content, vec!["@source './**/*.html'", "@source './dist/**/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ dist/ + │ ignored.html + ├── ✓ dist + │ └── ✓ index.html + ├── ✓ ignored.html + └── ✓ src + └── ✓ index.html + "); + assert_eq!( candidates, vec![ @@ -4178,7 +4294,10 @@ mod scanner { // match the glob must stay excluded, even though they are only ignored "via" their // parent directory. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "gen/"), @@ -4189,6 +4308,16 @@ mod scanner { vec!["@source '**/*'", "@source './gen/**/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ gen/ + ├── ✓ gen + │ ├── ✓ a.html + │ └── ✗ b.js + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['gen/a.html']", "content-['index.html']"] @@ -4201,7 +4330,10 @@ mod scanner { // A glob that doesn't pin an extension (`**/*`-style tail after a wildcard directory) // still applies the default extension rules, so binary files are not included. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("blog/2024/post/index.html", "content-['index.html']"), @@ -4211,6 +4343,16 @@ mod scanner { vec!["@source './blog/*/post/**/*'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + └── ✓ post + ├── ✗ image.png + ├── ✓ index.html + └── ✗ styles.scss + "); + assert_eq!(candidates, vec!["content-['index.html']"]); assert_eq!(files, vec!["blog/2024/post/index.html"]); } @@ -4222,7 +4364,10 @@ mod scanner { // wildcard part of a glob is not explicit, so git ignored directories inside the // walked subtree stay pruned, even when the whitelisted glob happens to match them. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("blog/.gitignore", "drafts/"), @@ -4235,6 +4380,17 @@ mod scanner { vec!["@source './blog/*/**/*'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + ├── ✗ .gitignore + │ drafts/ + └── ✓ 2024 + ├── ✗ drafts + │ └── ✗ secret.html + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['blog/2024/keep.html']"]); assert_eq!(files, vec!["blog/2024/keep.html"]); } @@ -4245,7 +4401,10 @@ mod scanner { // default auto source detection rules inside of it: nested `node_modules` are not // scanned. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "node_modules"), @@ -4265,6 +4424,20 @@ mod scanner { vec!["@source './node_modules/my-lib'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + └── ✓ node_modules + └── ✓ my-lib + ├── ✓ dist + │ └── ✓ index.html + ├── ✗ logo.png + └── ✗ node_modules + └── ✗ dep + └── ✗ index.html + "); + assert_eq!( candidates, vec!["content-['node_modules/my-lib/dist/index.html']"] @@ -4281,14 +4454,20 @@ mod scanner { // deliberately by whatever generates those files, so they still apply, just like they // do for auto and pattern sources. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "node_modules"), // The self-ignoring device at the base: bypassed ("node_modules/.generated/.gitignore", "*"), // A deliberate ignore deeper inside: applies - ("node_modules/.generated/ui/.gitignore", "ignored.tsx\ncache/"), + ( + "node_modules/.generated/ui/.gitignore", + "ignored.tsx\ncache/", + ), ( "node_modules/.generated/ui/button.tsx", "content-['button.tsx']", @@ -4309,6 +4488,25 @@ mod scanner { vec!["@source './node_modules/.generated'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + └── ✓ node_modules + └── ✓ .generated + ├── ✗ .gitignore + │ * + └── ✓ ui + ├── ✗ .gitignore + │ ignored.tsx + │ cache/ + ├── ✓ button.tsx + ├── ✗ cache + │ └── ✗ stale.tsx + ├── ✗ ignored.tsx + └── ✓ input.tsx + "); + assert_eq!( candidates, vec!["content-['button.tsx']", "content-['input.tsx']"] @@ -4328,7 +4526,10 @@ mod scanner { // `./src`: auto source detection applies, so `.gitignore` files and the default rules // are respected instead of the glob rescuing every git ignored file. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "src/secret.html"), @@ -4339,6 +4540,16 @@ mod scanner { vec!["@source './src/**'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ src/secret.html + └── ✓ src + ├── ✓ index.html + ├── ✗ logo.png + └── ✗ secret.html + "); + assert_eq!(candidates, vec!["content-['src/index.html']"]); assert_eq!(files, vec!["src/index.html"]); } @@ -4349,7 +4560,10 @@ mod scanner { // detection rules are independent of git: `node_modules`, binary files, etc. stay // ignored even when a `.gitignore` explicitly whitelists them. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "!node_modules/\n!*.png"), @@ -4363,6 +4577,18 @@ mod scanner { vec!["@source '**/*'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ !node_modules/ + │ !*.png + ├── ✓ index.html + ├── ✗ logo.png + └── ✗ node_modules + └── ✗ lib + └── ✗ index.html + "); + assert_eq!(candidates, vec!["content-['index.html']"]); assert_eq!(files, vec!["index.html"]); } @@ -4372,7 +4598,10 @@ mod scanner { // `.ignore` files (the gitignore-style files used by ripgrep and friends) work like // `.gitignore` files and rank above them within the same directory. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "by-git.html\nby-both.html"), @@ -4386,6 +4615,18 @@ mod scanner { vec!["@source '**/*'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ by-git.html + │ by-both.html + ├── ✗ .ignore + ├── ✓ by-both.html + ├── ✗ by-git.html + ├── ✗ by-ignore.html + └── ✓ keep.html + "); + assert_eq!( candidates, vec!["content-['by-both.html']", "content-['keep.html']"] @@ -4398,7 +4639,10 @@ mod scanner { // `.git/info/exclude` works like the repository root's `.gitignore` file, ranking // below it. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".git/info/exclude", "excluded.html\nreincluded.html"), @@ -4411,6 +4655,15 @@ mod scanner { vec!["@source '**/*'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ !reincluded.html + ├── ✗ excluded.html + ├── ✓ keep.html + └── ✓ reincluded.html + "); + assert_eq!( candidates, vec!["content-['keep.html']", "content-['reincluded.html']"] @@ -4421,7 +4674,10 @@ mod scanner { #[test] fn not_sources_apply_inside_external_sources() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "node_modules"), @@ -4440,6 +4696,18 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + └── ✓ node_modules + └── ✓ lib + ├── ✗ dist + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['node_modules/lib/src/index.html']"] @@ -4453,7 +4721,10 @@ mod scanner { // `.gitignore` sits in a directory between the glob's base and the ignored directory, // not at the base itself. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("blog/2024/.gitignore", "drafts/"), @@ -4466,6 +4737,17 @@ mod scanner { vec!["@source './blog/**/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + ├── ✗ .gitignore + │ drafts/ + ├── ✗ drafts + │ └── ✗ secret.html + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['blog/2024/keep.html']"]); assert_eq!(files, vec!["blog/2024/keep.html"]); } @@ -4474,7 +4756,9 @@ mod scanner { fn later_pattern_sources_override_earlier_not_sources() { // Order matters: a later positive `@source` wins from an earlier `@source not`, and // vice versa. - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ ("src/keep.html", "content-['src/keep.html']"), ("src/other.html", "content-['src/other.html']"), @@ -4486,9 +4770,18 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ keep.html + └── ✗ other.html + "); + assert_eq!(candidates, vec!["content-['src/keep.html']"]); - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ ("src/keep.html", "content-['src/keep.html']"), ("src/other.html", "content-['src/other.html']"), @@ -4500,22 +4793,34 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✗ src + ├── ✗ keep.html + └── ✗ other.html + "); + assert!(candidates.is_empty()); } #[test] fn later_directory_sources_override_earlier_not_sources() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[("src/index.html", "content-['src/index.html']")], - vec![ - "@source '**/*'", - "@source not './src'", - "@source './src'", - ], + vec!["@source '**/*'", "@source not './src'", "@source './src'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ index.html + "); + assert_eq!(candidates, vec!["content-['src/index.html']"]); assert_eq!(files, vec!["src/index.html"]); } @@ -4523,7 +4828,10 @@ mod scanner { #[test] fn wildcard_not_sources_exclude_matching_directories() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/bar/index.html", "content-['src/bar/index.html']"), @@ -4532,6 +4840,15 @@ mod scanner { vec!["@source './src/**/*.html'", "@source not './src/ba*'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar + │ └── ✗ index.html + └── ✓ foo + └── ✓ index.html + "); + assert_eq!(candidates, vec!["content-['src/foo/index.html']"]); assert_eq!(files, vec!["src/foo/index.html"]); } @@ -4539,12 +4856,24 @@ mod scanner { #[test] fn explicit_file_sources_include_default_ignored_files_without_extensions() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( - &[(".env", "content-['.env']"), ("index.html", "content-['index.html']")], + &[ + (".env", "content-['.env']"), + ("index.html", "content-['index.html']"), + ], vec!["@source '.env'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ .env + └── ✗ index.html + "); + assert_eq!(candidates, vec!["content-['.env']"]); assert_eq!(files, vec![".env"]); } @@ -4552,7 +4881,10 @@ mod scanner { #[test] fn unreachable_whitelists_do_not_reinclude_explicit_source_directories() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ (".gitignore", "parent/"), @@ -4566,6 +4898,19 @@ mod scanner { vec!["@source './parent/child'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ parent/ + └── ✓ parent + ├── ✗ .gitignore + │ !child/ + └── ✓ child + ├── ✗ .gitignore + │ index.html + └── ✓ index.html + "); + assert_eq!(candidates, vec!["content-['parent/child/index.html']"]); assert_eq!(files, vec!["parent/child/index.html"]); }