diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6bcb279a4..5ae59e80f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -27,8 +27,10 @@ jobs: - name: Linux os: namespace-profile-default + # Playwright 1.62+ dropped WebKit support for macOS 14, and hangs + # instead of failing when launching WebKit on a macos-14 runner. - name: macOS - os: macos-14 + os: macos-15 # Exclude windows and macos from being built on feature branches run-all: diff --git a/CHANGELOG.md b/CHANGELOG.md index 1fe55aa6c..e5e6f6f89 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,55 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] -- Nothing yet! +### Added + +- Add `@tailwindcss/turbopack` package to run Tailwind CSS with Next.js ([20367](https://github.com/tailwindlabs/tailwindcss/pull/20367)) + +### Fixed + +- Ensure watch mode detects changes to symlinked `@source` files whose real paths aren't otherwise scanned ([#20356](https://github.com/tailwindlabs/tailwindcss/pull/20356)) +- Ensure custom variants using `@scope` wrap the generated utilities instead of nesting inside them ([#20369](https://github.com/tailwindlabs/tailwindcss/pull/20369)) +- Fix flattening of `@scope` at-rules ([#20369](https://github.com/tailwindlabs/tailwindcss/pull/20369)) +- Fix standalone declarations in `@scope`, wrap them in `:where(:scope)` ([#20369](https://github.com/tailwindlabs/tailwindcss/pull/20369)) +- Always emit a space for empty fallback values in CSS variables (e.g. `var(--tw-blur,)` → `var(--tw-blur, )`) ([#20373](https://github.com/tailwindlabs/tailwindcss/pull/20373)) +- Canonicalization: convert arbitrary breakpoint and container query variants to named equivalents (e.g. `max-[64rem]` → `max-lg`) ([#20380](https://github.com/tailwindlabs/tailwindcss/pull/20380)) +- Prevent `@tailwindcss/vite` from crashing on every edit under Vite's experimental `bundledDev` mode ([#20379](https://github.com/tailwindlabs/tailwindcss/pull/20379)) +- Ensure `@tailwindcss/oxide` falls back to WASM on platforms without native bindings ([#20383](https://github.com/tailwindlabs/tailwindcss/pull/20383)) +- Detect classes in Ruby percent literals using angle brackets or custom delimiters (e.g. `%w`, `%w|flex|`), including in Slim and Haml templates ([#20387](https://github.com/tailwindlabs/tailwindcss/pull/20387)) +- Preserve whitespace in `--default(…)` values in custom functional utilities (e.g. `--default(box alphabetic)` no longer becomes `boxalphabetic`) ([#20392](https://github.com/tailwindlabs/tailwindcss/pull/20392)) +- Don't scan gitignored directories (e.g. `node_modules` and `.git`) when the project uses a safelist-style `.gitignore` (e.g. `/*` followed by `!/…` negations) ([#20397](https://github.com/tailwindlabs/tailwindcss/pull/20397)) +- Ensure root `theme('…')` namespace lookups in JavaScript plugins and config files return the full namespace object instead of the value of its `DEFAULT` key ([#20399](https://github.com/tailwindlabs/tailwindcss/pull/20399)) +- Skip ignored directories entirely when computing watch globs (`scanner.globs`), instead of walking their full contents on every rebuild ([#20408](https://github.com/tailwindlabs/tailwindcss/pull/20408)) +- Oxide: drop invalid UTF-8 candidates ([#20389](https://github.com/tailwindlabs/tailwindcss/pull/20389)) +- `@tailwindcss/vite` no longer forces a full page reload for external files (e.g.: `.php` files) ([#20414](https://github.com/tailwindlabs/tailwindcss/issues/20414)) +- Canonicalization: don't merge utilities that reference different theme variables set to CSS-wide keywords like `unset` ([#20417](https://github.com/tailwindlabs/tailwindcss/pull/20417)) +- Don't generate utilities when a modifier is used that would otherwise be silently ignored (e.g. `rounded-sm/[5]`, `shadow-sm/foo`, `stroke-2/50`) ([#20419](https://github.com/tailwindlabs/tailwindcss/pull/20419)) +- Only normalize top-level `and`, `or`, and `not` keywords in `supports-[…]` variants (e.g. `selector(a: not (.foo))` → `selector(a:not(.foo))`) ([#20420](https://github.com/tailwindlabs/tailwindcss/pull/20420)) +- Don't warn about Angular's `::ng-deep` and `:host-context()` when optimizing CSS ([#20434](https://github.com/tailwindlabs/tailwindcss/pull/20434)) +- Don't generate CSS for candidates containing an empty additional modifier (e.g. `bg-red-500/50/` and `group-hover/foo//bar:flex`) ([#20466](https://github.com/tailwindlabs/tailwindcss/pull/20466)) +- Sort `min-*`, `max-*`, and container query variants with decimal values numerically (e.g. `min-[40.25rem]` before `min-[40.5rem]`) ([#20512](https://github.com/tailwindlabs/tailwindcss/pull/20512)) +- Ensure CSS comments ending with `\*/` are closed correctly instead of swallowing the CSS that follows (e.g. `/* C:\temp\*/`) ([#20508](https://github.com/tailwindlabs/tailwindcss/pull/20508)) +- Improve style invalidation performance of `group-*` and `peer-*` variants ([#20513](https://github.com/tailwindlabs/tailwindcss/pull/20513)) + +## [4.3.3] - 2026-07-16 + +### Fixed + +- Support `--watch --poll[=ms]` in `@tailwindcss/cli` when filesystem events are unreliable or unavailable ([#20297](https://github.com/tailwindlabs/tailwindcss/pull/20297)) +- Canonicalization: match arbitrary hex colors against theme colors case-insensitively (e.g. `bg-[#fff]` and `bg-[#FFF]` → `bg-white`) ([#20298](https://github.com/tailwindlabs/tailwindcss/pull/20298)) +- Prevent Preflight from overriding Firefox's native `iframe:focus-visible` outline styles ([#20292](https://github.com/tailwindlabs/tailwindcss/pull/20292)) +- Ensure `theme('colors.foo')` in JS plugins resolves correctly when both `--color-foo` and `--color-foo-bar` exist ([#20299](https://github.com/tailwindlabs/tailwindcss/pull/20299)) +- Ensure fractional opacity modifiers work with named shadow sizes like `shadow-sm/12.5`, `text-shadow-sm/12.5`, `drop-shadow-sm/12.5`, and `inset-shadow-sm/12.5` ([#20302](https://github.com/tailwindlabs/tailwindcss/pull/20302)) +- Parse selectors like `[data-foo]div` as two selectors instead of one ([#20303](https://github.com/tailwindlabs/tailwindcss/pull/20303)) +- Ensure `@tailwindcss/postcss` rebuilds when a preprocessor like Sass changes the input CSS without changing the input file on disk ([#20310](https://github.com/tailwindlabs/tailwindcss/pull/20310)) +- Ensure CSS nesting is handled even when Lightning CSS isn't run, such as in `@tailwindcss/browser` and Tailwind Play ([#20124](https://github.com/tailwindlabs/tailwindcss/pull/20124)) +- Prevent achromatic theme colors from shifting hue when mixed in polar color spaces like `oklch` ([#20314](https://github.com/tailwindlabs/tailwindcss/pull/20314)) +- Ensure `--spacing(0)` is optimized to `0px` instead of `0` so it remains a `` when used in `calc(…)` ([#20319](https://github.com/tailwindlabs/tailwindcss/pull/20319)) +- Load `@parcel/watcher` only when needed in `@tailwindcss/cli --watch` mode, so one-off builds and `--watch --poll` work when `@parcel/watcher` can't be loaded ([#20325](https://github.com/tailwindlabs/tailwindcss/pull/20325)) +- Use explicit platform fonts instead of `system-ui` and `ui-sans-serif` so CJK text respects the page's `lang` attribute on Windows ([#20318](https://github.com/tailwindlabs/tailwindcss/pull/20318)) +- Prevent `@tailwindcss/upgrade` from rewriting ignored files when run from a subdirectory ([#20329](https://github.com/tailwindlabs/tailwindcss/pull/20329)) +- Ensure earlier `@source` rules pointing to nested files are scanned when later `@source` rules point to files in parent folders ([#20335](https://github.com/tailwindlabs/tailwindcss/pull/20335)) +- Prevent `@tailwindcss/vite` from triggering full page reloads when scanned files are processed by Vite but haven't been loaded as modules yet ([#20336](https://github.com/tailwindlabs/tailwindcss/pull/20336)) ## [4.3.2] - 2026-06-26 @@ -4075,7 +4123,8 @@ No release notes - Everything! -[unreleased]: https://github.com/tailwindlabs/tailwindcss/compare/v4.3.2...HEAD +[unreleased]: https://github.com/tailwindlabs/tailwindcss/compare/v4.3.3...HEAD +[4.3.3]: https://github.com/tailwindlabs/tailwindcss/compare/v4.3.2...v4.3.3 [4.3.2]: https://github.com/tailwindlabs/tailwindcss/compare/v4.3.1...v4.3.2 [4.3.1]: https://github.com/tailwindlabs/tailwindcss/compare/v4.3.0...v4.3.1 [4.3.0]: https://github.com/tailwindlabs/tailwindcss/compare/v4.2.4...v4.3.0 diff --git a/Cargo.lock b/Cargo.lock index d8459d235..73fd9e995 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -40,7 +40,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "531a9155a481e2ee699d4f98f43c0ca4ff8ee1bfd55c31e9e98fb29d2b176fe0" dependencies = [ "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "serde", ] @@ -59,6 +59,17 @@ dependencies = [ "syn", ] +[[package]] +name = "console" +version = "0.16.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4fe5f465a4f6fee88fad41b85d990f84c835335e85b5d9e6e63e0d06d28cba7c" +dependencies = [ + "encode_unicode", + "libc", + "windows-sys 0.61.2", +] + [[package]] name = "convert_case" version = "0.11.0" @@ -104,19 +115,9 @@ checksum = "22ec99545bb0ed0ea7bb9b8e1e9122ea386ff8a48c0922e43f36d45ab09e0e80" [[package]] name = "ctor" -version = "0.10.1" +version = "1.0.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83cf0d42651b16c6dfe68685716d18480d18a9c39c62d76e8cf3eb6ed5d8bcbf" -dependencies = [ - "ctor-proc-macro", - "dtor", -] - -[[package]] -name = "ctor-proc-macro" -version = "0.0.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7a949c44fcacbbbb7ada007dc7acb34603dd97cd47de5d054f2b6493ecebb483" +checksum = "2d83cb7e7a873830708d6b02a78cd36a592c6fa14bf267b68725103b85c0d77f" [[package]] name = "diff" @@ -124,21 +125,6 @@ version = "0.1.13" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "56254986775e3233ffa9c4d7d3faaf6d36a2c09d30b20687e9f88bc8bafc16c8" -[[package]] -name = "dtor" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "edf234dd1594d6dd434a8fb8cada51ddbbc593e40e4a01556a0b31c62da2775b" -dependencies = [ - "dtor-proc-macro", -] - -[[package]] -name = "dtor-proc-macro" -version = "0.0.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2647271c92754afcb174e758003cfd1cbf1e43e5a7853d7b1813e63e19e39a73" - [[package]] name = "dunce" version = "1.0.5" @@ -151,6 +137,12 @@ version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7fcaabb2fef8c910e7f4c7ce9f67a1283a1715879a7c230ca9d6d1ae31f16d91" +[[package]] +name = "encode_unicode" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" + [[package]] name = "errno" version = "0.3.9" @@ -266,14 +258,14 @@ dependencies = [ [[package]] name = "globset" -version = "0.4.17" +version = "0.4.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eab69130804d941f8075cfd713bf8848a2c3b3f201a9457a11e6f87e1ab62305" +checksum = "07c34a9410465b45bd9787443bc7370f37735bad04b0f0cd57ff1a3186c98988" dependencies = [ "aho-corasick", "bstr", "log", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "regex-syntax 0.8.5", ] @@ -298,7 +290,7 @@ dependencies = [ "globset", "log", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "same-file", "walkdir", "winapi-util", @@ -306,7 +298,7 @@ dependencies = [ [[package]] name = "ignore" -version = "0.4.24" +version = "0.4.33" dependencies = [ "bstr", "crossbeam-channel", @@ -315,12 +307,24 @@ dependencies = [ "globset", "log", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "same-file", "walkdir", "winapi-util", ] +[[package]] +name = "insta" +version = "1.48.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "86f0f8fee8c926415c58d6ae43a08523a26faccb2323f5e6b644fe7dd4ef6b82" +dependencies = [ + "console", + "once_cell", + "similar", + "tempfile", +] + [[package]] name = "itertools" version = "0.11.0" @@ -387,9 +391,9 @@ checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a" [[package]] name = "napi" -version = "3.8.5" +version = "3.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa73b028610e2b26e9e40bd2c8ff8a98e6d7ed5d67d89ebf4bfd2f992616b024" +checksum = "de33522036981030a75c231829566bc63414e08101a6f5ff4ac6cef19c8e0941" dependencies = [ "bitflags", "ctor", @@ -402,15 +406,15 @@ dependencies = [ [[package]] name = "napi-build" -version = "2.3.1" +version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d376940fd5b723c6893cd1ee3f33abbfd86acb1cd1ec079f3ab04a2a3bc4d3b1" +checksum = "c9c366d2c8c60b86fa632df75f745509b52f9128f91a6bad4c796e44abb505e1" [[package]] name = "napi-derive" -version = "3.5.4" +version = "3.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7430702d3cc05cf55f0a2c9e41d991c3b7a53f91e6146a8f282b1bfc7f3fd133" +checksum = "a49c513341a61a16a10af6efcce46b30d0822ba2d4fb197d24d33dfc199c78d5" dependencies = [ "convert_case", "ctor", @@ -422,9 +426,9 @@ dependencies = [ [[package]] name = "napi-derive-backend" -version = "5.0.3" +version = "6.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ca5a083f2c9b49a0c7d33ec75c083498849c6fcc46f5497317faa39ea77f5d5" +checksum = "d60b5d773ad46c698c8cc2cd9fde0b283d39cbb7f71c04bee633c7bdba4423bd" dependencies = [ "convert_case", "proc-macro2", @@ -435,9 +439,9 @@ dependencies = [ [[package]] name = "napi-sys" -version = "3.2.1" +version = "3.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8eb602b84d7c1edae45e50bbf1374696548f36ae179dfa667f577e384bb90c2b" +checksum = "85fbf1fa9f1babfe396d74bbbf52b3643770243e8f5b0b46715d4caf7f0dfc9a" dependencies = [ "libloading", ] @@ -470,9 +474,9 @@ dependencies = [ [[package]] name = "once_cell" -version = "1.19.0" +version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fdb12b2476b595f9358c5161aa467c2438859caa136dec86c26fdd2efe17b92" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "overload" @@ -516,9 +520,9 @@ dependencies = [ [[package]] name = "rayon" -version = "1.10.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b418a60154510ca1a002a752ca9714984e21e4241e804d32555251faf8b78ffa" +checksum = "fb39b166781f92d482534ef4b4b1b2568f42613b53e5b6c160e24cfbfa30926d" dependencies = [ "either", "rayon-core", @@ -526,9 +530,9 @@ dependencies = [ [[package]] name = "rayon-core" -version = "1.12.1" +version = "1.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1465873a3dfdaa8ae7cb14b4383657caab0b3e8a0aa9ae8e04b044854c8dfce2" +checksum = "22e18b0f0062d30d4230b2e85ff77fdfe4326feb054b9783a3460d8435c8ab91" dependencies = [ "crossbeam-deque", "crossbeam-utils", @@ -542,7 +546,7 @@ checksum = "b544ef1b4eac5dc2db33ea63606ae9ffcfac26c1416a2806ae0bf5f56b201191" dependencies = [ "aho-corasick", "memchr", - "regex-automata 0.4.8", + "regex-automata 0.4.18", "regex-syntax 0.8.5", ] @@ -557,9 +561,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.8" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "368758f23274712b504848e9d5a6f010445cc8b87a7cdb4d7cbee666c1288da3" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -627,6 +631,12 @@ dependencies = [ "lazy_static", ] +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + [[package]] name = "slab" version = "0.4.12" @@ -671,7 +681,8 @@ dependencies = [ "dunce", "fast-glob", "globwalk", - "ignore 0.4.24", + "ignore 0.4.33", + "insta", "log", "pretty_assertions", "rayon", @@ -857,6 +868,15 @@ dependencies = [ "windows-targets", ] +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + [[package]] name = "windows-targets" version = "0.52.6" diff --git a/crates/ignore/Cargo.toml b/crates/ignore/Cargo.toml index bd0352d15..d0af38ba9 100644 --- a/crates/ignore/Cargo.toml +++ b/crates/ignore/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "ignore" -version = "0.4.24" #:version +version = "0.4.33" #:version authors = ["Andrew Gallant "] description = """ A fast library for efficiently matching ignore files such as `.gitignore` @@ -12,7 +12,10 @@ repository = "https://github.com/BurntSushi/ripgrep/tree/master/crates/ignore" readme = "README.md" keywords = ["glob", "ignore", "gitignore", "pattern", "file"] license = "Unlicense OR MIT" +# CHANGED: Use an explicit edition instead of `edition.workspace = true` since this crate is +# vendored into the Tailwind CSS workspace. edition = "2024" +rust-version = "1.88" [lib] name = "ignore" @@ -20,15 +23,17 @@ bench = false [dependencies] crossbeam-deque = "0.8.3" -globset = "0.4.17" +# CHANGED: Use the published globset crate instead of a path dependency. +globset = "0.4.20" log = "0.4.20" memchr = "2.6.3" same-file = "1.0.6" walkdir = "2.4.0" +# CHANGED: Added `dunce` to canonicalize paths without UNC prefixes on Windows. dunce = "1.0.5" [dependencies.regex-automata] -version = "0.4.0" +version = "0.4.18" default-features = false features = ["std", "perf", "syntax", "meta", "nfa", "hybrid", "dfa-onepass"] diff --git a/crates/ignore/examples/walk.rs b/crates/ignore/examples/walk.rs index 9c627dc3e..c61d0515e 100644 --- a/crates/ignore/examples/walk.rs +++ b/crates/ignore/examples/walk.rs @@ -18,9 +18,7 @@ fn main() { let stdout_thread = std::thread::spawn(move || { let mut stdout = std::io::BufWriter::new(std::io::stdout()); for dent in rx { - stdout - .write_all(&Vec::from_path_lossy(dent.path())) - .unwrap(); + stdout.write_all(&Vec::from_path_lossy(dent.path())).unwrap(); stdout.write_all(b"\n").unwrap(); } }); diff --git a/crates/ignore/src/default_types.rs b/crates/ignore/src/default_types.rs index 4e060b76a..6b5bba0f0 100644 --- a/crates/ignore/src/default_types.rs +++ b/crates/ignore/src/default_types.rs @@ -47,6 +47,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["cml"], &["*.cml"]), (&["coffeescript"], &["*.coffee"]), (&["config"], &["*.cfg", "*.conf", "*.config", "*.ini"]), + (&["container"], &["*Containerfile*", "*Dockerfile*"]), (&["coq"], &["*.v"]), (&["cpp"], &[ "*.[ChH]", "*.cc", "*.[ch]pp", "*.[ch]xx", "*.hh", "*.inl", @@ -109,6 +110,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["hbs"], &["*.hbs"]), (&["hs"], &["*.hs", "*.lhs"]), (&["html"], &["*.htm", "*.html", "*.ejs"]), + (&["hurl"], &["*.hurl"]), (&["hy"], &["*.hy"]), (&["idris"], &["*.idr", "*.lidr"]), (&["janet"], &["*.janet"]), @@ -185,6 +187,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["mint"], &["*.mint"]), (&["mk"], &["mkfile"]), (&["ml"], &["*.ml"]), + (&["mojo"], &["*.mojo"]), (&["motoko"], &["*.mo"]), (&["msbuild"], &[ "*.csproj", "*.fsproj", "*.vcxproj", "*.proj", "*.props", "*.targets", @@ -206,11 +209,12 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ "*.php", "*.php3", "*.php4", "*.php5", "*.php7", "*.php8", "*.pht", "*.phtml" ]), + (&["pkgbuild"], &["PKGBUILD"]), (&["po"], &["*.po"]), (&["pod"], &["*.pod"]), (&["postscript"], &["*.eps", "*.ps"]), (&["prolog"], &["*.pl", "*.pro", "*.prolog", "*.P"]), - (&["protobuf"], &["*.proto"]), + (&["proto", "protobuf"], &["*.proto"]), (&["ps"], &["*.cdxml", "*.ps1", "*.ps1xml", "*.psd1", "*.psm1"]), (&["puppet"], &["*.epp", "*.erb", "*.pp", "*.rb"]), (&["purs"], &["*.purs"]), @@ -231,6 +235,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["red"], &["*.r", "*.red", "*.reds"]), (&["rescript"], &["*.res", "*.resi"]), (&["robot"], &["*.robot"]), + (&["rocq"], &["*.v"]), (&["rst"], &["*.rst"]), (&["ruby"], &[ // Idiomatic files @@ -274,6 +279,7 @@ pub(crate) const DEFAULT_TYPES: &[(&[&str], &[&str])] = &[ (&["spark"], &["*.spark"]), (&["spec"], &["*.spec"]), (&["sql"], &["*.sql", "*.psql"]), + (&["ssa"], &["*.ssa"]), (&["stylus"], &["*.styl"]), (&["sv"], &["*.v", "*.vg", "*.sv", "*.svh", "*.h"]), (&["svelte"], &["*.svelte", "*.svelte.ts"]), @@ -359,4 +365,14 @@ mod tests { previous_name = name; } } + + #[test] + fn default_types_aliases_are_sorted() { + for (aliases, _) in DEFAULT_TYPES.iter() { + assert!( + aliases.is_sorted(), + "this alias list is not sorted: {aliases:?}", + ); + } + } } diff --git a/crates/ignore/src/dir.rs b/crates/ignore/src/dir.rs index 11b58f8ca..6bee724c8 100644 --- a/crates/ignore/src/dir.rs +++ b/crates/ignore/src/dir.rs @@ -16,7 +16,7 @@ use std::{ collections::HashMap, ffi::{OsStr, OsString}, - fs::{File, FileType}, + fs::{self, File, FileType}, io::{self, BufRead}, path::{Path, PathBuf}, sync::{Arc, RwLock, Weak}, @@ -25,7 +25,7 @@ use std::{ use crate::{ gitignore::{self, Gitignore, GitignoreBuilder}, overrides::{self, Override}, - pathutil::{is_hidden, strip_prefix}, + pathutil::{is_hidden_entry, strip_prefix}, types::{self, Types}, walk::DirEntry, {Error, Match, PartialErrorBuilder}, @@ -91,7 +91,25 @@ struct IgnoreOptions { /// Ignore is a matcher useful for recursively walking one or more directories. #[derive(Clone, Debug)] -pub(crate) struct Ignore(Arc); +pub(crate) struct Ignore { + inner: Arc, + // Parent matchers are cached independently of the path being walked, but + // matching them still needs the canonicalized path originally passed to + // `add_parents`. For example, when walking `/tmp/project/src`, parent + // matchers use `/tmp/project/src` to rewrite `/tmp/project/src/foo.py` + // before matching it against ignore files from `/tmp/project` and its + // ancestors. + // + // For ripgrep itself, this means that `rg pat src tests` must rewrite + // `src/foo` relative to `.../src`, and not whatever root was prepared + // first. + // + // See: https://github.com/BurntSushi/ripgrep/pull/3420 + // See: https://github.com/BurntSushi/ripgrep/issues/3376 + // See: https://github.com/BurntSushi/ripgrep/issues/3419 + // See: https://github.com/BurntSushi/ripgrep/issues/3320 + absolute_base: Option>, +} #[derive(Clone, Debug)] struct IgnoreInner { @@ -112,12 +130,9 @@ struct IgnoreInner { /// /// If this is the root directory or there are otherwise no more /// directories to match, then `parent` is `None`. - parent: Option, + parent: Option>, /// Whether this is an absolute parent matcher, as added by add_parent. is_absolute_parent: bool, - /// The absolute base path of this matcher. Populated only if parent - /// directories are added. - absolute_base: Option>, /// The directory that gitignores should be interpreted relative to. /// /// Usually this is the directory containing the gitignore file. But in @@ -152,34 +167,36 @@ struct IgnoreInner { impl Ignore { /// Return the directory path of this matcher. + #[cfg(test)] pub(crate) fn path(&self) -> &Path { - &self.0.dir + &self.inner.dir } /// Return true if this matcher has no parent. pub(crate) fn is_root(&self) -> bool { - self.0.parent.is_none() - } - - /// Returns true if this matcher was added via the `add_parents` method. - pub(crate) fn is_absolute_parent(&self) -> bool { - self.0.is_absolute_parent + self.inner.parent.is_none() } /// Return this matcher's parent, if one exists. pub(crate) fn parent(&self) -> Option { - self.0.parent.clone() + self.inner.parent.as_ref().map(|parent| Ignore { + inner: parent.clone(), + absolute_base: self.absolute_base.clone(), + }) } /// Create a new `Ignore` matcher with the parent directories of `dir`. /// /// Note that this can only be called on an `Ignore` matcher with no /// parents (i.e., `is_root` returns `true`). This will panic otherwise. - pub(crate) fn add_parents>(&self, path: P) -> (Ignore, Option) { - if !self.0.opts.parents - && !self.0.opts.git_ignore - && !self.0.opts.git_exclude - && !self.0.opts.git_global + pub(crate) fn add_parents>( + &self, + path: P, + ) -> (Ignore, Option) { + if !self.inner.opts.parents + && !self.inner.opts.git_ignore + && !self.inner.opts.git_exclude + && !self.inner.opts.git_global { // If we never need info from parent directories, then don't do // anything. @@ -209,25 +226,34 @@ impl Ignore { let mut errs = PartialErrorBuilder::default(); let mut ig = self.clone(); for parent in parents.into_iter().rev() { - let mut compiled = self.0.compiled.write().unwrap(); + let mut compiled = self.inner.compiled.write().unwrap(); if let Some(weak) = compiled.get(parent.as_os_str()) { if let Some(prebuilt) = weak.upgrade() { - ig = Ignore(prebuilt); + ig = Ignore { + inner: prebuilt, + absolute_base: Some(absolute_base.clone()), + }; continue; } } let (mut igtmp, err) = ig.add_child_path(parent); errs.maybe_push(err); igtmp.is_absolute_parent = true; - igtmp.absolute_base = Some(absolute_base.clone()); - igtmp.has_git = if self.0.opts.require_git && self.0.opts.git_ignore { - parent.join(".git").exists() || parent.join(".jj").exists() - } else { - false - }; + igtmp.has_git = + if self.inner.opts.require_git && self.inner.opts.git_ignore { + parent.join(".git").exists() || parent.join(".jj").exists() + } else { + false + }; let ig_arc = Arc::new(igtmp); - ig = Ignore(ig_arc.clone()); - compiled.insert(parent.as_os_str().to_os_string(), Arc::downgrade(&ig_arc)); + ig = Ignore { + inner: ig_arc.clone(), + absolute_base: Some(absolute_base.clone()), + }; + compiled.insert( + parent.as_os_str().to_os_string(), + Arc::downgrade(&ig_arc), + ); } (ig, errs.into_error_option()) } @@ -240,60 +266,161 @@ impl Ignore { /// returned if it exists. /// /// Note that all I/O errors are completely ignored. - pub(crate) fn add_child>(&self, dir: P) -> (Ignore, Option) { + pub(crate) fn add_child>( + &self, + dir: P, + ) -> (Ignore, Option) { let (ig, err) = self.add_child_path(dir.as_ref()); - (Ignore(Arc::new(ig)), err) + ( + Ignore { + inner: Arc::new(ig), + absolute_base: self.absolute_base.clone(), + }, + err, + ) + } + + /// Like add_child, but uses successful read_dir entries to reduce + /// probing when discovering ignore files. + pub(crate) fn add_child_with_entries>( + &self, + dir: P, + entries: &[fs::DirEntry], + ) -> (Ignore, Option) { + let files = self.collect_ignore_files(entries); + let (ig, err) = self.add_child_path_with_found_ignore_files( + dir.as_ref(), + Some(&files), + ); + ( + Ignore { + inner: Arc::new(ig), + absolute_base: self.absolute_base.clone(), + }, + err, + ) } /// Like add_child, but takes a full path and returns an IgnoreInner. fn add_child_path(&self, dir: &Path) -> (IgnoreInner, Option) { - let git_type = - if self.0.opts.require_git && (self.0.opts.git_ignore || self.0.opts.git_exclude) { - dir.join(".git").metadata().ok().map(|md| md.file_type()) - } else { - None - }; - let has_git = git_type.is_some() || dir.join(".jj").exists(); + self.add_child_path_with_found_ignore_files(dir, None) + } + + fn collect_ignore_files( + &self, + entries: &[fs::DirEntry], + ) -> IgnoreFilesFound { + let custom_ignore_filenames = &self.inner.custom_ignore_filenames; + let mut files = IgnoreFilesFound { + has_ignore: false, + has_git_ignore: false, + has_git_dir: false, + has_jj_dir: false, + custom_ignore_files: vec![false; custom_ignore_filenames.len()], + }; + for entry in entries { + let file_name = entry.file_name(); + if file_name == OsStr::new(".ignore") { + files.has_ignore = true; + } else if file_name == OsStr::new(".gitignore") { + files.has_git_ignore = true; + } else if file_name == OsStr::new(".git") { + files.has_git_dir = true; + } else if file_name == OsStr::new(".jj") { + files.has_jj_dir = true; + } + for (i, name) in custom_ignore_filenames.iter().enumerate() { + if file_name == name.as_os_str() { + files.custom_ignore_files[i] = true; + } + } + } + files + } + + fn add_child_path_with_found_ignore_files( + &self, + dir: &Path, + ignore_files_list: Option<&IgnoreFilesFound>, + ) -> (IgnoreInner, Option) { + let check_vcs_dir = self.inner.opts.require_git + && (self.inner.opts.git_ignore || self.inner.opts.git_exclude); + let git_type = if check_vcs_dir + && ignore_files_list.is_none_or(|i| i.has_git_dir) + { + dir.join(".git").metadata().ok().map(|md| md.file_type()) + } else { + None + }; + let has_jj = check_vcs_dir + && ignore_files_list.is_none_or(|i| i.has_jj_dir) + && dir.join(".jj").exists(); + let has_git = check_vcs_dir && (git_type.is_some() || has_jj); let mut errs = PartialErrorBuilder::default(); - let custom_ig_matcher = if self.0.custom_ignore_filenames.is_empty() { + let custom_ig_matcher = if self + .inner + .custom_ignore_filenames + .is_empty() + { Gitignore::empty() } else { - let (m, err) = create_gitignore( - &dir, - &dir, - &self.0.custom_ignore_filenames, - self.0.opts.ignore_case_insensitive, - ); - errs.maybe_push(err); - m + let custom_ignore_names: Vec<&OsString> = match ignore_files_list { + None => self.inner.custom_ignore_filenames.iter().collect(), + Some(m) => self + .inner + .custom_ignore_filenames + .iter() + .zip(m.custom_ignore_files.iter()) + .filter(|&(_, &matched)| matched) + .map(|(name, _)| name) + .collect(), + }; + if custom_ignore_names.is_empty() { + Gitignore::empty() + } else { + let (m, err) = create_gitignore( + &dir, + &dir, + &custom_ignore_names, + self.inner.opts.ignore_case_insensitive, + ); + errs.maybe_push(err); + m + } }; - let ig_matcher = if !self.0.opts.ignore { + let ig_matcher = if !self.inner.opts.ignore + || !ignore_files_list.is_none_or(|i| i.has_ignore) + { Gitignore::empty() } else { let (m, err) = create_gitignore( &dir, &dir, &[".ignore"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m }; - let gi_matcher = if !self.0.opts.git_ignore { + let gi_matcher = if !self.inner.opts.git_ignore + || !ignore_files_list.is_none_or(|i| i.has_git_ignore) + { Gitignore::empty() } else { let (m, err) = create_gitignore( &dir, &dir, &[".gitignore"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m }; - let gi_exclude_matcher = if !self.0.opts.git_exclude { + let gi_exclude_matcher = if !self.inner.opts.git_exclude + || !ignore_files_list.is_none_or(|i| i.has_git_dir) + { Gitignore::empty() } else { match resolve_git_commondir(dir, git_type) { @@ -302,7 +429,7 @@ impl Ignore { &dir, &git_dir, &["info/exclude"], - self.0.opts.ignore_case_insensitive, + self.inner.opts.ignore_case_insensitive, ); errs.maybe_push(err); m @@ -314,32 +441,38 @@ impl Ignore { } }; let ig = IgnoreInner { - compiled: self.0.compiled.clone(), + compiled: self.inner.compiled.clone(), dir: dir.to_path_buf(), - overrides: self.0.overrides.clone(), - types: self.0.types.clone(), - parent: Some(self.clone()), + overrides: self.inner.overrides.clone(), + types: self.inner.types.clone(), + parent: Some(self.inner.clone()), is_absolute_parent: false, - absolute_base: self.0.absolute_base.clone(), - global_gitignores_relative_to: self.0.global_gitignores_relative_to.clone(), - explicit_ignores: self.0.explicit_ignores.clone(), - custom_ignore_filenames: self.0.custom_ignore_filenames.clone(), + global_gitignores_relative_to: self + .inner + .global_gitignores_relative_to + .clone(), + explicit_ignores: self.inner.explicit_ignores.clone(), + custom_ignore_filenames: self + .inner + .custom_ignore_filenames + .clone(), custom_ignore_matcher: custom_ig_matcher, ignore_matcher: ig_matcher, - git_global_matcher: self.0.git_global_matcher.clone(), + git_global_matcher: self.inner.git_global_matcher.clone(), git_ignore_matcher: gi_matcher, git_exclude_matcher: gi_exclude_matcher, has_git, - opts: self.0.opts, + opts: self.inner.opts, }; (ig, errs.into_error_option()) } /// Returns true if at least one type of ignore rule should be matched. fn has_any_ignore_rules(&self) -> bool { - let opts = self.0.opts; - let has_custom_ignore_files = !self.0.custom_ignore_filenames.is_empty(); - let has_explicit_ignores = !self.0.explicit_ignores.is_empty(); + let opts = self.inner.opts; + let has_custom_ignore_files = + !self.inner.custom_ignore_filenames.is_empty(); + let has_explicit_ignores = !self.inner.explicit_ignores.is_empty(); opts.ignore || opts.git_global @@ -350,9 +483,12 @@ impl Ignore { } /// Like `matched`, but works with a directory entry instead. - pub(crate) fn matched_dir_entry<'a>(&'a self, dent: &DirEntry) -> Match> { + pub(crate) fn matched_dir_entry<'a>( + &'a self, + dent: &DirEntry, + ) -> Match> { let m = self.matched(dent.path(), dent.is_dir()); - if m.is_none() && self.0.opts.hidden && is_hidden(dent) { + if m.is_none() && self.inner.opts.hidden && is_hidden_entry(dent) { return Match::Ignore(IgnoreMatch::hidden()); } m @@ -362,7 +498,11 @@ impl Ignore { /// ignored or not. /// /// The match contains information about its origin. - fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub(crate) fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { // We need to be careful with our path. If it has a leading ./, then // strip it because it causes nothing but trouble. let mut path = path.as_ref(); @@ -373,9 +513,9 @@ impl Ignore { // regardless of whether it's whitelist/ignore, then we quit and // return that result immediately. Overrides have the highest // precedence. - if !self.0.overrides.is_empty() { + if !self.inner.overrides.is_empty() { let mat = self - .0 + .inner .overrides .matched(path, is_dir) .map(IgnoreMatch::overrides); @@ -392,8 +532,9 @@ impl Ignore { whitelisted = mat; } } - if !self.0.types.is_empty() { - let mat = self.0.types.matched(path, is_dir).map(IgnoreMatch::types); + if !self.inner.types.is_empty() { + let mat = + self.inner.types.matched(path, is_dir).map(IgnoreMatch::types); if mat.is_ignore() { return mat; } else if mat.is_whitelist() { @@ -405,98 +546,143 @@ impl Ignore { /// Performs matching only on the ignore files for this directory and /// all parent directories. - fn matched_ignore<'a>(&'a self, path: &Path, is_dir: bool) -> Match> { - let (mut m_custom_ignore, mut m_ignore, mut m_gi, mut m_gi_exclude, mut m_explicit) = ( - Match::None, - Match::None, - Match::None, - Match::None, - Match::None, - ); - let any_git = !self.0.opts.require_git || self.parents().any(|ig| ig.0.has_git); + pub(crate) fn matched_ignore<'a>( + &'a self, + path: &Path, + is_dir: bool, + ) -> Match> { + let ( + mut m_custom_ignore, + mut m_ignore, + mut m_gi, + mut m_gi_exclude, + mut m_explicit, + ) = (Match::None, Match::None, Match::None, Match::None, Match::None); + let any_git = !self.inner.opts.require_git + || self.parents().any(|ig| ig.inner.has_git); let mut saw_git = false; - for ig in self.parents().take_while(|ig| !ig.0.is_absolute_parent) { + for ig in self.parents().take_while(|ig| !ig.inner.is_absolute_parent) + { if m_custom_ignore.is_none() { - m_custom_ignore = - ig.0.custom_ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_custom_ignore = ig + .inner + .custom_ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if m_ignore.is_none() { - m_ignore = - ig.0.ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_ignore = ig + .inner + .ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi.is_none() { - m_gi = - ig.0.git_ignore_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi = ig + .inner + .git_ignore_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi_exclude.is_none() { - m_gi_exclude = - ig.0.git_exclude_matcher - .matched(path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi_exclude = ig + .inner + .git_exclude_matcher + .matched(path, is_dir) + .map(IgnoreMatch::gitignore); } - saw_git = saw_git || ig.0.has_git; + saw_git = saw_git || ig.inner.has_git; } - if self.0.opts.parents { - if let Some(_) = self.absolute_base() { - // CHANGED: We removed a code path that rewrote the `path` to be relative to - // `self.absolute_base()` because it assumed that the every path is inside the base - // which is not the case for us as we use `WalkBuilder#add` to add roots outside of the - // base. - for ig in self.parents().skip_while(|ig| !ig.0.is_absolute_parent) { + if self.inner.opts.parents { + if let Some(abs_parent_path) = self.absolute_base() { + // What we want to do here is take the absolute base path of + // this directory and join it with the path we're searching. + // The main issue we want to avoid is accidentally duplicating + // directory components, so we try to strip any common prefix + // off of `path`. Overall, this seems a little ham-fisted, but + // it does fix a nasty bug. It should do fine until we overhaul + // this crate. + let path = abs_parent_path.join( + self.parents() + .take_while(|ig| !ig.inner.is_absolute_parent) + .last() + .map_or(path, |ig| { + // This is a weird special case when ripgrep users + // search with just a `.`, as some tools do + // automatically (like consult). In this case, if + // we don't bail out now, the code below will strip + // a leading `.` from `path`, which might mangle + // a hidden file name! + if ig.inner.dir.as_path() == Path::new(".") { + return path; + } + let without_dot_slash = strip_if_is_prefix( + "./", + ig.inner.dir.as_path(), + ); + let relative_base = + strip_if_is_prefix(without_dot_slash, path); + strip_if_is_prefix("/", relative_base) + }), + ); + + for ig in self + .parents() + .skip_while(|ig| !ig.inner.is_absolute_parent) + { if m_custom_ignore.is_none() { - m_custom_ignore = - ig.0.custom_ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_custom_ignore = ig + .inner + .custom_ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if m_ignore.is_none() { - m_ignore = - ig.0.ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_ignore = ig + .inner + .ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi.is_none() { - m_gi = - ig.0.git_ignore_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi = ig + .inner + .git_ignore_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } if any_git && !saw_git && m_gi_exclude.is_none() { - m_gi_exclude = - ig.0.git_exclude_matcher - .matched(&path, is_dir) - .map(IgnoreMatch::gitignore); + m_gi_exclude = ig + .inner + .git_exclude_matcher + .matched(&path, is_dir) + .map(IgnoreMatch::gitignore); } - saw_git = saw_git || ig.0.has_git; + saw_git = saw_git || ig.inner.has_git; } } } - for gi in self.0.explicit_ignores.iter().rev() { - // CHANGED: We need to make sure that the explicit gitignore rules apply to the path - // - // path = Is the current file/folder we are traversing - // gi.path() = Is the path of the custom gitignore file - // - // E.g.: If we have a custom rule for `/src/utils` with `**/*`, and we are looking at - // just `/src`, then the `**/*` rules do not apply to this folder, so we can - // ignore the current custom gitignore file. - // - if !path.starts_with(gi.path()) { - continue; - } + for gi in self.inner.explicit_ignores.iter().rev() { if !m_explicit.is_none() { break; } + // CHANGED: We need to make sure that the explicit gitignore rules + // apply to the path + // + // path = Is the current file/folder we are traversing + // gi.path() = Is the path of the custom gitignore file + // + // E.g.: If we have a custom rule for `/src/utils` with `**/*`, and + // we are looking at just `/src`, then the `**/*` rules do + // not apply to this folder, so we can ignore the current + // custom gitignore file. + if !path.starts_with(gi.path()) { + continue; + } m_explicit = gi.matched(&path, is_dir).map(IgnoreMatch::gitignore); } let m_global = if any_git { - self.0 + self.inner .git_global_matcher .matched(&path, is_dir) .map(IgnoreMatch::gitignore) @@ -504,59 +690,90 @@ impl Ignore { Match::None }; - // CHANGED: We added logic to configure an order in which the ignore files are respected and - // allowed a whitelist in a later file to overrule a block on an earlier file. + // CHANGED: We added logic to configure an order in which the ignore + // files are respected. Explicitly added ignores (via + // `WalkBuilder::add_gitignore`) take precedence over all ignore files + // found on disk, and the first source with a definitive answer wins. let order = [ // Manually added ignores - &m_explicit, + m_explicit, // .custom-ignore - &m_custom_ignore, + m_custom_ignore, // .ignore - &m_ignore, + m_ignore, // .gitignore - &m_gi, + m_gi, // .git/info/exclude - &m_gi_exclude, + m_gi_exclude, // Global gitignore - &m_global, + m_global, ]; - for check in order.into_iter() { - if check.is_none() { - continue; + if !check.is_none() { + return check; } - - return check.clone(); } - - m_explicit + Match::None } /// Returns an iterator over parent ignore matchers, including this one. pub(crate) fn parents(&self) -> Parents<'_> { - Parents(Some(self)) + Parents(Some(IgnoreRef { inner: &self.inner })) } /// Returns the first absolute path of the first absolute parent, if /// one exists. fn absolute_base(&self) -> Option<&Path> { - self.0.absolute_base.as_ref().map(|p| &***p) + self.absolute_base.as_ref().map(|p| &***p) + } +} + +/// State for tracking what kinds of files ripgrep is interested in for a +/// given directory. +/// +/// This is computed over the entire set of files in a directory instead of +/// trying to stat each file individually. If a file is present, it's only then +/// that we stat it for more information, instead of relying on the stat to +/// determine its existence. +#[derive(Debug)] +struct IgnoreFilesFound { + has_ignore: bool, + has_git_ignore: bool, + has_git_dir: bool, + has_jj_dir: bool, + custom_ignore_files: Vec, +} + +#[derive(Clone, Copy)] +pub(crate) struct IgnoreRef<'a> { + inner: &'a IgnoreInner, +} + +impl IgnoreRef<'_> { + pub(crate) fn path(&self) -> &Path { + &self.inner.dir + } + + pub(crate) fn is_absolute_parent(&self) -> bool { + self.inner.is_absolute_parent } } /// An iterator over all parents of an ignore matcher, including itself. -/// -/// The lifetime `'a` refers to the lifetime of the initial `Ignore` matcher. -pub(crate) struct Parents<'a>(Option<&'a Ignore>); +pub(crate) struct Parents<'a>(Option>); impl<'a> Iterator for Parents<'a> { - type Item = &'a Ignore; + type Item = IgnoreRef<'a>; - fn next(&mut self) -> Option<&'a Ignore> { + fn next(&mut self) -> Option> { match self.0.take() { None => None, Some(ig) => { - self.0 = ig.0.parent.as_ref(); + self.0 = ig + .inner + .parent + .as_deref() + .map(|inner| IgnoreRef { inner }); Some(ig) } } @@ -645,33 +862,42 @@ impl IgnoreBuilder { } gi } else { - log::debug!("ignoring global gitignore file because CWD is not known"); + log::debug!( + "ignoring global gitignore file because CWD is not known" + ); Gitignore::empty() }; - Ignore(Arc::new(IgnoreInner { - compiled: Arc::new(RwLock::new(HashMap::new())), - dir: self.dir.clone(), - overrides: self.overrides.clone(), - types: self.types.clone(), - parent: None, - is_absolute_parent: true, + Ignore { + inner: Arc::new(IgnoreInner { + compiled: Arc::new(RwLock::new(HashMap::new())), + dir: self.dir.clone(), + overrides: self.overrides.clone(), + types: self.types.clone(), + parent: None, + is_absolute_parent: true, + global_gitignores_relative_to, + explicit_ignores: Arc::new(self.explicit_ignores.clone()), + custom_ignore_filenames: Arc::new( + self.custom_ignore_filenames.clone(), + ), + custom_ignore_matcher: Gitignore::empty(), + ignore_matcher: Gitignore::empty(), + git_global_matcher: Arc::new(git_global_matcher), + git_ignore_matcher: Gitignore::empty(), + git_exclude_matcher: Gitignore::empty(), + has_git: false, + opts: self.opts, + }), absolute_base: None, - global_gitignores_relative_to, - explicit_ignores: Arc::new(self.explicit_ignores.clone()), - custom_ignore_filenames: Arc::new(self.custom_ignore_filenames.clone()), - custom_ignore_matcher: Gitignore::empty(), - ignore_matcher: Gitignore::empty(), - git_global_matcher: Arc::new(git_global_matcher), - git_ignore_matcher: Gitignore::empty(), - git_exclude_matcher: Gitignore::empty(), - has_git: false, - opts: self.opts, - })) + } } /// Set the current directory used for matching global gitignores. - pub(crate) fn current_dir(&mut self, cwd: impl Into) -> &mut IgnoreBuilder { + pub(crate) fn current_dir( + &mut self, + cwd: impl Into, + ) -> &mut IgnoreBuilder { self.global_gitignores_relative_to = Some(cwd.into()); self } @@ -681,7 +907,10 @@ impl IgnoreBuilder { /// By default, no override matcher is used. /// /// This overrides any previous setting. - pub(crate) fn overrides(&mut self, overrides: Override) -> &mut IgnoreBuilder { + pub(crate) fn overrides( + &mut self, + overrides: Override, + ) -> &mut IgnoreBuilder { self.overrides = Arc::new(overrides); self } @@ -712,8 +941,7 @@ impl IgnoreBuilder { &mut self, file_name: S, ) -> &mut IgnoreBuilder { - self.custom_ignore_filenames - .push(file_name.as_ref().to_os_string()); + self.custom_ignore_filenames.push(file_name.as_ref().to_os_string()); self } @@ -725,6 +953,11 @@ impl IgnoreBuilder { self } + /// Whether ignoring hidden files is enabled or not. + pub(crate) fn is_hidden(&self) -> bool { + self.opts.hidden + } + /// Enables reading `.ignore` files. /// /// `.ignore` files have the same semantics as `gitignore` files and are @@ -795,7 +1028,10 @@ impl IgnoreBuilder { /// Process ignore files case insensitively /// /// This is disabled by default. - pub(crate) fn ignore_case_insensitive(&mut self, yes: bool) -> &mut IgnoreBuilder { + pub(crate) fn ignore_case_insensitive( + &mut self, + yes: bool, + ) -> &mut IgnoreBuilder { self.opts.ignore_case_insensitive = yes; self } @@ -854,7 +1090,10 @@ pub(crate) fn create_gitignore>( /// them when multiple repositories are searched. /// /// Some I/O errors are ignored. -fn resolve_git_commondir(dir: &Path, git_type: Option) -> Result> { +fn resolve_git_commondir( + dir: &Path, + git_type: Option, +) -> Result> { let git_dir_path = || dir.join(".git"); let git_dir = git_dir_path(); if !git_type.map_or(false, |ft| ft.is_file()) { @@ -899,15 +1138,20 @@ fn resolve_git_commondir(dir: &Path, git_type: Option) -> Result + ?Sized>(prefix: &'a P, path: &'a Path) -> &'a Path { +fn strip_if_is_prefix<'a, P: AsRef + ?Sized>( + prefix: &'a P, + path: &'a Path, +) -> &'a Path { strip_prefix(prefix, path).map_or(path, |p| p) } #[cfg(test)] mod tests { - use std::{io::Write, path::Path}; + use std::{io::Write, path::Path, sync::Arc}; - use crate::{Error, dir::IgnoreBuilder, gitignore::Gitignore, tests::TempDir}; + use crate::{ + Error, dir::IgnoreBuilder, gitignore::Gitignore, tests::TempDir, + }; fn wfile>(path: P, contents: &str) { let mut file = std::fs::File::create(path).unwrap(); @@ -936,11 +1180,11 @@ mod tests { let (gi, err) = Gitignore::new(td.path().join("not-an-ignore")); assert!(err.is_none()); - let (ig, err) = IgnoreBuilder::new() - .add_ignore(gi) - .build() - .add_child(td.path()); + let (ig, err) = + IgnoreBuilder::new().add_ignore(gi).build().add_child(td.path()); assert!(err.is_none()); + // CHANGED: Explicit ignores only apply to paths inside the directory + // of the ignore file, so we have to match against full paths. assert!(ig.matched(td.path().join("foo"), false).is_ignore()); assert!(ig.matched(td.path().join("bar"), false).is_whitelist()); assert!(ig.matched(td.path().join("baz"), false).is_none()); @@ -1211,15 +1455,152 @@ mod tests { let (ig2, err) = ig1.add_child("src"); assert!(err.is_none()); - // CHANGED: These test cases do not make sense for us as we never call the Ignore with - // relative paths. - assert!(ig1.matched("llvm", true).is_ignore()); - assert!(ig2.matched("llvm", true).is_ignore()); + assert!(ig1.matched("llvm", true).is_none()); + assert!(ig2.matched("llvm", true).is_none()); assert!(ig2.matched("src/llvm", true).is_none()); assert!(ig2.matched("foo", false).is_ignore()); assert!(ig2.matched("src/foo", false).is_ignore()); } + #[test] + fn absolute_parent_matchers_are_cached_across_roots() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("src/build")); + mkdirp(td.path().join("tests/build")); + wfile(td.path().join(".gitignore"), "tests/**/build/\n"); + + let ig0 = IgnoreBuilder::new().build(); + let (src_parents, err) = ig0.add_parents(td.path().join("src")); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(td.path().join("src")); + assert!(err.is_none()); + let (tests_parents, err) = ig0.add_parents(td.path().join("tests")); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(td.path().join("tests")); + assert!(err.is_none()); + + assert!(Arc::ptr_eq(&src_parents.inner, &tests_parents.inner)); + assert!(src.matched("build", true).is_none()); + assert!(tests.matched("build", true).is_ignore()); + } + + /// Parent matchers are shared across search roots, but path rewriting for + /// absolute parents must use each root's own base path. Otherwise a rule + /// like `src/invalid` is matched against the wrong absolute path when + /// `src` is searched before a sibling root (e.g. `tests`). + /// + /// Paths passed to `matched` use the same relative layout as `Walk` when + /// roots are given as relative directory names. + /// + /// Regression for: https://github.com/BurntSushi/ripgrep/issues/3376 + /// and https://github.com/BurntSushi/ripgrep/issues/3419 + #[test] + fn multi_root_gitignore_order_independent() { + let td = tmpdir(); + let cwd = std::env::current_dir().unwrap(); + // Use paths relative to CWD like the CLI walk does for `rg pat src tests`. + let root = td.path().strip_prefix(&cwd).unwrap_or(td.path()); + let src_root = root.join("src"); + let tests_root = root.join("tests"); + + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("src")); + mkdirp(td.path().join("tests")); + wfile(td.path().join(".gitignore"), "src/invalid\n"); + wfile(td.path().join("src/invalid"), "x"); + wfile(td.path().join("src/valid"), "x"); + wfile(td.path().join("tests/valid"), "x"); + + let ig0 = IgnoreBuilder::new().build(); + + // Historically buggy order: search `src` first, then `tests`. + let (src_parents, err) = ig0.add_parents(&src_root); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(&src_root); + assert!(err.is_none()); + let (tests_parents, err) = ig0.add_parents(&tests_root); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(&tests_root); + assert!(err.is_none()); + + assert!(Arc::ptr_eq(&src_parents.inner, &tests_parents.inner)); + // Each root must carry its own absolute_base even though inners are shared. + assert_ne!( + src.absolute_base.as_ref().unwrap().as_path(), + tests.absolute_base.as_ref().unwrap().as_path() + ); + assert!( + src.matched(src_root.join("invalid"), false).is_ignore(), + "parent .gitignore must apply for the src root even when \ + another root was prepared in the same process" + ); + assert!(src.matched(src_root.join("valid"), false).is_none()); + assert!(tests.matched(tests_root.join("valid"), false).is_none()); + + // Reverse order should behave the same way. + let ig0 = IgnoreBuilder::new().build(); + let (tests_parents, err) = ig0.add_parents(&tests_root); + assert!(err.is_none()); + let (tests, err) = tests_parents.add_child(&tests_root); + assert!(err.is_none()); + let (src_parents, err) = ig0.add_parents(&src_root); + assert!(err.is_none()); + let (src, err) = src_parents.add_child(&src_root); + assert!(err.is_none()); + + assert!(src.matched(src_root.join("invalid"), false).is_ignore()); + assert!(src.matched(src_root.join("valid"), false).is_none()); + assert!(tests.matched(tests_root.join("valid"), false).is_none()); + } + + /// Same multi-root / order issue for non-git ignore files (e.g. `.rgignore` + /// via custom ignore names). + /// + /// Regression for: https://github.com/BurntSushi/ripgrep/issues/3320 + #[test] + fn multi_root_custom_ignore_order_independent() { + let td = tmpdir(); + let cwd = std::env::current_dir().unwrap(); + let root = td.path().strip_prefix(&cwd).unwrap_or(td.path()); + let alpha_root = root.join("alpha"); + let beta_root = root.join("beta"); + + mkdirp(td.path().join("alpha")); + mkdirp(td.path().join("beta")); + wfile(td.path().join(".rgignore"), "beta/**/*.svg\n"); + wfile(td.path().join("alpha/a.txt"), "x"); + wfile(td.path().join("beta/x.svg"), "x"); + + let ig0 = IgnoreBuilder::new() + .add_custom_ignore_filename(".rgignore") + .ignore(false) + .git_ignore(false) + .git_global(false) + .git_exclude(false) + .build(); + + let (alpha_parents, err) = ig0.add_parents(&alpha_root); + assert!(err.is_none()); + let (alpha, err) = alpha_parents.add_child(&alpha_root); + assert!(err.is_none()); + let (beta_parents, err) = ig0.add_parents(&beta_root); + assert!(err.is_none()); + let (beta, err) = beta_parents.add_child(&beta_root); + assert!(err.is_none()); + + assert_ne!( + alpha.absolute_base.as_ref().unwrap().as_path(), + beta.absolute_base.as_ref().unwrap().as_path() + ); + assert!(alpha.matched(alpha_root.join("a.txt"), false).is_none()); + assert!( + beta.matched(beta_root.join("x.svg"), false).is_ignore(), + "parent .rgignore must apply for the beta root regardless of \ + which root was set up first" + ); + } + #[test] fn git_info_exclude_in_linked_worktree() { let td = tmpdir(); @@ -1227,16 +1608,14 @@ mod tests { mkdirp(git_dir.join("info")); wfile(git_dir.join("info/exclude"), "ignore_me"); mkdirp(git_dir.join("worktrees/linked-worktree")); - let commondir_path = || git_dir.join("worktrees/linked-worktree/commondir"); + let commondir_path = + || git_dir.join("worktrees/linked-worktree/commondir"); mkdirp(td.path().join("linked-worktree")); let worktree_git_dir_abs = format!( "gitdir: {}", git_dir.join("worktrees/linked-worktree").to_str().unwrap(), ); - wfile( - td.path().join("linked-worktree/.git"), - &worktree_git_dir_abs, - ); + wfile(td.path().join("linked-worktree/.git"), &worktree_git_dir_abs); // relative commondir wfile(commondir_path(), "../.."); diff --git a/crates/ignore/src/gitignore.rs b/crates/ignore/src/gitignore.rs index f822d8390..8824139b5 100644 --- a/crates/ignore/src/gitignore.rs +++ b/crates/ignore/src/gitignore.rs @@ -102,7 +102,9 @@ impl Gitignore { /// /// Note that I/O errors are ignored. For more granular control over /// errors, use `GitignoreBuilder`. - pub fn new>(gitignore_path: P) -> (Gitignore, Option) { + pub fn new>( + gitignore_path: P, + ) -> (Gitignore, Option) { let path = gitignore_path.as_ref(); let parent = path.parent().unwrap_or(Path::new("/")); let mut builder = GitignoreBuilder::new(parent); @@ -123,6 +125,17 @@ impl Gitignore { /// The global config file path is specified by git's `core.excludesFile` /// config option. /// + /// # Behavior + /// + /// This routine does its best to discover any global git exclude files. + /// This will try to parse out the `excludesFile` config option in your + /// global git configuration, if necessary. + /// + /// The specific things this routine tries (which are subject to change + /// based on how git behaves) are: + /// + /// + /// /// Git's config file location is `$HOME/.gitconfig`. If `$HOME/.gitconfig` /// does not exist or does not specify `core.excludesFile`, then /// `$XDG_CONFIG_HOME/git/ignore` is read. If `$XDG_CONFIG_HOME` is not @@ -145,7 +158,8 @@ impl Gitignore { num_ignores: 0, num_whitelists: 0, matches: None, - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: false, } } @@ -190,7 +204,11 @@ impl Gitignore { /// determined by a common suffix of the directory containing this /// gitignore) is stripped. If there is no common suffix/prefix overlap, /// then `path` is assumed to be relative to this matcher. - pub fn matched>(&self, path: P, is_dir: bool) -> Match<&Glob> { + pub fn matched>( + &self, + path: P, + is_dir: bool, + ) -> Match<&Glob> { if self.is_empty() { return Match::None; } @@ -243,11 +261,16 @@ impl Gitignore { } /// Like matched, but takes a path that has already been stripped. - fn matched_stripped>(&self, path: P, is_dir: bool) -> Match<&Glob> { + fn matched_stripped>( + &self, + path: P, + is_dir: bool, + ) -> Match<&Glob> { if self.is_empty() { return Match::None; } - // CHANGED: Rules marked as only_on_files can not match against directories. + // CHANGED: Rules marked as only_on_files can not match against + // directories. if self.only_on_files && is_dir { return Match::None; } @@ -270,7 +293,10 @@ impl Gitignore { /// Strips the given path such that it's suitable for matching with this /// gitignore matcher. - fn strip<'a, P: 'a + AsRef + ?Sized>(&'a self, path: &'a P) -> &'a Path { + fn strip<'a, P: 'a + AsRef + ?Sized>( + &'a self, + path: &'a P, + ) -> &'a Path { let mut path = path.as_ref(); // A leading ./ is completely superfluous. We also strip it from // our gitignore root path, so we need to strip it from our candidate @@ -326,7 +352,8 @@ impl GitignoreBuilder { globs: vec![], case_insensitive: false, allow_unclosed_class: true, - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: false, } } @@ -337,18 +364,21 @@ impl GitignoreBuilder { pub fn build(&self) -> Result { let nignore = self.globs.iter().filter(|g| !g.is_whitelist()).count(); let nwhite = self.globs.iter().filter(|g| g.is_whitelist()).count(); - let set = self.builder.build().map_err(|err| Error::Glob { - glob: None, - err: err.to_string(), - })?; + let set = self + .builder + .build() + .map_err(|err| Error::Glob { glob: None, err: err.to_string() })?; Ok(Gitignore { set, root: self.root.clone(), globs: self.globs.clone(), num_ignores: nignore as u64, num_whitelists: nwhite as u64, - matches: Some(Arc::new(Pool::new(|| vec![]))), - // CHANGED: Add a flag to have Gitignore rules that apply only to files. + matches: Some(Arc::new( + Pool::with_available_parallelism_capacity(|| vec![]), + )), + // CHANGED: Add a flag to have Gitignore rules that apply only to + // files. only_on_files: self.only_on_files, }) } @@ -411,11 +441,8 @@ impl GitignoreBuilder { // Match Git's handling of .gitignore files that begin with the Unicode BOM const UTF8_BOM: &str = "\u{feff}"; - let line = if i == 0 { - line.trim_start_matches(UTF8_BOM) - } else { - &line - }; + let line = + if i == 0 { line.trim_start_matches(UTF8_BOM) } else { &line }; if let Err(err) = self.add_line(Some(path.to_path_buf()), &line) { errs.push(err.tagged(path, lineno)); @@ -537,7 +564,10 @@ impl GitignoreBuilder { /// affected. /// /// This is disabled by default. - pub fn case_insensitive(&mut self, yes: bool) -> Result<&mut GitignoreBuilder, Error> { + pub fn case_insensitive( + &mut self, + yes: bool, + ) -> Result<&mut GitignoreBuilder, Error> { // TODO: This should not return a `Result`. Fix this in the next semver // release. self.case_insensitive = yes; @@ -556,7 +586,10 @@ impl GitignoreBuilder { /// modes since the glob parser becomes more permissive. You might want to /// enable this when compatibility (e.g., with POSIX glob implementations) /// is more important than good error messages. - pub fn allow_unclosed_class(&mut self, yes: bool) -> &mut GitignoreBuilder { + pub fn allow_unclosed_class( + &mut self, + yes: bool, + ) -> &mut GitignoreBuilder { self.allow_unclosed_class = yes; self } @@ -576,32 +609,56 @@ impl GitignoreBuilder { /// /// Note that the file path returned may not exist. pub fn gitconfig_excludes_path() -> Option { - // git supports $HOME/.gitconfig and $XDG_CONFIG_HOME/git/config. Notably, - // both can be active at the same time, where $HOME/.gitconfig takes - // precedent. So if $HOME/.gitconfig defines a `core.excludesFile`, then - // we're done. - match gitconfig_home_contents().and_then(|x| parse_excludes_file(&x)) { - Some(path) => return Some(path), - None => {} + // When GIT_CONFIG_GLOBAL is set, it replaces both $HOME/.gitconfig and + // $XDG_CONFIG_HOME/git/config (per git 2.32+). Otherwise, git supports + // $HOME/.gitconfig and $XDG_CONFIG_HOME/git/config simultaneously, where + // $HOME/.gitconfig takes precedent. + gitconfig_global_env_contents() + .and_then(|x| parse_excludes_file(&x)) + .or_else(|| { + gitconfig_home_contents().and_then(|x| parse_excludes_file(&x)) + }) + .or_else(|| { + gitconfig_xdg_contents().and_then(|x| parse_excludes_file(&x)) + }) + // System-level config has the lowest priority for core.excludesFile. + // GIT_CONFIG_SYSTEM overrides the default /etc/gitconfig path. + .or_else(|| { + gitconfig_system_contents().and_then(|x| parse_excludes_file(&x)) + }) + .or_else(excludes_file_default) +} + +/// Returns the file contents of git's global config file from the path +/// specified by the `GIT_CONFIG_GLOBAL` environment variable. +fn gitconfig_global_env_contents() -> Option> { + let path = std::env::var_os("GIT_CONFIG_GLOBAL").map(PathBuf::from)?; + if path.as_os_str().is_empty() { + return None; } - match gitconfig_xdg_contents().and_then(|x| parse_excludes_file(&x)) { - Some(path) => return Some(path), - None => {} - } - excludes_file_default() + let mut file = BufReader::new(File::open(path).ok()?); + let mut contents = vec![]; + file.read_to_end(&mut contents).ok().map(|_| contents) +} + +/// Returns the file contents of git's system-level config file. +/// +/// Checks `GIT_CONFIG_SYSTEM` first, then falls back to `/etc/gitconfig`. +fn gitconfig_system_contents() -> Option> { + let path = std::env::var_os("GIT_CONFIG_SYSTEM") + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) + .unwrap_or_else(|| PathBuf::from("/etc/gitconfig")); + let mut file = BufReader::new(File::open(path).ok()?); + let mut contents = vec![]; + file.read_to_end(&mut contents).ok().map(|_| contents) } /// Returns the file contents of git's global config file, if one exists, in /// the user's home directory. fn gitconfig_home_contents() -> Option> { - let home = match home_dir() { - None => return None, - Some(home) => home, - }; - let mut file = match File::open(home.join(".gitconfig")) { - Err(_) => return None, - Ok(file) => BufReader::new(file), - }; + let home = home_dir()?; + let mut file = BufReader::new(File::open(home.join(".gitconfig")).ok()?); let mut contents = vec![]; file.read_to_end(&mut contents).ok().map(|_| contents) } @@ -610,19 +667,11 @@ fn gitconfig_home_contents() -> Option> { /// the user's XDG_CONFIG_HOME directory. fn gitconfig_xdg_contents() -> Option> { let path = std::env::var_os("XDG_CONFIG_HOME") - .and_then(|x| { - if x.is_empty() { - None - } else { - Some(PathBuf::from(x)) - } - }) + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) .or_else(|| home_dir().map(|p| p.join(".config"))) - .map(|x| x.join("git/config")); - let mut file = match path.and_then(|p| File::open(p).ok()) { - None => return None, - Some(file) => BufReader::new(file), - }; + .map(|x| x.join("git/config"))?; + let mut file = BufReader::new(File::open(path).ok()?); let mut contents = vec![]; file.read_to_end(&mut contents).ok().map(|_| contents) } @@ -632,13 +681,8 @@ fn gitconfig_xdg_contents() -> Option> { /// Specifically, this respects XDG_CONFIG_HOME. fn excludes_file_default() -> Option { std::env::var_os("XDG_CONFIG_HOME") - .and_then(|x| { - if x.is_empty() { - None - } else { - Some(PathBuf::from(x)) - } - }) + .map(PathBuf::from) + .filter(|x| !x.as_os_str().is_empty()) .or_else(|| home_dir().map(|p| p.join(".config"))) .map(|x| x.join("git/ignore")) } @@ -667,9 +711,7 @@ fn parse_excludes_file(data: &[u8]) -> Option { re.captures(data, &mut caps); let span = caps.get_group(1)?; let candidate = &data[span]; - std::str::from_utf8(candidate) - .ok() - .map(|s| PathBuf::from(expand_tilde(s))) + std::str::from_utf8(candidate).ok().map(|s| PathBuf::from(expand_tilde(s))) } /// Expands ~ in file paths to the value of $HOME. @@ -831,7 +873,10 @@ mod tests { fn parse_excludes_file4() { let data = bytes("[core]\nexcludesFile = \"~/foo/bar\""); let got = super::parse_excludes_file(&data); - assert_eq!(path_string(got.unwrap()), super::expand_tilde("~/foo/bar")); + assert_eq!( + path_string(got.unwrap()), + super::expand_tilde("~/foo/bar") + ); } #[test] diff --git a/crates/ignore/src/incremental.rs b/crates/ignore/src/incremental.rs new file mode 100644 index 000000000..5030b0d00 --- /dev/null +++ b/crates/ignore/src/incremental.rs @@ -0,0 +1,1286 @@ +use std::{ + collections::HashMap, + path::{Path, PathBuf}, + sync::OnceLock, +}; + +use crate::{ + Error, Match, PartialErrorBuilder, dir::Ignore, pathutil::is_hidden_path, +}; + +/// A cached matcher for checking paths against hierarchical ignore files. +/// +/// An `IncrementalIgnore` is built from a [`crate::WalkBuilder`]. Unlike a +/// recursive walk, it can check individual paths while still respecting the +/// ignore files in every relevant parent directory. Matchers for directories +/// are compiled on first use and then retained for later queries. +/// Each matcher corresponds to exactly one root configured on the builder, +/// and paths passed to it are interpreted relative to that root. +/// A matcher for the special `-` root representing standard input is inert +/// and always returns a non-match. +/// +/// The matcher checks path-based filters in the same precedence order as +/// a traversal. This includes glob overrides, `.ignore`, `.gitignore`, +/// `.git/info/exclude`, global and explicitly added ignore files, custom +/// ignore file names and file type selections. It does not apply filters that +/// require a directory entry or other traversal state, such as custom entry +/// predicates. Hidden-file detection, minimum and maximum depth limits and the +/// maximum file size are applied. +/// +/// A matcher is a snapshot at directory granularity. Once the ignore files in +/// a directory have been loaded, edits to those files are not observed. Build +/// a new matcher to reload them. +/// +/// # Warning +/// +/// The incremental path checking here necessarily needs to do a lot more work +/// per path matched. Callers should _not_ use this to run directory traversal. +/// This is intended to avoid the work of re-traversing an entire directory +/// tree when only a few changes are detected. (For example, in response to +/// file additions or deletions.) +/// +/// # Example +/// +/// ```rust,no_run +/// use ignore::WalkBuilder; +/// +/// let mut builder = WalkBuilder::new("."); +/// builder.add_custom_ignore_filename(".rgignore"); +/// let mut matchers = builder.build_matchers(); +/// let matcher = &mut matchers[0]; +/// +/// if matcher.matched("src/generated.rs", false).is_ignore() { +/// println!("ignored"); +/// } +/// ``` +#[derive(Clone, Debug)] +pub struct IncrementalIgnore { + /// The root exactly as it was given to `WalkBuilder`. + root: PathBuf, + /// The normalized root used only by the opt-in normalization routine. + normalized_root: OnceLock>, + /// The matcher for the configured root directory, loaded on first use. + ignore: RootIgnore, + /// Directory paths relative to `root`, excluding the root itself. + dirs: HashMap, + /// Options for additional filtering beyond gitignore. + options: IncrementalIgnoreOptions, +} + +/// The options for a matcher, mostly meant to duplicate as much as we can from +/// `WalkParallel`. +#[derive(Clone, Debug)] +pub(crate) struct IncrementalIgnoreOptions { + pub(crate) min_depth: Option, + pub(crate) max_depth: Option, + pub(crate) max_filesize: Option, + pub(crate) hidden: bool, + pub(crate) follow_links: bool, +} + +#[derive(Clone, Debug)] +enum RootIgnore { + Unloaded(Ignore), + Loaded(Ignore), + NotDirectory, + Stdin, +} + +/// Cached traversal state for a directory relative to the configured root. +/// +/// The presence of an entry means that the directory and every ancestor +/// between it and the root have already been checked. +#[derive(Clone, Debug)] +enum CachedDir { + /// The directory may be descended into. The matcher includes the ignore + /// rules loaded through this directory and is therefore the matcher to use + /// for its children. + Allowed(Ignore), + /// The directory may not be descended into because it was ignored by a + /// path rule or hidden-file filtering. Every descendant is consequently + /// ignored, and ignore files inside this directory are not loaded. + Ignored, +} + +impl IncrementalIgnore { + pub(crate) fn new( + root: PathBuf, + ignore: Ignore, + options: IncrementalIgnoreOptions, + ) -> IncrementalIgnore { + // File traversal special cases `-` to search stdin, so we recognize + // it here for completeness too. In particular, we really want + // `WalkBuilder::build_matchers` to return a matcher for every root, + // even when it's a simple file (handled automatically) or when it's + // stdin (necessarily special cased). + // + // If callers need to search a file or directory named `-`, then they + // can use `./-`. As is the case for file traversal too. + let ignore = if root == Path::new("-") { + RootIgnore::Stdin + } else { + RootIgnore::Unloaded(ignore) + }; + IncrementalIgnore { + root, + normalized_root: OnceLock::new(), + ignore, + dirs: HashMap::new(), + options, + } + } + + /// Return the root that paths matched by this matcher are relative to. + pub fn root(&self) -> &Path { + &self.root + } + + /// Normalize `path` and return it relative to this matcher's root. + /// + /// This returns `None` when `path` cannot be made absolute or + /// when it is known to be outside this matcher's root. Unlike + /// [`IncrementalIgnore::matched`], this performs absolute path conversion, + /// lexical normalization and allocation. It is intended as an opt-in + /// convenience for callers that do not already have root-relative paths. + /// + /// Note that `.` is interpreted relative to the process level current + /// working directory. It is _not_ interpreted relative to the root of + /// this matcher. + /// + /// Note also that this may reject paths that only differ in casing. For + /// example, if the root path for this matcher is `/FOO` but the provided + /// path is `/foo/bar`, then this may return `None`. Callers must ensure + /// casing is consistent between the path provided and the root path for + /// this matcher. + pub fn normalize>(&self, path: P) -> Option { + if matches!(self.ignore, RootIgnore::Stdin) { + return None; + } + let path = normalize_absolute(path.as_ref())?; + let root = self + .normalized_root + .get_or_init(|| normalize_absolute(&self.root)) + .as_ref()?; + path.strip_prefix(root).ok().map(Path::to_path_buf) + } + + /// Match a root-relative path against ignore files in its directory and + /// all relevant parent directories. + /// + /// `is_dir` should be true when `path` should be matched as a directory. + /// + /// For the return value, use [`IncrementalMatch::is_ignore`], + /// [`IncrementalMatch::is_whitelist`] or [`IncrementalMatch::is_none`] to + /// inspect it. + /// + /// Matchers for previously unseen directories are loaded and cached during + /// this call. Errors encountered while loading ignore files are logged. To + /// receive those errors, use [`IncrementalIgnore::matched_with_errors`]. + /// + /// `path` must be relative to this matcher's root and must not contain a + /// parent directory (`..`) component. Behavior is unspecified when these + /// preconditions are violated. Callers with an absolute path or with a + /// path containing `.` or `..` may use [`IncrementalIgnore::normalize`] + /// to get a path satisfying these preconditions. In all cases, a relative + /// path is *assumed* to be relative to the root of this ignore matcher. + /// + /// In general, it is intended that callers doing recursive directory + /// traversal on the root of this matcher may provide relative paths to + /// this routine *without* calling [`IncrementalIgnore::normalize`]. + /// + /// The empty path represents the explicitly configured root and also + /// returns non-match, consistent with recursive traversal where a root is + /// always treated as being at depth zero. + pub fn matched>( + &mut self, + path: P, + is_dir: bool, + ) -> IncrementalMatch { + let (matched, err) = self.matched_with_errors(path, is_dir); + if let Some(err) = err { + log::debug!("error while loading ignore files: {err}"); + } + matched + } + + /// Match a root-relative path and return errors encountered while loading + /// ignore files. + /// + /// This is equivalent to [`IncrementalIgnore::matched`], except that + /// it returns any errors from newly loaded ignore files. Loading can + /// partially succeed, so valid rules are always applied to the returned + /// match even when an error is present. + pub fn matched_with_errors>( + &mut self, + path: P, + is_dir: bool, + ) -> (IncrementalMatch, Option) { + let mut errs = PartialErrorBuilder::default(); + let matched = + self.matched_with_errors_impl(path.as_ref(), is_dir, &mut errs); + (matched, errs.into_error_option()) + } + + fn matched_with_errors_impl( + &mut self, + relative: &Path, + is_dir: bool, + errs: &mut PartialErrorBuilder, + ) -> IncrementalMatch { + // We short-circuit here when our matcher corresponds to `Stdin` in + // order to always return a non-match. This is somewhat redundant with + // `root_ignore()` which does this too, but we do it here so that it + // always happens, e.g., before depth filtering. + if relative.is_absolute() || matches!(self.ignore, RootIgnore::Stdin) { + return IncrementalMatch::none(is_dir); + } + + let (mut satisfies_min, mut edge_max) = (true, false); + if self.options.min_depth.is_some() || self.options.max_depth.is_some() + { + let depth = relative + .components() + .filter_map(|component| match component { + std::path::Component::CurDir => None, + component => Some(component), + }) + .count(); + satisfies_min = + self.options.min_depth.is_none_or(|min| depth >= min); + let satisfies_max = + self.options.max_depth.is_none_or(|max| depth <= max); + edge_max = self.options.max_depth.is_some_and(|max| depth == max); + // When we have a file that isn't past our min depth, we can give + // up right away. + if !is_dir && !satisfies_min { + return IncrementalMatch::ignore().not_within_depth(); + } + // Same for *anything* that exceeds our max depth. + if !satisfies_max { + return IncrementalMatch::ignore().not_within_depth(); + } + } + + // When the path is invalid in some way, we bail out early to avoid + // potentially doing a stat call below. + let (mut mat, valid) = self + .matched_with_errors_ignore(relative, is_dir, errs) + .map(|mat| (mat, true)) + .unwrap_or_else(|| (IncrementalMatch::none(is_dir), false)); + + if !is_dir + && !mat.is_ignore() + && valid + && let Some(max_filesize) = self.options.max_filesize + { + let path = self.root().join(relative); + let result = if self.options.follow_links { + path.metadata() + } else { + path.symlink_metadata() + }; + match result { + Ok(md) if md.len() > max_filesize => { + return IncrementalMatch::ignore(); + } + Ok(_) => {} + Err(err) => { + // Record the error but otherwise fall through + let err = Error::from(err); + errs.push(err.with_path(path)); + } + } + } + + // We still need to tag our `mat` if it doesn't pass the depth filter, + // or if it's a directory on the edge of a depth filter. This can only + // happen when the path is reported as a directory. Otherwise regular + // files are always handled above. + if is_dir { + if !satisfies_min { + mat = mat.not_within_depth(); + } + if edge_max { + mat = mat.no_descent(); + } + } + mat + } + + fn matched_with_errors_ignore( + &mut self, + relative: &Path, + is_dir: bool, + errs: &mut PartialErrorBuilder, + ) -> Option { + let mut components = relative + .components() + .filter_map(|component| match component { + std::path::Component::CurDir => None, + component => Some(component), + }) + .peekable(); + components.peek()?; + + // If the exact parent is cached, then all of its ancestors have + // already been checked. An allowed cache entry is the matcher to use + // for children of that directory, while an ignored cache entry is + // terminal for every descendant. In the usual case, this avoids both + // walking every component and allocating a relative directory path. + // + // Only try the fast path when the final component is a normal path + // component. Invalid paths are handled by the component walk below. + let has_normal_final_component = + relative.components().next_back().is_some_and(|component| { + matches!(component, std::path::Component::Normal(_)) + }); + if has_normal_final_component && let Some(parent) = relative.parent() { + match self.dirs.get(parent) { + Some(CachedDir::Allowed(ignore)) => { + return Some(self.match_path(ignore, relative, is_dir)); + } + Some(CachedDir::Ignored) => { + return Some(IncrementalMatch::ignore()); + } + None => {} + } + } + + let mut ignore = self.root_ignore(errs)?; + let mut dir = PathBuf::new(); + while let Some(component) = components.next() { + match component { + std::path::Component::ParentDir + | std::path::Component::RootDir + | std::path::Component::Prefix(_) => { + return None; + } + std::path::Component::CurDir => continue, + std::path::Component::Normal(_) => {} + } + if components.peek().is_none() { + break; + } + dir.push(component.as_os_str()); + match self.dirs.get(&dir) { + Some(CachedDir::Allowed(cached)) => { + ignore = cached.clone(); + continue; + } + Some(CachedDir::Ignored) => { + return Some(IncrementalMatch::ignore()); + } + None => {} + } + + let path = self.root.join(&dir); + let mat = ignore.matched(&path, true); + let is_hidden = + self.options.hidden && mat.is_none() && is_hidden_path(&path); + if mat.is_ignore() || is_hidden { + self.dirs.insert(dir.clone(), CachedDir::Ignored); + return Some(IncrementalMatch::ignore()); + } + let (child, err) = ignore.add_child(&path); + errs.maybe_push(err); + self.dirs.insert(dir.clone(), CachedDir::Allowed(child.clone())); + ignore = child; + } + + Some(self.match_path(&ignore, relative, is_dir)) + } + + fn match_path( + &self, + ignore: &Ignore, + relative: &Path, + is_dir: bool, + ) -> IncrementalMatch { + let path = self.root.join(relative); + let mut mat = IncrementalMatch::from_match( + ignore.matched(&path, is_dir).map(|_| ()), + is_dir, + ); + // Whether a file is hidden or not has low precedence in filtering. We + // only check it if we haven't matched anything above. This permits + // callers to whitelist hidden files or directories. + if self.options.hidden && mat.is_none() && is_hidden_path(&path) { + mat = IncrementalMatch::ignore(); + } + mat + } + + fn root_ignore( + &mut self, + errs: &mut PartialErrorBuilder, + ) -> Option { + let ignore = match self.ignore { + RootIgnore::Unloaded(ref ignore) => ignore.clone(), + RootIgnore::Loaded(ref ignore) => return Some(ignore.clone()), + RootIgnore::NotDirectory | RootIgnore::Stdin => return None, + }; + if !self.root.is_dir() { + self.ignore = RootIgnore::NotDirectory; + return None; + } + + let (parents, err) = ignore.add_parents(&self.root); + errs.maybe_push(err); + let (root, err) = parents.add_child(&self.root); + errs.maybe_push(err); + self.ignore = RootIgnore::Loaded(root.clone()); + Some(root) + } +} + +/// The result of an incremental match. +/// +/// This is similar to [`Match`] in that it reports whether a file path should +/// be ignored, whitelisted or didn't match anything at all. It also has extra +/// data, such as whether a directory matched but should not be descended into. +/// +/// Generally speaking, callers that only care about specific files can stick +/// to the `is_none()`, `is_ignore()` or `is_whitelist()` predicates. Directory +/// entries are more complicated because we sometimes want to yield directories +/// to descend into, but not actually visit (e.g., for the minimum depth +/// filter). Or, we may want to yield a directory to visit but not descend into +/// (e.g., for the maximum depth filter). +#[derive(Clone, Debug)] +pub struct IncrementalMatch { + mat: Match<()>, + should_descend: bool, + is_within_depth: bool, +} + +impl IncrementalMatch { + fn none(is_dir: bool) -> IncrementalMatch { + IncrementalMatch { + mat: Match::None, + should_descend: is_dir, + is_within_depth: true, + } + } + + fn ignore() -> IncrementalMatch { + IncrementalMatch { + mat: Match::Ignore(()), + should_descend: false, + is_within_depth: true, + } + } + + fn from_match(mat: Match<()>, is_dir: bool) -> IncrementalMatch { + let should_descend = is_dir && !mat.is_ignore(); + IncrementalMatch { mat, should_descend, is_within_depth: true } + } + + fn no_descent(self) -> IncrementalMatch { + IncrementalMatch { should_descend: false, ..self } + } + + fn not_within_depth(self) -> IncrementalMatch { + IncrementalMatch { is_within_depth: false, ..self } + } + + /// Returns true if the match result didn't match anything. + pub fn is_none(&self) -> bool { + self.mat.is_none() + } + + /// Returns true if the match result implies the path should be ignored. + pub fn is_ignore(&self) -> bool { + self.mat.is_ignore() + } + + /// Returns true if the match result implies the path should be + /// whitelisted. + pub fn is_whitelist(&self) -> bool { + self.mat.is_whitelist() + } + + /// Returns true only when this match corresponds to a directory *and* + /// whether the caller should look inside this directory for additional + /// results. + /// + /// It is possible for a match to report `false` for + /// [`IncrementalMatch::is_ignore`] _and_ `false` for + /// [`IncrementalMatch::should_descend`]. This can occur, for example, when + /// a maximum depth setting allows a directory through, but where none of + /// its children entries should be visited. + /// + /// This is always `false` for a file path that does *not* correspond to a + /// directory. + pub fn should_descend(&self) -> bool { + self.should_descend + } + + /// Returns true only when this result corresponds to an entry that is + /// within the depth filter. + /// + /// This is `false` when a path corresponds to a directory and is less than + /// the minimum depth. In this case, callers should continue looking inside + /// that directory. + /// + /// This is always `true` for a match corresponding to a file path that + /// isn't a directory. + pub fn is_within_depth(&self) -> bool { + self.is_within_depth + } + + /// Inverts the match so that `Ignore` becomes `Whitelist` and + /// `Whitelist` becomes `Ignore`. A non-match remains the same. + pub fn invert(self) -> IncrementalMatch { + IncrementalMatch { mat: self.mat.invert(), ..self } + } +} + +/// Return a lexically normalized absolute representation of `path`. +/// +/// This collapses `.` and `..`, but intentionally does not canonicalize or +/// resolve symlinks. Ignore rules apply to the lexical path, and resolving a +/// symlink below a root could move the result outside of that root. +fn normalize_absolute(path: &Path) -> Option { + let absolute = std::path::absolute(path).ok()?; + let mut normalized = PathBuf::new(); + for component in absolute.components() { + match component { + std::path::Component::CurDir => {} + std::path::Component::ParentDir => { + normalized.pop(); + } + _ => normalized.push(component.as_os_str()), + } + } + Some(normalized) +} + +#[cfg(test)] +mod tests { + use std::{ + fs::{self, File}, + io::Write, + path::{Path, PathBuf}, + }; + + use crate::{ + IncrementalIgnore, IncrementalMatch, WalkBuilder, + overrides::OverrideBuilder, tests::TempDir, types::TypesBuilder, + }; + + use super::CachedDir; + + fn wfile>(path: P, contents: &str) { + let mut file = File::create(path).unwrap(); + file.write_all(contents.as_bytes()).unwrap(); + } + + fn mkdirp>(path: P) { + fs::create_dir_all(path).unwrap(); + } + + fn tmpdir() -> TempDir { + TempDir::new().unwrap() + } + + fn builder>(path: P) -> WalkBuilder { + let mut builder = WalkBuilder::new(path); + builder.git_global(false); + builder + } + + fn one_matcher(builder: &WalkBuilder) -> IncrementalIgnore { + let mut matchers = builder.build_matchers(); + assert_eq!(matchers.len(), 1); + matchers.pop().unwrap() + } + + fn builders, P: AsRef>( + paths: I, + ) -> WalkBuilder { + let mut builder = WalkBuilder::from_iter(paths); + builder.git_global(false); + builder + } + + fn matchers(builder: &WalkBuilder) -> Vec { + builder.build_matchers() + } + + fn matchedf>( + matcher: &mut IncrementalIgnore, + path: P, + ) -> IncrementalMatch { + let (matched, err) = matcher.matched_with_errors(path, false); + assert!(err.is_none(), "unexpected matcher error: {err:?}"); + matched + } + + fn matchedd>( + matcher: &mut IncrementalIgnore, + path: P, + ) -> IncrementalMatch { + let (matched, err) = matcher.matched_with_errors(path, true); + assert!(err.is_none(), "unexpected matcher error: {err:?}"); + matched + } + + // Test that multiple parent ignore files, when nested, are respected. + #[test] + fn nested_parent_gitignores() { + let td = tmpdir(); + let root = td.path().join("project/work"); + mkdirp(td.path().join(".git")); + mkdirp(root.join("src")); + wfile(td.path().join(".gitignore"), "*.tmp\n"); + wfile(td.path().join("project/.gitignore"), "!keep.tmp\nnested.log\n"); + + let mut m = one_matcher(&builder(&root)); + assert_eq!(m.root(), root); + assert!(matchedf(&mut m, "src/drop.tmp").is_ignore()); + assert!(matchedf(&mut m, "src/keep.tmp").is_whitelist()); + assert!(matchedf(&mut m, "src/nested.log").is_ignore()); + assert!(matchedf(&mut m, "src/ok.rs").is_none()); + } + + // Test that an anchored rule in a child directory is matched relative to + // that directory, not relative to the configured root. + #[test] + fn anchored_child_rule_uses_child_root() { + let td = tmpdir(); + let root = td.path().join("root"); + mkdirp(root.join("a")); + wfile(root.join("a/.ignore"), "/foo\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "a/foo").is_ignore()); + assert!(matchedf(&mut m, "a/b/foo").is_none()); + } + + // Test that a leading `./` impacts how the rules are matched. + #[cfg(not(windows))] + #[test] + fn leading_dot_slash_impacts_matching() { + let td = tmpdir(); + let root = td.path().join("root"); + mkdirp(root.join("a")); + wfile(root.join(".ignore"), "/foo\n"); + wfile(root.join("a/.ignore"), "/foo\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "foo").is_ignore()); + assert!(matchedf(&mut m, "./foo").is_none()); + assert!(matchedf(&mut m, "a/allowed").is_none()); + assert!(matchedf(&mut m, "a/foo").is_ignore()); + assert!(matchedf(&mut m, "./a/foo").is_none()); + } + + // Test that custom ignore files are respected. + #[test] + fn parent_ignore_and_custom_ignore() { + let td = tmpdir(); + let root = td.path().join("project/work"); + mkdirp(root.join("src")); + wfile(td.path().join(".ignore"), "*.cache\n"); + wfile(td.path().join(".rgignore"), "*.svg\n"); + wfile(td.path().join("project/.ignore"), "!keep.cache\n"); + wfile(td.path().join("project/.rgignore"), "!keep.svg\n"); + + let mut builder = builder(&root); + builder.add_custom_ignore_filename(".rgignore"); + let mut m = one_matcher(&builder); + assert!(matchedf(&mut m, "src/drop.cache").is_ignore()); + assert!(matchedf(&mut m, "src/keep.cache").is_whitelist()); + assert!(matchedf(&mut m, "src/drop.svg").is_ignore()); + assert!(matchedf(&mut m, "src/keep.svg").is_whitelist()); + } + + // Test that glob overrides take precedence over ignore files. + #[test] + fn glob_overrides_are_applied() { + let td = tmpdir(); + wfile(td.path().join(".ignore"), "keep.rs\n!drop.rs\n"); + + let mut overrides = OverrideBuilder::new(td.path()); + overrides.add("keep.rs").unwrap(); + overrides.add("!drop.rs").unwrap(); + let mut b = builder(td.path()); + b.overrides(overrides.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "keep.rs").is_whitelist()); + assert!(matchedf(&mut m, "drop.rs").is_ignore()); + } + + // Test that an override-ignored directory prevents matching rules below + // it, just as it prevents a traversal from descending into the directory. + #[test] + fn glob_overrides_apply_to_ancestors() { + let td = tmpdir(); + mkdirp(td.path().join("blocked")); + wfile(td.path().join("blocked/.ignore"), "!keep.rs\n"); + + let mut overrides = OverrideBuilder::new(td.path()); + overrides.add("!blocked/").unwrap(); + let mut b = builder(td.path()); + b.overrides(overrides.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "blocked/keep.rs").is_ignore()); + } + + // Test that file type selections are applied to files, but not + // directories. + #[test] + fn file_types_are_applied() { + let td = tmpdir(); + mkdirp(td.path().join("src")); + + let mut types = TypesBuilder::new(); + types.add("rust", "*.rs").unwrap(); + types.select("rust"); + let mut b = builder(td.path()); + b.types(types.build().unwrap()); + let mut m = one_matcher(&b); + + assert!(matchedd(&mut m, "src").is_none()); + assert!(matchedf(&mut m, "src/lib.rs").is_whitelist()); + assert!(matchedf(&mut m, "README.md").is_ignore()); + } + + #[test] + fn directly_ignored_directory_is_not_descended() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("blocked")); + wfile(td.path().join(".gitignore"), "blocked/\n"); + + let mut m = one_matcher(&builder(td.path())); + let dir = matchedd(&mut m, "blocked"); + assert!(dir.is_ignore()); + assert!(!dir.should_descend()); + } + + // Test that when a directory is ignored, anything below it always ignored + // even when there are explicit whitelist rules. This matches directory + // traversal semantics. + #[test] + fn ignored_ancestor_wins() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join("blocked")); + mkdirp(td.path().join("open")); + // This is the key line: since the entire `blocked` directory is + // ignored, a proper file traversal won't ever descend into it. So + // `blocked/keep.rs` should be ignored even if there are ignore rules + // "beneath" it that whitelist it. + wfile(td.path().join(".gitignore"), "blocked/\n!blocked/keep.rs\n"); + wfile(td.path().join("blocked/.gitignore"), "!keep.rs\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "blocked/keep.rs").is_ignore()); + assert!(matchedf(&mut m, "blocked/other.rs").is_ignore()); + assert!(matchedf(&mut m, "open/keep.rs").is_none()); + } + + // Test that we respect git boundaries. And that we don't respect git + // boundaries when not configured to do so. + #[test] + fn respects_git_repository_boundaries() { + let td = tmpdir(); + let root = td.path().join("repo/src"); + mkdirp(td.path().join("repo/.git")); + mkdirp(&root); + wfile(td.path().join(".gitignore"), "outside-rule\n"); + wfile(td.path().join(".ignore"), "tool-rule\n"); + wfile(td.path().join("repo/.gitignore"), "inside-rule\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "inside-rule").is_ignore()); + assert!(matchedf(&mut m, "outside-rule").is_none()); + assert!(matchedf(&mut m, "tool-rule").is_ignore()); + + let mut no_git_required = builder(&root); + no_git_required.require_git(false); + let mut m = one_matcher(&no_git_required); + assert!(matchedf(&mut m, "outside-rule").is_ignore()); + } + + // Test that when a gitignore matcher in a parent directory is created, we + // reuse that matcher from memory even if it's changed on disk. + #[test] + fn compiled_matchers_are_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a")); + wfile(td.path().join("a/.ignore"), "*.tmp\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/first.tmp").is_ignore()); + + // Below demonstrates that this new ignore file contents + // aren't actually picked up because it was already loaded. + wfile(td.path().join("a/.ignore"), "!*.tmp\n*.rs\n"); + assert!(matchedf(&mut m, "a/first.tmp").is_ignore()); + assert!(matchedf(&mut m, "a/second.tmp").is_ignore()); + assert!(matchedf(&mut m, "a/keep.rs").is_none()); + // To get the new ignore file, we need to rebuild the matcher. + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/first.tmp").is_whitelist()); + assert!(matchedf(&mut m, "a/second.tmp").is_whitelist()); + assert!(matchedf(&mut m, "a/keep.rs").is_ignore()); + } + + #[test] + fn cached_allowed_parent_matches_children() { + let td = tmpdir(); + mkdirp(td.path().join("a/b")); + wfile(td.path().join("a/b/.ignore"), "ignored\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/b/allowed").is_none()); + assert!(matches!( + m.dirs.get(Path::new("a/b")), + Some(CachedDir::Allowed(_)) + )); + assert!(matchedf(&mut m, "a/b/ignored").is_ignore()); + } + + #[test] + fn cached_ignored_parent_matches_children() { + let td = tmpdir(); + mkdirp(td.path().join("blocked")); + wfile(td.path().join(".ignore"), "blocked/\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "blocked/first").is_ignore()); + assert!(matches!( + m.dirs.get(Path::new("blocked")), + Some(CachedDir::Ignored) + )); + assert!(matchedf(&mut m, "blocked/second").is_ignore()); + } + + #[test] + fn cached_parent_matcher_is_used_for_directory() { + let td = tmpdir(); + mkdirp(td.path().join("a/b")); + wfile(td.path().join("a/b/.ignore"), "**\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, "a/b/file").is_ignore()); + assert!(matches!( + m.dirs.get(Path::new("a/b")), + Some(CachedDir::Allowed(_)) + )); + let dir = matchedd(&mut m, "a/b"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + // Like `compiled_matchers_are_reused`, but with multiple roots. + #[test] + fn compiled_multi_matchers_are_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c")); + + wfile(td.path().join("a/.ignore"), "*.tmp\n"); + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + assert!(matchedf(&mut mats[0], "first.tmp").is_ignore()); + assert!(matchedf(&mut mats[0], "c/first.tmp").is_ignore()); + + // Even though we haven't used the second matcher, it should still + // reuse the `a/.ignore` above. If it didn't, then `a/b/c/first.tmp` + // below would be whitelisted. + wfile(td.path().join("a/.ignore"), "!*.tmp\n"); + assert!(matchedf(&mut mats[1], "first.tmp").is_ignore()); + } + + #[test] + fn compiled_multi_child_matchers_are_not_reused() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c/d/e")); + + wfile(td.path().join("a/b/c/d/e/.ignore"), "*.tmp\n"); + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + // Some sanity checking first. + assert!(matchedf(&mut mats[0], "c/first.tmp").is_none()); + assert!(matchedf(&mut mats[0], "c/d/first.tmp").is_none()); + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_ignore()); + + // Now write a new ignore file at the same location as above + // and check that the other matcher still uses the "stale" data. + wfile(td.path().join("a/b/c/d/e/.ignore"), "!*.tmp\n"); + assert!(matchedf(&mut mats[1], "first.tmp").is_none()); + assert!(matchedf(&mut mats[1], "d/first.tmp").is_none()); + // This is the punch line: because we didn't load `mats[1]` before + // changing the ignore file, it loads it here and thus this path gets + // whitelisted. + assert!(matchedf(&mut mats[1], "d/e/first.tmp").is_whitelist()); + // ... but `mats[0]` still uses the stale gitignore matcher cached in + // memory! + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_ignore()); + + // If we rebuilder the matcher... then we force reloading and they're + // now consistent with one another. + let mut mats = matchers(&builders([ + td.path().join("a/b"), + td.path().join("a/b/c"), + ])); + assert!(matchedf(&mut mats[0], "c/d/e/first.tmp").is_whitelist()); + assert!(matchedf(&mut mats[1], "d/e/first.tmp").is_whitelist()); + } + + // Tests that even when there is an error with a glob pattern, we still + // respect other glob patterns that are valid. + #[test] + fn partial_errors_keep_valid_rules() { + let td = tmpdir(); + let root = td.path().join("work"); + mkdirp(&root); + wfile(td.path().join(".ignore"), "{bad\n*.tmp\n"); + + let mut builder = builder(&root); + builder.git_ignore(false).git_exclude(false); + let mut m = one_matcher(&builder); + let (matched, err) = m.matched_with_errors("drop.tmp", false); + assert!(err.is_some()); + assert!(matched.is_ignore()); + } + + // Tests that parent rules are not respected if the matcher is configured + // not to do so. + #[test] + fn parent_loading_can_be_disabled() { + let td = tmpdir(); + let root = td.path().join("work"); + mkdirp(&root); + wfile(td.path().join(".ignore"), "parent-rule\n"); + wfile(root.join(".ignore"), "root-rule\n"); + + let mut b = builder(&root); + b.parents(false); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "parent-rule").is_none()); + assert!(matchedf(&mut m, "root-rule").is_ignore()); + + // Sanity check that without `parents(false)`, the parent rule is + // respected. + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "parent-rule").is_ignore()); + assert!(matchedf(&mut m, "root-rule").is_ignore()); + } + + // Tests that we can normalize a file path that isn't already in "normal" + // relative form, and then use that to match on ignore files. + #[test] + fn paths_are_relative_to_the_root() { + let td = tmpdir(); + let root = td.path().join("root"); + let outside = td.path().join("outside"); + mkdirp(&root); + mkdirp(&outside); + wfile(root.join(".ignore"), "file\n"); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedd(&mut m, "").is_none()); + assert!(matchedd(&mut m, ".").is_none()); + assert!(matchedf(&mut m, "file").is_ignore()); + // Doesn't work because it isn't relative to the root. It's absolute. + assert!(matchedf(&mut m, root.join("file")).is_none()); + // Also doesn't work because while it's relative, it contains `..`. + assert!(matchedf(&mut m, "dir/../file").is_none()); + + let norm = m.normalize(root.join("dir/../file")).unwrap(); + assert_eq!(norm, Path::new("file")); + assert!(matchedf(&mut m, "file").is_ignore()); + + // Doesn't normalize because its an absolute path outside of our root. + assert_eq!(m.normalize(outside.join("file")), None); + } + + // Tests that two matchers in two different directories correctly interpret + // the same parent gitignore file. + #[test] + fn multiple_roots_keep_their_own_context() { + let td = tmpdir(); + let root_a = td.path().join("a"); + let root_b = td.path().join("b"); + mkdirp(td.path().join(".git")); + mkdirp(&root_a); + mkdirp(&root_b); + wfile(td.path().join(".gitignore"), "/a/*.tmp\n/b/*.log\n"); + + let mut builder = + builders([root_a.as_path(), Path::new("-"), root_b.as_path()]); + builder.git_global(false); + let mut ms = matchers(&builder); + assert_eq!(ms.len(), 3); + assert_eq!(ms[0].root(), root_a); + assert_eq!(ms[1].root(), Path::new("-")); + assert_eq!(ms[2].root(), root_b); + assert!(matchedf(&mut ms[0], "drop.tmp").is_ignore()); + assert!(matchedf(&mut ms[0], "keep.log").is_none()); + assert!(matchedf(&mut ms[1], "anything").is_none()); + assert_eq!(ms[1].normalize("anything"), None); + assert!(matchedf(&mut ms[2], "keep.tmp").is_none()); + assert!(matchedf(&mut ms[2], "drop.log").is_ignore()); + } + + // Tests that only the exact `-` root represents standard input. + #[test] + fn dot_dash_root_is_not_stdin() { + let m = one_matcher(&builder("./-")); + assert_eq!(m.root(), Path::new("./-")); + assert_eq!(m.normalize("./-/file"), Some(PathBuf::from("file"))); + } + + #[test] + fn stdin_is_inert_with_depth_limits() { + let mut b = builder("-"); + b.min_depth(Some(2)); + let mut m = one_matcher(&b); + let mat = matchedf(&mut m, "file"); + assert!(mat.is_none()); + assert!(mat.is_within_depth()); + + let mut b = builder("-"); + b.max_depth(Some(0)); + let mut m = one_matcher(&b); + let mat = matchedd(&mut m, "dir"); + assert!(mat.is_none()); + assert!(mat.is_within_depth()); + assert!(mat.should_descend()); + } + + // Tests that ignore matching works lexically, and doesn't accidentally + // resolve symbolic links. + #[cfg(unix)] + #[test] + fn symlink_path_stays_under_lexical_root() { + use std::os::unix::fs::symlink; + + let td = tmpdir(); + let root = td.path().join("root"); + let outside = td.path().join("outside"); + mkdirp(&root); + mkdirp(&outside); + wfile(root.join(".ignore"), "link\n"); + wfile(outside.join("target"), ""); + symlink(outside.join("target"), root.join("link")).unwrap(); + + let mut m = one_matcher(&builder(&root)); + assert!(matchedf(&mut m, "link").is_ignore()); + } + + #[test] + fn depth_limits() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c/d")); + + let mut b = builder(td.path()); + b.min_depth(Some(2)).max_depth(Some(3)); + let mut m = one_matcher(&b); + + let dir = matchedd(&mut m, "a"); + assert!(!dir.is_within_depth()); + assert!(dir.should_descend()); + assert!(matchedf(&mut m, "file").is_ignore()); + + let dir = matchedd(&mut m, "a/b"); + assert!(dir.is_within_depth()); + assert!(dir.should_descend()); + assert!(!matchedf(&mut m, "a/file").is_ignore()); + + let dir = matchedd(&mut m, "a/b/c"); + assert!(dir.is_within_depth()); + assert!(!dir.should_descend()); + assert!(!matchedf(&mut m, "a/b/file").is_ignore()); + assert!(matchedf(&mut m, "a/b/c/file").is_ignore()); + + let dir = matchedd(&mut m, "a/b/c/d"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + assert!(!dir.should_descend()); + assert!(matchedf(&mut m, "a/b/c/d/file").is_ignore()); + } + + #[test] + fn depth_limits_apply_to_root() { + let td = tmpdir(); + + let mut b = builder(td.path()); + b.min_depth(Some(1)); + let mut m = one_matcher(&b); + for path in ["", "."] { + let root = matchedd(&mut m, path); + assert!(root.is_none()); + assert!(!root.is_within_depth()); + assert!(root.should_descend()); + } + + let mut b = builder(td.path()); + b.max_depth(Some(0)); + let mut m = one_matcher(&b); + for path in ["", "."] { + let root = matchedd(&mut m, path); + assert!(root.is_none()); + assert!(root.is_within_depth()); + assert!(!root.should_descend()); + } + } + + #[test] + fn max_filesize() { + let td = tmpdir(); + mkdirp(td.path().join("dir")); + wfile(td.path().join("empty"), ""); + wfile(td.path().join("at-limit"), "12345"); + wfile(td.path().join("over-limit"), "123456"); + + let mut b = builder(td.path()); + b.max_filesize(Some(5)); + let mut m = one_matcher(&b); + + assert!(matchedf(&mut m, "empty").is_none()); + assert!(matchedf(&mut m, "at-limit").is_none()); + assert!(matchedf(&mut m, "over-limit").is_ignore()); + + let dir = matchedd(&mut m, "dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + #[test] + fn max_filesize_does_not_stat_ignored_file() { + let td = tmpdir(); + wfile(td.path().join(".ignore"), "ignored\n"); + + let mut b = builder(td.path()); + b.max_filesize(Some(0)); + let mut m = one_matcher(&b); + let (mat, err) = m.matched_with_errors("ignored", false); + assert!(mat.is_ignore()); + assert!(err.is_none(), "ignored missing file was statted: {err:?}"); + } + + #[cfg(unix)] + #[test] + fn max_filesize_respects_follow_links() { + use std::os::unix::fs::symlink; + + let td = tmpdir(); + wfile( + td.path().join("target"), + "target contents are much longer than the size limit", + ); + symlink("target", td.path().join("link")).unwrap(); + + let mut b = builder(td.path()); + b.max_filesize(Some(10)); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "link").is_none()); + + b.follow_links(true); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, "link").is_ignore()); + } + + #[test] + fn hidden_files_and_directories() { + let td = tmpdir(); + mkdirp(td.path().join(".hidden-dir")); + mkdirp(td.path().join("visible-dir")); + wfile(td.path().join(".hidden-file"), ""); + wfile(td.path().join("visible-file"), ""); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden-file").is_ignore()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_ignore()); + assert!(!dir.should_descend()); + assert!(matchedf(&mut m, "visible-file").is_none()); + let dir = matchedd(&mut m, "visible-dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + + let mut b = builder(td.path()); + b.hidden(false); + let mut m = one_matcher(&b); + assert!(matchedf(&mut m, ".hidden-file").is_none()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_none()); + assert!(dir.should_descend()); + } + + #[test] + fn gitignore_whitelist_overrides_hidden_filter() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden-dir")); + wfile(td.path().join(".hidden-file"), ""); + wfile(td.path().join(".gitignore"), "!.hidden-file\n!.hidden-dir/\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden-file").is_whitelist()); + let dir = matchedd(&mut m, ".hidden-dir"); + assert!(dir.is_whitelist()); + assert!(dir.should_descend()); + } + + #[test] + fn descendants_of_hidden_directories_are_ignored() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden/nested")); + mkdirp(td.path().join("visible/.hidden")); + wfile(td.path().join(".hidden/file"), ""); + wfile(td.path().join(".hidden/nested/file"), ""); + wfile(td.path().join("visible/.hidden/file"), ""); + wfile(td.path().join(".hidden/.gitignore"), "!file\n"); + + let mut m = one_matcher(&builder(td.path())); + assert!(matchedf(&mut m, ".hidden/file").is_ignore()); + assert!(matchedf(&mut m, ".hidden/nested/file").is_ignore()); + assert!(matchedf(&mut m, "visible/.hidden/file").is_ignore()); + } + + #[test] + fn whitelisted_hidden_directory_allows_descendants() { + let td = tmpdir(); + mkdirp(td.path().join(".git")); + mkdirp(td.path().join(".hidden")); + wfile(td.path().join(".hidden/file"), ""); + wfile(td.path().join(".gitignore"), "!.hidden/\n"); + + let mut m = one_matcher(&builder(td.path())); + let dir = matchedd(&mut m, ".hidden"); + assert!(dir.is_whitelist()); + assert!(dir.should_descend()); + assert!(matchedf(&mut m, ".hidden/file").is_none()); + } + + #[test] + fn hidden_directories_outside_depth_limits() { + let td = tmpdir(); + mkdirp(td.path().join("a/.hidden")); + + let mut b = builder(td.path()); + b.min_depth(Some(3)); + let mut m = one_matcher(&b); + let dir = matchedd(&mut m, "a/.hidden"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + + let mut b = builder(td.path()); + b.max_depth(Some(1)); + let mut m = one_matcher(&b); + let dir = matchedd(&mut m, "a/.hidden"); + assert!(dir.is_ignore()); + assert!(!dir.is_within_depth()); + } +} diff --git a/crates/ignore/src/lib.rs b/crates/ignore/src/lib.rs index 609004c4e..b9c9a3ece 100644 --- a/crates/ignore/src/lib.rs +++ b/crates/ignore/src/lib.rs @@ -48,13 +48,16 @@ See the documentation for `WalkBuilder` for many other options. use std::path::{Path, PathBuf}; +pub use crate::incremental::{IncrementalIgnore, IncrementalMatch}; pub use crate::walk::{ - DirEntry, ParallelVisitor, ParallelVisitorBuilder, Walk, WalkBuilder, WalkParallel, WalkState, + DirEntry, ParallelVisitor, ParallelVisitorBuilder, Walk, WalkBuilder, + WalkParallel, WalkState, }; mod default_types; mod dir; pub mod gitignore; +mod incremental; pub mod overrides; mod pathutil; pub mod types; @@ -120,34 +123,31 @@ impl Clone for Error { fn clone(&self) -> Error { match *self { Error::Partial(ref errs) => Error::Partial(errs.clone()), - Error::WithLineNumber { line, ref err } => Error::WithLineNumber { - line, - err: err.clone(), - }, - Error::WithPath { ref path, ref err } => Error::WithPath { - path: path.clone(), - err: err.clone(), - }, - Error::WithDepth { depth, ref err } => Error::WithDepth { - depth, - err: err.clone(), - }, - Error::Loop { - ref ancestor, - ref child, - } => Error::Loop { + Error::WithLineNumber { line, ref err } => { + Error::WithLineNumber { line, err: err.clone() } + } + Error::WithPath { ref path, ref err } => { + Error::WithPath { path: path.clone(), err: err.clone() } + } + Error::WithDepth { depth, ref err } => { + Error::WithDepth { depth, err: err.clone() } + } + Error::Loop { ref ancestor, ref child } => Error::Loop { ancestor: ancestor.clone(), child: child.clone(), }, Error::Io(ref err) => match err.raw_os_error() { Some(e) => Error::Io(std::io::Error::from_raw_os_error(e)), - None => Error::Io(std::io::Error::new(err.kind(), err.to_string())), + None => { + Error::Io(std::io::Error::new(err.kind(), err.to_string())) + } }, - Error::Glob { ref glob, ref err } => Error::Glob { - glob: glob.clone(), - err: err.clone(), - }, - Error::UnrecognizedFileType(ref err) => Error::UnrecognizedFileType(err.clone()), + Error::Glob { ref glob, ref err } => { + Error::Glob { glob: glob.clone(), err: err.clone() } + } + Error::UnrecognizedFileType(ref err) => { + Error::UnrecognizedFileType(err.clone()) + } Error::InvalidDefinition => Error::InvalidDefinition, } } @@ -269,19 +269,14 @@ impl Error { /// Turn an error into a tagged error with the given depth. fn with_depth(self, depth: usize) -> Error { - Error::WithDepth { - depth, - err: Box::new(self), - } + Error::WithDepth { depth, err: Box::new(self) } } /// Turn an error into a tagged error with the given file path and line /// number. If path is empty, then it is omitted from the error. fn tagged>(self, path: P, lineno: u64) -> Error { - let errline = Error::WithLineNumber { - line: lineno, - err: Box::new(self), - }; + let errline = + Error::WithLineNumber { line: lineno, err: Box::new(self) }; if path.as_ref().as_os_str().is_empty() { return errline; } @@ -301,12 +296,12 @@ impl Error { }; } let path = err.path().map(|p| p.to_path_buf()); - let mut ig_err = Error::Io(std::io::Error::from(err)); + let mut ig_err = Error::WithDepth { + depth, + err: Box::new(Error::Io(std::io::Error::from(err))), + }; if let Some(path) = path { - ig_err = Error::WithPath { - path, - err: Box::new(ig_err), - }; + ig_err = Error::WithPath { path, err: Box::new(ig_err) }; } ig_err } @@ -333,7 +328,8 @@ impl std::fmt::Display for Error { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { match *self { Error::Partial(ref errs) => { - let msgs: Vec = errs.iter().map(|err| err.to_string()).collect(); + let msgs: Vec = + errs.iter().map(|err| err.to_string()).collect(); write!(f, "{}", msgs.join("\n")) } Error::WithLineNumber { line, ref err } => { @@ -343,10 +339,7 @@ impl std::fmt::Display for Error { write!(f, "{}: {}", path.display(), err) } Error::WithDepth { ref err, .. } => err.fmt(f), - Error::Loop { - ref ancestor, - ref child, - } => write!( + Error::Loop { ref ancestor, ref child } => write!( f, "File system loop found: \ {} points to an ancestor {}", @@ -354,14 +347,8 @@ impl std::fmt::Display for Error { ancestor.display() ), Error::Io(ref err) => err.fmt(f), - Error::Glob { - glob: None, - ref err, - } => write!(f, "{}", err), - Error::Glob { - glob: Some(ref glob), - ref err, - } => { + Error::Glob { glob: None, ref err } => write!(f, "{}", err), + Error::Glob { glob: Some(ref glob), ref err } => { write!(f, "error parsing glob '{}': {}", glob, err) } Error::UnrecognizedFileType(ref ty) => { @@ -507,7 +494,8 @@ mod tests { }; /// A convenient result type alias. - pub(crate) type Result = std::result::Result>; + pub(crate) type Result = + std::result::Result>; macro_rules! err { ($($tt:tt)*) => { @@ -545,8 +533,9 @@ mod tests { if path.is_dir() { continue; } - fs::create_dir_all(&path) - .map_err(|e| err!("failed to create {}: {}", path.display(), e))?; + fs::create_dir_all(&path).map_err(|e| { + err!("failed to create {}: {}", path.display(), e) + })?; return Ok(TempDir(path)); } Err(err!("failed to create temp dir after {} tries", TRIES)) diff --git a/crates/ignore/src/overrides.rs b/crates/ignore/src/overrides.rs index afb9f16ce..005cae8f2 100644 --- a/crates/ignore/src/overrides.rs +++ b/crates/ignore/src/overrides.rs @@ -94,7 +94,11 @@ impl Override { /// given) is stripped. If there is no common suffix/prefix overlap, then /// `path` is assumed to reside in the same directory as the root path for /// this set of overrides. - pub fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { if self.is_empty() { return Match::None; } @@ -146,7 +150,10 @@ impl OverrideBuilder { /// affected. /// /// This is disabled by default. - pub fn case_insensitive(&mut self, yes: bool) -> Result<&mut OverrideBuilder, Error> { + pub fn case_insensitive( + &mut self, + yes: bool, + ) -> Result<&mut OverrideBuilder, Error> { // TODO: This should not return a `Result`. Fix this in the next semver // release. self.builder.case_insensitive(yes)?; @@ -276,11 +283,8 @@ mod tests { #[test] fn default_case_sensitive() { - let ov = OverrideBuilder::new(ROOT) - .add("*.html") - .unwrap() - .build() - .unwrap(); + let ov = + OverrideBuilder::new(ROOT).add("*.html").unwrap().build().unwrap(); assert!(ov.matched("foo.html", false).is_whitelist()); assert!(ov.matched("foo.HTML", false).is_ignore()); assert!(ov.matched("foo.htm", false).is_ignore()); diff --git a/crates/ignore/src/pathutil.rs b/crates/ignore/src/pathutil.rs index 0ceb5a356..5de8c106c 100644 --- a/crates/ignore/src/pathutil.rs +++ b/crates/ignore/src/pathutil.rs @@ -2,55 +2,89 @@ use std::{ffi::OsStr, path::Path}; use crate::walk::DirEntry; -/// Returns true if and only if this entry is considered to be hidden. +/// Returns true if and only if this path is considered to be hidden. /// -/// This only returns true if the base name of the path starts with a `.`. +/// # Platform behavior /// -/// On Unix, this implements a more optimized check. -#[cfg(unix)] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - use std::os::unix::ffi::OsStrExt; - - if let Some(name) = file_name(dent.path()) { - name.as_bytes().get(0) == Some(&b'.') - } else { - false - } -} - -/// Returns true if and only if this entry is considered to be hidden. +/// ## Windows /// -/// On Windows, this returns true if one of the following is true: +/// This returns true if one of the following is true: /// /// * The base name of the path starts with a `.`. /// * The file attributes have the `HIDDEN` property set. -#[cfg(windows)] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - use std::os::windows::fs::MetadataExt; - use winapi_util::file; - - // This looks like we're doing an extra stat call, but on Windows, the - // directory traverser reuses the metadata retrieved from each directory - // entry and stores it on the DirEntry itself. So this is "free." - if let Ok(md) = dent.metadata() { - if file::is_hidden(md.file_attributes() as u64) { - return true; - } - } - if let Some(name) = file_name(dent.path()) { - name.to_str().map(|s| s.starts_with(".")).unwrap_or(false) - } else { - false - } -} - -/// Returns true if and only if this entry is considered to be hidden. +/// +/// ## All other platforms /// /// This only returns true if the base name of the path starts with a `.`. -#[cfg(not(any(unix, windows)))] -pub(crate) fn is_hidden(dent: &DirEntry) -> bool { - if let Some(name) = file_name(dent.path()) { - name.to_str().map(|s| s.starts_with(".")).unwrap_or(false) +pub(crate) fn is_hidden_path(dent: &Path) -> bool { + #[cfg(not(windows))] + fn imp(path: &Path) -> bool { + is_hidden_path_only(path) + } + + #[cfg(windows)] + fn imp(path: &Path) -> bool { + use std::os::windows::fs::MetadataExt; + use winapi_util::file; + + if let Ok(md) = path.metadata() { + if file::is_hidden(md.file_attributes() as u64) { + return true; + } + } + is_hidden_path_only(path) + } + + imp(dent) +} + +/// Returns true if and only if this directory entry is considered to be +/// hidden. +/// +/// # Platform behavior +/// +/// ## Windows +/// +/// This returns true if one of the following is true: +/// +/// * The base name of the path starts with a `.`. +/// * The file attributes have the `HIDDEN` property set. +/// +/// ## All other platforms +/// +/// This only returns true if the base name of the path starts with a `.`. +pub(crate) fn is_hidden_entry(dent: &DirEntry) -> bool { + #[cfg(not(windows))] + fn imp(dent: &DirEntry) -> bool { + is_hidden_path_only(dent.path()) + } + + #[cfg(windows)] + fn imp(dent: &DirEntry) -> bool { + use std::os::windows::fs::MetadataExt; + use winapi_util::file; + + // This looks like we're doing an extra stat call, but on Windows, the + // directory traverser reuses the metadata retrieved from each directory + // entry and stores it on the DirEntry itself. So this is "free." + if let Ok(md) = dent.metadata() { + if file::is_hidden(md.file_attributes() as u64) { + return true; + } + } + is_hidden_path_only(dent.path()) + } + + imp(dent) +} + +/// Returns true if and only if this path is considered to be hidden from only +/// the path itself. +/// +/// This has the same behavior on all platforms. +fn is_hidden_path_only(path: &Path) -> bool { + if let Some(name) = file_name(path) { + name.as_encoded_bytes().starts_with(b".") } else { false } @@ -59,83 +93,79 @@ pub(crate) fn is_hidden(dent: &DirEntry) -> bool { /// Strip `prefix` from the `path` and return the remainder. /// /// If `path` doesn't have a prefix `prefix`, then return `None`. -#[cfg(unix)] pub(crate) fn strip_prefix<'a, P: AsRef + ?Sized>( prefix: &'a P, path: &'a Path, ) -> Option<&'a Path> { - use std::os::unix::ffi::OsStrExt; + #[cfg(unix)] + fn imp<'a>(prefix: &'a Path, path: &'a Path) -> Option<&'a Path> { + use std::os::unix::ffi::OsStrExt; - let prefix = prefix.as_ref().as_os_str().as_bytes(); - let path = path.as_os_str().as_bytes(); - if prefix.len() > path.len() || prefix != &path[0..prefix.len()] { - None - } else { - Some(&Path::new(OsStr::from_bytes(&path[prefix.len()..]))) + let prefix = prefix.as_os_str().as_bytes(); + let path = path.as_os_str().as_bytes(); + if prefix.len() > path.len() || prefix != &path[0..prefix.len()] { + None + } else { + Some(&Path::new(OsStr::from_bytes(&path[prefix.len()..]))) + } } -} -/// Strip `prefix` from the `path` and return the remainder. -/// -/// If `path` doesn't have a prefix `prefix`, then return `None`. -#[cfg(not(unix))] -pub(crate) fn strip_prefix<'a, P: AsRef + ?Sized>( - prefix: &'a P, - path: &'a Path, -) -> Option<&'a Path> { - path.strip_prefix(prefix).ok() + #[cfg(not(unix))] + fn imp<'a>(prefix: &'a Path, path: &'a Path) -> Option<&'a Path> { + path.strip_prefix(prefix).ok() + } + + imp(prefix.as_ref(), path) } /// Returns true if this file path is just a file name. i.e., Its parent is /// the empty string. -#[cfg(unix)] pub(crate) fn is_file_name>(path: P) -> bool { - use std::os::unix::ffi::OsStrExt; - - use memchr::memchr; - - let path = path.as_ref().as_os_str().as_bytes(); - memchr(b'/', path).is_none() -} - -/// Returns true if this file path is just a file name. i.e., Its parent is -/// the empty string. -#[cfg(not(unix))] -pub(crate) fn is_file_name>(path: P) -> bool { - path.as_ref() - .parent() - .map(|p| p.as_os_str().is_empty()) - .unwrap_or(false) -} - -/// The final component of the path, if it is a normal file. -/// -/// If the path terminates in ., .., or consists solely of a root of prefix, -/// file_name will return None. -#[cfg(unix)] -pub(crate) fn file_name<'a, P: AsRef + ?Sized>(path: &'a P) -> Option<&'a OsStr> { - use memchr::memrchr; - use std::os::unix::ffi::OsStrExt; - - let path = path.as_ref().as_os_str().as_bytes(); - if path.is_empty() { - return None; - } else if path.len() == 1 && path[0] == b'.' { - return None; - } else if path.last() == Some(&b'.') { - return None; - } else if path.len() >= 2 && &path[path.len() - 2..] == &b".."[..] { - return None; + #[cfg(unix)] + { + memchr::memchr(b'/', path.as_ref().as_os_str().as_encoded_bytes()) + .is_none() + } + #[cfg(not(unix))] + { + path.as_ref() + .parent() + .map(|p| p.as_os_str().is_empty()) + .unwrap_or(false) } - let last_slash = memrchr(b'/', path).map(|i| i + 1).unwrap_or(0); - Some(OsStr::from_bytes(&path[last_slash..])) } /// The final component of the path, if it is a normal file. /// -/// If the path terminates in ., .., or consists solely of a root of prefix, -/// file_name will return None. -#[cfg(not(unix))] -pub(crate) fn file_name<'a, P: AsRef + ?Sized>(path: &'a P) -> Option<&'a OsStr> { - path.as_ref().file_name() +/// If the path terminates in `.`, `..`, or consists solely of a root of +/// prefix, this will return `None`. +pub(crate) fn file_name<'a, P: AsRef + ?Sized>( + path: &'a P, +) -> Option<&'a OsStr> { + #[cfg(unix)] + fn imp(path: &Path) -> Option<&OsStr> { + use std::os::unix::ffi::OsStrExt; + + use memchr::memrchr; + + let path = path.as_os_str().as_bytes(); + if path.is_empty() { + return None; + } else if path.len() == 1 && path[0] == b'.' { + return None; + } else if path.last() == Some(&b'.') { + return None; + } else if path.len() >= 2 && &path[path.len() - 2..] == &b".."[..] { + return None; + } + let last_slash = memrchr(b'/', path).map(|i| i + 1).unwrap_or(0); + Some(OsStr::from_bytes(&path[last_slash..])) + } + + #[cfg(not(unix))] + fn imp(path: &Path) -> Option<&OsStr> { + path.file_name() + } + + imp(path.as_ref()) } diff --git a/crates/ignore/src/types.rs b/crates/ignore/src/types.rs index aa23999c0..313cf5c0e 100644 --- a/crates/ignore/src/types.rs +++ b/crates/ignore/src/types.rs @@ -204,8 +204,12 @@ impl Selection { fn map U>(self, f: F) -> Selection { match self { - Selection::Select(name, inner) => Selection::Select(name, f(inner)), - Selection::Negate(name, inner) => Selection::Negate(name, f(inner)), + Selection::Select(name, inner) => { + Selection::Select(name, f(inner)) + } + Selection::Negate(name, inner) => { + Selection::Negate(name, f(inner)) + } } } @@ -227,7 +231,9 @@ impl Types { has_selected: false, glob_to_selection: vec![], set: GlobSetBuilder::new().build().unwrap(), - matches: Arc::new(Pool::new(|| vec![])), + matches: Arc::new(Pool::with_available_parallelism_capacity( + || vec![], + )), } } @@ -254,7 +260,11 @@ impl Types { /// The path is considered ignored if it matches a negated file type. /// If at least one file type is selected and `path` doesn't match, then /// the path is also considered ignored. - pub fn matched<'a, P: AsRef>(&'a self, path: P, is_dir: bool) -> Match> { + pub fn matched<'a, P: AsRef>( + &'a self, + path: P, + is_dir: bool, + ) -> Match> { // File types don't apply to directories, and we can't do anything // if our glob set is empty. if is_dir || self.set.is_empty() { @@ -306,10 +316,7 @@ impl TypesBuilder { /// of default type definitions can be added with `add_defaults`, and /// additional type definitions can be added with `select` and `negate`. pub fn new() -> TypesBuilder { - TypesBuilder { - types: HashMap::new(), - selections: vec![], - } + TypesBuilder { types: HashMap::new(), selections: vec![] } } /// Build the current set of file type definitions *and* selections into @@ -343,17 +350,18 @@ impl TypesBuilder { } selections.push(selection.clone().map(move |_| def)); } - let set = build_set.build().map_err(|err| Error::Glob { - glob: None, - err: err.to_string(), - })?; + let set = build_set + .build() + .map_err(|err| Error::Glob { glob: None, err: err.to_string() })?; Ok(Types { defs, selections, has_selected, glob_to_selection, set, - matches: Arc::new(Pool::new(|| vec![])), + matches: Arc::new(Pool::with_available_parallelism_capacity( + || vec![], + )), }) } @@ -377,12 +385,10 @@ impl TypesBuilder { pub fn select(&mut self, name: &str) -> &mut TypesBuilder { if name == "all" { for name in self.types.keys() { - self.selections - .push(Selection::Select(name.to_string(), ())); + self.selections.push(Selection::Select(name.to_string(), ())); } } else { - self.selections - .push(Selection::Select(name.to_string(), ())); + self.selections.push(Selection::Select(name.to_string(), ())); } self } @@ -393,12 +399,10 @@ impl TypesBuilder { pub fn negate(&mut self, name: &str) -> &mut TypesBuilder { if name == "all" { for name in self.types.keys() { - self.selections - .push(Selection::Negate(name.to_string(), ())); + self.selections.push(Selection::Negate(name.to_string(), ())); } } else { - self.selections - .push(Selection::Negate(name.to_string(), ())); + self.selections.push(Selection::Negate(name.to_string(), ())); } self } @@ -453,7 +457,10 @@ impl TypesBuilder { 3 => { let name = parts[0]; let types_string = parts[2]; - if name.is_empty() || parts[1] != "include" || types_string.is_empty() { + if name.is_empty() + || parts[1] != "include" + || types_string.is_empty() + { return Err(Error::InvalidDefinition); } let types = types_string.split(','); @@ -463,7 +470,8 @@ impl TypesBuilder { return Err(Error::InvalidDefinition); } for type_name in types { - let globs = self.types.get(type_name).unwrap().globs.clone(); + let globs = + self.types.get(type_name).unwrap().globs.clone(); for glob in globs { self.add(name, &glob)?; } @@ -549,30 +557,9 @@ mod tests { matched!(not, matchnot1, types(), vec!["rust"], vec![], "index.html"); matched!(not, matchnot2, types(), vec![], vec!["rust"], "main.rs"); - matched!( - not, - matchnot3, - types(), - vec!["foo"], - vec!["rust"], - "main.rs" - ); - matched!( - not, - matchnot4, - types(), - vec!["rust"], - vec!["foo"], - "main.rs" - ); - matched!( - not, - matchnot5, - types(), - vec!["rust"], - vec!["foo"], - "main.foo" - ); + matched!(not, matchnot3, types(), vec!["foo"], vec!["rust"], "main.rs"); + matched!(not, matchnot4, types(), vec!["rust"], vec!["foo"], "main.rs"); + matched!(not, matchnot5, types(), vec!["rust"], vec!["foo"], "main.foo"); matched!(not, matchnot6, types(), vec!["combo"], vec![], "leftpad.js"); matched!(not, matchnot7, types(), vec!["py"], vec![], "index.html"); matched!(not, matchnot8, types(), vec!["python"], vec![], "doc.md"); diff --git a/crates/ignore/src/walk.rs b/crates/ignore/src/walk.rs index ebf20947a..2dff78852 100644 --- a/crates/ignore/src/walk.rs +++ b/crates/ignore/src/walk.rs @@ -17,7 +17,9 @@ use { use crate::{ Error, PartialErrorBuilder, dir::{Ignore, IgnoreBuilder}, + // CHANGED: Also import `Gitignore` for `WalkBuilder::add_gitignore`. gitignore::{Gitignore, GitignoreBuilder}, + incremental::{IncrementalIgnore, IncrementalIgnoreOptions}, overrides::Override, types::Types, }; @@ -104,24 +106,15 @@ impl DirEntry { } fn new_stdin() -> DirEntry { - DirEntry { - dent: DirEntryInner::Stdin, - err: None, - } + DirEntry { dent: DirEntryInner::Stdin, err: None } } fn new_walkdir(dent: walkdir::DirEntry, err: Option) -> DirEntry { - DirEntry { - dent: DirEntryInner::Walkdir(dent), - err, - } + DirEntry { dent: DirEntryInner::Walkdir(dent), err } } fn new_raw(dent: DirEntryRaw, err: Option) -> DirEntry { - DirEntry { - dent: DirEntryInner::Raw(dent), - err, - } + DirEntry { dent: DirEntryInner::Raw(dent), err } } } @@ -187,9 +180,11 @@ impl DirEntryInner { )); Err(err.with_path("")) } - Walkdir(ref x) => x - .metadata() - .map_err(|err| Error::Io(io::Error::from(err)).with_path(x.path())), + Walkdir(ref x) => x.metadata().map_err(|err| { + Error::Io(io::Error::from(err)) + .with_depth(x.depth()) + .with_path(x.path()) + }), Raw(ref x) => x.metadata(), } } @@ -308,7 +303,9 @@ impl DirEntryRaw { } else { fs::symlink_metadata(&self.path) } - .map_err(|err| Error::Io(io::Error::from(err)).with_path(&self.path)) + .map_err(|err| { + Error::Io(err).with_depth(self.depth).with_path(&self.path) + }) } fn file_type(&self) -> FileType { @@ -316,9 +313,7 @@ impl DirEntryRaw { } fn file_name(&self) -> &OsStr { - self.path - .file_name() - .unwrap_or_else(|| self.path.as_os_str()) + self.path.file_name().unwrap_or_else(|| self.path.as_os_str()) } fn depth(&self) -> usize { @@ -330,13 +325,13 @@ impl DirEntryRaw { self.ino } - fn from_entry(depth: usize, ent: &fs::DirEntry) -> Result { + fn from_entry( + depth: usize, + ent: &fs::DirEntry, + ) -> Result { let ty = ent.file_type().map_err(|err| { - let err = Error::Io(io::Error::from(err)).with_path(ent.path()); - Error::WithDepth { - depth, - err: Box::new(err), - } + let err = Error::Io(err).with_depth(depth).with_path(ent.path()); + Error::WithDepth { depth, err: Box::new(err) } })?; DirEntryRaw::from_entry_os(depth, ent, ty) } @@ -348,11 +343,8 @@ impl DirEntryRaw { ty: fs::FileType, ) -> Result { let md = ent.metadata().map_err(|err| { - let err = Error::Io(io::Error::from(err)).with_path(ent.path()); - Error::WithDepth { - depth, - err: Box::new(err), - } + let err = Error::Io(err).with_depth(depth).with_path(ent.path()); + Error::WithDepth { depth, err: Box::new(err) } })?; Ok(DirEntryRaw { path: ent.path(), @@ -395,8 +387,13 @@ impl DirEntryRaw { } #[cfg(windows)] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { - let md = fs::metadata(&pb).map_err(|err| Error::Io(err).with_path(&pb))?; + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { + let md = fs::metadata(&pb) + .map_err(|err| Error::Io(err).with_depth(depth).with_path(&pb))?; Ok(DirEntryRaw { path: pb, ty: md.file_type(), @@ -407,10 +404,15 @@ impl DirEntryRaw { } #[cfg(unix)] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { use std::os::unix::fs::MetadataExt; - let md = fs::metadata(&pb).map_err(|err| Error::Io(err).with_path(&pb))?; + let md = fs::metadata(&pb) + .map_err(|err| Error::Io(err).with_depth(depth).with_path(&pb))?; Ok(DirEntryRaw { path: pb, ty: md.file_type(), @@ -423,7 +425,11 @@ impl DirEntryRaw { // Placeholder implementation to allow compiling on non-standard platforms // (e.g. wasm32). #[cfg(not(any(windows, unix)))] - fn from_path(depth: usize, pb: PathBuf, link: bool) -> Result { + fn from_path( + depth: usize, + pb: PathBuf, + link: bool, + ) -> Result { Err(Error::Io(io::Error::new( io::ErrorKind::Other, "unsupported platform", @@ -502,7 +508,8 @@ pub struct WalkBuilder { /// /// When `None`, the CWD is fetched from `std::env::current_dir()`. If /// that fails, then global gitignores are ignored (an error is logged). - global_gitignores_relative_to: OnceLock>>, + global_gitignores_relative_to: + OnceLock>>, } #[derive(Clone)] @@ -544,8 +551,16 @@ impl WalkBuilder { /// is better to call `add` on this builder than to create multiple /// `Walk` values. pub fn new>(path: P) -> WalkBuilder { + WalkBuilder::from_iter([path]) + } + + /// Create an empty builder to which paths can be added. + /// + /// Note that if you call `build` on this instance before calling `add` + /// on it, it will return exactly zero items during iteration. + pub fn empty() -> WalkBuilder { WalkBuilder { - paths: vec![path.as_ref().to_path_buf()], + paths: vec![], ig_builder: IgnoreBuilder::new(), max_depth: None, min_depth: None, @@ -560,6 +575,21 @@ impl WalkBuilder { } } + /// Create a new builder for a recursive directory iterator from the + /// sequence of paths. + /// + /// Note that if the iterator is empty, this is the same as + /// `WalkBuilder::empty`. + pub fn from_iter, I: IntoIterator>( + paths: I, + ) -> WalkBuilder { + let mut builder = WalkBuilder::empty(); + for path in paths.into_iter() { + builder.add(path); + } + builder + } + /// Build a new `Walk` iterator. pub fn build(&self) -> Walk { let follow_links = self.follow_links; @@ -585,10 +615,14 @@ impl WalkBuilder { if let Some(ref sorter) = sorter { match sorter.clone() { Sorter::ByName(cmp) => { - wd = wd.sort_by(move |a, b| cmp(a.file_name(), b.file_name())); + wd = wd.sort_by(move |a, b| { + cmp(a.file_name(), b.file_name()) + }); } Sorter::ByPath(cmp) => { - wd = wd.sort_by(move |a, b| cmp(a.path(), b.path())); + wd = wd.sort_by(move |a, b| { + cmp(a.path(), b.path()) + }); } } } @@ -597,31 +631,76 @@ impl WalkBuilder { }) .collect::>() .into_iter(); - let ig_root = self - .get_or_set_current_dir() - .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) - .unwrap_or_else(|| self.ig_builder.build()); + let ig_root = self.build_ignore(); Walk { its, it: None, ig_root: ig_root.clone(), ig: ig_root.clone(), + max_depth: self.max_depth, max_filesize: self.max_filesize, skip: self.skip.clone(), filter: self.filter.clone(), } } + /// Build matchers for checking paths against ignore files without + /// recursively walking the configured roots. + /// + /// The returned matchers use the path-based filtering configuration + /// on this builder, including glob overrides, file type selections, + /// parent ignore files, `.ignore`, `.gitignore`, global Git + /// ignore files, explicitly added ignore files and custom ignore + /// file names. For example, ripgrep configures `.rgignore` via + /// [`WalkBuilder::add_custom_ignore_filename`]. Minimum and maximum depth + /// limits, maximum file size and hidden-file filtering are also applied. + /// Other options that only control traversal or require a directory entry, + /// such as custom entry predicates, are not applied. + /// + /// One matcher is returned for each configured path, in the same order as + /// the paths were added to this builder. Each matcher accepts paths + /// relative to its own [`IncrementalIgnore::root`]. The matcher for the + /// special `-` path representing standard input always returns a non-match + /// for all inputs. + /// + /// Ignore matchers are loaded lazily and cached by directory. + /// Thus, the first query may read ignore files from the root and + /// its parents, while later queries reuse the compiled matchers. + /// Errors encountered while loading ignore files are returned by + /// [`IncrementalIgnore::matched_with_errors`]. Once an ignore file has + /// been loaded, changes to it are not observed. Build new matchers to + /// reload changed ignore files. + /// + /// Matchers built together share the builder's base ignore configuration + /// and compiled parent matchers. + pub fn build_matchers(&self) -> Vec { + let ignore = self.build_ignore(); + let options = IncrementalIgnoreOptions { + min_depth: self.min_depth, + max_depth: self.max_depth, + max_filesize: self.max_filesize, + hidden: self.ig_builder.is_hidden(), + follow_links: self.follow_links, + }; + self.paths + .iter() + .map(move |path| { + IncrementalIgnore::new( + path.clone(), + ignore.clone(), + options.clone(), + ) + }) + .collect() + } + /// Build a new `WalkParallel` iterator. /// /// Note that this *doesn't* return something that implements `Iterator`. /// Instead, the returned value must be run with a closure. e.g., /// `builder.build_parallel().run(|| |path| { println!("{path:?}"); WalkState::Continue })`. pub fn build_parallel(&self) -> WalkParallel { - let ig_root = self - .get_or_set_current_dir() - .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) - .unwrap_or_else(|| self.ig_builder.build()); + let ig_root = self.build_ignore(); WalkParallel { paths: self.paths.clone().into_iter(), ig_root, @@ -651,7 +730,10 @@ impl WalkBuilder { /// The default, `None`, imposes no depth restriction. pub fn max_depth(&mut self, depth: Option) -> &mut WalkBuilder { self.max_depth = depth; - if self.min_depth.is_some() && self.max_depth.is_some() && self.max_depth < self.min_depth { + if self.min_depth.is_some() + && self.max_depth.is_some() + && self.max_depth < self.min_depth + { self.max_depth = self.min_depth; } self @@ -662,7 +744,10 @@ impl WalkBuilder { /// The default, `None`, imposes no minimum depth restriction. pub fn min_depth(&mut self, depth: Option) -> &mut WalkBuilder { self.min_depth = depth; - if self.max_depth.is_some() && self.min_depth.is_some() && self.min_depth > self.max_depth { + if self.max_depth.is_some() + && self.min_depth.is_some() + && self.min_depth > self.max_depth + { self.min_depth = self.max_depth; } self @@ -705,7 +790,12 @@ impl WalkBuilder { /// An error will also occur if this walker could not get the current /// working directory (and `WalkBuilder::current_dir` isn't set). pub fn add_ignore>(&mut self, path: P) -> Option { - // CHANGED: Dropped this code + // CHANGED: Root the ignore file at `""` instead of the current working + // directory. Explicit ignores are scoped to the directory of the + // ignore file (see `matched_ignore`), and a root of `""` makes the + // rules apply to every walked path regardless of the walk root. This + // also avoids depending on the current working directory entirely. + // // let path = path.as_ref(); // let Some(cwd) = self.get_or_set_current_dir() else { // let err = std::io::Error::other(format!( @@ -729,7 +819,11 @@ impl WalkBuilder { errs.into_error_option() } - /// CHANGED: Add a Gitignore to the builder. + /// CHANGED: Add a prebuilt Gitignore to the builder. + /// + /// Like the ignore file added via `add_ignore`, these rules are matched + /// against the full path of each walked entry, scoped to the `Gitignore`'s + /// root path. pub fn add_gitignore(&mut self, gi: Gitignore) { self.ig_builder.add_ignore(gi); } @@ -982,7 +1076,10 @@ impl WalkBuilder { /// /// Global gitignore files come from things like a user's git configuration /// or from gitignore files added via [`WalkBuilder::add_ignore`]. - pub fn current_dir(&mut self, cwd: impl Into) -> &mut WalkBuilder { + pub fn current_dir( + &mut self, + cwd: impl Into, + ) -> &mut WalkBuilder { let cwd = cwd.into(); self.ig_builder.current_dir(cwd.clone()); if let Err(cwd) = self.global_gitignores_relative_to.set(Ok(cwd)) { @@ -1002,7 +1099,10 @@ impl WalkBuilder { let result = std::env::current_dir().map_err(Arc::new); match result { Ok(ref path) => { - log::trace!("automatically discovered CWD: {}", path.display()); + log::trace!( + "automatically discovered CWD: {}", + path.display() + ); } Err(ref err) => { log::debug!( @@ -1016,6 +1116,13 @@ impl WalkBuilder { }); result.as_ref().ok().map(|path| &**path) } + + /// Build the root ignore matcher shared by all consumers of this builder. + fn build_ignore(&self) -> Ignore { + self.get_or_set_current_dir() + .map(|cwd| self.ig_builder.build_with_cwd(Some(cwd.to_path_buf()))) + .unwrap_or_else(|| self.ig_builder.build()) + } } /// Walk is a recursive directory iterator over file paths in one or more @@ -1029,6 +1136,7 @@ pub struct Walk { it: Option, ig_root: Ignore, ig: Ignore, + max_depth: Option, max_filesize: Option, skip: Option>, filter: Option, @@ -1044,6 +1152,17 @@ impl Walk { WalkBuilder::new(path).build() } + /// Create a new recursive directory iterator from the sequence of paths + /// given. + /// + /// Note that if the provided iterator is empty, then `Walk` is guaranteed + /// to yield zero entries. + pub fn from_iter, I: IntoIterator>( + paths: I, + ) -> Walk { + WalkBuilder::from_iter(paths).build() + } + fn skip_entry(&self, ent: &DirEntry) -> Result { if ent.depth() == 0 { return Ok(false); @@ -1128,12 +1247,17 @@ impl Iterator for Walk { self.it.as_mut().unwrap().it.skip_current_dir(); // Still need to push this on the stack because // we'll get a WalkEvent::Exit event for this dir. - // We don't care if it errors though. - let (igtmp, _) = self.ig.add_child(ent.path()); + // Its ignore files cannot apply to any visited entry. + let (igtmp, _) = + self.ig.add_child_with_entries(ent.path(), &[]); self.ig = igtmp; continue; } - let (igtmp, err) = self.ig.add_child(ent.path()); + let (igtmp, err) = if self.max_depth == Some(ent.depth()) { + self.ig.add_child_with_entries(ent.path(), &[]) + } else { + self.ig.add_child(ent.path()) + }; self.ig = igtmp; ent.err = err; return Some(Ok(ent)); @@ -1175,11 +1299,7 @@ enum WalkEvent { impl From for WalkEventIter { fn from(it: WalkDir) -> WalkEventIter { - WalkEventIter { - depth: 0, - it: it.into_iter(), - next: None, - } + WalkEventIter { depth: 0, it: it.into_iter(), next: None } } } @@ -1252,7 +1372,9 @@ pub trait ParallelVisitorBuilder<'s> { fn build(&mut self) -> Box; } -impl<'a, 's, P: ParallelVisitorBuilder<'s>> ParallelVisitorBuilder<'s> for &'a mut P { +impl<'a, 's, P: ParallelVisitorBuilder<'s>> ParallelVisitorBuilder<'s> + for &'a mut P +{ fn build(&mut self) -> Box { (**self).build() } @@ -1273,14 +1395,17 @@ struct FnBuilder { builder: F, } -impl<'s, F: FnMut() -> FnVisitor<'s>> ParallelVisitorBuilder<'s> for FnBuilder { +impl<'s, F: FnMut() -> FnVisitor<'s>> ParallelVisitorBuilder<'s> + for FnBuilder +{ fn build(&mut self) -> Box { let visitor = (self.builder)(); Box::new(FnVisitorImp { visitor }) } } -type FnVisitor<'s> = Box) -> WalkState + Send + 's>; +type FnVisitor<'s> = + Box) -> WalkState + Send + 's>; struct FnVisitorImp<'s> { visitor: FnVisitor<'s>, @@ -1370,7 +1495,9 @@ impl WalkParallel { } }; match DirEntryRaw::from_path(0, path, false) { - Ok(dent) => (DirEntry::new_raw(dent, None), root_device), + Ok(dent) => { + (DirEntry::new_raw(dent, None), root_device) + } Err(err) => { if visitor.visit(Err(err)).is_quit() { return; @@ -1394,21 +1521,28 @@ impl WalkParallel { let quit_now = Arc::new(AtomicBool::new(false)); let active_workers = Arc::new(AtomicUsize::new(threads)); let stacks = Stack::new_for_each_thread(threads, stack); + // Collect all of the workers first. In the case that + // `builder.build()` panics, we want that to happen and + // propagate before we actually start to run any of the + // workers. + let workers: Vec<_> = stacks + .into_iter() + .map(|stack| Worker { + visitor: builder.build(), + stack, + quit_now: quit_now.clone(), + active_workers: active_workers.clone(), + max_depth: self.max_depth, + min_depth: self.min_depth, + max_filesize: self.max_filesize, + follow_links: self.follow_links, + skip: self.skip.clone(), + filter: self.filter.clone(), + }) + .collect(); std::thread::scope(|s| { - let handles: Vec<_> = stacks + let handles: Vec<_> = workers .into_iter() - .map(|stack| Worker { - visitor: builder.build(), - stack, - quit_now: quit_now.clone(), - active_workers: active_workers.clone(), - max_depth: self.max_depth, - min_depth: self.min_depth, - max_filesize: self.max_filesize, - follow_links: self.follow_links, - skip: self.skip.clone(), - filter: self.filter.clone(), - }) .map(|worker| s.spawn(|| worker.run())) .collect(); for handle in handles { @@ -1419,9 +1553,7 @@ impl WalkParallel { fn threads(&self) -> usize { if self.threads == 0 { - std::thread::available_parallelism() - .map_or(1, |n| n.get()) - .min(12) + std::thread::available_parallelism().map_or(1, |n| n.get()).min(12) } else { self.threads } @@ -1452,6 +1584,12 @@ struct Work { root_device: Option, } +#[derive(Default)] +struct ReadDirResult { + entries: Vec, + errors: Vec, +} + impl Work { /// Returns true if and only if this work item is a directory. fn is_dir(&self) -> bool { @@ -1478,6 +1616,13 @@ impl Work { err } + /// Adds ignore rules for this directory without reading its contents. + fn add_ignore(&mut self) { + let (ig, err) = self.ignore.add_child(self.dent.path()); + self.ignore = ig; + self.dent.err = err; + } + /// Reads the directory contents of this work item and adds ignore /// rules for this directory. /// @@ -1485,7 +1630,7 @@ impl Work { /// an error is returned. If there was a problem reading the ignore /// rules for this directory, then the error is attached to this /// work item's directory entry. - fn read_dir(&mut self) -> Result { + fn read_dir(&mut self) -> Result { let readdir = match fs::read_dir(self.dent.path()) { Ok(readdir) => readdir, Err(err) => { @@ -1495,10 +1640,24 @@ impl Work { return Err(err); } }; - let (ig, err) = self.ignore.add_child(self.dent.path()); + // Actually descend into the directory and read its contents + let mut result = ReadDirResult::default(); + for entry in readdir { + match entry { + Ok(entry) => result.entries.push(entry), + Err(err) => result.errors.push( + Error::from(err) + .with_path(self.dent.path()) + .with_depth(self.dent.depth() + 1), + ), + } + } + let (ig, err) = self + .ignore + .add_child_with_entries(self.dent.path(), &result.entries); self.ignore = ig; self.dent.err = err; - Ok(readdir) + Ok(result) } } @@ -1522,11 +1681,11 @@ impl Stack { // breadth-first. We do depth-first because a breadth first traversal // on wide directories with a lot of gitignores is disastrous (for // example, searching a directory tree containing all of crates.io). - let deques: Vec> = std::iter::repeat_with(Deque::new_lifo) - .take(threads) - .collect(); - let stealers = - Arc::<[Stealer]>::from(deques.iter().map(Deque::stealer).collect::>()); + let deques: Vec> = + std::iter::repeat_with(Deque::new_lifo).take(threads).collect(); + let stealers = Arc::<[Stealer]>::from( + deques.iter().map(Deque::stealer).collect::>(), + ); let stacks: Vec = deques .into_iter() .enumerate() @@ -1668,8 +1827,13 @@ impl<'s> Worker<'s> { // have sufficient read permissions to list the directory. // In that case we still want to provide the closure with a valid // entry before passing the error value. - let readdir = work.read_dir(); let depth = work.dent.depth(); + let readdir = if descend && self.max_depth.is_none_or(|m| depth < m) { + Some(work.read_dir()) + } else { + work.add_ignore(); + None + }; if should_visit { let state = self.visitor.visit(Ok(work.dent)); if !state.is_continue() { @@ -1680,6 +1844,10 @@ impl<'s> Worker<'s> { return WalkState::Skip; } + let readdir = match readdir { + Some(readdir) => readdir, + None => return WalkState::Skip, + }; let readdir = match readdir { Ok(readdir) => readdir, Err(err) => { @@ -1687,11 +1855,19 @@ impl<'s> Worker<'s> { } }; - if self.max_depth.map_or(false, |max| depth >= max) { - return WalkState::Skip; + for result in readdir.entries { + let state = self.generate_work( + &work.ignore, + depth + 1, + work.root_device, + result, + ); + if state.is_quit() { + return state; + } } - for result in readdir { - let state = self.generate_work(&work.ignore, depth + 1, work.root_device, result); + for err in readdir.errors { + let state = self.visitor.visit(Err(err)); if state.is_quit() { return state; } @@ -1717,14 +1893,8 @@ impl<'s> Worker<'s> { ig: &Ignore, depth: usize, root_device: Option, - result: Result, + fs_dent: fs::DirEntry, ) -> WalkState { - let fs_dent = match result { - Ok(fs_dent) => fs_dent, - Err(err) => { - return self.visitor.visit(Err(Error::from(err).with_depth(depth))); - } - }; let mut dent = match DirEntryRaw::from_entry(depth, &fs_dent) { Ok(dent) => DirEntry::new_raw(dent, None), Err(err) => { @@ -1760,26 +1930,24 @@ impl<'s> Worker<'s> { return WalkState::Continue; } } - let should_skip_filesize = if self.max_filesize.is_some() && !dent.is_dir() { - skip_filesize( - self.max_filesize.unwrap(), - dent.path(), - &dent.metadata().ok(), - ) - } else { - false - }; - let should_skip_filtered = if let Some(Filter(predicate)) = &self.filter { - !predicate(&dent) - } else { - false - }; + let should_skip_filesize = + if self.max_filesize.is_some() && !dent.is_dir() { + skip_filesize( + self.max_filesize.unwrap(), + dent.path(), + &dent.metadata().ok(), + ) + } else { + false + }; + let should_skip_filtered = + if let Some(Filter(predicate)) = &self.filter { + !predicate(&dent) + } else { + false + }; if !should_skip_filesize && !should_skip_filtered { - self.send(Work { - dent, - ignore: ig.clone(), - root_device, - }); + self.send(Work { dent, ignore: ig.clone(), root_device }); } WalkState::Continue } @@ -1820,6 +1988,9 @@ impl<'s> Worker<'s> { } // Wait for next `Work` or `Quit` message. loop { + if self.is_quit_now() { + return None; + } if let Some(v) = self.recv() { self.activate_worker(); value = Some(v); @@ -1873,24 +2044,25 @@ impl<'s> Worker<'s> { } } +impl<'s> Drop for Worker<'s> { + fn drop(&mut self) { + if std::thread::panicking() { + self.quit_now(); + } + } +} + fn check_symlink_loop( ig_parent: &Ignore, child_path: &Path, child_depth: usize, ) -> Result<(), Error> { let hchild = Handle::from_path(child_path).map_err(|err| { - Error::from(err) - .with_path(child_path) - .with_depth(child_depth) + Error::from(err).with_path(child_path).with_depth(child_depth) })?; - for ig in ig_parent - .parents() - .take_while(|ig| !ig.is_absolute_parent()) - { + for ig in ig_parent.parents().take_while(|ig| !ig.is_absolute_parent()) { let h = Handle::from_path(ig.path()).map_err(|err| { - Error::from(err) - .with_path(child_path) - .with_depth(child_depth) + Error::from(err).with_path(child_path).with_depth(child_depth) })?; if hchild == h { return Err(Error::Loop { @@ -1905,7 +2077,11 @@ fn check_symlink_loop( // Before calling this function, make sure that you ensure that is really // necessary as the arguments imply a file stat. -fn skip_filesize(max_filesize: u64, path: &Path, ent: &Option) -> bool { +fn skip_filesize( + max_filesize: u64, + path: &Path, + ent: &Option, +) -> bool { let filesize = match *ent { Some(ref md) => Some(md.len()), None => None, @@ -1977,9 +2153,9 @@ fn path_equals(dent: &DirEntry, handle: &Handle) -> Result { if dent.is_stdin() || never_equal(dent, handle) { return Ok(false); } - Handle::from_path(dent.path()) - .map(|h| &h == handle) - .map_err(|err| Error::Io(err).with_path(dent.path())) + Handle::from_path(dent.path()).map(|h| &h == handle).map_err(|err| { + Error::Io(err).with_depth(dent.depth()).with_path(dent.path()) + }) } /// Returns true if the given walkdir entry corresponds to a directory. @@ -1997,16 +2173,14 @@ fn walkdir_is_dir(dent: &walkdir::DirEntry) -> bool { if !dent.file_type().is_symlink() || dent.depth() > 0 { return false; } - dent.path() - .metadata() - .ok() - .map_or(false, |md| md.file_type().is_dir()) + dent.path().metadata().ok().map_or(false, |md| md.file_type().is_dir()) } /// Returns true if and only if the given path is on the same device as the /// given root device. fn is_same_file_system(root_device: u64, path: &Path) -> Result { - let dent_device = device_num(path).map_err(|err| Error::Io(err).with_path(path))?; + let dent_device = + device_num(path).map_err(|err| Error::Io(err).with_path(path))?; Ok(root_device == dent_device) } @@ -2065,11 +2239,7 @@ mod tests { } fn normal_path(unix: &str) -> String { - if cfg!(windows) { - unix.replace("\\", "/") - } else { - unix.to_string() - } + if cfg!(windows) { unix.replace("\\", "/") } else { unix.to_string() } } fn walk_collect(prefix: &Path, builder: &WalkBuilder) -> Vec { @@ -2089,7 +2259,10 @@ mod tests { paths } - fn walk_collect_parallel(prefix: &Path, builder: &WalkBuilder) -> Vec { + fn walk_collect_parallel( + prefix: &Path, + builder: &WalkBuilder, + ) -> Vec { let mut paths = vec![]; for dent in walk_collect_entries_parallel(builder) { let path = dent.path().strip_prefix(prefix).unwrap(); @@ -2278,6 +2451,27 @@ mod tests { ); } + #[test] + fn max_depth_does_not_load_unreachable_ignore_files() { + let td = tmpdir(); + let leaf = td.path().join("leaf"); + mkdirp(&leaf); + wfile(leaf.join(".ignore"), "{invalid\n"); + + let mut builder = WalkBuilder::new(td.path()); + builder.max_depth(Some(1)); + let entry = builder + .build() + .find_map(|result| { + let entry = result.unwrap(); + (entry.path() == leaf).then_some(entry) + }) + .unwrap(); + + assert!(entry.error().is_none()); + assert_paths(td.path(), &builder, &["leaf"]); + } + #[test] fn min_depth() { let td = tmpdir(); @@ -2388,7 +2582,9 @@ mod tests { assert_eq!(1, dents.len()); assert!(!dents[0].path_is_symlink()); - let dents = walk_collect_entries_parallel(&WalkBuilder::new(td.path().join("foo"))); + let dents = walk_collect_entries_parallel(&WalkBuilder::new( + td.path().join("foo"), + )); assert_eq!(1, dents.len()); assert!(!dents[0].path_is_symlink()); } @@ -2474,8 +2670,88 @@ mod tests { assert_paths( td.path(), - &WalkBuilder::new(td.path()).filter_entry(|entry| entry.file_name() != OsStr::new("a")), + &WalkBuilder::new(td.path()) + .filter_entry(|entry| entry.file_name() != OsStr::new("a")), &["x", "x/y", "x/y/foo"], ); } + + #[test] + fn empty() { + let td = tmpdir(); + assert_paths(td.path(), &WalkBuilder::empty(), &[]); + + let empty_paths: Vec<&OsStr> = Vec::new(); + assert_paths(td.path(), &WalkBuilder::from_iter(empty_paths), &[]); + } + + #[test] + fn from_iter() { + let td = tmpdir(); + mkdirp(td.path().join("a/b/c")); + mkdirp(td.path().join("d/e/f")); + mkdirp(td.path().join("x/y")); + wfile(td.path().join("a/b/foo"), ""); + wfile(td.path().join("d/e/f/foo"), ""); + wfile(td.path().join("x/y/foo"), ""); + + let paths = vec![ + td.path().join("a"), + td.path().join("d"), + td.path().join("x"), + ]; + + assert_paths( + td.path(), + &WalkBuilder::from_iter(paths), + &[ + "x", + "x/y", + "x/y/foo", + "d", + "d/e", + "d/e/f", + "d/e/f/foo", + "a", + "a/b", + "a/b/foo", + "a/b/c", + ], + ); + } + + // This should always panic and never hang. + // + // Ref: https://github.com/BurntSushi/ripgrep/issues/3009 + #[test] + #[should_panic] + fn panic_in_parallel() { + let td = tmpdir(); + wfile(td.path().join("foo.txt"), ""); + + WalkBuilder::new(td.path()) + .threads(40) + .build_parallel() + .run(|| Box::new(|_| panic!("oops!"))); + } + + // This should always panic and never hang. The first call to the visitor + // builder is used while processing the root paths. Previously, a panic on + // the third call occurred after the first worker had already been spawned, + // leaving it waiting indefinitely for workers that were never created. + #[test] + #[should_panic(expected = "builder panic")] + fn panic_in_parallel_builder() { + let td = tmpdir(); + wfile(td.path().join("foo.txt"), ""); + + let mut builds = 0; + WalkBuilder::new(td.path()).threads(2).build_parallel().run(|| { + builds += 1; + if builds == 3 { + panic!("builder panic"); + } + Box::new(|_| WalkState::Continue) + }); + } } diff --git a/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs b/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs index b7b7c6f95..ecb7b47e3 100644 --- a/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs +++ b/crates/ignore/tests/gitignore_matched_path_or_any_parents_tests.rs @@ -2,7 +2,8 @@ use std::path::Path; use ignore::gitignore::{Gitignore, GitignoreBuilder}; -const IGNORE_FILE: &'static str = "tests/gitignore_matched_path_or_any_parents_tests.gitignore"; +const IGNORE_FILE: &'static str = + "tests/gitignore_matched_path_or_any_parents_tests.gitignore"; fn get_gitignore() -> Gitignore { let mut builder = GitignoreBuilder::new("ROOT"); @@ -23,7 +24,9 @@ fn test_path_should_be_under_root() { #[test] fn test_files_in_root() { let gitignore = get_gitignore(); - let m = |path: &str| gitignore.matched_path_or_any_parents(Path::new(path), false); + let m = |path: &str| { + gitignore.matched_path_or_any_parents(Path::new(path), false) + }; // 0x assert!(m("ROOT/file_root_00").is_ignore()); @@ -53,7 +56,9 @@ fn test_files_in_root() { #[test] fn test_files_in_deep() { let gitignore = get_gitignore(); - let m = |path: &str| gitignore.matched_path_or_any_parents(Path::new(path), false); + let m = |path: &str| { + gitignore.matched_path_or_any_parents(Path::new(path), false) + }; // 0x assert!(m("ROOT/parent_dir/file_deep_00").is_ignore()); @@ -83,8 +88,9 @@ fn test_files_in_deep() { #[test] fn test_dirs_in_root() { let gitignore = get_gitignore(); - let m = - |path: &str, is_dir: bool| gitignore.matched_path_or_any_parents(Path::new(path), is_dir); + let m = |path: &str, is_dir: bool| { + gitignore.matched_path_or_any_parents(Path::new(path), is_dir) + }; // 00 assert!(m("ROOT/dir_root_00", true).is_ignore()); @@ -186,20 +192,25 @@ fn test_dirs_in_root() { #[test] fn test_dirs_in_deep() { let gitignore = get_gitignore(); - let m = - |path: &str, is_dir: bool| gitignore.matched_path_or_any_parents(Path::new(path), is_dir); + let m = |path: &str, is_dir: bool| { + gitignore.matched_path_or_any_parents(Path::new(path), is_dir) + }; // 00 assert!(m("ROOT/parent_dir/dir_deep_00", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_00/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_00/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_00/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_00/child_dir/file", false).is_ignore() + ); // 01 assert!(m("ROOT/parent_dir/dir_deep_01", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_01/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_01/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_01/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_01/child_dir/file", false).is_ignore() + ); // 02 assert!(m("ROOT/parent_dir/dir_deep_02", true).is_none()); @@ -241,51 +252,67 @@ fn test_dirs_in_deep() { assert!(m("ROOT/parent_dir/dir_deep_20", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_20/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_20/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_20/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_20/child_dir/file", false).is_ignore() + ); // 21 assert!(m("ROOT/parent_dir/dir_deep_21", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_21/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_21/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_21/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_21/child_dir/file", false).is_ignore() + ); // 22 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_22", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_22/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_22/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_22/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_22/child_dir/file", false).is_ignore() + ); // 23 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_23", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_23/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_23/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_23/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_23/child_dir/file", false).is_ignore() + ); // 30 assert!(m("ROOT/parent_dir/dir_deep_30", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_30/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_30/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_30/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_30/child_dir/file", false).is_ignore() + ); // 31 assert!(m("ROOT/parent_dir/dir_deep_31", true).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_31/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_31/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_31/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_31/child_dir/file", false).is_ignore() + ); // 32 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_32", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_32/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_32/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_32/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_32/child_dir/file", false).is_ignore() + ); // 33 // dir itself doesn't match assert!(m("ROOT/parent_dir/dir_deep_33", true).is_none()); assert!(m("ROOT/parent_dir/dir_deep_33/file", false).is_ignore()); assert!(m("ROOT/parent_dir/dir_deep_33/child_dir", true).is_ignore()); - assert!(m("ROOT/parent_dir/dir_deep_33/child_dir/file", false).is_ignore()); + assert!( + m("ROOT/parent_dir/dir_deep_33/child_dir/file", false).is_ignore() + ); } diff --git a/crates/node/.gitignore b/crates/node/.gitignore index d2200a62e..c3f2ec696 100644 --- a/crates/node/.gitignore +++ b/crates/node/.gitignore @@ -202,5 +202,6 @@ index.js browser.js tailwindcss-oxide.wasi-browser.js tailwindcss-oxide.wasi.cjs +tailwindcss-oxide.wasi.d.cts wasi-worker-browser.mjs wasi-worker.mjs diff --git a/crates/node/Cargo.toml b/crates/node/Cargo.toml index 610de5ccd..a3df7b813 100644 --- a/crates/node/Cargo.toml +++ b/crates/node/Cargo.toml @@ -8,10 +8,10 @@ crate-type = ["cdylib"] [dependencies] # Default enable napi4 feature, see https://nodejs.org/api/n-api.html#node-api-version-matrix -napi = { version = "3.8.5", default-features = false, features = ["napi4"] } -napi-derive = "3.5.4" +napi = { version = "3.11.0", default-features = false, features = ["napi4"] } +napi-derive = "3.6.0" tailwindcss-oxide = { path = "../oxide" } -rayon = "1.10.0" +rayon = "1.12.0" [build-dependencies] -napi-build = "2.3.1" +napi-build = "2.3.2" diff --git a/crates/node/npm/android-arm-eabi/package.json b/crates/node/npm/android-arm-eabi/package.json index 6e2b40ee5..d46ceb237 100644 --- a/crates/node/npm/android-arm-eabi/package.json +++ b/crates/node/npm/android-arm-eabi/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-android-arm-eabi", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/android-arm64/package.json b/crates/node/npm/android-arm64/package.json index 95d84f4fe..e6e9d577d 100644 --- a/crates/node/npm/android-arm64/package.json +++ b/crates/node/npm/android-arm64/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-android-arm64", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/darwin-arm64/package.json b/crates/node/npm/darwin-arm64/package.json index 382d9e6ba..1cb737c2e 100644 --- a/crates/node/npm/darwin-arm64/package.json +++ b/crates/node/npm/darwin-arm64/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-darwin-arm64", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/darwin-x64/package.json b/crates/node/npm/darwin-x64/package.json index 15851163e..ffd75e54b 100644 --- a/crates/node/npm/darwin-x64/package.json +++ b/crates/node/npm/darwin-x64/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-darwin-x64", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/freebsd-x64/package.json b/crates/node/npm/freebsd-x64/package.json index fb890a730..e248f39d1 100644 --- a/crates/node/npm/freebsd-x64/package.json +++ b/crates/node/npm/freebsd-x64/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-freebsd-x64", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/linux-arm-gnueabihf/package.json b/crates/node/npm/linux-arm-gnueabihf/package.json index 21bbaee93..50d2aef73 100644 --- a/crates/node/npm/linux-arm-gnueabihf/package.json +++ b/crates/node/npm/linux-arm-gnueabihf/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-linux-arm-gnueabihf", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/linux-arm64-gnu/package.json b/crates/node/npm/linux-arm64-gnu/package.json index 5fd91c422..5875213f8 100644 --- a/crates/node/npm/linux-arm64-gnu/package.json +++ b/crates/node/npm/linux-arm64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-linux-arm64-gnu", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/linux-arm64-musl/package.json b/crates/node/npm/linux-arm64-musl/package.json index 18ca072c4..9da37f3fd 100644 --- a/crates/node/npm/linux-arm64-musl/package.json +++ b/crates/node/npm/linux-arm64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-linux-arm64-musl", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/linux-x64-gnu/package.json b/crates/node/npm/linux-x64-gnu/package.json index 8cfc4d852..d409743fa 100644 --- a/crates/node/npm/linux-x64-gnu/package.json +++ b/crates/node/npm/linux-x64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-linux-x64-gnu", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/linux-x64-musl/package.json b/crates/node/npm/linux-x64-musl/package.json index fcf648f3f..5b404fd70 100644 --- a/crates/node/npm/linux-x64-musl/package.json +++ b/crates/node/npm/linux-x64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-linux-x64-musl", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/wasm32-wasi/README.md b/crates/node/npm/wasm32-wasi/README.md index 9ec3a7b3f..2a5368149 100644 --- a/crates/node/npm/wasm32-wasi/README.md +++ b/crates/node/npm/wasm32-wasi/README.md @@ -1,3 +1,3 @@ # `@tailwindcss/oxide-wasm32-wasi` -This is the **wasm32-wasip1-threads** binary for `@tailwindcss/oxide` +This is the **wasm32-wasip1-threads** build of `@tailwindcss/oxide` diff --git a/crates/node/npm/wasm32-wasi/package.json b/crates/node/npm/wasm32-wasi/package.json index c38562783..cee9fb05a 100644 --- a/crates/node/npm/wasm32-wasi/package.json +++ b/crates/node/npm/wasm32-wasi/package.json @@ -1,9 +1,6 @@ { "name": "@tailwindcss/oxide-wasm32-wasi", - "version": "4.3.2", - "cpu": [ - "wasm32" - ], + "version": "4.3.3", "main": "tailwindcss-oxide.wasi.cjs", "files": [ "tailwindcss-oxide.wasm32-wasi.wasm", @@ -14,7 +11,7 @@ ], "license": "MIT", "engines": { - "node": ">=14.0.0" + "node": "^20.19.0 || ^22.13.0 || >=23.5.0" }, "publishConfig": { "provenance": true, @@ -27,11 +24,11 @@ }, "browser": "tailwindcss-oxide.wasi-browser.js", "dependencies": { - "@napi-rs/wasm-runtime": "^1.1.4", - "@emnapi/core": "^1.11.1", - "@emnapi/runtime": "^1.11.1", - "@tybys/wasm-util": "^0.10.2", - "@emnapi/wasi-threads": "^1.2.2", + "@napi-rs/wasm-runtime": "^1.2.2", + "@emnapi/core": "^1.11.3", + "@emnapi/runtime": "^1.11.3", + "@tybys/wasm-util": "^0.10.3", + "@emnapi/wasi-threads": "^1.2.3", "tslib": "^2.8.1" }, "bundledDependencies": [ diff --git a/crates/node/npm/win32-arm64-msvc/package.json b/crates/node/npm/win32-arm64-msvc/package.json index a994d8d9d..991727a32 100644 --- a/crates/node/npm/win32-arm64-msvc/package.json +++ b/crates/node/npm/win32-arm64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-win32-arm64-msvc", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/npm/win32-x64-msvc/package.json b/crates/node/npm/win32-x64-msvc/package.json index 2c6bfedb0..82d75c300 100644 --- a/crates/node/npm/win32-x64-msvc/package.json +++ b/crates/node/npm/win32-x64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide-win32-x64-msvc", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", diff --git a/crates/node/package.json b/crates/node/package.json index 40ab70649..be09427fb 100644 --- a/crates/node/package.json +++ b/crates/node/package.json @@ -1,6 +1,6 @@ { "name": "@tailwindcss/oxide", - "version": "4.3.2", + "version": "4.3.3", "repository": { "type": "git", "url": "git+https://github.com/tailwindlabs/tailwindcss.git", @@ -33,9 +33,11 @@ }, "license": "MIT", "devDependencies": { - "@napi-rs/cli": "3.7.0", - "@napi-rs/wasm-runtime": "^1.1.5", - "emnapi": "1.11.1" + "@emnapi/core": "1.11.3", + "@emnapi/runtime": "1.11.3", + "@napi-rs/cli": "3.7.4", + "@napi-rs/wasm-runtime": "^1.2.2", + "emnapi": "1.11.3" }, "engines": { "node": ">= 20" diff --git a/crates/node/src/lib.rs b/crates/node/src/lib.rs index 11ff25b0a..fb32a818c 100644 --- a/crates/node/src/lib.rs +++ b/crates/node/src/lib.rs @@ -165,6 +165,11 @@ impl Scanner { self.scanner.get_files() } + #[napi(getter)] + pub fn scanned_files(&self) -> Vec { + self.scanner.get_scanned_files() + } + #[napi(getter)] pub fn globs(&mut self) -> Vec { self diff --git a/crates/oxide/Cargo.toml b/crates/oxide/Cargo.toml index 0965320c9..f8a5f6137 100644 --- a/crates/oxide/Cargo.toml +++ b/crates/oxide/Cargo.toml @@ -20,6 +20,7 @@ ignore = { path = "../ignore" } regex = "1.11.1" [dev-dependencies] +insta = "1.48.0" tempfile = "3.13.0" pretty_assertions = "1.4.1" unicode-width = "0.2.0" diff --git a/crates/oxide/src/extractor/bracket_stack.rs b/crates/oxide/src/extractor/bracket_stack.rs index f34e56ed3..2b8d3a13f 100644 --- a/crates/oxide/src/extractor/bracket_stack.rs +++ b/crates/oxide/src/extractor/bracket_stack.rs @@ -25,6 +25,7 @@ impl BracketStack { b'(' => b')', b'[' => b']', b'{' => b'}', + b'<' => b'>', _ => std::hint::unreachable_unchecked(), }; } diff --git a/crates/oxide/src/extractor/pre_processors/haml.rs b/crates/oxide/src/extractor/pre_processors/haml.rs index 99b7bd052..ec8be0817 100644 --- a/crates/oxide/src/extractor/pre_processors/haml.rs +++ b/crates/oxide/src/extractor/pre_processors/haml.rs @@ -203,6 +203,81 @@ impl PreProcessor for Haml { } } + // Handle Ruby syntax with `%w[]` arrays embedded in Haml attribute hashes. E.g.: + // + // ```haml + // %div{class: %w[bg-blue-500 w-10 h-10]} + // ``` + // + // A `%` that follows a value is not a percent literal. E.g.: the `50%w` in + // `hit rate 50%w.` + b'%' if matches!(cursor.next(), b'w' | b'W') + && !cursor.prev().is_ascii_alphanumeric() + && !matches!(cursor.prev(), b'_' | b')' | b']' | b'}') => + { + // Boundary characters + let (open, close) = match cursor.input.get(cursor.pos + 2) { + Some(b'[') => (b'[', b']'), + Some(b'(') => (b'(', b')'), + Some(b'{') => (b'{', b'}'), + Some(b'<') => (b'<', b'>'), + + // Any other ASCII punctuation can be used as a custom delimiter + Some(&c) if c.is_ascii_punctuation() => (c, c), + + // Everything else is not a valid delimiter + _ => { + cursor.advance(); + continue; + } + }; + + result[cursor.pos] = b' '; // Replace `%` + cursor.advance(); + result[cursor.pos] = b' '; // Replace `w` + cursor.advance(); + result[cursor.pos] = b' '; // Replace the opening delimiter + cursor.advance(); + + // Paired delimiters can be nested as long as they are balanced. E.g.: + // `%w[foo[bar]baz]` produces a flat array. + let mut depth = 1_usize; + + while cursor.pos < len { + match cursor.curr() { + // Skip escaped characters, unless the backslash is the delimiter + // itself + b'\\' if close != b'\\' => { + // Use backslash to embed spaces in the strings. + if cursor.next() == b' ' { + result[cursor.pos] = b' '; + } + + cursor.advance(); + } + + // Start of a nested delimiter pair + c if c == open && open != close => depth += 1, + + // Closing delimiter + c if c == close => { + depth -= 1; + + // End of the literal, replace the closing delimiter with a space + if depth == 0 { + result[cursor.pos] = b' '; + break; + } + } + + // Everything else is valid content + _ => {} + } + + cursor.advance(); + } + } + // Replace following characters with spaces if they are not inside of brackets b'#' | b'=' if bracket_stack.is_empty() => { result[cursor.pos] = b' '; @@ -424,6 +499,81 @@ mod tests { ); } + // https://github.com/tailwindlabs/tailwindcss/issues/20386 + #[test] + fn test_embedded_ruby_percent_w_delimiters() { + for (input, expected) in [ + // %w[…] in an attribute hash + ( + "%div{class: %w[flex px-2.5]}", + "%div class: flex px-2.5 ", + ), + // %w<…> + ( + "%div{class: %w}", + "%div class: flex px-2.5 ", + ), + // Nested `<…>` does not end the literal + ( + "%div{class: %w px-2.5>}", + "%div class: flex px-2.5 ", + ), + // Custom delimiters + ( + "%div{class: %w|flex px-2.5|}", + "%div class: flex px-2.5 ", + ), + ( + "%div{class: %W!flex px-2.5!}", + "%div class: flex px-2.5 ", + ), + ( + "%div{class: %w#text-sm leading-6#}", + "%div class: text-sm leading-6 ", + ), + ( + "%div{class: %w=italic tracking-wide=}", + "%div class: italic tracking-wide ", + ), + // Nested paired delimiters stay balanced inside the literal + ( + "%div{class: %w[content-['[hello]'] p-4]}", + "%div class: content-['[hello]'] p-4 ", + ), + // Escaped spaces embed a space in a single array element + ( + r#"%div{class: %w[foo\ bar baz-1]}"#, + r#"%div class: foo bar baz-1 "#, + ), + // A `%` that follows a value is not a percent literal + ("%p hit rate 50%w.", "%p hit rate 50%w "), + ] { + Haml::test(input, expected); + } + + let input = r#" + %div{class: %w[bg-blue-500 w-10 h-10]} + %div{class: %w} + %div{class: %w|underline font-bold|} + - classes = %w + "#; + + Haml::test_extract_contains( + input, + vec![ + "bg-blue-500", + "w-10", + "h-10", + "flex", + "px-2.5", + "underline", + "font-bold", + "mt-4", + "grid", + ], + ); + } + // https://github.com/tailwindlabs/tailwindcss/pull/17051#issuecomment-2711181352 #[test] fn test_haml_full_file_17051() { diff --git a/crates/oxide/src/extractor/pre_processors/ruby.rs b/crates/oxide/src/extractor/pre_processors/ruby.rs index 4345f2a1e..d9a57ab77 100644 --- a/crates/oxide/src/extractor/pre_processors/ruby.rs +++ b/crates/oxide/src/extractor/pre_processors/ruby.rs @@ -158,6 +158,15 @@ impl PreProcessor for Ruby { continue; } + // A `%` that follows a value is a modulo operation, not a percent literal. E.g.: the + // `50%w` in `hit rate 50%w.` + if cursor.prev().is_ascii_alphanumeric() + || matches!(cursor.prev(), b'_' | b')' | b']' | b'}') + { + cursor.advance(); + continue; + } + cursor.advance_twice(); // Boundary character @@ -165,8 +174,13 @@ impl PreProcessor for Ruby { b'[' => b']', b'(' => b')', b'{' => b'}', - b'#' => b'#', + b'<' => b'>', b' ' => b'\n', + + // Any other ASCII punctuation can be used as a custom delimiter + c if c.is_ascii_punctuation() => c, + + // Everything else is not a valid delimiter _ => { cursor.advance(); continue; @@ -183,8 +197,8 @@ impl PreProcessor for Ruby { while cursor.pos < len { match cursor.curr() { - // Skip escaped characters - b'\\' => { + // Skip escaped characters, unless the backslash is the delimiter itself + b'\\' if boundary != b'\\' => { // Use backslash to embed spaces in the strings. if cursor.next() == b' ' { result[cursor.pos] = b' '; @@ -198,6 +212,11 @@ impl PreProcessor for Ruby { bracket_stack.push(cursor.curr()); } + // Start of a nested `<…>`, which Ruby allows inside a `%w<…>` literal + b'<' if boundary == b'>' => { + bracket_stack.push(cursor.curr()); + } + // End of a nested bracket b']' | b')' | b'}' if !bracket_stack.is_empty() => { if !bracket_stack.pop(cursor.curr()) { @@ -206,6 +225,14 @@ impl PreProcessor for Ruby { } } + // End of a nested `<…>` + b'>' if boundary == b'>' && !bracket_stack.is_empty() => { + if !bracket_stack.pop(cursor.curr()) { + // Unbalanced + cursor.advance(); + } + } + // End of the pattern, replace the boundary character with a space _ if cursor.curr() == boundary => { if boundary != b'\n' { @@ -253,6 +280,24 @@ mod tests { "%w(flex data-[state=pending]:bg-(--my-color) flex-col)", "%w flex data-[state=pending]:bg-(--my-color) flex-col ", ), + // %w<…> + ("%w", "%w flex px-2.5 "), + ( + "%w", + "%w flex data-[state=pending]:bg-(--my-color) flex-col ", + ), + // Nested `<…>` does not end the literal + ("%w px-2.5>", "%w flex px-2.5 "), + // %w|…|, %w:…:, %w!…! + ("%w|flex px-2.5|", "%w flex px-2.5 "), + ("%w:flex px-2.5:", "%w flex px-2.5 "), + ("%w!flex px-2.5!", "%w flex px-2.5 "), + (r#"%w\flex px-2.5\"#, r#"%w flex px-2.5 "#), + // A `%` that follows a value is a modulo operation, not a percent literal + ( + "hit rate 50%w.\n%w[flex px-2.5]", + "hit rate 50%w.\n%w flex px-2.5 ", + ), // %w …\n ("%w flex px-2.5\n", "%w flex px-2.5\n"), @@ -330,6 +375,20 @@ mod tests { "%w(flex data-[state=pending]:bg-(--my-color) flex-col)", vec!["flex", "data-[state=pending]:bg-(--my-color)", "flex-col"], ), + // %w<…> + ("%w", vec!["flex", "px-2.5"]), + ("%w", vec!["flex", "px-2.5"]), + ("%w<2xl:flex>", vec!["2xl:flex"]), + ( + "%w", + vec!["flex", "data-[state=pending]:bg-(--my-color)", "flex-col"], + ), + // Nested `<…>` does not end the literal + ("%w px-2.5>", vec!["flex", "px-2.5"]), + // %w|…|, %w:…:, %w!…! + ("%w|flex px-2.5|", vec!["flex", "px-2.5"]), + ("%w:flex px-2.5:", vec!["flex", "px-2.5"]), + ("%w!flex px-2.5!", vec!["flex", "px-2.5"]), ( "# test\n# test\n# {ActiveRecord::Base#save!}[rdoc-ref:Persistence#save!]\n%w[flex px-2.5]", diff --git a/crates/oxide/src/extractor/pre_processors/slim.rs b/crates/oxide/src/extractor/pre_processors/slim.rs index eacfb55ee..5683858d5 100644 --- a/crates/oxide/src/extractor/pre_processors/slim.rs +++ b/crates/oxide/src/extractor/pre_processors/slim.rs @@ -65,17 +65,74 @@ impl PreProcessor for Slim { // class=%w[bg-blue-500 w-10 h-10] // ] // ``` + // + // A `%` that follows a value is not a percent literal. E.g.: the `50%w` in + // `hit rate 50%w.` b'%' if matches!(cursor.next(), b'w' | b'W') - && matches!(cursor.input.get(cursor.pos + 2), Some(b'[' | b'(' | b'{')) => + && !cursor.prev().is_ascii_alphanumeric() + && !matches!(cursor.prev(), b'_' | b')' | b']' | b'}') => { + // Boundary characters + let (open, close) = match cursor.input.get(cursor.pos + 2) { + Some(b'[') => (b'[', b']'), + Some(b'(') => (b'(', b')'), + Some(b'{') => (b'{', b'}'), + Some(b'<') => (b'<', b'>'), + + // Any other ASCII punctuation can be used as a custom delimiter + Some(&c) if c.is_ascii_punctuation() => (c, c), + + // Everything else is not a valid delimiter + _ => { + cursor.advance(); + continue; + } + }; + result[cursor.pos] = b' '; // Replace `%` cursor.advance(); result[cursor.pos] = b' '; // Replace `w` cursor.advance(); - result[cursor.pos] = b' '; // Replace `[` or `(` or `{` - bracket_stack.push(cursor.curr()); - cursor.advance(); // Move past the bracket - continue; + result[cursor.pos] = b' '; // Replace the opening delimiter + cursor.advance(); + + // Paired delimiters can be nested as long as they are balanced. E.g.: + // `%w[foo[bar]baz]` produces a flat array. + let mut depth = 1_usize; + + while cursor.pos < len { + match cursor.curr() { + // Skip escaped characters, unless the backslash is the delimiter + // itself + b'\\' if close != b'\\' => { + // Use backslash to embed spaces in the strings. + if cursor.next() == b' ' { + result[cursor.pos] = b' '; + } + + cursor.advance(); + } + + // Start of a nested delimiter pair + c if c == open && open != close => depth += 1, + + // Closing delimiter + c if c == close => { + depth -= 1; + + // End of the literal, replace the closing delimiter with a space + if depth == 0 { + result[cursor.pos] = b' '; + break; + } + } + + // Everything else is valid content + _ => {} + } + + cursor.advance(); + } } // Any `[` preceded by an alphanumeric value will not be part of a candidate. @@ -316,16 +373,77 @@ mod tests { ] "#; - let expected = r#" - div - class= bg-blue-500 w-10 h-10] - ] - div - class= w-10 bg-green-500 h-10] - ] - "#; + let expected = " + div \n class= bg-blue-500 w-10 h-10 \n ] + div \n class= w-10 bg-green-500 h-10 \n ] + "; Slim::test(input, expected); Slim::test_extract_contains(input, vec!["bg-blue-500", "bg-green-500", "w-10", "h-10"]); } + + // https://github.com/tailwindlabs/tailwindcss/issues/20386 + #[test] + fn test_embedded_ruby_percent_w_delimiters() { + for (input, expected) in [ + // %w<…>, Slim only counts `[({` nesting in attribute values, so the code must be + // wrapped in parentheses to contain spaces + ( + "div class=(%w)", + "div class= bg-blue-500 w-10 h-10 )", + ), + // Nested `<…>` does not end the literal + ( + "div class=(%w px-2.5>)", + "div class= flex px-2.5 )", + ), + // Custom delimiters + ("div class=(%w|flex px-2.5|)", "div class= flex px-2.5 )"), + ("div class=(%W!flex px-2.5!)", "div class= flex px-2.5 )"), + ( + "div class=(%w#text-sm leading-6#)", + "div class= text-sm leading-6 )", + ), + ( + "div class=(%w=italic tracking-wide=)", + "div class= italic tracking-wide )", + ), + // Nested paired delimiters stay balanced inside the literal + ( + "div class=(%w[content-['[hello]'] p-4])", + "div class= content-['[hello]'] p-4 )", + ), + // Escaped spaces embed a space in a single array element + ( + r#"div class=(%w[foo\ bar baz-1])"#, + r#"div class= foo bar baz-1 )"#, + ), + // Ruby control line, which is plain Ruby code + ("- classes = %w", "- classes = mt-4 flex "), + // A `%` that follows a value is not a percent literal + ("| hit rate 50%w.", "| hit rate 50%w "), + ] { + Slim::test(input, expected); + } + + let input = r#" + div[ + class=(%w) + ] + - classes = %w|w-10 bg-green-500 h-10| + = tag.div class: %W!px-2.5 flex! + "#; + + Slim::test_extract_contains( + input, + vec![ + "bg-blue-500", + "bg-green-500", + "w-10", + "h-10", + "px-2.5", + "flex", + ], + ); + } } diff --git a/crates/oxide/src/scanner/detect_sources.rs b/crates/oxide/src/scanner/detect_sources.rs index c519cd00d..6cfdd49cc 100644 --- a/crates/oxide/src/scanner/detect_sources.rs +++ b/crates/oxide/src/scanner/detect_sources.rs @@ -112,6 +112,7 @@ pub fn resolve_globs( } if !dirs.contains(path) { + it.skip_current_dir(); continue; } diff --git a/crates/oxide/src/scanner/init_tracing.rs b/crates/oxide/src/scanner/init_tracing.rs index 2a570ff20..dd4c1d071 100644 --- a/crates/oxide/src/scanner/init_tracing.rs +++ b/crates/oxide/src/scanner/init_tracing.rs @@ -37,20 +37,52 @@ pub fn init_tracing() { return; } - let file_path = format!("tailwindcss-{}.log", std::process::id()); - let file = OpenOptions::new() + let root = Path::new(".tailwindcss"); + let logs_dir = root.join("logs"); + if let Err(err) = std::fs::create_dir_all(&logs_dir) { + eprintln!( + "{} Failed to create {}, skipping debug logs ({err})", + dim("[DEBUG]"), + highlight(&logs_dir.display().to_string()) + ); + return; + } + + // Ensure everything inside `.tailwindcss/` is ignored by git. The file is only created if it + // doesn't exist yet, an existing `.gitignore` is left untouched. + if let Ok(mut file) = std::fs::File::create_new(root.join(".gitignore")) { + _ = file.write_all(b"*\n"); + } + + let file_path = logs_dir.join(format!( + "scanner-{}-{}.log", + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis()) + .unwrap_or(0), + std::process::id() + )); + let file = match OpenOptions::new() .create(true) .append(true) .open(&file_path) - .unwrap_or_else(|_| panic!("Failed to open {file_path}")); + { + Ok(file) => file, + Err(err) => { + eprintln!( + "{} Failed to create {}, skipping debug logs ({err})", + dim("[DEBUG]"), + highlight(&file_path.display().to_string()) + ); + return; + } + }; - let file_path = Path::new(&file_path); - let absolute_file_path = dunce::canonicalize(file_path) - .unwrap_or_else(|_| panic!("Failed to canonicalize {file_path:?}")); + let absolute_file_path = dunce::canonicalize(&file_path).unwrap_or_else(|_| file_path.clone()); eprintln!( "{} Writing debug info to: {}\n", dim("[DEBUG]"), - highlight(absolute_file_path.as_path().to_str().unwrap()) + highlight(&absolute_file_path.display().to_string()) ); let file = Arc::new(Mutex::new(file)); diff --git a/crates/oxide/src/scanner/mod.rs b/crates/oxide/src/scanner/mod.rs index 02830555d..b7da9aa17 100644 --- a/crates/oxide/src/scanner/mod.rs +++ b/crates/oxide/src/scanner/mod.rs @@ -78,11 +78,12 @@ pub struct Scanner { /// Track unique set of candidates candidates: FxHashSet, - /// Track mtimes for files so re-scans can skip unchanged files. - /// Only populated after the first scan completes (to avoid unnecessary - /// metadata calls on initial build). + /// Track mtimes for files so incremental scans can skip unchanged files. mtimes: FxHashMap, + /// Files that were scanned during the last `scan()` call. + scanned_files: Vec, + /// Whether we've completed at least one full scan. When false, we skip /// mtime tracking entirely so the initial build stays fast. has_scanned_once: bool, @@ -122,9 +123,10 @@ impl Scanner { pub fn scan(&mut self) -> Vec { self.sources_scanned = false; - let (scanned_blobs, css_files) = self.discover_sources(); + let (scanned_blobs, css_files, files) = self.discover_sources(); self.extract_candidates(scanned_blobs, css_files); + self.scanned_files = files; // Return all candidates sorted let mut result = self.candidates.iter().cloned().collect::>(); @@ -179,6 +181,11 @@ impl Scanner { continue; } + // The walked path can contain symlinks, while the changed files have already + // been canonicalized. Lazily canonicalize the walked path so we can compare + // the real paths as well. + let mut canonical_path: Option = None; + let mut drop_file_indexes = vec![]; for (idx, changed_file) in new_unknown_files.iter().enumerate().rev() { let ChangedContent::File(file, _) = changed_file else { @@ -187,7 +194,15 @@ impl Scanner { // When the file is found on disk it means that all the rules pass. We can // extract the current file and remove it from the list of passed in files. - if file == path { + let matches = file == path + || (file.file_name() == path.file_name() && { + if canonical_path.is_none() { + canonical_path = dunce::canonicalize(path).ok(); + } + canonical_path.as_deref() == Some(file.as_path()) + }); + + if matches { self.files.insert(path.to_path_buf()); // Track for future use content_to_scan.push(changed_file.clone()); // Track for parsing drop_file_indexes.push(idx); @@ -256,6 +271,11 @@ impl Scanner { .collect() } + #[tracing::instrument(skip_all)] + pub fn get_scanned_files(&self) -> Vec { + self.scanned_files.clone() + } + #[tracing::instrument(skip_all)] pub fn get_globs(&mut self) -> Vec { if let Some(globs) = &self.globs { @@ -336,15 +356,14 @@ impl Scanner { let i = s.as_ptr() as usize - offset; let original = &original_content[i..i + s.len()]; if original.contains_str("-[]") { - return Some(unsafe { - (String::from_utf8_unchecked(original.to_vec()), i) - }); + return String::from_utf8(original.to_vec()) + .ok() + .map(|candidate| (candidate, i)); } - // SAFETY: When we parsed the candidates, we already guaranteed that the byte - // slices are valid, therefore we don't have to re-check here when we want to - // convert it back to a string. - Some(unsafe { (String::from_utf8_unchecked(s.to_vec()), i) }) + String::from_utf8(s.to_vec()) + .ok() + .map(|candidate| (candidate, i)) } _ => None, @@ -353,14 +372,14 @@ impl Scanner { } #[tracing::instrument(skip_all)] - fn discover_sources(&mut self) -> (Vec>, Vec) { + fn discover_sources(&mut self) -> (Vec>, Vec, Vec) { if self.sources_scanned { - return (vec![], vec![]); + return (vec![], vec![], vec![]); } self.sources_scanned = true; let Some(walker) = &mut self.walker else { - return (vec![], vec![]); + return (vec![], vec![], vec![]); }; // Use synchronous walk for the initial build (lower overhead) and parallel @@ -372,7 +391,8 @@ impl Scanner { }; let mut css_files: Vec = vec![]; - let mut content_paths: Vec<(PathBuf, String)> = Vec::new(); + let mut content_paths: Vec<(PathBuf, String)> = vec![]; + let mut changed_files = vec![]; // Fresh state self.files.clear(); @@ -380,42 +400,93 @@ impl Scanner { self.extensions.clear(); self.globs = None; - for (path, is_dir, extension) in all_entries { - if is_dir { - self.dirs.insert(path); - } else { - // Deduplicate: parallel walk can visit the same file from multiple threads - if !self.files.insert(path.clone()) { - continue; + // Cache canonicalized folders in case a file itself is not symlinked, but any of the parent + // folders are symlinked. + let mut cached_canonical_dirs: FxHashMap = FxHashMap::default(); + + for entry in all_entries { + match entry { + WalkEntry::Dir(path) => { + self.dirs.insert(path); } - self.extensions.insert(extension.clone()); - - // On re-scans, check mtime to skip unchanged files. - // On the first scan we skip this entirely to avoid extra - // metadata syscalls. - let changed = if self.has_scanned_once { - let current_mtime = path.metadata().ok().and_then(|m| m.modified().ok()); - - match current_mtime { - Some(mtime) => { - let prev = self.mtimes.insert(path.clone(), mtime); - prev.is_none_or(|prev| prev != mtime) - } - None => true, + WalkEntry::File { + path, + mtime, + is_symlink, + } => { + // Deduplicate: parallel walk can visit the same file from multiple threads + if !self.files.insert(path.clone()) { + continue; } - } else { - true - }; - if !changed { - continue; - } + // Track canonicalized paths in addition to potentially symlinked file paths + let canonical = if is_symlink { + dunce::canonicalize(&path).ok() + } else { + path.parent().and_then(|parent| { + // Perf: cache the canonicalized parent path such that sibling files don't + // have to canonicalize over and over again. + let canonical_parent = cached_canonical_dirs + .entry(parent.to_path_buf()) + .or_insert_with(|| { + dunce::canonicalize(parent) + .unwrap_or_else(|_| parent.to_path_buf()) + }); - match extension.as_str() { - // Special handing for CSS files, we don't want to extract candidates from - // these files, but we do want to extract used CSS variables. - "css" => css_files.push(path), - _ => content_paths.push((path, extension)), + if canonical_parent.as_path() != parent { + path.file_name() + .map(|file_name| canonical_parent.join(file_name)) + } else { + None + } + }) + }; + + if let Some(canonical) = canonical { + if canonical != path { + self.files.insert(canonical); + } + } + let extension = path + .extension() + .and_then(|x| x.to_str()) + .unwrap_or_default() + .to_owned(); + + self.extensions.insert(extension.to_owned()); + + // On incremental scans, check mtime to skip unchanged files. + // On the first scan, track mtimes while still scanning every file. + let changed = if self.has_scanned_once { + match mtime { + Some(mtime) => { + let prev = self.mtimes.insert(path.clone(), mtime); + prev.is_none_or(|prev| prev != mtime) + } + None => true, + } + } else { + if let Some(mtime) = mtime { + self.mtimes.insert(path.clone(), mtime); + } + + true + }; + + if !changed { + continue; + } + + if let Ok(file) = path.clone().into_os_string().into_string() { + changed_files.push(file); + } + + match extension.as_str() { + // Special handing for CSS files, we don't want to extract candidates from + // these files, but we do want to extract used CSS variables. + "css" => css_files.push(path), + _ => content_paths.push((path, extension)), + } } } } @@ -442,7 +513,9 @@ impl Scanner { self.has_scanned_once = true; } - (scanned_blobs, css_files) + changed_files.par_sort_unstable(); + + (scanned_blobs, css_files, changed_files) } } @@ -543,35 +616,50 @@ where a }) .into_iter() - .map(|s| unsafe { String::from_utf8_unchecked(s.to_vec()) }) + .filter_map(|s| String::from_utf8(s.to_vec()).ok()) .collect() } -type WalkEntry = (PathBuf, bool, String); +#[derive(Debug)] +enum WalkEntry { + Dir(PathBuf), + File { + path: PathBuf, + mtime: Option, + + /// Whether the path itself is a symlink + is_symlink: bool, + }, +} + +impl From for WalkEntry { + fn from(entry: ignore::DirEntry) -> Self { + let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); + let is_symlink = entry.path_is_symlink(); + let path = entry.into_path(); + + if is_dir { + WalkEntry::Dir(path) + } else { + let mtime = path.metadata().ok().and_then(|m| m.modified().ok()); + WalkEntry::File { + path, + mtime, + is_symlink, + } + } + } +} /// Walk the file system synchronously. Used for the initial build where the overhead of spawning /// parallel walker threads is not worth it. #[tracing::instrument(skip_all)] fn walk_synchronous(walker: &mut WalkBuilder) -> Vec { - let mut entries = Vec::new(); - - for entry in walker.build().filter_map(Result::ok) { - let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); - let path = entry.into_path(); - - if is_dir { - entries.push((path, true, String::new())); - } else { - let ext = path - .extension() - .and_then(|x| x.to_str()) - .unwrap_or_default() - .to_owned(); - entries.push((path, false, ext)); - } - } - - entries + walker + .build() + .filter_map(Result::ok) + .map(WalkEntry::from) + .collect() } /// Walk the file system in parallel. Used in watch mode where the parallel walker overhead is @@ -591,7 +679,7 @@ fn walk_parallel(walker: &mut WalkBuilder) -> Vec { } } - let collected: Arc>> = Arc::new(Mutex::new(Vec::new())); + let collected: Arc>> = Arc::new(Mutex::new(vec![])); walker.build_parallel().run(|| { let mut buf = FlushOnDrop { @@ -604,19 +692,7 @@ fn walk_parallel(walker: &mut WalkBuilder) -> Vec { return ignore::WalkState::Continue; }; - let is_dir = entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false); - let path = entry.into_path(); - - if is_dir { - buf.local.push((path, true, String::new())); - } else { - let ext = path - .extension() - .and_then(|x| x.to_str()) - .unwrap_or_default() - .to_owned(); - buf.local.push((path, false, ext)); - } + buf.local.push(WalkEntry::from(entry)); if buf.local.len() >= 256 { buf.shared.lock().unwrap().append(&mut buf.local); diff --git a/crates/oxide/src/scanner/sources.rs b/crates/oxide/src/scanner/sources.rs index 36ccf3b7e..53c09387f 100644 --- a/crates/oxide/src/scanner/sources.rs +++ b/crates/oxide/src/scanner/sources.rs @@ -148,6 +148,16 @@ fn expand_restricted_patterns(sources: Vec) -> Vec { }) .collect::>(); + // Bases of restricted patterns. Each of these becomes its own walk root with its own + // `*` + `!` rules, so an ancestor base must not ignore them recursively. + let pattern_roots = sources + .iter() + .filter_map(|source| match source { + SourceEntry::Pattern { base, .. } => Some(base.clone()), + _ => None, + }) + .collect::>(); + let mut restricted_roots: FxHashSet = FxHashSet::default(); let mut expanded = vec![]; @@ -167,14 +177,17 @@ fn expand_restricted_patterns(sources: Vec) -> Vec { } // Ignore everything in the directory. We will later add the specific patterns we are - // interested in. When another source root is nested in this base, only ignore direct - // children so the nested source can still be walked from its own root. + // interested in. if restricted_roots.insert(base.clone()) { - let pattern = if unrestricted_roots.iter().any(|root| root.starts_with(base)) { - "/*" - } else { - "*" - }; + // When another source root is nested inside this base — an unrestricted root, or the + // base of another restricted pattern (which is walked from its own root with its own + // rules) — only ignore direct children so the nested root can still be walked. + let has_nested_root = unrestricted_roots.iter().any(|root| root.starts_with(base)) + || pattern_roots + .iter() + .any(|root| root != base && root.starts_with(base)); + + let pattern = if has_nested_root { "/*" } else { "*" }; expanded.push(SourceEntry::Ignored { base: base.clone(), @@ -750,12 +763,11 @@ pub fn public_source_entries_to_private_source_entries( .collect::>(); // Compiled `.gitignore` matchers are cached per directory so we read and parse each - // `.gitignore` file at most once, even though entries commonly share ancestor directories - // (e.g. the repository root). A cached `None` means the directory has no `.gitignore` file. + // `.gitignore` file at most once, even though entries commonly share ancestor directories (e.g. + // the repository root). A cached `None` means the directory has no `.gitignore` file. let mut gitignores: FxHashMap> = FxHashMap::default(); - // Boundary for the `.gitignore` walk when a source is not inside a git repository (see - // below). + // Boundary for the `.gitignore` walk when a source is not inside a git repository (see below). let cwd = std::env::current_dir() .map(|cwd| dunce::canonicalize(&cwd).unwrap_or(cwd)) .ok(); @@ -777,18 +789,39 @@ pub fn public_source_entries_to_private_source_entries( let gitignore = gitignores.entry(dir.to_path_buf()).or_insert_with(|| { let path = dir.join(".gitignore"); - // `Gitignore::new` roots the matcher at the directory - // containing the file, so patterns match relative to it. + // `Gitignore::new` roots the matcher at the directory containing the file, + // so patterns match relative to it. path.is_file().then(|| Gitignore::new(&path).0) }); - if let Some(gitignore) = gitignore { - if gitignore - .matched_path_or_any_parents(&base, true) - .is_ignore() - { - source = SourceEntry::External { base: base.into() }; - break; + // Only `.gitignore` files in ancestors of `base` can ignore `base` itself. + // Patterns in `base`'s own `.gitignore` only match paths _inside_ `base`, never + // `base` itself (the file walker still applies them to `base`'s contents). + // + // Skipping `base`'s own `.gitignore` also prevents a false positive for + // whitelist style `.gitignore` files, because relativizing `base` against + // itself yields the empty path, which incorrectly matches `/*`. + // + // E.g.: + // + // ```gitignore + // /* + // !/.gitignore + // !/app + // !/public + // ``` + // + // Everything inside `base` except `.gitignore`, `app` and `public` is ignored, + // but `base` itself is not. + if dir != base { + if let Some(gitignore) = gitignore { + if gitignore + .matched_path_or_any_parents(&base, true) + .is_ignore() + { + source = SourceEntry::External { base: base.into() }; + break; + } } } @@ -797,12 +830,11 @@ pub fn public_source_entries_to_private_source_entries( break; } - // Without a git repository there is no repository root to stop at. Stop - // once the directory contains the current working directory instead, so - // `.gitignore` files outside of the project (e.g. in the user's home - // directory) can never promote a source to an external source. Note that - // the file walker still applies those `.gitignore` files when deciding - // which files to scan. + // Without a git repository there is no repository root to stop at. Stop once + // the directory contains the current working directory instead, so `.gitignore` + // files outside of the project (e.g. in the user's home directory) can never + // promote a source to an external source. Note that the file walker still + // applies those `.gitignore` files when deciding which files to scan. if !inside_git_repo && cwd.as_ref().is_some_and(|cwd| cwd.starts_with(dir)) { break; } diff --git a/crates/oxide/tests/scanner.rs b/crates/oxide/tests/scanner.rs index 1139196e7..fb5354a87 100644 --- a/crates/oxide/tests/scanner.rs +++ b/crates/oxide/tests/scanner.rs @@ -1,5 +1,6 @@ #[cfg(test)] mod scanner { + use insta::assert_snapshot; use pretty_assertions::assert_eq; use std::path::{Path, PathBuf}; use std::process::Command; @@ -55,6 +56,157 @@ mod scanner { globs: Vec, normalized_sources: Vec, candidates: Vec, + tree: String, + } + + /// Renders the directory as a tree, annotating every file and folder with an indicator that + /// shows whether the scanner picked it up: + /// + /// - `✓` — scanned + /// - `✗` — ignored / skipped + /// + /// Symlinks are rendered as `link → target`, where the target is shown relative to the folder + /// containing the symlink. The contents of every `.gitignore` file are printed right below the + /// file itself. + /// + /// Folders that are a git repository root (they contain a `.git` folder) are marked with + /// `(git)`, because ignore rules behave differently inside and outside of a repository. + fn fs_tree(root: &Path, scanned: &[String]) -> String { + /// Computes the relative path from `parent` to `target`, where both are relative to the + /// same root, e.g.: `b/c` seen from `b` is `c`, and the root itself seen from `b` is `..`. + fn relative_to(target: &Path, parent: &Path) -> String { + let target = target.components().collect::>(); + let parent = parent.components().collect::>(); + + let common = target + .iter() + .zip(&parent) + .take_while(|(a, b)| **a == **b) + .count(); + + let mut parts = vec![".."; parent.len() - common]; + parts.extend( + target[common..] + .iter() + .map(|component| component.as_os_str().to_str().unwrap()), + ); + + if parts.is_empty() { + // The symlink points to its own parent folder + ".".into() + } else { + parts.join("/") + } + } + + fn walk( + dir: &Path, + root: &Path, + prefix: &str, + scanned: &[String], + visited: &mut Vec, + out: &mut String, + ) { + let mut entries = fs::read_dir(dir) + .unwrap() + .filter_map(Result::ok) + .filter(|entry| entry.file_name() != ".git") + .collect::>(); + entries.sort_by_key(|entry| entry.file_name()); + + for (i, entry) in entries.iter().enumerate() { + let last = i == entries.len() - 1; + let connector = if last { "└── " } else { "├── " }; + let child_prefix = if last { " " } else { "│ " }; + + let path = entry.path(); + let name = entry.file_name().to_string_lossy().to_string(); + let rel = path + .strip_prefix(root) + .unwrap() + .to_string_lossy() + .replace('\\', "/"); + + let file_type = entry.file_type().unwrap(); + + // a file is scanned when it shows up in the scanned files list. a folder is + // considered scanned when any scanned file lives inside of it. paths that go + // through a symlink count towards the symlink, not towards the target directory. + let dir_prefix = format!("{rel}/"); + let is_scanned = scanned + .iter() + .any(|file| file == &rel || file.starts_with(&dir_prefix)); + let indicator = if is_scanned { "✓" } else { "✗" }; + + let mut display_name = name.clone(); + if path.join(".git").exists() { + display_name = format!("{display_name} (git)"); + } + if file_type.is_symlink() { + if let Ok(target) = fs::read_link(&path) { + // Show the target relative to the folder containing the symlink, or as-is + // when it points outside of the tree. + let target = match (target.strip_prefix(root), dir.strip_prefix(root)) { + (Ok(target), Ok(parent)) => relative_to(target, parent), + _ => target.to_string_lossy().replace('\\', "/"), + }; + display_name = format!("{name} → {target}"); + + // The symlink points to a target that doesn't exist + if path.metadata().is_err() { + display_name = format!("{display_name} (broken)"); + } + } + } + + out.push_str(&format!("{prefix}{connector}{indicator} {display_name}\n")); + + // Print the contents of `.gitignore` files below the file itself, so the ignore + // rules are visible in the same output. + if name == ".gitignore" { + if let Ok(contents) = fs::read_to_string(&path) { + for line in contents.lines() { + let line = format!("{prefix}{child_prefix} {line}"); + out.push_str(line.trim_end()); + out.push('\n'); + } + } + } + + // Descend into directories, including symlinked ones. The paths inside a symlinked + // directory are computed through the symlink, so the indicators show what the + // scanner saw via that route. A visited stack prevents symlink cycles from + // recursing forever. + if path.metadata().map(|meta| meta.is_dir()).unwrap_or(false) { + let Ok(canonical) = dunce::canonicalize(&path) else { + continue; + }; + if visited.contains(&canonical) { + continue; + } + + visited.push(canonical); + walk( + &path, + root, + &format!("{prefix}{child_prefix}"), + scanned, + visited, + out, + ); + visited.pop(); + } + } + } + + let mut visited = vec![dunce::canonicalize(root).unwrap()]; + let mut out = if root.join(".git").exists() { + String::from(". (git)\n") + } else { + String::from(".\n") + }; + walk(root, root, "", scanned, &mut visited, &mut out); + out } fn create_files_in(dir: &path::Path, paths: &[(&str, &str)]) { @@ -106,6 +258,18 @@ mod scanner { globs } + fn normalize_files(files: Vec, base: &Path) -> Vec { + let base_dir = + format!("{}{}", dunce::canonicalize(base).unwrap().display(), "/").replace('\\', "/"); + + let mut files = files + .iter() + .map(|file| file.replace('\\', "/").replace(&base_dir, "")) + .collect::>(); + files.sort(); + files + } + fn scan_with_globs( paths_with_content: &[(&str, &str)], source_directives: Vec<&str>, @@ -155,11 +319,14 @@ mod scanner { .collect::>(); normalized_sources.sort(); + let tree = fs_tree(&dir, &files); + ScanResult { files, globs, normalized_sources, candidates, + tree, } } @@ -173,6 +340,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -180,6 +348,14 @@ mod scanner { ("b.html", ""), ("c.html", ""), ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + └── ✓ index.html + "); assert_eq!(files, vec!["a.html", "b.html", "c.html", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -191,6 +367,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ (".gitignore", "b.html"), @@ -199,6 +376,17 @@ mod scanner { ("b.html", ""), ("c.html", ""), ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ b.html + ├── ✓ a.html + ├── ✗ b.html + ├── ✓ c.html + └── ✓ index.html + "); + assert_eq!(files, vec!["a.html", "c.html", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -210,6 +398,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -220,6 +409,20 @@ mod scanner { ("public/deeply/nested/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ public + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + ├── ✓ deeply + │ └── ✓ nested + │ └── ✓ c.html + └── ✓ nested + └── ✓ c.html + "); + assert_eq!( files, vec![ @@ -241,6 +444,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -254,6 +458,25 @@ mod scanner { ("public/very/deeply/nested/a.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ public + ├── ✓ a.html + ├── ✓ b.html + ├── ✓ c.html + ├── ✓ nested + │ ├── ✓ a.html + │ ├── ✓ again + │ │ └── ✓ a.html + │ ├── ✓ b.html + │ └── ✓ c.html + └── ✓ very + └── ✓ deeply + └── ✓ nested + └── ✓ a.html + "); + assert_eq!( files, vec![ @@ -278,6 +501,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ (".gitignore", "public/b.html\na.html"), @@ -287,6 +511,18 @@ mod scanner { ("public/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ public/b.html + │ a.html + ├── ✓ index.html + └── ✓ public + ├── ✗ a.html + ├── ✗ b.html + └── ✓ c.html + "); + assert_eq!(files, vec!["index.html", "public/c.html",]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -298,6 +534,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -306,6 +543,15 @@ mod scanner { ("src/c.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + └── ✓ src + ├── ✓ a.html + ├── ✓ b.html + └── ✓ c.html + "); + assert_eq!( files, vec!["index.html", "src/a.html", "src/b.html", "src/c.html"] @@ -323,6 +569,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -331,6 +578,14 @@ mod scanner { ("c.lock", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ a.mp4 + ├── ✗ b.png + ├── ✗ c.lock + └── ✓ index.html + "); + assert_eq!(files, vec!["index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -343,6 +598,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ // Looks like `.pages` binary extension, but it's a folder @@ -356,6 +612,16 @@ mod scanner { ), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ other.pages + ├── ✗ other.pages + │ └── ✗ index.html + └── ✓ some.pages + └── ✓ index.html + "); + assert_eq!(files, vec!["some.pages/index.html"]); assert_eq!(globs, vec!["*", "some.pages/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -367,6 +633,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -375,6 +642,14 @@ mod scanner { ("c.less", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ a.css + ├── ✗ b.sass + ├── ✗ c.less + └── ✓ index.html + "); + assert_eq!(files, vec!["a.css", "index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -386,9 +661,16 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[("src/index.my-extension", "")]); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ index.my-extension + "); + assert_eq!(files, vec!["src/index.my-extension"]); assert_eq!(globs, vec!["*", "src/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,my-extension,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -400,6 +682,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ ("index.html", ""), @@ -407,6 +690,13 @@ mod scanner { ("yarn.lock", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.html + ├── ✗ package-lock.json + └── ✗ yarn.lock + "); + assert_eq!(files, vec!["index.html"]); assert_eq!(globs, vec!["*"]); assert_eq!(normalized_sources, vec!["**/*"]); @@ -418,6 +708,7 @@ mod scanner { files, globs, normalized_sources, + tree, .. } = scan(&[ // Explicitly listed root files @@ -462,6 +753,58 @@ mod scanner { ("nested-d/very/deeply/nested/directory/again/foo.html", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✓ foo.html + ├── ✓ nested-a + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + ├── ✓ nested-b + │ └── ✓ deeply-nested + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + ├── ✓ nested-c + │ ├── ✗ .gitignore + │ │ ignored-folder/ + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ ├── ✓ foo.html + │ ├── ✗ ignored-folder + │ │ ├── ✗ bar.html + │ │ ├── ✗ baz.html + │ │ └── ✗ foo.html + │ └── ✓ sibling-folder + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + └── ✓ nested-d + ├── ✗ .gitignore + │ deep/ + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✓ foo.html + └── ✓ very + └── ✓ deeply + └── ✓ nested + ├── ✓ bar.html + ├── ✓ baz.html + ├── ✗ deep + │ ├── ✗ bar.html + │ ├── ✗ baz.html + │ └── ✗ foo.html + ├── ✓ directory + │ ├── ✓ again + │ │ └── ✓ foo.html + │ ├── ✓ bar.html + │ ├── ✓ baz.html + │ └── ✓ foo.html + └── ✓ foo.html + "); + assert_eq!( files, vec![ @@ -516,6 +859,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan(&[ // The gitignore file is used to filter out files but not scanned for candidates @@ -538,6 +882,21 @@ mod scanner { ("index4.svelte", ""), ]); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ # md:font-bold + │ foo.html + ├── ✗ foo.html + ├── ✗ foo.jpg + ├── ✓ index.angular.html + ├── ✓ index.html + ├── ✓ index.svelte + ├── ✓ index2.svelte + ├── ✓ index3.svelte + └── ✓ index4.svelte + "); + assert_eq!( candidates, vec![ @@ -561,12 +920,21 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[("foo/bar/baz/foo.html", "content-['foo.html']")], vec!["@source '**/*'", "@source './foo/bar/baz/..'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ foo + └── ✓ bar + └── ✓ baz + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['foo.html']"]); assert_eq!(normalized_sources, vec!["**/*", "foo/bar/**/*"]); } @@ -598,6 +966,19 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ project-a + │ └── ✗ src + │ └── ✗ index.css + └── ✓ project-b + ├── ✓ ignored + │ ├── ✓ except.html + │ └── ✗ ignored.html + └── ✓ keep + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['GOOD-1']", "content-['GOOD-2']"]); } @@ -607,12 +988,18 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[("my-file", "content-['my-file']")], vec!["@source '**/*'", "@source './my-file'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ my-file + "); + assert_eq!(candidates, vec!["content-['my-file']"]); assert_eq!(normalized_sources, vec!["**/*", "my-file"]); } @@ -623,6 +1010,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -642,6 +1030,14 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ my-folder.bin + │ └── ✓ foo.html + └── ✓ my-folder.templates + └── ✓ foo.html + "); + assert_eq!( candidates, vec![ @@ -660,6 +1056,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -670,10 +1067,111 @@ mod scanner { vec!["@source '**/*'", "@source '*.styl'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ foo.styl + "); + assert_eq!(candidates, vec!["content-['foo.styl']"]); assert_eq!(normalized_sources, vec!["**/*", "*.styl"]); } + #[test] + fn it_should_drop_invalid_utf8_candidates() { + let dir = tempdir().unwrap(); + fs::write(dir.path().join("index.html"), b"flex bg-[\x80] block").unwrap(); + + let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( + dir.path().to_path_buf(), + "@source '*.html'", + )]); + + let candidates = scanner + .scan() + .into_iter() + .map(String::into_bytes) + .collect::>(); + + assert_eq!(candidates, vec![b"block".to_vec(), b"flex".to_vec()]); + } + + #[test] + fn it_should_not_store_invalid_utf8_candidates_during_incremental_scans() { + let dir = tempdir().unwrap(); + let file = dir.path().join("index.html"); + fs::write(&file, b"flex bg-[\x80]").unwrap(); + + let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( + dir.path().to_path_buf(), + "@source '*.html'", + )]); + + let candidates = scanner + .scan_content(vec![ChangedContent::File(file, "html".into())]) + .into_iter() + .map(String::into_bytes) + .collect::>(); + assert_eq!(candidates, vec![b"flex".to_vec()]); + + let candidates = + scanner.scan_content(vec![ChangedContent::Content("block".into(), "html".into())]); + assert_eq!(candidates, vec!["block"]); + + let candidates = scanner + .scan() + .into_iter() + .map(String::into_bytes) + .collect::>(); + assert_eq!(candidates, vec![b"block".to_vec(), b"flex".to_vec()]); + } + + #[test] + fn it_should_drop_invalid_utf8_candidates_with_positions() { + let dir = tempdir().unwrap(); + let file = dir.path().join("index.html"); + fs::write( + &file, + b"flex bg-[\x80] group-[]:block group-[]:bg-[\x80] grid", + ) + .unwrap(); + + let mut scanner = Scanner::new(vec![]); + let candidates = scanner + .get_candidates_with_positions(ChangedContent::File(file, "html".into())) + .into_iter() + .map(|(candidate, position)| (candidate.into_bytes(), position)) + .collect::>(); + + assert_eq!( + candidates, + vec![ + (b"flex".to_vec(), 0), + (b"group-[]:block".to_vec(), 12), + (b"grid".to_vec(), 43), + ] + ); + } + + #[test] + fn it_should_preserve_valid_utf8_candidates() { + let dir = tempdir().unwrap(); + fs::write( + dir.path().join("index.html"), + "before:content-['💩'] bg-[é] font-[中文]".as_bytes(), + ) + .unwrap(); + + let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( + dir.path().to_path_buf(), + "@source '*.html'", + )]); + + assert_eq!( + scanner.scan(), + vec!["before:content-['💩']", "bg-[é]", "font-[中文]"] + ); + } + #[test] fn it_should_preserve_paths_for_sources_ending_in_a_deep_glob() { let ScanResult { @@ -681,6 +1179,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ( @@ -695,6 +1195,18 @@ mod scanner { vec!["@source './blog/*/foo/bar/baz/**/*'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ blog + └── ✓ 2024 + └── ✓ foo + └── ✓ bar + ├── ✓ baz + │ └── ✓ index.html + └── ✗ qux + └── ✗ index.html + "); + assert_eq!( candidates, vec!["content-['blog/2024/foo/bar/baz/index.html']"] @@ -709,6 +1221,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -722,6 +1235,19 @@ mod scanner { vec!["@source '**/*'", "@source './**/*.{styl}'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ app + ├── ✓ (theme) + │ └── ✓ page.styl + ├── ✓ [...slug] + │ └── ✓ page.styl + ├── ✓ [[...slug]] + │ └── ✓ page.styl + └── ✓ [slug] + └── ✓ page.styl + "); + assert_eq!( candidates, vec![ @@ -763,6 +1289,14 @@ mod scanner { let mut scanner = Scanner::new(sources); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ └── ✓ index.html + └── ✓ project-b + └── ✓ index.html + "); + // We've done the initial scan and found the files assert_eq!( candidates, @@ -778,6 +1312,7 @@ mod scanner { let ScanResult { candidates, normalized_sources, + tree, .. } = scan_with_globs( &[ @@ -790,6 +1325,13 @@ mod scanner { vec!["@source '**/*'", "@source 'foo.styl'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ foo.styl + └── ✓ foo.styl + "); + assert_eq!(candidates, vec!["content-['foo.styl']"]); assert_eq!(normalized_sources, vec!["**/*", "foo.styl"]); } @@ -819,6 +1361,14 @@ mod scanner { let mut scanner = Scanner::new(sources); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ └── ✓ index.html + └── ✓ project-b + └── ✓ index.html + "); + // We've done the initial scan and found the files assert_eq!( candidates, @@ -923,6 +1473,24 @@ mod scanner { "content-['project-b/sub1/sub2/new.html']" ] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ project-a + │ ├── ✓ index.html + │ ├── ✓ new.html + │ └── ✓ sub1 + │ └── ✓ sub2 + │ ├── ✓ index.html + │ └── ✓ new.html + └── ✓ project-b + ├── ✓ index.html + ├── ✓ new.html + └── ✓ sub1 + └── ✓ sub2 + ├── ✓ index.html + └── ✓ new.html + "); } #[test] @@ -954,6 +1522,14 @@ mod scanner { vec!["src/index.html", "src/keep.html", "src/remove.html"] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + ├── ✓ keep.html + └── ✓ remove.html + "); + fs::remove_file(dir.join("src/remove.html")).unwrap(); scanner.scan(); @@ -961,6 +1537,13 @@ mod scanner { scanned_files(&mut scanner, &dir), vec!["src/index.html", "src/keep.html"] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✓ keep.html + "); } #[test] @@ -1001,6 +1584,18 @@ mod scanner { ] ); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + ├── ✓ keep + │ └── ✓ index.html + └── ✓ remove + ├── ✓ index.html + └── ✓ nested + └── ✓ index.html + "); + fs::remove_dir_all(dir.join("src/remove")).unwrap(); scanner.scan(); @@ -1008,6 +1603,14 @@ mod scanner { scanned_files(&mut scanner, &dir), vec!["src/index.html", "src/keep/index.html"] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ src + ├── ✓ index.html + └── ✓ keep + └── ✓ index.html + "); } #[test] @@ -1034,6 +1637,13 @@ mod scanner { scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + ├── ✓ index.html + └── ✓ src + └── ✓ index.html + "); + let globs = scanned_globs(&mut scanner, &dir); assert!(globs.iter().any(|glob| glob.starts_with("src/**/*"))); @@ -1041,10 +1651,62 @@ mod scanner { scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . (git) + └── ✓ index.html + "); + let globs = scanned_globs(&mut scanner, &dir); assert!(!globs.iter().any(|glob| glob.starts_with("src/**/*"))); } + #[test] + fn it_should_track_files_scanned_by_the_last_scan() { + let dir = tempdir().unwrap().into_path(); + + let _ = Command::new("git").arg("init").current_dir(&dir).output(); + + create_files_in( + &dir, + &[ + ("src/index.html", "content-['src/index.html']"), + ("src/keep.html", "content-['src/keep.html']"), + ], + ); + + let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( + dir.clone(), + "@source '**/*'", + )]); + + assert_eq!( + scanner.scan(), + vec!["content-['src/index.html']", "content-['src/keep.html']"] + ); + + assert_eq!( + scanner.scan(), + vec!["content-['src/index.html']", "content-['src/keep.html']"] + ); + assert_eq!(scanner.get_scanned_files(), Vec::::new()); + + sleep(Duration::from_millis(10)); + fs::write(dir.join("src/index.html"), "content-['src/changed.html']").unwrap(); + + assert_eq!( + scanner.scan(), + vec![ + "content-['src/changed.html']", + "content-['src/index.html']", + "content-['src/keep.html']", + ] + ); + assert_eq!( + normalize_files(scanner.get_scanned_files(), &dir), + vec!["src/index.html"] + ); + } + #[test] fn it_should_ignore_negated_custom_sources() { let ScanResult { @@ -1052,6 +1714,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/index.ts", "content-['src/index.ts']"), @@ -1079,6 +1743,25 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ admin + │ └── ✓ foo + │ └── ✓ template.html + ├── ✗ colors + │ ├── ✗ blue.tsx + │ ├── ✗ green.tsx + │ └── ✗ red.jsx + ├── ✗ index.ts + ├── ✓ templates + │ └── ✓ index.html + └── ✗ utils + ├── ✗ date.ts + ├── ✗ file.ts + └── ✗ string.ts + "); + assert_eq!( candidates, vec![ @@ -1124,6 +1807,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( // Typically skipped &[ @@ -1135,6 +1820,15 @@ mod scanner { vec!["@source '**/*'", "@source 'src/**/*.{exe,bin}'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ out + │ └── ✗ out.exe + └── ✓ src + ├── ✓ index.bin + └── ✓ index.exe + "); + assert_eq!( candidates, vec!["content-['src/index.bin']", "content-['src/index.exe']",] @@ -1162,6 +1856,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("index.html", "content-['index.html']"), @@ -1186,6 +1882,21 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ index.html + └── ✓ src + ├── ✓ admin + │ ├── ✗ ignore.html + │ └── ✓ index.html + ├── ✓ dashboard + │ ├── ✗ ignore.html + │ └── ✓ index.html + ├── ✗ ignore.html + ├── ✗ index.html + └── ✗ lib.ts + "); + assert_eq!( candidates, vec![ @@ -1205,7 +1916,10 @@ mod scanner { #[test] fn it_should_restrict_explicit_file_sources_to_the_matching_file() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1214,6 +1928,13 @@ mod scanner { vec!["@source './src/foo.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar.html + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['src/foo.html']"]); assert_eq!(files, vec!["src/foo.html"]); } @@ -1221,7 +1942,10 @@ mod scanner { #[test] fn it_should_combine_multiple_restricted_sources_for_the_same_base() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1231,6 +1955,14 @@ mod scanner { vec!["@source './src/foo.html'", "@source './src/bar.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar.html + ├── ✗ baz.html + └── ✓ foo.html + "); + assert_eq!( candidates, vec!["content-['src/bar.html']", "content-['src/foo.html']"] @@ -1238,15 +1970,113 @@ mod scanner { assert_eq!(files, vec!["src/bar.html", "src/foo.html"]); } + // https://github.com/tailwindlabs/tailwindcss/issues/20333 + #[test] + fn it_should_combine_nested_and_root_restricted_sources() { + // The restriction (`*`) added for the root-level file must not prevent walking into the + // `nested` folder that another explicit source points into. At the same time, relaxing + // that restriction to `/*` must not accidentally open up sibling folders (`ignore-me`) + // or files that no explicit source points at. + let paths_with_content = &[ + ("nested/component.html", "content-['nested/component.html']"), + ("nested/ignore-me.html", "content-['nested/ignore-me.html']"), + ("ignore-me/component.html", "content-['ignore-me']"), + ("component-sources.classes.txt", "content-['classes.txt']"), + ("ignore-me.txt", "content-['ignore-me.txt']"), + ]; + + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + paths_with_content, + vec![ + "@source './nested/component.html'", + "@source './component-sources.classes.txt'", + ], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ component-sources.classes.txt + ├── ✗ ignore-me + │ └── ✗ component.html + ├── ✗ ignore-me.txt + └── ✓ nested + ├── ✓ component.html + └── ✗ ignore-me.html + "); + + assert_eq!( + candidates, + vec![ + "content-['classes.txt']", + "content-['nested/component.html']" + ] + ); + assert_eq!( + files, + vec!["component-sources.classes.txt", "nested/component.html"] + ); + + // Same setup, but with the root-level source declared first + let ScanResult { + candidates, + files, + tree, + .. + } = scan_with_globs( + paths_with_content, + vec![ + "@source './component-sources.classes.txt'", + "@source './nested/component.html'", + ], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✓ component-sources.classes.txt + ├── ✗ ignore-me + │ └── ✗ component.html + ├── ✗ ignore-me.txt + └── ✓ nested + ├── ✓ component.html + └── ✗ ignore-me.html + "); + + assert_eq!( + candidates, + vec![ + "content-['classes.txt']", + "content-['nested/component.html']" + ] + ); + assert_eq!( + files, + vec!["component-sources.classes.txt", "nested/component.html"] + ); + } + #[test] fn it_should_allow_later_ignores_to_override_restricted_sources() { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[("src/foo.html", "content-['src/foo.html']")], vec!["@source './src/foo.html'", "@source not './src/foo.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✗ src + └── ✗ foo.html + "); + assert!(candidates.is_empty()); assert!(files.is_empty()); } @@ -1255,7 +2085,10 @@ mod scanner { fn it_should_handle_sources_with_parent_patterns() { { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1265,6 +2098,15 @@ mod scanner { vec!["@source './src/ba*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✓ ignore.html + │ └── ✓ index.html + └── ✗ foo.html + "); + assert_eq!( candidates, vec![ @@ -1277,7 +2119,10 @@ mod scanner { { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1290,13 +2135,25 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✗ ignore.html + │ └── ✓ index.html + └── ✗ foo.html + "); + assert_eq!(candidates, vec!["content-['src/bar/index.html']"]); assert_eq!(files, vec!["src/bar/index.html"]); } { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1306,13 +2163,25 @@ mod scanner { vec!["@source '**/*'", "@source not './src/ba*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✗ bar + │ ├── ✗ ignore.html + │ └── ✗ index.html + └── ✓ foo.html + "); + assert_eq!(candidates, vec!["content-['src/foo.html']"]); assert_eq!(files, vec!["src/foo.html"]); } { let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/foo.html", "content-['src/foo.html']"), @@ -1326,6 +2195,15 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ bar + │ ├── ✗ ignore.html + │ └── ✓ index.html + └── ✓ foo.html + "); + assert_eq!( candidates, vec!["content-['src/bar/index.html']", "content-['src/foo.html']"] @@ -1340,7 +2218,10 @@ mod scanner { // specific `@source` points at a single file inside a subdirectory. The restriction // added for the explicit file must not hide its siblings from the broad source. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/components/button.html", "content-['button']"), @@ -1349,6 +2230,14 @@ mod scanner { vec!["@source '**/*'", "@source './src/components/button.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ components + ├── ✓ button.html + └── ✓ card.html + "); + assert_eq!(candidates, vec!["content-['button']", "content-['card']"]); assert_eq!( files, @@ -1361,7 +2250,10 @@ mod scanner { // Same as above, but the broad source is an auto-detected directory (`@source "src"`) // and the explicit file lives in a nested subdirectory of it. let ScanResult { - candidates, files, .. + candidates, + files, + tree, + .. } = scan_with_globs( &[ ("src/components/button.html", "content-['button']"), @@ -1370,6 +2262,14 @@ mod scanner { vec!["@source 'src'", "@source './src/components/button.html'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + └── ✓ components + ├── ✓ button.html + └── ✓ card.html + "); + assert_eq!(candidates, vec!["content-['button']", "content-['card']"]); assert_eq!( files, @@ -1379,7 +2279,9 @@ mod scanner { #[test] fn root_file_source_should_not_suppress_sibling_source_roots() { - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ ("index.css", ""), ("src/index.html", "content-['src/index.html']"), @@ -1393,6 +2295,17 @@ mod scanner { ], ); + assert_snapshot!(tree, @" + . (git) + ├── ✓ index.css + ├── ✓ pages + │ ├── ✓ foo.html + │ └── ✓ nested + │ └── ✓ foo.html + └── ✓ src + └── ✓ index.html + "); + assert_eq!( candidates, vec![ @@ -1410,6 +2323,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "ignore-1.html\nweb/ignore-2.html"), @@ -1421,6 +2336,19 @@ mod scanner { vec!["@source './src'", "@source './web'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ ignore-1.html + │ web/ignore-2.html + ├── ✓ src + │ └── ✓ index.html + └── ✓ web + ├── ✗ ignore-1.html + ├── ✗ ignore-2.html + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['src/index.html']", "content-['web/index.html']",] @@ -1438,6 +2366,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/logo.jpg", "content-['/src/logo.jpg']"), @@ -1446,6 +2376,13 @@ mod scanner { vec!["@source './src/logo.{jpg,png}'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ logo.jpg + └── ✓ logo.png + "); + assert_eq!( candidates, vec!["content-['/src/logo.jpg']", "content-['/src/logo.png']"] @@ -1462,6 +2399,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "ignore-1.html\n/web/ignore-2.html"), @@ -1471,6 +2410,17 @@ mod scanner { ], vec!["@source './web'", "@source './web/ignore-1.html'"], ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ ignore-1.html + │ /web/ignore-2.html + └── ✓ web + ├── ✓ ignore-1.html + ├── ✗ ignore-2.html + └── ✓ index.html + "); assert_eq!( candidates, vec![ @@ -1484,12 +2434,173 @@ mod scanner { assert_eq!(normalized_sources, vec!["web/**/*", "web/ignore-1.html"]); } + #[test] + fn it_respects_gitignore_files_with_whitelist_patterns() { + // https://github.com/tailwindlabs/tailwindcss/discussions/20382 + // + // A `.gitignore` that ignores everything (`/*`) and then whitelists specific + // directories and files using negated patterns. + let ScanResult { + files, + candidates, + tree, + .. + } = scan(&[ + ( + ".gitignore", + "/*\n!/app\n!/public\n!/package.json\n!/.gitignore\n", + ), + ("app/index.html", "content-['app/index.html']"), + ("public/index.html", "content-['public/index.html']"), + ("package.json", ""), + // These are all ignored by the `/*` rule because they are not whitelisted + ("build/generated.html", "content-['build/generated.html']"), + ("logs/dev.log", "content-['logs/dev.log']"), + ( + "node_modules/my-ui-lib/index.html", + "content-['node_modules/my-ui-lib/index.html']", + ), + ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /* + │ !/app + │ !/public + │ !/package.json + │ !/.gitignore + ├── ✓ app + │ └── ✓ index.html + ├── ✗ build + │ └── ✗ generated.html + ├── ✗ logs + │ └── ✗ dev.log + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + ├── ✓ package.json + └── ✓ public + └── ✓ index.html + "); + + assert_eq!( + files, + vec!["app/index.html", "package.json", "public/index.html"] + ); + assert_eq!( + candidates, + vec![ + "content-['app/index.html']", + "content-['public/index.html']" + ] + ); + } + + #[test] + fn it_respects_gitignore_files_with_nested_whitelist_patterns() { + // Example from the official `.gitignore` documentation: + // + // ```gitignore + // # exclude everything except directory foo/bar + // /* + // !/foo + // /foo/* + // !/foo/bar + // ``` + let ScanResult { + files, + candidates, + tree, + .. + } = scan(&[ + ( + ".gitignore", + "# exclude everything except directory foo/bar\n/*\n!/foo\n/foo/*\n!/foo/bar\n", + ), + ("foo/bar/index.html", "content-['foo/bar/index.html']"), + ( + "foo/bar/nested/index.html", + "content-['foo/bar/nested/index.html']", + ), + // These are all ignored because they are not inside `foo/bar` + ("index.html", "content-['index.html']"), + ("foo/index.html", "content-['foo/index.html']"), + ("foo/baz/index.html", "content-['foo/baz/index.html']"), + ]); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ # exclude everything except directory foo/bar + │ /* + │ !/foo + │ /foo/* + │ !/foo/bar + ├── ✓ foo + │ ├── ✓ bar + │ │ ├── ✓ index.html + │ │ └── ✓ nested + │ │ └── ✓ index.html + │ ├── ✗ baz + │ │ └── ✗ index.html + │ └── ✗ index.html + └── ✗ index.html + "); + + assert_eq!( + files, + vec!["foo/bar/index.html", "foo/bar/nested/index.html"] + ); + assert_eq!( + candidates, + vec![ + "content-['foo/bar/index.html']", + "content-['foo/bar/nested/index.html']" + ] + ); + } + + #[test] + fn explicit_source_directories_respect_their_own_gitignore() { + // A directory referenced via `@source` behaves like an auto source detection + // root. The `.gitignore` file _inside_ that directory still applies to its + // contents. Only when the directory itself is ignored (by an ancestor + // `.gitignore`, see the tests above) do we bypass the ignore rules. + let ScanResult { + files, + candidates, + tree, + .. + } = scan_with_globs( + &[ + ("vendor/.gitignore", "ignored.html"), + ("vendor/index.html", "content-['vendor/index.html']"), + ("vendor/ignored.html", "content-['vendor/ignored.html']"), + ], + vec!["@source 'vendor'"], + ); + + assert_snapshot!(tree, @" + . (git) + └── ✓ vendor + ├── ✗ .gitignore + │ ignored.html + ├── ✗ ignored.html + └── ✓ index.html + "); + + assert_eq!(files, vec!["vendor/index.html"]); + assert_eq!(candidates, vec!["content-['vendor/index.html']"]); + } + #[test] fn explicit_sources_can_include_files_inside_gitignored_parent_directories() { let ScanResult { candidates, files, globs, + tree, .. } = scan_with_globs( &[ @@ -1501,6 +2612,18 @@ mod scanner { vec!["@source './src/ef*/*.html'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ src/efgh/ + └── ✓ src + ├── ✗ abcd + │ └── ✗ index.html + └── ✓ efgh + ├── ✗ ignore.js + └── ✓ index.html + "); + assert_eq!(candidates, vec!["content-['src/efgh/index.html']"]); assert_eq!(files, vec!["src/efgh/index.html"]); assert_eq!(globs, vec!["src/ef*/*.html"]); @@ -1576,7 +2699,33 @@ mod scanner { ), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✗ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); // All ignore files are applied because there's no git repo assert_eq!( @@ -1594,7 +2743,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home (git) + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✗ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1613,7 +2788,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home/project")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project (git) + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✗ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1633,7 +2834,33 @@ mod scanner { .arg("init") .current_dir(dir.join("home/project/apps")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps (git) + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web + ├── ✗ .gitignore + │ ignore-web.html + ├── ✗ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✓ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1655,7 +2882,33 @@ mod scanner { .current_dir(dir.join("home/project/apps/web")) .output(); - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ home + ├── ✗ .gitignore + │ ignore-home.html + └── ✓ project + ├── ✗ .gitignore + │ ignore-project.html + └── ✓ apps + ├── ✗ .gitignore + │ ignore-apps.html + ├── ✓ admin + │ └── ✓ index.html + ├── ✓ dashboard + │ └── ✓ index.html + └── ✓ web (git) + ├── ✗ .gitignore + │ ignore-web.html + ├── ✓ ignore-apps.html + ├── ✓ ignore-home.html + ├── ✓ ignore-project.html + ├── ✗ ignore-web.html + └── ✓ index.html + "); assert_eq!( candidates, @@ -1689,7 +2942,15 @@ mod scanner { public_source_entry_from_pattern(dir.clone(), "@source not 'src/ignore-me.html'"), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + ├── ✗ ignore-me.html + └── ✓ keep-me.html + "); assert_eq!(candidates, vec!["content-['keep-me.html']"]); } @@ -1716,7 +2977,17 @@ mod scanner { ), ]; - let candidates = Scanner::new(sources.clone()).scan(); + let mut scanner = Scanner::new(sources.clone()); + let candidates = scanner.scan(); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + └── ✓ app + └── ✓ [foo] + ├── ✗ ignore-me.html + └── ✓ keep-me.html + "); assert_eq!(candidates, vec!["content-['keep-me.html']"]); } @@ -1743,6 +3014,12 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/keep-me.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ src + └── ✓ keep-me.html + "); + // Create new files that should definitely be ignored create_files_in( &dir, @@ -1770,6 +3047,18 @@ mod scanner { let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ src/ignored-by-gitignore.html + └── ✓ src + ├── ✗ ignore-by-extension.bin + ├── ✓ ignored-by-gitignore.html + ├── ✗ ignored-by-source-not.html + ├── ✓ keep-me.html + └── ✓ new-file.html + "); + assert_eq!( candidates, vec![ @@ -1799,6 +3088,11 @@ mod scanner { let candidates = scanner.scan(); assert!(candidates.is_empty()); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✗ foo.styl + "); + // Explicitly allow `.styl` files let mut scanner = Scanner::new(vec![ public_source_entry_from_pattern(dir.clone(), "@source '**/*'"), @@ -1807,6 +3101,11 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['foo.styl']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + └── ✓ foo.styl + "); } #[test] @@ -1831,6 +3130,13 @@ mod scanner { let candidates = scanner.scan(); assert!(candidates.is_empty()); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ index.html + └── ✗ index.html + "); + let mut scanner = Scanner::new(vec![ public_source_entry_from_pattern(dir.clone(), "@source '**/*'"), public_source_entry_from_pattern(dir.clone(), "@source './*.html'"), @@ -1838,6 +3144,13 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['index.html']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ index.html + └── ✓ index.html + "); } #[test] @@ -1872,6 +3185,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + // Explicitly listing all `*.html` files, should not include `node_modules` because it's // ignored let sources = vec![public_source_entry_from_pattern( @@ -1883,6 +3207,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/index.html']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✗ node_modules + │ └── ✗ my-ui-lib + │ └── ✗ index.html + └── ✓ src + └── ✓ index.html + "); + // Explicitly listing all `*.html` files // Explicitly list the `node_modules/my-ui-lib` // @@ -1900,12 +3235,25 @@ mod scanner { "content-['src/index.html']" ] ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + ├── ✓ node_modules + │ └── ✓ my-ui-lib + │ └── ✓ index.html + └── ✓ src + └── ✓ index.html + "); } // https://github.com/tailwindlabs/tailwindcss/issues/19844 #[test] fn test_allow_explicit_sources_ignored_by_allow_list_gitignore() { - let ScanResult { candidates, .. } = scan_with_globs( + let ScanResult { + candidates, tree, .. + } = scan_with_globs( &[ (".gitignore", "*\n!/app\n!/app/design\n!/app/design/**\n"), ( @@ -1920,6 +3268,27 @@ mod scanner { vec!["@source 'vendor/acme/theme'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ * + │ !/app + │ !/app/design + │ !/app/design/** + ├── ✗ app + │ └── ✗ design + │ └── ✗ frontend + │ └── ✗ theme + │ └── ✗ templates + │ └── ✗ component.phtml + └── ✓ vendor + └── ✓ acme + └── ✓ theme + └── ✓ module + └── ✓ templates + └── ✓ component.phtml + "); + assert_eq!( candidates, vec!["content-['vendor/acme/theme/module/templates/component.phtml']"] @@ -1933,6 +3302,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ( @@ -1951,6 +3322,17 @@ mod scanner { vec!["@source '**/*'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ node_modules + │ └── ✗ index.html + └── ✓ packages + └── ✓ web + ├── ✓ index.html + └── ✗ node_modules + └── ✗ index.html + "); + assert_eq!(candidates, vec!["content-['packages/web/index.html']"]); assert_eq!(files, vec!["packages/web/index.html",]); @@ -1965,6 +3347,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ (".gitignore", "node_modules\ndist"), @@ -1980,6 +3364,18 @@ mod scanner { vec!["@source 'node_modules/my-ui-lib'"], ); + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + └── ✓ my-ui-lib + ├── ✓ dist + │ └── ✓ index.html + └── ✗ node.exe + "); + assert_eq!( candidates, vec!["content-['node_modules/my-ui-lib/dist/index.html']"] @@ -2021,6 +3417,17 @@ mod scanner { let candidates = scanner.scan(); assert_eq!(candidates, vec!["content-['src/components/button.tsx']"]); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ *.jsx + │ generated/ + └── ✓ src + └── ✓ components + ├── ✗ button.jsx + └── ✓ button.tsx + "); + // Create 2 new files, one "good" and one "bad" file, and manually scan them. This should // only return the "good" file because the "bad" one is ignored by a `.gitignore` file. create_files_in( @@ -2081,6 +3488,8 @@ mod scanner { files, globs, normalized_sources, + tree, + .. } = scan_with_globs( &[ ("src/💩.js", "content-['src/💩.js']"), @@ -2090,6 +3499,15 @@ mod scanner { vec!["@source '**/*'", "@source not 'src/🤦‍♂️'"], ); + assert_snapshot!(tree, @" + . (git) + └── ✓ src + ├── ✓ 💩.js + ├── ✗ 🤦‍♂️ + │ └── ✗ foo.tsx + └── ✓ 🤦‍♂️.tsx + "); + assert_eq!( candidates, vec!["content-['src/💩.js']", "content-['src/🤦‍♂️.tsx']"] @@ -2126,6 +3544,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✓ .gitkeep + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2137,6 +3572,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✗ .gitkeep + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2148,6 +3600,23 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ .gitignore + │ node_modules + │ dist + └── ✓ node_modules + ├── ✓ .pnpm + │ └── ✓ @org+my-ui-library + │ └── ✓ dist + │ └── ✓ index.ts + └── ✓ @org + ├── ✓ .gitkeep + └── ✓ my-ui-library → ../.pnpm/@org+my-ui-library + └── ✓ dist + └── ✓ index.ts + "); + assert_eq!( candidates, vec!["content-['node_modules/.pnpm/@org+my-ui-library/dist/index.ts']"] @@ -2165,11 +3634,14 @@ mod scanner { ], ); - // Create recursive symlinks - let _ = symlink(dir.join("a"), dir.join("b")); - let _ = symlink(dir.join("b/c"), dir.join("c")); - let _ = symlink(dir.join("b/root"), &dir); - let _ = symlink(dir.join("c"), dir.join("a")); + // Create recursive symlinks: + // + // - `a → b`, `b/c → c`, `c → a` form a cycle + // - `b/root → .` points back at the root directory + let _ = symlink(dir.join("b"), dir.join("a")); + let _ = symlink(dir.join("c"), dir.join("b/c")); + let _ = symlink(&dir, dir.join("b/root")); + let _ = symlink(dir.join("a"), dir.join("c")); let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( dir.clone(), @@ -2177,6 +3649,24 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ a → b + │ ├── ✗ c → ../c + │ ├── ✓ index.html + │ └── ✗ root → .. + ├── ✓ b + │ ├── ✗ c → ../c + │ ├── ✓ index.html + │ └── ✗ root → .. + ├── ✓ c → a + │ ├── ✗ c → . + │ ├── ✓ index.html + │ └── ✗ root → .. + └── ✓ z + └── ✓ index.html + "); + assert_eq!( candidates, vec!["content-['b/index.html']", "content-['z/index.html']"] @@ -2201,6 +3691,14 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ abcd + │ └── ✓ xyz.html + └── ✓ efgh → abcd + └── ✓ xyz.html + "); + assert_eq!(candidates, vec!["content-['abcd/xyz.html']"]); // Partially referencing the symlinked folder with a glob, should find the file @@ -2241,6 +3739,18 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ actual-dir + │ └── ✗ ignore.html + ├── ✗ actual-file.html + ├── ✗ linked-dir → actual-dir + │ └── ✗ ignore.html + ├── ✗ linked-file.html → actual-file.html + └── ✓ src + └── ✓ keep.html + "); + assert_eq!(candidates, vec!["content-['src/keep.html']"]); let mut scanner = Scanner::new(vec![ @@ -2278,6 +3788,16 @@ mod scanner { )]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ actual-dir + │ └── ✓ ignore.html + └── ✓ project + ├── ✓ keep.html + └── ✓ linked-dir → ../actual-dir + └── ✓ ignore.html + "); + assert_eq!( candidates, vec![ @@ -2292,9 +3812,232 @@ mod scanner { ]); let candidates = scanner.scan(); + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✗ actual-dir + │ └── ✗ ignore.html + └── ✓ project + ├── ✓ keep.html + └── ✗ linked-dir → ../actual-dir + └── ✗ ignore.html + "); + assert_eq!(candidates, vec!["content-['project/keep.html']"]); } + #[test] + fn test_files_behind_symlinks_are_tracked_with_their_canonical_paths() { + let dir = tempdir().unwrap().into_path(); + create_files_in( + &dir, + &[ + ( + "packages/repro/source.html", + "content-['packages/repro/source.html']", + ), + ( + "packages/repro/nested/deep.html", + "content-['packages/repro/nested/deep.html']", + ), + ( + "packages/other/index.html", + "content-['packages/other/index.html']", + ), + ], + ); + + // Mimic a pnpm workspace where `node_modules` contains a symlink to the actual package + fs::create_dir_all(dir.join("node_modules")).unwrap(); + let _ = symlink(dir.join("packages/repro"), dir.join("node_modules/repro")); + + // A directly symlinked file + let _ = symlink_file( + dir.join("packages/other/index.html"), + dir.join("linked.html"), + ); + + let mut scanner = Scanner::new(vec![ + public_source_entry_from_pattern( + dir.clone(), + "@source 'node_modules/repro/source.html'", + ), + // The symlink sits multiple levels up from the file + public_source_entry_from_pattern( + dir.clone(), + "@source 'node_modules/repro/nested/deep.html'", + ), + public_source_entry_from_pattern(dir.clone(), "@source 'linked.html'"), + ]); + + let candidates = scanner.scan(); + assert_eq!( + candidates, + vec![ + "content-['packages/other/index.html']", + "content-['packages/repro/nested/deep.html']", + "content-['packages/repro/source.html']", + ] + ); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ linked.html → packages/other/index.html + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ ├── ✓ nested + │ │ └── ✓ deep.html + │ └── ✓ source.html + └── ✓ packages + ├── ✓ other + │ └── ✓ index.html + └── ✓ repro + ├── ✓ nested + │ └── ✓ deep.html + └── ✓ source.html + "); + + // Both the symlinked paths and the canonical paths should be tracked, such that file + // watchers watching the returned files also watch the real files on disk. + let files = scanned_files(&mut scanner, &dir); + assert_eq!( + files, + vec![ + "linked.html", + "node_modules/repro/nested/deep.html", + "node_modules/repro/source.html", + "packages/other/index.html", + "packages/repro/nested/deep.html", + "packages/repro/source.html", + ] + ); + } + + #[test] + fn test_changes_to_the_canonical_path_of_a_symlinked_file_are_detected() { + let dir = tempdir().unwrap().into_path(); + create_files_in(&dir, &[("packages/repro/source.html", "content-['v1']")]); + + // Mimic a pnpm workspace where `node_modules` contains a symlink to the actual package + fs::create_dir_all(dir.join("node_modules")).unwrap(); + let _ = symlink(dir.join("packages/repro"), dir.join("node_modules/repro")); + + let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( + dir.clone(), + "@source 'node_modules/repro/source.html'", + )]); + + let candidates = scanner.scan(); + assert_eq!(candidates, vec!["content-['v1']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ └── ✓ source.html + └── ✓ packages + └── ✓ repro + └── ✓ source.html + "); + + // Update the real file on disk. This is the path file watchers will report changes for. + create_files_in(&dir, &[("packages/repro/source.html", "content-['v2']")]); + + let candidates = scanner.scan_content(vec![ChangedContent::File( + dir.join("packages/repro/source.html"), + "html".into(), + )]); + assert_eq!(candidates, vec!["content-['v2']"]); + } + + #[test] + fn test_new_files_behind_symlinks_are_detected_at_their_canonical_path() { + let dir = tempdir().unwrap().into_path(); + create_files_in(&dir, &[("packages/repro/a.html", "content-['a']")]); + + // Mimic a pnpm workspace where `node_modules` contains a symlink to the actual package + fs::create_dir_all(dir.join("node_modules")).unwrap(); + let _ = symlink(dir.join("packages/repro"), dir.join("node_modules/repro")); + + let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( + dir.clone(), + "@source 'node_modules/repro/*.html'", + )]); + + let candidates = scanner.scan(); + assert_eq!(candidates, vec!["content-['a']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ └── ✓ a.html + └── ✓ packages + └── ✓ repro + └── ✓ a.html + "); + + // Create a new file in the real directory. File watchers watching the real directory + // will report the new file with its canonical path. + create_files_in(&dir, &[("packages/repro/b.html", "content-['b']")]); + + let candidates = scanner.scan_content(vec![ChangedContent::File( + dir.join("packages/repro/b.html"), + "html".into(), + )]); + assert_eq!(candidates, vec!["content-['b']"]); + + assert_snapshot!(fs_tree(&dir, &scanned_files(&mut scanner, &dir)), @" + . + ├── ✓ node_modules + │ └── ✓ repro → ../packages/repro + │ ├── ✓ a.html + │ └── ✓ b.html + └── ✓ packages + └── ✓ repro + ├── ✓ a.html + └── ✗ b.html + "); + } + + // https://github.com/tailwindlabs/tailwindcss/pull/20408 + #[test] + fn test_resolving_globs_does_not_traverse_gitignored_directories() { + let ScanResult { + files, globs, tree, .. + } = scan_with_globs( + &[ + (".gitignore", "/vendor\n"), + ("vendor/pkg/canary/index.html", ""), + ("src/index.html", ""), + ], + vec!["@source '**/*'", "@source './vendor/pkg/canary'"], + ); + + assert_snapshot!(tree, @" + . (git) + ├── ✗ .gitignore + │ /vendor + ├── ✓ src + │ └── ✓ index.html + └── ✓ vendor + └── ✓ pkg + └── ✓ canary + └── ✓ index.html + "); + + assert_eq!( + files, + vec!["src/index.html", "vendor/pkg/canary/index.html"] + ); + assert_eq!(globs, vec![ + "*", + "src/**/*.{aspx,astro,cjs,cts,eex,erb,gjs,gts,haml,handlebars,hbs,heex,html,jade,js,jsx,liquid,md,mdx,mjs,mts,mustache,njk,nunjucks,php,pug,py,razor,rb,rhtml,rs,slim,svelte,tpl,ts,tsx,twig,vue}", + + // This should not include `**` or `**.*.{aspx,...}` otherwise this might be scanned recursively. + "vendor/pkg/canary/*", + ]); + } + #[test] fn test_extract_used_css_variables_from_css() { let dir = tempdir().unwrap().into_path(); @@ -2320,6 +4063,11 @@ mod scanner { ("src/defined-at-start.css", "--color-defined-at-start: red;"), ], ); + fs::write( + dir.join("src/invalid.css"), + b".button { color: var(--color-\x80); }", + ) + .unwrap(); let mut scanner = Scanner::new(vec![public_source_entry_from_pattern( dir.clone(), diff --git a/integrations/cli/index.test.ts b/integrations/cli/index.test.ts index ba11d546a..07f8bcb68 100644 --- a/integrations/cli/index.test.ts +++ b/integrations/cli/index.test.ts @@ -375,6 +375,40 @@ describe.each([ }, ) + test( + 'watch mode with polling', + { + fs: { + 'package.json': json` + { + "dependencies": { + "tailwindcss": "workspace:^", + "@tailwindcss/cli": "workspace:^" + } + } + `, + 'src/index.css': css`@import 'tailwindcss/utilities';`, + 'src/index.html': html` +
+ `, + }, + }, + async ({ fs, spawn }) => { + let process = await spawn( + `${command} --input src/index.css --output dist/out.css --watch --poll=50`, + ) + await process.onStderr((m) => m.includes('Done in')) + + await fs.expectFileToContain('dist/out.css', [candidate`underline`]) + + await fs.write('src/index.html', html` +
+ `) + + await fs.expectFileToContain('dist/out.css', [candidate`flex`]) + }, + ) + test( "watch mode with unknown @source paths shouldn't crash on Windows", { @@ -2242,6 +2276,70 @@ test( }, ) +// https://github.com/tailwindlabs/tailwindcss/discussions/20382 +test( + 'auto source detection respects allow-list .gitignore files', + { + fs: { + 'package.json': json` + { + "dependencies": { + "tailwindcss": "workspace:^", + "@tailwindcss/cli": "workspace:^" + } + } + `, + '.gitignore': txt` + /* + !/.gitignore + !/app + !/public + !/package.json + `, + 'app/root.css': css` @import 'tailwindcss/utilities'; `, + // Included by the `!/app` pattern in `.gitignore` + // + // → Should be included + 'app/index.html': html` +
+ `, + // Included by the `!/public` pattern in `.gitignore` + // + // → Should be included + 'public/index.html': html` +
+ `, + // Ignored by the `/*` in `.gitignore` + // + // → Should be ignored + 'build/index.html': html` +
+ `, + // Ignored by the `/*` in `.gitignore` and the default `node_modules` rules + // + // → Should be ignored + 'node_modules/my-ui-lib/index.html': html` +
+ `, + }, + }, + async ({ fs, exec }) => { + await exec('pnpm tailwindcss --input app/root.css --output dist/out.css') + + await fs.expectFileToContain('dist/out.css', [ + candidate`content-['app/index.html']`, + candidate`content-['public/index.html']`, + ]) + + await fs.expectFileNotToContain('dist/out.css', [ + candidate`content-['build/index.html']`, + candidate`content-['node_modules/my-ui-lib/index.html']`, + ]) + }, +) + test( '@source works with symlinks (referencing folder in current folder)', { @@ -2797,8 +2895,8 @@ test( @layer theme, base, components, utilities; @layer theme { :root, :host { - --font-sans: ui-sans-serif, system-ui, sans-serif, 'Apple Color Emoji', 'Segoe UI Emoji', 'Segoe UI Symbol', - 'Noto Color Emoji'; + --font-sans: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, 'Helvetica Neue', 'Noto Sans', Arial, + sans-serif, 'Apple Color Emoji', 'Segoe UI Emoji', 'Segoe UI Symbol', 'Noto Color Emoji'; --font-mono: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, 'Liberation Mono', 'Courier New', monospace; --color-red-500: oklch(63.7% 0.237 25.331); @@ -2817,7 +2915,7 @@ test( line-height: 1.5; -webkit-text-size-adjust: 100%; tab-size: 4; - font-family: var(--default-font-family, ui-sans-serif, system-ui, sans-serif, 'Apple Color Emoji', 'Segoe UI Emoji', 'Segoe UI Symbol', 'Noto Color Emoji'); + font-family: var(--default-font-family, -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, 'Helvetica Neue', 'Noto Sans', Arial, sans-serif, 'Apple Color Emoji', 'Segoe UI Emoji', 'Segoe UI Symbol', 'Noto Color Emoji'); font-feature-settings: var(--default-font-feature-settings, normal); font-variation-settings: var(--default-font-variation-settings, normal); -webkit-tap-highlight-color: transparent; @@ -2869,7 +2967,7 @@ test( border-color: inherit; border-collapse: collapse; } - :-moz-focusring { + :-moz-focusring:where(:not(iframe)) { outline: auto; } progress { diff --git a/integrations/cli/plugins.test.ts b/integrations/cli/plugins.test.ts index 4cda7955a..0b810c538 100644 --- a/integrations/cli/plugins.test.ts +++ b/integrations/cli/plugins.test.ts @@ -109,13 +109,11 @@ test( expect(await fs.dumpFiles('dist/out.css')).toMatchInlineSnapshot(` " --- dist/out.css --- - .prose-custom { - :where(hr):not(:where([class~="not-prose"], [class~="not-prose"] *)) { - --x: 1; - @media (width >= 640px) { - &.custom { - --x: 2; - } + .prose-custom :where(hr):not(:where([class~="not-prose"], [class~="not-prose"] *)) { + --x: 1; + @media (width >= 640px) { + &.custom { + --x: 2; } } } diff --git a/integrations/oxide/wasm.test.ts b/integrations/oxide/wasm.test.ts index ac2231672..2d63c707a 100644 --- a/integrations/oxide/wasm.test.ts +++ b/integrations/oxide/wasm.test.ts @@ -1,4 +1,4 @@ -import { css, js, json, test } from '../utils' +import { css, js, json, test, yaml } from '../utils' // This test runs the wasm build using the `node:wasi` runtime. // @@ -57,3 +57,187 @@ testFn( `) }, ) + +testFn( + '`@tailwindcss/oxide` falls back to the wasm build when no native binding is available', + { + fs: { + 'package.json': json` + { + "dependencies": { + "@tailwindcss/oxide": "workspace:^" + } + } + `, + 'pnpm-workspace.yaml': yaml` + # Trick pnpm in only supporting an architecture that @tailwindcss/oxide + # doesn't support, and therefore should fallback to the wasm version. + supportedArchitectures: + os: + - openbsd + cpu: + - x64 + `, + 'src/index.js': js` + const className = "content-['src/index.js']" + module.exports = { className } + `, + 'index.mjs': js` + import { createRequire } from 'node:module' + import { join } from 'node:path' + + let require = createRequire(import.meta.url) + let { Scanner } = require('@tailwindcss/oxide') + + let loaded = Object.keys(require.cache) + + let scanner = new Scanner({ + sources: [ + { + base: join(process.cwd(), 'src'), + pattern: '**/*', + negated: false, + }, + ], + }) + + console.log( + JSON.stringify({ + native: loaded.filter((file) => file.endsWith('.node')), + wasi: loaded.some((file) => file.endsWith('tailwindcss-oxide.wasi.cjs')), + candidates: scanner.scan(), + }), + ) + process.exit() + `, + }, + }, + async ({ expect, exec }) => { + // Since vitest runs under `pnpm run`, pnpm's bin shims export a NODE_PATH + // that includes the repository's hidden hoist directory + // (`node_modules/.pnpm/node_modules`), which links every workspace package, + // including all native `@tailwindcss/oxide-*` bindings. + // + // Node uses `NODE_PATH` exactly when the local `node_modules` lookup fails, + // which would defeat the simulated unsupported platform, so clear it. + let output = await exec(`node index.mjs`, { env: { NODE_PATH: '' } }) + let { native, wasi, candidates } = JSON.parse(output) + + // No native binding was installed or loaded, ... + expect(native).toEqual([]) + + // ... the wasm32-wasi binding is what actually loaded, ... + expect(wasi).toBe(true) + + // ... and scanning real files on disk works through it. + expect(candidates).toMatchInlineSnapshot(` + [ + "className", + "const", + "content-['src/index.js']", + "exports", + ] + `) + }, +) + +testFn( + 'the wasm build loads even when preopening the filesystem root is denied', + { + fs: { + 'package.json': json` + { + "dependencies": { + "@tailwindcss/oxide": "workspace:^" + } + } + `, + 'pnpm-workspace.yaml': yaml` + # Trick pnpm in only supporting an architecture that @tailwindcss/oxide + # doesn't support, and therefore should fallback to the wasm version. + supportedArchitectures: + os: + - openbsd + cpu: + - x64 + `, + // The wasm bindings generated by `@napi-rs/cli` preopen the filesystem + // root, which sandboxed platforms (e.g. OpenHarmony, Android) deny with + // `UVWASI_EACCES`, making the wasm fallback fail to load on exactly the + // platforms that need it. + // + // We patch `@napi-rs/cli`'s templates to retry with narrower preopens + // (see `patches/@napi-rs__cli@3.7.4.patch`). Simulate such a sandbox by + // denying `/` preopens. + 'preload.cjs': js` + const wasi = require('node:wasi') + const RealWASI = wasi.WASI + + wasi.WASI = class WASI extends RealWASI { + constructor(options) { + if (options?.preopens?.['/'] !== undefined) { + const error = new Error('UVWASI_EACCES, uvwasi_init') + error.code = 'UVWASI_EACCES' + error.syscall = 'uvwasi_init' + throw error + } + super(options) + } + } + `, + 'src/index.js': js` + const className = "content-['src/index.js']" + module.exports = { className } + `, + 'index.mjs': js` + import { createRequire } from 'node:module' + import { join } from 'node:path' + + let require = createRequire(import.meta.url) + let { Scanner } = require('@tailwindcss/oxide') + + let loaded = Object.keys(require.cache) + + let scanner = new Scanner({ + sources: [ + { + base: join(process.cwd(), 'src'), + pattern: '**/*', + negated: false, + }, + ], + }) + + console.log( + JSON.stringify({ + wasi: loaded.some((file) => file.endsWith('tailwindcss-oxide.wasi.cjs')), + candidates: scanner.scan(), + }), + ) + process.exit() + `, + }, + }, + async ({ expect, exec }) => { + // See the note about NODE_PATH in the test above. + let output = await exec(`node --require ./preload.cjs index.mjs`, { env: { NODE_PATH: '' } }) + + // Only parse the first line, because Node prints an `ExperimentalWarning` + // about WASI to stderr, which `exec` appends to the output. + let { wasi, candidates } = JSON.parse(output.trim().split('\n')[0]) + + // The wasm32-wasi binding loaded despite `/` being denied, ... + expect(wasi).toBe(true) + + // ... and scanning files under the current working directory still works + // through the narrower preopen. + expect(candidates).toMatchInlineSnapshot(` + [ + "className", + "const", + "content-['src/index.js']", + "exports", + ] + `) + }, +) diff --git a/integrations/turbopack/loader.test.ts b/integrations/turbopack/loader.test.ts new file mode 100644 index 000000000..755cd60df --- /dev/null +++ b/integrations/turbopack/loader.test.ts @@ -0,0 +1,83 @@ +import { candidate, css, fetchStyles, js, json, jsx, retryAssertion, test } from '../utils' + +test( + '@tailwindcss/turbopack loader (dev)', + { + timeout: 120_000, + fs: { + 'package.json': json` + { + "dependencies": { + "next": "^16.2.7", + "react": "^19.2.7", + "react-dom": "^19.2.7", + "tailwindcss": "workspace:^", + "@tailwindcss/turbopack": "workspace:^" + } + } + `, + 'next.config.mjs': js` + export default { + turbopack: { + rules: { + '*.css': { + loaders: ['@tailwindcss/turbopack'], + as: '*.css', + }, + }, + }, + } + `, + 'app/layout.js': jsx` + import './globals.css' + + export default function RootLayout({ children }) { + return ( + + {children} + + ) + } + `, + 'app/page.js': jsx` + export default function Page() { + return
Hello, Next.js!
+ } + `, + 'app/globals.css': css` @import 'tailwindcss'; `, + }, + }, + async ({ fs, spawn, expect }) => { + let process = await spawn('pnpm next dev') + + let url = '' + await process.onStdout((message) => { + let match = /Local:\s*(http.*)/.exec(message) + if (match) url = match[1] + return Boolean(url) + }) + + await process.onStdout((message) => message.includes('Ready in')) + + await retryAssertion(async () => { + let styles = await fetchStyles(url) + expect(styles).toContain(candidate`flex`) + expect(styles).not.toContain(candidate`underline`) + }) + + await fs.write( + 'app/page.js', + jsx` + export default function Page() { + return
Hello, Next.js!
+ } + `, + ) + + await retryAssertion(async () => { + let styles = await fetchStyles(url) + expect(styles).toContain(candidate`flex`) + expect(styles).toContain(candidate`underline`) + }) + }, +) diff --git a/integrations/upgrade/index.test.ts b/integrations/upgrade/index.test.ts index 9e236e8fd..efac411db 100644 --- a/integrations/upgrade/index.test.ts +++ b/integrations/upgrade/index.test.ts @@ -2557,7 +2557,7 @@ test( @import './utilities.css'; @import './generated/ignore-me.css'; `, - 'src/generated/.gitignore': ` + 'src/generated/.gitignore': txt` * !.gitignore `, @@ -2817,7 +2817,7 @@ test( 
--- ./src/input.css --- - @import 'tailwindcss'; + @import 'tailwindcss'; /* The default border color has changed to \`currentcolor\` in Tailwind CSS v4, @@ -3371,6 +3371,61 @@ test( }, ) +// https://github.com/tailwindlabs/tailwindcss/issues/20328 +test( + 'ignored files should not be touched when upgrading from a nested directory', + { + fs: { + 'pnpm-workspace.yaml': yaml` + # + packages: + - packages/* + `, + 'package.json': json` + { + "name": "root", + "private": true + } + `, + 'packages/css/package.json': json` + { + "name": "css-pkg", + "private": true, + "devDependencies": { + "tailwindcss": "^4", + "@tailwindcss/upgrade": "workspace:^" + } + } + `, + 'packages/css/src/input.css': css` + @import 'tailwindcss/utilities.css' layer(utilities) source(none); + `, + 'packages/css/src/index.html': html`
Hi
`, + + // Ensure files/folders ignored by a `.gitignore` in the root take effect + // when executing the upgrade tool from a sub-package. + '.gitignore': txt` + node_modules/ + `, + }, + }, + async ({ root, exec, fs, expect }) => { + await exec('git init', { cwd: root }) + + await exec('pnpm exec upgrade --force', { + cwd: path.join(root, 'packages/css'), + }) + + expect(await fs.dumpFiles('packages/css/node_modules/tailwindcss/utilities.css')) + .toMatchInlineSnapshot(` + " + --- packages/css/node_modules/tailwindcss/utilities.css --- + @tailwind utilities; + " + `) + }, +) + test( 'upgrade