From 55daf8e0f55348788af18ca590e86aabd51e5d9b Mon Sep 17 00:00:00 2001
From: Robin Malfait
Date: Wed, 7 Jun 2023 17:44:35 +0200
Subject: [PATCH] Ensure the oxide parser has feature parity with the stable
RegEx parser (#11389)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
* WIP
* use `parse` instead of `defaultExtractor`
* skip `Vue` describe block
* add a few more dedicated arbitrary values/properties tests
* use parallel parsing
* splitup Vue tests
* add some Rust specific tests
* setup parse candidate strings test system
These tests will run against the `Regex` and `Rust` based parsers. We
have groups of classes of various shapes and forms + variants and
rendered in various template situation (plain, html, Vue, ...)
+ enable all skipped tests
* ensure we also validate the classes with variants
The classes with variants are built in the `templateTable` function, so
we get them out again by using the potional arguments of the `test.each`
cb function.
* cleanup test suite
* add "anti-test" tests
To make sure that we are _not_ parsing out certain values given a
certain input.
* Add ParseAction enum
* Restart parsing following an arbitrary parse failure
* Split variants off before validating the uility part
* Collapse candidate from the end when validation fails
* Support `<`, and `>` in variant position
* fix error
* format parser.rs
* Refactor
* Update editorconfig
* wip
* wip
* Refactor
* Refactor
* Simplify
* wip
* wip
* wip
* wip
* wip
* wip
* wip
* run `cargo clippy --fix`
* run `cargo fmt`
* implement `cargo clippy` suggestions
These were not applied using `cargo clippy --fix`
* only allow `.` in the candidate part when surrounded by 0-9
This is only in the candidate part, not the arbitrary part.
* % characters can only appear at the end after digits
* > and < should only be part of variants (start OR end)
It can technically be inside the candidate when we have stacked
variants:
```
dark::underline
```
* handle parsing utilities within quotes, parans or brackets
* mark `pt-1.5` as an expected value sliced out from `["pt-1.5"]`
* Add cursor abstraction
* wip
* disable the oxideParser if using a custom `prefix` or `separator`
* update tests
* Use cursor abstraction
* Refactor more code toward use of global cursor
* wip
* simplify
* Simplify
* Simplify
* Simplify
* Cleanup
* wip
* Simplify
* wip
* Simplify
* Handle candidates ending with % sign
* Tweak code a bit
* fmt
* Simplify
* Add cursor details to trace
* cargo fmt
* use preferred `zoom-0.5` name instead of `zoom-.5`
* drop over-extracted utilities in oxide parser
The RegEx parser does extract `underline` from
```html
```
... but that's not needed and is not happening in the oxide parser
This means that we have to make the output check a little bit different
but they are explicit based on the feature flag.
* allow extracting variants+utilities inside `{}` for the oxide parser
* characters in candidates such as `group-${id}` should not be allowed
* do not extract any of the following candidate `w-[foo-bar]w-[bar-baz]`
* ensure we can consume the full candidate and discard it
* Add fast skipping of whitespace
* Use fast skipping whenever possible
* Add fast skipping to benchmark
* Hand-tune to generate more optimized assembly
* Move code around a bit
This makes sure all the fancy SIMD stuff is as early as possible. This results in an extremely minor perf increase.
* Undo tweak
no meaningful perf difference in real world scenarios
* Disable fast skipping for now
It needs to be done in a different spot so it doesn’t affect how things are returned
* Change test names
* Fix normalize config error
* cleanup a bit
* Cleanup
* Extract validation result enum
* Cleanup comments
* Simplify
* Fix formatting
* Run clippy
* wip
* add `md>` under the special characters test set
---------
Co-authored-by: Adam Wathan <4323180+adamwathan@users.noreply.github.com>
Co-authored-by: Jordan Pittman
---
.editorconfig | 8 +
oxide/crates/core/benches/parse_candidates.rs | 12 +
oxide/crates/core/src/cursor.rs | 159 ++++
oxide/crates/core/src/fast_skip.rs | 89 +++
oxide/crates/core/src/lib.rs | 2 +
oxide/crates/core/src/parser.rs | 711 ++++++++++++++----
src/featureFlags.js | 2 +-
src/util/normalizeConfig.js | 17 +-
tests/animations.test.js | 15 +-
tests/arbitrary-variants.test.js | 113 ++-
tests/basic-usage.test.js | 55 +-
tests/parse-candidate-strings.test.js | 423 +++++++++++
12 files changed, 1384 insertions(+), 222 deletions(-)
create mode 100644 oxide/crates/core/src/cursor.rs
create mode 100644 oxide/crates/core/src/fast_skip.rs
create mode 100644 tests/parse-candidate-strings.test.js
diff --git a/.editorconfig b/.editorconfig
index c6c8b3621..f6d83b7c7 100644
--- a/.editorconfig
+++ b/.editorconfig
@@ -7,3 +7,11 @@ end_of_line = lf
charset = utf-8
trim_trailing_whitespace = true
insert_final_newline = true
+
+[*.rs]
+indent_style = space
+indent_size = 4
+end_of_line = lf
+charset = utf-8
+trim_trailing_whitespace = true
+insert_final_newline = true
diff --git a/oxide/crates/core/benches/parse_candidates.rs b/oxide/crates/core/benches/parse_candidates.rs
index 55c919471..aacc1469e 100644
--- a/oxide/crates/core/benches/parse_candidates.rs
+++ b/oxide/crates/core/benches/parse_candidates.rs
@@ -31,6 +31,18 @@ pub fn criterion_benchmark(c: &mut Criterion) {
c.bench_function("parse_candidate_strings (real world)", |b| {
b.iter(|| parse(include_bytes!("./fixtures/template-499.html")))
});
+
+ let mut group = c.benchmark_group("sample-size-example");
+ group.sample_size(10);
+
+ group.bench_function("parse_candidate_strings (fast space skipping)", |b| {
+ let count = 10_000;
+ let crazy1 = format!("{}underline", " ".repeat(count));
+ let crazy2 = crazy1.repeat(count);
+ let crazy3 = crazy2.as_bytes();
+
+ b.iter(|| parse(black_box(crazy3)))
+ });
}
criterion_group!(benches, criterion_benchmark);
diff --git a/oxide/crates/core/src/cursor.rs b/oxide/crates/core/src/cursor.rs
new file mode 100644
index 000000000..0c422c6ff
--- /dev/null
+++ b/oxide/crates/core/src/cursor.rs
@@ -0,0 +1,159 @@
+use std::{ascii::escape_default, fmt::Display};
+
+#[derive(Debug, Clone)]
+pub struct Cursor<'a> {
+ // The input we're scanning
+ pub input: &'a [u8],
+
+ // The location of the cursor in the input
+ pub pos: usize,
+
+ /// Is the cursor at the start of the input
+ pub at_start: bool,
+
+ /// Is the cursor at the end of the input
+ pub at_end: bool,
+
+ /// The previously consumed character
+ /// If `at_start` is true, this will be NUL
+ pub prev: u8,
+
+ /// The current character
+ pub curr: u8,
+
+ /// The upcoming character (if any)
+ /// If `at_end` is true, this will be NUL
+ pub next: u8,
+}
+
+impl<'a> Cursor<'a> {
+ pub fn new(input: &'a [u8]) -> Self {
+ let mut cursor = Self {
+ input,
+ pos: 0,
+ at_start: true,
+ at_end: false,
+ prev: 0x00,
+ curr: 0x00,
+ next: 0x00,
+ };
+ cursor.move_to(0);
+ cursor
+ }
+
+ pub fn rewind_by(&mut self, amount: usize) {
+ self.move_to(self.pos.saturating_sub(amount));
+ }
+
+ pub fn advance_by(&mut self, amount: usize) {
+ self.move_to(self.pos.saturating_add(amount));
+ }
+
+ pub fn move_to(&mut self, pos: usize) {
+ let len = self.input.len();
+ let pos = pos.clamp(0, len);
+
+ self.pos = pos;
+ self.at_start = pos == 0;
+ self.at_end = pos + 1 >= len;
+
+ self.prev = if pos > 0 { self.input[pos - 1] } else { 0x00 };
+ self.curr = if pos < len { self.input[pos] } else { 0x00 };
+ self.next = if pos + 1 < len {
+ self.input[pos + 1]
+ } else {
+ 0x00
+ };
+ }
+}
+
+impl<'a> Display for Cursor<'a> {
+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+ let len = self.input.len().to_string();
+
+ let pos = format!("{: >len_count$}", self.pos, len_count = len.len());
+ write!(f, "{}/{} ", pos, len)?;
+
+ if self.at_start {
+ write!(f, "S ")?;
+ } else if self.at_end {
+ write!(f, "E ")?;
+ } else {
+ write!(f, "M ")?;
+ }
+
+ fn to_str(c: u8) -> String {
+ if c == 0x00 {
+ "NUL".into()
+ } else {
+ format!("{:?}", escape_default(c).to_string())
+ }
+ }
+
+ write!(
+ f,
+ "[{} {} {}]",
+ to_str(self.prev),
+ to_str(self.curr),
+ to_str(self.next)
+ )
+ }
+}
+
+#[cfg(test)]
+mod test {
+ use super::*;
+
+ #[test]
+ fn test_cursor() {
+ let mut cursor = Cursor::new(b"hello world");
+ assert_eq!(cursor.pos, 0);
+ assert!(cursor.at_start);
+ assert!(!cursor.at_end);
+ assert_eq!(cursor.prev, 0x00);
+ assert_eq!(cursor.curr, b'h');
+ assert_eq!(cursor.next, b'e');
+
+ cursor.advance_by(1);
+ assert_eq!(cursor.pos, 1);
+ assert!(!cursor.at_start);
+ assert!(!cursor.at_end);
+ assert_eq!(cursor.prev, b'h');
+ assert_eq!(cursor.curr, b'e');
+ assert_eq!(cursor.next, b'l');
+
+ // Advancing too far should stop at the end
+ cursor.advance_by(10);
+ assert_eq!(cursor.pos, 11);
+ assert!(!cursor.at_start);
+ assert!(cursor.at_end);
+ assert_eq!(cursor.prev, b'd');
+ assert_eq!(cursor.curr, 0x00);
+ assert_eq!(cursor.next, 0x00);
+
+ // Can't advance past the end
+ cursor.advance_by(1);
+ assert_eq!(cursor.pos, 11);
+ assert!(!cursor.at_start);
+ assert!(cursor.at_end);
+ assert_eq!(cursor.prev, b'd');
+ assert_eq!(cursor.curr, 0x00);
+ assert_eq!(cursor.next, 0x00);
+
+ cursor.rewind_by(1);
+ assert_eq!(cursor.pos, 10);
+ assert!(!cursor.at_start);
+ assert!(cursor.at_end);
+ assert_eq!(cursor.prev, b'l');
+ assert_eq!(cursor.curr, b'd');
+ assert_eq!(cursor.next, 0x00);
+
+ cursor.rewind_by(10);
+ assert_eq!(cursor.pos, 0);
+ assert!(cursor.at_start);
+ assert!(!cursor.at_end);
+ assert_eq!(cursor.prev, 0x00);
+ assert_eq!(cursor.curr, b'h');
+ assert_eq!(cursor.next, b'e');
+ }
+}
diff --git a/oxide/crates/core/src/fast_skip.rs b/oxide/crates/core/src/fast_skip.rs
new file mode 100644
index 000000000..488e61bd6
--- /dev/null
+++ b/oxide/crates/core/src/fast_skip.rs
@@ -0,0 +1,89 @@
+use crate::cursor::Cursor;
+
+const STRIDE: usize = 16;
+type Mask = [bool; STRIDE];
+
+#[inline(always)]
+pub fn fast_skip(cursor: &Cursor) -> Option {
+ // If we don't have enough bytes left to check then bail early
+ if cursor.pos + STRIDE >= cursor.input.len() {
+ return None;
+ }
+
+ if !cursor.curr.is_ascii_whitespace() {
+ return None;
+ }
+
+ let mut offset = 1;
+
+ // SAFETY: We've already checked (indirectly) that this index is valid
+ let remaining = unsafe { cursor.input.get_unchecked(cursor.pos..) };
+
+ // NOTE: This loop uses primitives designed to be auto-vectorized
+ // Do not change this loop without benchmarking the results
+ // And checking the generated assembly using godbolt.org
+ for (i, chunk) in remaining.chunks_exact(STRIDE).enumerate() {
+ let value = load(chunk);
+ let is_whitespace = is_ascii_whitespace(value);
+ let is_all_whitespace = all_true(is_whitespace);
+
+ if is_all_whitespace {
+ offset = (i + 1) * STRIDE;
+ } else {
+ break;
+ }
+ }
+
+ Some(cursor.pos + offset)
+}
+
+#[inline(always)]
+fn load(input: &[u8]) -> [u8; STRIDE] {
+ let mut value = [0u8; STRIDE];
+ value.copy_from_slice(input);
+ value
+}
+
+#[inline(always)]
+fn eq(input: [u8; STRIDE], val: u8) -> Mask {
+ let mut res = [false; STRIDE];
+ for n in 0..STRIDE {
+ res[n] = input[n] == val
+ }
+ res
+}
+
+#[inline(always)]
+fn or(a: [bool; STRIDE], b: [bool; STRIDE]) -> [bool; STRIDE] {
+ let mut res = [false; STRIDE];
+ for n in 0..STRIDE {
+ res[n] = a[n] | b[n];
+ }
+ res
+}
+
+#[inline(always)]
+fn all_true(a: [bool; STRIDE]) -> bool {
+ let mut res = true;
+ for item in a.iter().take(STRIDE) {
+ res &= item;
+ }
+ res
+}
+
+#[inline(always)]
+fn is_ascii_whitespace(value: [u8; STRIDE]) -> [bool; STRIDE] {
+ let whitespace_1 = eq(value, b'\t');
+ let whitespace_2 = eq(value, b'\n');
+ let whitespace_3 = eq(value, b'\x0C');
+ let whitespace_4 = eq(value, b'\r');
+ let whitespace_5 = eq(value, b' ');
+
+ or(
+ or(
+ or(or(whitespace_1, whitespace_2), whitespace_3),
+ whitespace_4,
+ ),
+ whitespace_5,
+ )
+}
diff --git a/oxide/crates/core/src/lib.rs b/oxide/crates/core/src/lib.rs
index 858e17a2d..9596c6855 100644
--- a/oxide/crates/core/src/lib.rs
+++ b/oxide/crates/core/src/lib.rs
@@ -9,6 +9,8 @@ use tracing::event;
use walkdir::WalkDir;
pub mod candidate;
+pub mod cursor;
+pub mod fast_skip;
pub mod glob;
pub mod location;
pub mod modifier;
diff --git a/oxide/crates/core/src/parser.rs b/oxide/crates/core/src/parser.rs
index acee57022..92740847b 100644
--- a/oxide/crates/core/src/parser.rs
+++ b/oxide/crates/core/src/parser.rs
@@ -2,6 +2,39 @@ use bstr::ByteSlice;
use fxhash::FxHashSet;
use tracing::trace;
+use crate::{cursor::Cursor, fast_skip::fast_skip};
+
+#[derive(Debug, PartialEq, Eq, Clone)]
+pub enum ParseAction<'a> {
+ Consume,
+ Skip,
+ RestartAt(usize),
+
+ SingleCandidate(&'a [u8]),
+ MultipleCandidates(Vec<&'a [u8]>),
+ Done,
+}
+
+#[derive(Debug, PartialEq, Eq, Clone)]
+pub enum Bracketing<'a> {
+ Included(&'a [u8]),
+ Wrapped(&'a [u8]),
+ None,
+}
+
+#[derive(Debug, PartialEq, Eq, Clone, Copy)]
+pub struct SplitCandidate<'a> {
+ variant: &'a [u8],
+ utility: &'a [u8],
+}
+
+#[derive(Debug, PartialEq, Eq, Clone, Copy)]
+pub enum ValidationResult {
+ Invalid,
+ Valid,
+ Restart,
+}
+
#[derive(Default)]
pub struct ExtractorOptions {
pub preserve_spaces_in_arbitrary: bool,
@@ -11,9 +44,7 @@ pub struct Extractor<'a> {
opts: ExtractorOptions,
input: &'a [u8],
- pos: usize,
-
- prev: u8,
+ cursor: Cursor<'a>,
idx_start: usize,
idx_end: usize,
@@ -24,24 +55,24 @@ pub struct Extractor<'a> {
in_candidate: bool,
in_escape: bool,
+ discard_next: bool,
+
quote_stack: Vec,
bracket_stack: Vec,
- // buffer: [Option<&'a [u8]>; 8],
}
impl<'a> Extractor<'a> {
pub fn all(input: &'a [u8], opts: ExtractorOptions) -> Vec<&'a [u8]> {
- Self::new(input, opts).collect()
+ Self::new(input, opts).flatten().collect()
}
pub fn unique(input: &'a [u8], opts: ExtractorOptions) -> FxHashSet<&'a [u8]> {
let mut candidates: FxHashSet<&[u8]> = Default::default();
candidates.reserve(100);
- candidates.extend(Self::new(input, opts));
+ candidates.extend(Self::new(input, opts).flatten());
candidates
}
- #[cfg(test)]
pub fn unique_ord(input: &'a [u8], opts: ExtractorOptions) -> Vec<&'a [u8]> {
// This is an inefficient way to get an ordered, unique
// list as a Vec but it is only meant for testing.
@@ -59,8 +90,7 @@ impl<'a> Extractor<'a> {
Self {
opts,
input,
- pos: 0,
- prev: 0,
+ cursor: Cursor::new(input),
idx_start: 0,
idx_end: 0,
@@ -70,10 +100,11 @@ impl<'a> Extractor<'a> {
in_candidate: false,
in_escape: false,
+ discard_next: false,
+
idx_last: input.len(),
quote_stack: Vec::with_capacity(8),
bracket_stack: Vec::with_capacity(8),
- // buffer: [None; 8],
}
}
}
@@ -86,126 +117,212 @@ impl<'a> Extractor<'a> {
}
#[inline(always)]
- fn get_current_candidate(&mut self) -> Option<&'a [u8]> {
- let candidate = &self.input[self.idx_start..=self.idx_end];
+ fn get_current_candidate(&mut self) -> ParseAction<'a> {
+ if self.discard_next {
+ return ParseAction::Skip;
+ }
- if Extractor::is_valid_candidate_string(candidate) {
- Some(candidate)
- } else {
- None
+ let mut candidate = &self.input[self.idx_start..=self.idx_end];
+
+ while !candidate.is_empty() {
+ match Extractor::is_valid_candidate_string(candidate) {
+ ValidationResult::Valid => return ParseAction::SingleCandidate(candidate),
+ ValidationResult::Restart => return ParseAction::RestartAt(self.idx_start + 1),
+ _ => {}
+ }
+
+ match candidate.split_last() {
+ // At this point the candidate is technically invalid, however it can be that it
+ // has a few dangling characters attached to it. For example, think about a
+ // JavaScript object:
+ //
+ // ```js
+ // { underline: true }
+ // ```
+ //
+ // The candidate at this point will be `underline:`, which is invalid. However, we
+ // can assume in this case that the `:` should not be there, and therefore we can
+ // try to slice it off and retry the validation.
+ Some((b':' | b'/' | b'.', head)) => {
+ candidate = head;
+ }
+
+ // It could also be that we have the candidate is nested inside of bracket or quote
+ // pairs. In this case we want to retrieve the inner part and try to validate that
+ // inner part instead. For example, in a JavaScript array:
+ //
+ // ```js
+ // let myClasses = ["underline"]
+ // ```
+ //
+ // The `underline` is nested inside of quotes and in square brackets. Let's try to
+ // get the inner part and validate that instead.
+ _ => match Self::slice_surrounding(candidate) {
+ Some(shorter) if shorter != candidate => {
+ candidate = shorter;
+ }
+ _ => break,
+ },
+ }
+ }
+
+ ParseAction::Consume
+ }
+
+ #[inline(always)]
+ fn split_candidate(candidate: &'a [u8]) -> SplitCandidate {
+ let mut brackets = 0;
+ let mut idx_end = 0;
+
+ for (n, c) in candidate.iter().enumerate() {
+ match c {
+ b'[' => brackets += 1,
+ b']' if brackets > 0 => brackets -= 1,
+ b':' if brackets == 0 => idx_end = n + 1,
+ _ => {}
+ }
+ }
+
+ SplitCandidate {
+ variant: &candidate[0..idx_end],
+ utility: &candidate[idx_end..],
}
}
#[inline(always)]
- fn is_valid_candidate_string(candidate: &'a [u8]) -> bool {
- let original_candidate = &candidate;
- let mut offset = 0;
+ fn contains_in_constrained(candidate: &'a [u8], bytes: Vec) -> bool {
+ let mut brackets = 0;
- // Some special cases that we can ignore while testing the validations.
- if candidate.starts_with(b"!-") {
+ for c in candidate {
+ match c {
+ b'[' => brackets += 1,
+ b']' if brackets > 0 => brackets -= 1,
+ _ if brackets == 0 && bytes.contains(c) => return true,
+ _ => {}
+ }
+ }
+
+ false
+ }
+
+ #[inline(always)]
+ fn is_valid_candidate_string(candidate: &'a [u8]) -> ValidationResult {
+ let split_candidate = Extractor::split_candidate(candidate);
+
+ let mut offset = 0;
+ let utility = &split_candidate.utility;
+ let original_utility = &utility;
+
+ // Some special cases that we can ignore while validating
+ if utility.starts_with(b"!-") {
offset += 2;
- } else if candidate.starts_with(b"!") || candidate.starts_with(b"-") {
+ } else if utility.starts_with(b"!") || utility.starts_with(b"-") {
offset += 1;
}
+ // These are allowed in arbitrary values and in variants but nowhere else
+ if Extractor::contains_in_constrained(utility, vec![b'<', b'>']) {
+ return ValidationResult::Restart;
+ }
+
// Pluck out the part that we are interested in.
- let candidate = &candidate[offset..];
+ let utility = &utility[offset..];
// Validations
// We should have _something_
- if candidate.is_empty() {
- return false;
+ if utility.is_empty() {
+ return ValidationResult::Invalid;
}
// = b'0' && candidate[0] <= b'9' && !candidate.contains(&b':') {
- return false;
+ if utility[0] >= b'0' && utility[0] <= b'9' && !utility.contains(&b':') {
+ return ValidationResult::Invalid;
}
// In case of an arbitrary property, we should have at least this structure: [a:b]
- if candidate.starts_with(b"[") && candidate.ends_with(b"]") {
+ if utility.starts_with(b"[") && utility.ends_with(b"]") {
// [a:b] is at least 5 characters long
- if candidate.len() < 5 {
- return false;
+ if utility.len() < 5 {
+ return ValidationResult::Invalid;
}
// Should contain a `:`
- if !candidate.contains(&b':') {
- return false;
+ if !utility.contains(&b':') {
+ return ValidationResult::Invalid;
}
// Now that we validated that the candidate is technically fine, let's ensure that it
// doesn't start with a `-` because that would make it invalid for arbitrary properties.
- if original_candidate.starts_with(b"-") || original_candidate.starts_with(b"!-") {
- return false;
+ if original_utility.starts_with(b"-") || original_utility.starts_with(b"!-") {
+ return ValidationResult::Invalid;
}
// The ':` must be preceded by a-Z0-9 because it represents a property name.
- let colon = candidate.find(":").unwrap();
+ let colon = utility.find(":").unwrap();
- if !candidate
+ if !utility
.chars()
.nth(colon - 1)
.map_or_else(|| false, |c| c.is_ascii_alphanumeric())
{
- return false;
+ return ValidationResult::Invalid;
}
}
// In case of an arbitrary property, we should not have a modifier of any kind
- if candidate.starts_with(b"[") {
+ if utility.starts_with(b"[") {
if let Some(first_non_escaped_closing_bracket_idx) =
- candidate.char_indices().position(|(s, _, other)| {
+ utility.char_indices().position(|(s, _, other)| {
other == ']'
- && candidate
+ && utility
.get((s - 1)..s)
.map(|c| !c.eq(b"\\"))
.unwrap_or(false)
})
{
- if let Some(next) = candidate
+ if let Some(next) = utility
.chars()
.nth(first_non_escaped_closing_bracket_idx + 1)
{
// `/` is the indicator for a modifier, this is not allowed after arbitrary
// properties
if next == '/' {
- return false;
+ return ValidationResult::Invalid;
}
}
}
}
- true
+ ValidationResult::Valid
}
#[inline(always)]
- fn parse_escaped(&mut self) -> bool {
+ fn parse_escaped(&mut self) -> ParseAction<'a> {
// If this character is escaped, we don't care about it.
// It gets consumed.
trace!("Escape::Consume");
self.in_escape = false;
- true
+ ParseAction::Consume
}
#[inline(always)]
- fn parse_arbitrary(&mut self, curr: u8, pos: usize) -> bool {
+ fn parse_arbitrary(&mut self) -> ParseAction<'a> {
// In this we could technically use memchr 6 times (then looped) to find the indexes / bounds of arbitrary valuesq
if self.in_escape {
return self.parse_escaped();
}
- match curr {
+ match self.cursor.curr {
b'\\' => {
// The `\` character is used to escape characters in arbitrary content _and_ to prevent the starting of arbitrary content
trace!("Arbitrary::Escape");
@@ -213,7 +330,7 @@ impl<'a> Extractor<'a> {
}
// Make sure the brackets are balanced
- b'[' => self.bracket_stack.push(curr),
+ b'[' => self.bracket_stack.push(self.cursor.curr),
b']' => match self.bracket_stack.last() {
// We've ended a nested bracket
Some(&last_bracket) if last_bracket == b'[' => {
@@ -222,12 +339,16 @@ impl<'a> Extractor<'a> {
// This is the last bracket meaning the end of arbitrary content
_ if !self.in_quotes() => {
+ if matches!(self.cursor.next, b'a'..=b'z' | b'A'..=b'Z' | b'0'..=b'9') {
+ return ParseAction::Consume;
+ }
+
trace!("Arbitrary::End\t");
self.in_arbitrary = false;
- if pos - self.idx_arbitrary_start == 1 {
+ if self.cursor.pos - self.idx_arbitrary_start == 1 {
// We have an empty arbitrary value, which is not allowed
- return false;
+ return ParseAction::Skip;
}
}
@@ -239,19 +360,22 @@ impl<'a> Extractor<'a> {
// These can "escape" the arbitrary value mode
// switching of `[` and `]` characters
b'"' | b'\'' | b'`' => match self.quote_stack.last() {
- Some(&last_quote) if last_quote == curr => {
+ Some(&last_quote) if last_quote == self.cursor.curr => {
trace!("Quote::End\t");
self.quote_stack.pop();
}
_ => {
trace!("Quote::Start\t");
- self.quote_stack.push(curr);
+ self.quote_stack.push(self.cursor.curr);
}
},
b' ' if !self.opts.preserve_spaces_in_arbitrary => {
trace!("Arbitrary::SkipAndEndEarly\t");
- return false;
+
+ // Restart the parser ahead of the arbitrary value
+ // It may pick up more candidates
+ return ParseAction::RestartAt(self.idx_arbitrary_start + 1);
}
// Arbitrary values allow any character inside them
@@ -262,51 +386,55 @@ impl<'a> Extractor<'a> {
}
}
- true
+ ParseAction::Consume
}
#[inline(always)]
- fn parse_start(&mut self, curr: u8, pos: usize) -> bool {
- match curr {
+ fn parse_start(&mut self) -> ParseAction<'a> {
+ match self.cursor.curr {
// Enter arbitrary value mode
b'[' => {
trace!("Arbitrary::Start\t");
self.in_arbitrary = true;
- self.idx_arbitrary_start = pos;
+ self.idx_arbitrary_start = self.cursor.pos;
- true
+ ParseAction::Consume
}
// Allowed first characters.
- b'@' | b'!' | b'-' | b'<' | b'0'..=b'9' | b'a'..=b'z' | b'A'..=b'Z' => {
+ b'@' | b'!' | b'-' | b'<' | b'>' | b'0'..=b'9' | b'a'..=b'z' | b'A'..=b'Z' => {
// TODO: A bunch of characters that we currently support but maybe we only want it behind
// a flag. E.g.: '' | '$' | '^' | '_'
+ // | '$' | '^' | '_'
+
+ // When the new candidate is preceeded by a `:`, then we want to keep parsing, but
+ // throw away the full candidate because it can not be a valid candidate at the end
+ // of the day.
+ if self.cursor.prev == b':' {
+ self.discard_next = true;
+ }
trace!("Candidate::Start\t");
- true
+ ParseAction::Consume
}
- _ => false,
+ _ => ParseAction::Skip,
}
}
#[inline(always)]
- fn parse_continue(&mut self, prev: u8, curr: u8, pos: usize) -> bool {
- match curr {
+ fn parse_continue(&mut self) -> ParseAction<'a> {
+ match self.cursor.curr {
// Enter arbitrary value mode
- b'[' if prev == b'@'
- || prev == b'-'
- || prev == b' '
- || prev == b':' // Variant separator
- || prev == b'/' // Modifier separator
- || prev == b'!'
- || prev == b'\0' =>
+ b'[' if matches!(
+ self.cursor.prev,
+ b'@' | b'-' | b' ' | b':' | b'/' | b'!' | b'\0'
+ ) =>
{
trace!("Arbitrary::Start\t");
self.in_arbitrary = true;
- self.idx_arbitrary_start = pos;
+ self.idx_arbitrary_start = self.cursor.pos;
}
// Can't enter arbitrary value mode
@@ -314,154 +442,305 @@ impl<'a> Extractor<'a> {
b'[' => {
trace!("Arbitrary::Skip_Start\t");
- return false;
+ return ParseAction::Skip;
+ }
+
+ // A % can only appear at the end of the candidate itself. It can also only be after a
+ // digit 0-9. This covers the following cases:
+ // - from-15%
+ b'%' if self.cursor.prev.is_ascii_digit() => {
+ return match (self.cursor.at_end, self.cursor.next) {
+ // End of string == end of candidate == okay
+ (true, _) => ParseAction::Consume,
+
+ // Looks like the end of a candidate == okay
+ (_, b' ' | b'\'' | b'"' | b'`') => ParseAction::Consume,
+
+ // Otherwise, not a valid character in a candidate
+ _ => ParseAction::Skip,
+ };
+ }
+ b'%' => return ParseAction::Skip,
+
+ // < and > can only be part of a variant and only be the first or last character
+ b'<' | b'>' => {
+ // Can only be the first or last character
+ // E.g.:
+ // - :underline
+ // ^
+ if self.cursor.pos == self.idx_start || self.cursor.pos == self.idx_last {
+ trace!("Candidate::Consume\t");
+ }
+ // If it is in the middle, it can only be part of a stacked variant
+ // - dark::underline
+ // ^
+ else if self.cursor.prev == b':' || self.cursor.next == b':' {
+ trace!("Candidate::Consume\t");
+ } else {
+ return ParseAction::Skip;
+ }
}
// Allowed characters in the candidate itself
// None of these can come after a closing bracket `]`
- b'a'..=b'z'
- | b'A'..=b'Z'
- | b'0'..=b'9'
- | b'-'
- | b'_'
- | b'('
- | b')'
- | b'!'
- | b'@'
- | b'%'
- if prev != b']' =>
+ b'a'..=b'z' | b'A'..=b'Z' | b'0'..=b'9' | b'-' | b'_' | b'!' | b'@'
+ if self.cursor.prev != b']' =>
{
/* TODO: The `b'@'` is necessary for custom separators like _@, maybe we can handle this in a better way... */
trace!("Candidate::Consume\t");
}
+ // A dot (.) can only appear in the candidate itself (not the arbitrary part), if the previous
+ // and next characters are both digits. This covers the following cases:
+ // - p-1.5
+ b'.' if self.cursor.prev.is_ascii_digit() => match self.cursor.next {
+ next if next.is_ascii_digit() => {
+ trace!("Candidate::Consume\t");
+ }
+ _ => return ParseAction::Skip,
+ },
+
// Allowed characters in the candidate itself
// These MUST NOT appear at the end of the candidate
- b'/' | b':' | b'.' if pos + 1 < self.idx_last => {
+ b'/' | b':' if !self.cursor.at_end => {
trace!("Candidate::Consume\t");
}
- _ => return false,
+ _ => return ParseAction::Skip,
}
- true
+ ParseAction::Consume
}
#[inline(always)]
- fn can_be_candidate(&mut self, c: u8) -> bool {
+ fn can_be_candidate(&mut self) -> bool {
self.in_candidate
&& !self.in_arbitrary
- && (0..=127).contains(&c)
+ && (0..=127).contains(&self.cursor.curr)
&& (self.idx_start == 0 || self.input[self.idx_start - 1] <= 127)
}
#[inline(always)]
- fn handle_skip(&mut self, pos: usize) {
+ fn handle_skip(&mut self) {
// In all other cases, we skip characters and reset everything so we can make new candidates
trace!("Characters::Skip\t");
- self.idx_start = pos;
- self.idx_end = pos;
+ self.idx_start = self.cursor.pos;
+ self.idx_end = self.cursor.pos;
self.in_candidate = false;
self.in_arbitrary = false;
self.in_escape = false;
}
#[inline(always)]
- fn parse_char(&mut self, prev: u8, curr: u8, pos: usize) -> bool {
+ fn parse_char(&mut self) -> ParseAction<'a> {
if self.in_arbitrary {
- self.parse_arbitrary(curr, pos)
+ self.parse_arbitrary()
} else if self.in_candidate {
- self.parse_continue(prev, curr, pos)
- } else if self.parse_start(curr, pos) {
+ self.parse_continue()
+ } else if self.parse_start() == ParseAction::Consume {
self.in_candidate = true;
- self.idx_start = pos;
- self.idx_end = pos;
+ self.idx_start = self.cursor.pos;
+ self.idx_end = self.cursor.pos;
- true
+ ParseAction::Consume
} else {
- false
+ ParseAction::Skip
}
}
#[inline(always)]
- fn yield_candidate(&mut self, pos: usize, curr: u8, did_consume: bool) -> Option<&'a [u8]> {
- // If we're still consuming characters, we keep going
- // Only exception is if we've hit the end of the input
- if did_consume && pos + 1 < self.idx_last {
- return None;
- }
-
- let candidate = if self.can_be_candidate(curr) {
+ fn yield_candidate(&mut self) -> ParseAction<'a> {
+ if self.can_be_candidate() {
self.get_current_candidate()
} else {
- None
- };
-
- self.handle_skip(pos);
-
- candidate
+ ParseAction::Consume
+ }
}
#[inline(always)]
- fn read(&mut self) -> (usize, u8) {
- if self.pos == self.idx_last {
- return (usize::MAX, 0);
- }
+ fn restart(&mut self, pos: usize) {
+ trace!("Parser::Restart\t{}", pos);
- let r = (self.pos, self.input[self.pos]);
- self.pos += 1;
+ self.idx_start = pos;
+ self.idx_end = pos;
+ self.idx_arbitrary_start = 0;
- r
+ self.in_arbitrary = false;
+ self.in_candidate = false;
+ self.in_escape = false;
+
+ self.discard_next = false;
+
+ self.quote_stack.clear();
+ self.bracket_stack.clear();
+ self.cursor.move_to(pos);
}
#[inline(always)]
- fn parse_and_yield(&mut self) -> Option