From ff0a5e3f7bec87b22db6d9e3bc576c1c9ff88d36 Mon Sep 17 00:00:00 2001 From: Jordan Pittman Date: Sat, 12 Oct 2024 18:46:33 -0400 Subject: [PATCH] wip: decode --- crates/core/src/util/decode.rs | 122 ++++++++++++++++++++++++++ crates/core/src/util/math.rs | 154 +++++++++++++++++++++++++++++++++ crates/core/src/util/mod.rs | 8 ++ 3 files changed, 284 insertions(+) create mode 100644 crates/core/src/util/decode.rs create mode 100644 crates/core/src/util/math.rs diff --git a/crates/core/src/util/decode.rs b/crates/core/src/util/decode.rs new file mode 100644 index 000000000..3501011d4 --- /dev/null +++ b/crates/core/src/util/decode.rs @@ -0,0 +1,122 @@ +// import { addWhitespaceAroundMathOperators } from './math-operators' + +use std::mem; + +pub fn decode_arbitrary_value(input: &[u8]) -> Vec { + // We do not want to normalize anything inside of a url() because if we + // replace `_` with ` `, then it will very likely break the url. + if input.starts_with(b"url(") { + return input.to_vec() + } + + let input = convert_underscores_to_whitespace(input); + // let input = addWhitespaceAroundMathOperators(input); + + return input.to_vec(); +} + +/// Convert `_` to ` ` unless escaped (`\_`) in which case they +/// should be converted to `_` instead. +pub fn convert_underscores_to_whitespace(input: &[u8]) -> Vec { + let mut result = Vec::::with_capacity(input.len()); + + let input = write_decoded_8(input, &mut result); + write_decoded_scalar(input, &mut result); + + result +} + +pub fn write_decoded_8<'a, 'b>(input: &'a [u8], result: &'b mut Vec) -> &'a [u8] { + const CHUNK_SIZE: usize = mem::size_of::(); + const NUL_8: [u8; CHUNK_SIZE] = [0x00; CHUNK_SIZE]; + const SPACE_8: [u8; CHUNK_SIZE] = [b' '; CHUNK_SIZE]; + const ESCAPE_8: [u8; CHUNK_SIZE] = [b'\\'; CHUNK_SIZE]; + const UNDERSCORE_8: [u8; CHUNK_SIZE] = [b'_'; CHUNK_SIZE]; + + let mut chunks = input.chunks_exact(CHUNK_SIZE); + + while let Some(chunk) = chunks.next() { + let mut chunk: [u8; CHUNK_SIZE] = chunk.try_into().unwrap(); + let mut chunk: u64 = u64::from_ne_bytes(chunk); + + let mut is_escape = [false; CHUNK_SIZE]; + for j in 0..CHUNK_SIZE { + is_escape[j] = chunk[j] == b'\\'; + } + + let mut is_underscore = [false; CHUNK_SIZE]; + for j in 0..CHUNK_SIZE { + is_underscore[j] = chunk[j] == b'_'; + } + + // Replace underscores with spaces in the chunk + for j in 0..CHUNK_SIZE { + chunk[j] = if is_underscore[j] { + SPACE_8[j] + } else { + chunk[j] + }; + } + + // Replace escaped underscores with underscores + for j in 0..(CHUNK_SIZE - 1) { + chunk[j] = if is_escape[j] && is_underscore[j + 1] { + UNDERSCORE_8[j] + } else { + chunk[j] + }; + } + + // Replace escapes with NUL bytes + for j in 0..CHUNK_SIZE { + chunk[j] = if is_escape[j] { + NUL_8[j] + } else { + chunk[j] + }; + } + + result.extend(&chunk); + } + + return chunks.remainder(); +} + + +pub fn write_decoded_scalar(input: &[u8], result: &mut Vec) { + let mut i = 0; + + while i < input.len() { + match input[i] { + b'\\' => { + if i + 1 == input.len() { + // We've hit the end of the string and there's no character to escape + result.extend(b"\\"); + } else if input[i + 1] == b'_' { + // We've hit an escaped underscore + result.extend(b"_"); + } else { + // We've hit an "escaped" character that isn't an underscore + // which means its not actually escaped and should be treated + // as a literal character. Since we've already read the next + // character, we'll just write both of them out here. + result.extend(&input[i..i+1]); + } + + i += 2; + }, + + b'_' => { + result.push(b' '); + i += 1; + }, + + _ => { + result.push(input[i]); + i += 1; + } + } + } +} + +// a\b\c diff --git a/crates/core/src/util/math.rs b/crates/core/src/util/math.rs new file mode 100644 index 000000000..58dba1625 --- /dev/null +++ b/crates/core/src/util/math.rs @@ -0,0 +1,154 @@ +// const mathFunctions = [ +// 'calc', +// 'min', +// 'max', +// 'clamp', +// 'mod', +// 'rem', +// 'sin', +// 'cos', +// 'tan', +// 'asin', +// 'acos', +// 'atan', +// 'atan2', +// 'pow', +// 'sqrt', +// 'hypot', +// 'log', +// 'exp', +// 'round', +// ] + +// export function hasMathFn(input: string) { +// return input.indexOf('(') !== -1 && mathFunctions.some((fn) => input.includes(`${fn}(`)) +// } + +// export function addWhitespaceAroundMathOperators(input: string) { +// // There's definitely no functions in the input, so bail early +// if (input.indexOf('(') === -1) { +// return input +// } + +// // Bail early if there are no math functions in the input +// if (!mathFunctions.some((fn) => input.includes(fn))) { +// return input +// } + +// let result = '' +// let formattable: boolean[] = [] + +// for (let i = 0; i < input.length; i++) { +// let char = input[i] + +// // Determine if we're inside a math function +// if (char === '(') { +// result += char + +// // Scan backwards to determine the function name. This assumes math +// // functions are named with lowercase alphanumeric characters. +// let start = i + +// for (let j = i - 1; j >= 0; j--) { +// let inner = input.charCodeAt(j) + +// if (inner >= 48 && inner <= 57) { +// start = j // 0-9 +// } else if (inner >= 97 && inner <= 122) { +// start = j // a-z +// } else { +// break +// } +// } + +// let fn = input.slice(start, i) + +// // This is a known math function so start formatting +// if (mathFunctions.includes(fn)) { +// formattable.unshift(true) +// continue +// } + +// // We've encountered nested parens inside a math function, record that and +// // keep formatting until we've closed all parens. +// else if (formattable[0] && fn === '') { +// formattable.unshift(true) +// continue +// } + +// // This is not a known math function so don't format it +// formattable.unshift(false) +// continue +// } + +// // We've exited the function so format according to the parent function's +// // type. +// else if (char === ')') { +// result += char +// formattable.shift() +// } + +// // Add spaces after commas in math functions +// else if (char === ',' && formattable[0]) { +// result += `, ` +// continue +// } + +// // Skip over consecutive whitespace +// else if (char === ' ' && formattable[0] && result[result.length - 1] === ' ') { +// continue +// } + +// // Add whitespace around operators inside math functions +// else if ((char === '+' || char === '*' || char === '/' || char === '-') && formattable[0]) { +// let trimmed = result.trimEnd() +// let prev = trimmed[trimmed.length - 1] + +// // If we're preceded by an operator don't add spaces +// if (prev === '+' || prev === '*' || prev === '/' || prev === '-') { +// result += char +// continue +// } + +// // If we're at the beginning of an argument don't add spaces +// else if (prev === '(' || prev === ',') { +// result += char +// continue +// } + +// // Add spaces only after the operator if we already have spaces before it +// else if (input[i - 1] === ' ') { +// result += `${char} ` +// } + +// // Add spaces around the operator +// else { +// result += ` ${char} ` +// } +// } + +// // Skip over `to-zero` when in a math function. +// // +// // This is specifically to handle this value in the round(…) function: +// // +// // ``` +// // round(to-zero, 1px) +// // ^^^^^^^ +// // ``` +// // +// // This is because the first argument is optionally a keyword and `to-zero` +// // contains a hyphen and we want to avoid adding spaces inside it. +// else if (formattable[0] && input.startsWith('to-zero', i)) { +// let start = i +// i += 7 +// result += input.slice(start, i + 1) +// } + +// // Handle all other characters +// else { +// result += char +// } +// } + +// return result +// } diff --git a/crates/core/src/util/mod.rs b/crates/core/src/util/mod.rs index 82934ee25..a5b9de7d6 100644 --- a/crates/core/src/util/mod.rs +++ b/crates/core/src/util/mod.rs @@ -1,9 +1,17 @@ +mod decode; mod escape; mod fast_stack; mod gurantee; +mod math; mod segment; mod throughput; +#[allow(unused_imports)] +pub use crate::util::decode::*; + +#[allow(unused_imports)] +pub use crate::util::math::*; + #[allow(unused_imports)] pub use crate::util::escape::*;