wip: decode
This commit is contained in:
parent
8c5aec02c4
commit
ff0a5e3f7b
3 changed files with 284 additions and 0 deletions
122
crates/core/src/util/decode.rs
Normal file
122
crates/core/src/util/decode.rs
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
// import { addWhitespaceAroundMathOperators } from './math-operators'
|
||||
|
||||
use std::mem;
|
||||
|
||||
pub fn decode_arbitrary_value(input: &[u8]) -> Vec<u8> {
|
||||
// We do not want to normalize anything inside of a url() because if we
|
||||
// replace `_` with ` `, then it will very likely break the url.
|
||||
if input.starts_with(b"url(") {
|
||||
return input.to_vec()
|
||||
}
|
||||
|
||||
let input = convert_underscores_to_whitespace(input);
|
||||
// let input = addWhitespaceAroundMathOperators(input);
|
||||
|
||||
return input.to_vec();
|
||||
}
|
||||
|
||||
/// Convert `_` to ` ` unless escaped (`\_`) in which case they
|
||||
/// should be converted to `_` instead.
|
||||
pub fn convert_underscores_to_whitespace(input: &[u8]) -> Vec<u8> {
|
||||
let mut result = Vec::<u8>::with_capacity(input.len());
|
||||
|
||||
let input = write_decoded_8(input, &mut result);
|
||||
write_decoded_scalar(input, &mut result);
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
pub fn write_decoded_8<'a, 'b>(input: &'a [u8], result: &'b mut Vec<u8>) -> &'a [u8] {
|
||||
const CHUNK_SIZE: usize = mem::size_of::<u64>();
|
||||
const NUL_8: [u8; CHUNK_SIZE] = [0x00; CHUNK_SIZE];
|
||||
const SPACE_8: [u8; CHUNK_SIZE] = [b' '; CHUNK_SIZE];
|
||||
const ESCAPE_8: [u8; CHUNK_SIZE] = [b'\\'; CHUNK_SIZE];
|
||||
const UNDERSCORE_8: [u8; CHUNK_SIZE] = [b'_'; CHUNK_SIZE];
|
||||
|
||||
let mut chunks = input.chunks_exact(CHUNK_SIZE);
|
||||
|
||||
while let Some(chunk) = chunks.next() {
|
||||
let mut chunk: [u8; CHUNK_SIZE] = chunk.try_into().unwrap();
|
||||
let mut chunk: u64 = u64::from_ne_bytes(chunk);
|
||||
|
||||
let mut is_escape = [false; CHUNK_SIZE];
|
||||
for j in 0..CHUNK_SIZE {
|
||||
is_escape[j] = chunk[j] == b'\\';
|
||||
}
|
||||
|
||||
let mut is_underscore = [false; CHUNK_SIZE];
|
||||
for j in 0..CHUNK_SIZE {
|
||||
is_underscore[j] = chunk[j] == b'_';
|
||||
}
|
||||
|
||||
// Replace underscores with spaces in the chunk
|
||||
for j in 0..CHUNK_SIZE {
|
||||
chunk[j] = if is_underscore[j] {
|
||||
SPACE_8[j]
|
||||
} else {
|
||||
chunk[j]
|
||||
};
|
||||
}
|
||||
|
||||
// Replace escaped underscores with underscores
|
||||
for j in 0..(CHUNK_SIZE - 1) {
|
||||
chunk[j] = if is_escape[j] && is_underscore[j + 1] {
|
||||
UNDERSCORE_8[j]
|
||||
} else {
|
||||
chunk[j]
|
||||
};
|
||||
}
|
||||
|
||||
// Replace escapes with NUL bytes
|
||||
for j in 0..CHUNK_SIZE {
|
||||
chunk[j] = if is_escape[j] {
|
||||
NUL_8[j]
|
||||
} else {
|
||||
chunk[j]
|
||||
};
|
||||
}
|
||||
|
||||
result.extend(&chunk);
|
||||
}
|
||||
|
||||
return chunks.remainder();
|
||||
}
|
||||
|
||||
|
||||
pub fn write_decoded_scalar(input: &[u8], result: &mut Vec<u8>) {
|
||||
let mut i = 0;
|
||||
|
||||
while i < input.len() {
|
||||
match input[i] {
|
||||
b'\\' => {
|
||||
if i + 1 == input.len() {
|
||||
// We've hit the end of the string and there's no character to escape
|
||||
result.extend(b"\\");
|
||||
} else if input[i + 1] == b'_' {
|
||||
// We've hit an escaped underscore
|
||||
result.extend(b"_");
|
||||
} else {
|
||||
// We've hit an "escaped" character that isn't an underscore
|
||||
// which means its not actually escaped and should be treated
|
||||
// as a literal character. Since we've already read the next
|
||||
// character, we'll just write both of them out here.
|
||||
result.extend(&input[i..i+1]);
|
||||
}
|
||||
|
||||
i += 2;
|
||||
},
|
||||
|
||||
b'_' => {
|
||||
result.push(b' ');
|
||||
i += 1;
|
||||
},
|
||||
|
||||
_ => {
|
||||
result.push(input[i]);
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// a\b\c
|
||||
154
crates/core/src/util/math.rs
Normal file
154
crates/core/src/util/math.rs
Normal file
|
|
@ -0,0 +1,154 @@
|
|||
// const mathFunctions = [
|
||||
// 'calc',
|
||||
// 'min',
|
||||
// 'max',
|
||||
// 'clamp',
|
||||
// 'mod',
|
||||
// 'rem',
|
||||
// 'sin',
|
||||
// 'cos',
|
||||
// 'tan',
|
||||
// 'asin',
|
||||
// 'acos',
|
||||
// 'atan',
|
||||
// 'atan2',
|
||||
// 'pow',
|
||||
// 'sqrt',
|
||||
// 'hypot',
|
||||
// 'log',
|
||||
// 'exp',
|
||||
// 'round',
|
||||
// ]
|
||||
|
||||
// export function hasMathFn(input: string) {
|
||||
// return input.indexOf('(') !== -1 && mathFunctions.some((fn) => input.includes(`${fn}(`))
|
||||
// }
|
||||
|
||||
// export function addWhitespaceAroundMathOperators(input: string) {
|
||||
// // There's definitely no functions in the input, so bail early
|
||||
// if (input.indexOf('(') === -1) {
|
||||
// return input
|
||||
// }
|
||||
|
||||
// // Bail early if there are no math functions in the input
|
||||
// if (!mathFunctions.some((fn) => input.includes(fn))) {
|
||||
// return input
|
||||
// }
|
||||
|
||||
// let result = ''
|
||||
// let formattable: boolean[] = []
|
||||
|
||||
// for (let i = 0; i < input.length; i++) {
|
||||
// let char = input[i]
|
||||
|
||||
// // Determine if we're inside a math function
|
||||
// if (char === '(') {
|
||||
// result += char
|
||||
|
||||
// // Scan backwards to determine the function name. This assumes math
|
||||
// // functions are named with lowercase alphanumeric characters.
|
||||
// let start = i
|
||||
|
||||
// for (let j = i - 1; j >= 0; j--) {
|
||||
// let inner = input.charCodeAt(j)
|
||||
|
||||
// if (inner >= 48 && inner <= 57) {
|
||||
// start = j // 0-9
|
||||
// } else if (inner >= 97 && inner <= 122) {
|
||||
// start = j // a-z
|
||||
// } else {
|
||||
// break
|
||||
// }
|
||||
// }
|
||||
|
||||
// let fn = input.slice(start, i)
|
||||
|
||||
// // This is a known math function so start formatting
|
||||
// if (mathFunctions.includes(fn)) {
|
||||
// formattable.unshift(true)
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // We've encountered nested parens inside a math function, record that and
|
||||
// // keep formatting until we've closed all parens.
|
||||
// else if (formattable[0] && fn === '') {
|
||||
// formattable.unshift(true)
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // This is not a known math function so don't format it
|
||||
// formattable.unshift(false)
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // We've exited the function so format according to the parent function's
|
||||
// // type.
|
||||
// else if (char === ')') {
|
||||
// result += char
|
||||
// formattable.shift()
|
||||
// }
|
||||
|
||||
// // Add spaces after commas in math functions
|
||||
// else if (char === ',' && formattable[0]) {
|
||||
// result += `, `
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // Skip over consecutive whitespace
|
||||
// else if (char === ' ' && formattable[0] && result[result.length - 1] === ' ') {
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // Add whitespace around operators inside math functions
|
||||
// else if ((char === '+' || char === '*' || char === '/' || char === '-') && formattable[0]) {
|
||||
// let trimmed = result.trimEnd()
|
||||
// let prev = trimmed[trimmed.length - 1]
|
||||
|
||||
// // If we're preceded by an operator don't add spaces
|
||||
// if (prev === '+' || prev === '*' || prev === '/' || prev === '-') {
|
||||
// result += char
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // If we're at the beginning of an argument don't add spaces
|
||||
// else if (prev === '(' || prev === ',') {
|
||||
// result += char
|
||||
// continue
|
||||
// }
|
||||
|
||||
// // Add spaces only after the operator if we already have spaces before it
|
||||
// else if (input[i - 1] === ' ') {
|
||||
// result += `${char} `
|
||||
// }
|
||||
|
||||
// // Add spaces around the operator
|
||||
// else {
|
||||
// result += ` ${char} `
|
||||
// }
|
||||
// }
|
||||
|
||||
// // Skip over `to-zero` when in a math function.
|
||||
// //
|
||||
// // This is specifically to handle this value in the round(…) function:
|
||||
// //
|
||||
// // ```
|
||||
// // round(to-zero, 1px)
|
||||
// // ^^^^^^^
|
||||
// // ```
|
||||
// //
|
||||
// // This is because the first argument is optionally a keyword and `to-zero`
|
||||
// // contains a hyphen and we want to avoid adding spaces inside it.
|
||||
// else if (formattable[0] && input.startsWith('to-zero', i)) {
|
||||
// let start = i
|
||||
// i += 7
|
||||
// result += input.slice(start, i + 1)
|
||||
// }
|
||||
|
||||
// // Handle all other characters
|
||||
// else {
|
||||
// result += char
|
||||
// }
|
||||
// }
|
||||
|
||||
// return result
|
||||
// }
|
||||
|
|
@ -1,9 +1,17 @@
|
|||
mod decode;
|
||||
mod escape;
|
||||
mod fast_stack;
|
||||
mod gurantee;
|
||||
mod math;
|
||||
mod segment;
|
||||
mod throughput;
|
||||
|
||||
#[allow(unused_imports)]
|
||||
pub use crate::util::decode::*;
|
||||
|
||||
#[allow(unused_imports)]
|
||||
pub use crate::util::math::*;
|
||||
|
||||
#[allow(unused_imports)]
|
||||
pub use crate::util::escape::*;
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue