/src/rust-cssparser/src/parser.rs
Line | Count | Source |
1 | | /* This Source Code Form is subject to the terms of the Mozilla Public |
2 | | * License, v. 2.0. If a copy of the MPL was not distributed with this |
3 | | * file, You can obtain one at http://mozilla.org/MPL/2.0/. */ |
4 | | |
5 | | use crate::cow_rc_str::CowRcStr; |
6 | | use crate::tokenizer::{SeenStatus, SourceLocation, SourcePosition, Token}; |
7 | | use smallvec::SmallVec; |
8 | | use std::fmt; |
9 | | use std::ops::BitOr; |
10 | | |
11 | | /// A capture of the internal state of a `Parser` (including the position within the input), |
12 | | /// obtained from the `Parser::position` method. |
13 | | /// |
14 | | /// Can be used with the `Parser::reset` method to restore that state. |
15 | | /// Should only be used with the `Parser` instance it came from. |
16 | | #[derive(Debug, Clone, Default)] |
17 | | pub struct ParserState { |
18 | | pub(crate) position: usize, |
19 | | pub(crate) current_line_start_position: usize, |
20 | | pub(crate) current_line_number: u32, |
21 | | pub(crate) at_start_of: Option<BlockType>, |
22 | | } |
23 | | |
24 | | impl ParserState { |
25 | | /// The position from the start of the input, counted in UTF-8 bytes. |
26 | | #[inline] |
27 | 483k | pub fn position(&self) -> SourcePosition { |
28 | 483k | SourcePosition(self.position) |
29 | 483k | } |
30 | | |
31 | | /// The line number and column number |
32 | | #[inline] |
33 | | pub fn source_location(&self) -> SourceLocation { |
34 | | SourceLocation { |
35 | | line: self.current_line_number, |
36 | | column: (self.position - self.current_line_start_position + 1) as u32, |
37 | | } |
38 | | } |
39 | | } |
40 | | |
41 | | /// When parsing until a given token, sometimes the caller knows that parsing is going to restart |
42 | | /// at some earlier point, and consuming until we find a top level delimiter is just wasted work. |
43 | | /// |
44 | | /// In that case, callers can pass ParseUntilErrorBehavior::Stop to avoid doing all that wasted |
45 | | /// work. |
46 | | /// |
47 | | /// This is important for things like CSS nesting, where something like: |
48 | | /// |
49 | | /// foo:is(..) { |
50 | | /// ... |
51 | | /// } |
52 | | /// |
53 | | /// Would need to scan the whole {} block to find a semicolon, only for parsing getting restarted |
54 | | /// as a qualified rule later. |
55 | | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
56 | | pub enum ParseUntilErrorBehavior { |
57 | | /// Consume until we see the relevant delimiter or the end of the stream. |
58 | | Consume, |
59 | | /// Eagerly error. |
60 | | Stop, |
61 | | } |
62 | | |
63 | | /// Details about a `BasicParseError` |
64 | | #[derive(Clone, Debug, PartialEq)] |
65 | | pub enum BasicParseErrorKind { |
66 | | /// An unexpected token was encountered. |
67 | | /// |
68 | | /// The token itself is deliberately not stored: it made this enum 32 bytes, |
69 | | /// which pushed `Result<&Token, BasicParseError>` (returned from every token |
70 | | /// fetch) to 40 bytes and therefore out of registers and into memory. |
71 | | /// Callers that want to name the token can recover it from the source text |
72 | | /// they already carry for the error message. |
73 | | UnexpectedToken, |
74 | | /// The end of the input was encountered unexpectedly. |
75 | | EndOfInput, |
76 | | /// An `@` rule was encountered that was invalid. See `UnexpectedToken` for |
77 | | /// why the rule name is not stored. |
78 | | AtRuleInvalid, |
79 | | /// The body of an '@' rule was invalid. |
80 | | AtRuleBodyInvalid, |
81 | | /// A qualified rule was encountered that was invalid. |
82 | | QualifiedRuleInvalid, |
83 | | /// We've gone over the nesting limit. |
84 | | TooManyNestedBlocks, |
85 | | } |
86 | | |
87 | | impl fmt::Display for BasicParseErrorKind { |
88 | 0 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
89 | 0 | match self { |
90 | | BasicParseErrorKind::TooManyNestedBlocks => { |
91 | 0 | write!(f, "nesting block limit reached") |
92 | | } |
93 | 0 | BasicParseErrorKind::UnexpectedToken => write!(f, "unexpected token"), |
94 | 0 | BasicParseErrorKind::EndOfInput => write!(f, "unexpected end of input"), |
95 | 0 | BasicParseErrorKind::AtRuleInvalid => write!(f, "invalid @ rule encountered"), |
96 | 0 | BasicParseErrorKind::AtRuleBodyInvalid => write!(f, "invalid @ rule body encountered"), |
97 | | BasicParseErrorKind::QualifiedRuleInvalid => { |
98 | 0 | write!(f, "invalid qualified rule encountered") |
99 | | } |
100 | | } |
101 | 0 | } |
102 | | } |
103 | | |
104 | | /// The fundamental parsing errors that can be triggered by built-in parsing routines. |
105 | | #[derive(Clone, Debug, PartialEq)] |
106 | | pub struct BasicParseError { |
107 | | /// Details of this error |
108 | | pub kind: BasicParseErrorKind, |
109 | | } |
110 | | |
111 | | impl BasicParseError { |
112 | | /// Create a new BasicParseError of the given kind. |
113 | | #[inline] |
114 | 744k | pub fn new(kind: BasicParseErrorKind) -> Self { |
115 | 744k | Self { kind } |
116 | 744k | } Unexecuted instantiation: <cssparser::parser::BasicParseError>::new <cssparser::parser::BasicParseError>::new Line | Count | Source | 114 | 744k | pub fn new(kind: BasicParseErrorKind) -> Self { | 115 | 744k | Self { kind } | 116 | 744k | } |
|
117 | | |
118 | | /// Create a new BasicParseError for an unexpected token. |
119 | | #[inline] |
120 | 0 | pub fn unexpected_token() -> Self { |
121 | 0 | Self::new(BasicParseErrorKind::UnexpectedToken) |
122 | 0 | } Unexecuted instantiation: <cssparser::parser::BasicParseError>::unexpected_token Unexecuted instantiation: <cssparser::parser::BasicParseError>::unexpected_token |
123 | | } |
124 | | |
125 | | impl<T> From<BasicParseError> for ParseError<T> { |
126 | | #[inline] |
127 | 0 | fn from(this: BasicParseError) -> ParseError<T> { |
128 | 0 | ParseError { |
129 | 0 | kind: ParseErrorKind::Basic(this.kind), |
130 | 0 | } |
131 | 0 | } |
132 | | } |
133 | | |
134 | | /// Details of a `ParseError` |
135 | | #[derive(Clone, Debug, PartialEq)] |
136 | | pub enum ParseErrorKind<T> { |
137 | | /// A fundamental parse error from a built-in parsing routine. |
138 | | Basic(BasicParseErrorKind), |
139 | | /// A parse error reported by downstream consumer code. |
140 | | Custom(T), |
141 | | } |
142 | | |
143 | | impl<T> ParseErrorKind<T> { |
144 | | /// Like `std::convert::Into::into` |
145 | | pub fn into<U>(self) -> ParseErrorKind<U> |
146 | | where |
147 | | T: Into<U>, |
148 | | { |
149 | | match self { |
150 | | ParseErrorKind::Basic(basic) => ParseErrorKind::Basic(basic), |
151 | | ParseErrorKind::Custom(custom) => ParseErrorKind::Custom(custom.into()), |
152 | | } |
153 | | } |
154 | | } |
155 | | |
156 | | impl<E: fmt::Display> fmt::Display for ParseErrorKind<E> { |
157 | | fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { |
158 | | match self { |
159 | | ParseErrorKind::Basic(basic) => basic.fmt(f), |
160 | | ParseErrorKind::Custom(custom) => custom.fmt(f), |
161 | | } |
162 | | } |
163 | | } |
164 | | |
165 | | /// Extensible parse errors that can be encountered by client parsing implementations. |
166 | | #[derive(Clone, Debug, PartialEq)] |
167 | | pub struct ParseError<E> { |
168 | | /// Details of this error |
169 | | pub kind: ParseErrorKind<E>, |
170 | | } |
171 | | |
172 | | impl<T> ParseError<T> { |
173 | | /// Create a new ParseError from a basic error kind. |
174 | | #[inline] |
175 | 1.17k | pub fn from_basic_kind(kind: BasicParseErrorKind) -> Self { |
176 | 1.17k | Self { |
177 | 1.17k | kind: ParseErrorKind::Basic(kind), |
178 | 1.17k | } |
179 | 1.17k | } |
180 | | |
181 | | /// Create a new ParseError for an unexpected token. |
182 | | #[inline] |
183 | 1.16k | pub fn unexpected_token() -> Self { |
184 | 1.16k | Self::from_basic_kind(BasicParseErrorKind::UnexpectedToken) |
185 | 1.16k | } |
186 | | |
187 | | /// Create a new ParseError from a consumer-defined error. |
188 | | #[inline] |
189 | | pub fn custom<E: Into<T>>(error: E) -> Self { |
190 | | Self { |
191 | | kind: ParseErrorKind::Custom(error.into()), |
192 | | } |
193 | | } |
194 | | |
195 | | /// Extract the fundamental parse error from an extensible error. |
196 | | pub fn basic(self) -> BasicParseError { |
197 | | match self.kind { |
198 | | ParseErrorKind::Basic(kind) => BasicParseError { kind }, |
199 | | ParseErrorKind::Custom(_) => panic!("Not a basic parse error"), |
200 | | } |
201 | | } |
202 | | |
203 | | /// Like `std::convert::Into::into` |
204 | | pub fn into<U>(self) -> ParseError<U> |
205 | | where |
206 | | T: Into<U>, |
207 | | { |
208 | | ParseError { |
209 | | kind: self.kind.into(), |
210 | | } |
211 | | } |
212 | | } |
213 | | |
214 | | impl<E: fmt::Display> fmt::Display for ParseError<E> { |
215 | | fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { |
216 | | self.kind.fmt(f) |
217 | | } |
218 | | } |
219 | | |
220 | | impl<E: fmt::Display + fmt::Debug> std::error::Error for ParseError<E> {} |
221 | | |
222 | | /// A CSS parser that borrows its `&str` input, yields `Token`s, and keeps track of nested blocks |
223 | | /// and functions. |
224 | | pub struct Parser<'i> { |
225 | | pub(crate) input: &'i str, |
226 | | pub(crate) state: ParserState, |
227 | | cached_token: CachedToken<'i>, |
228 | | current_block_depth: u8, |
229 | | nested_block_limit: u8, |
230 | | /// For parsers from `parse_until` or `parse_nested_block` |
231 | | stop_before: Delimiters, |
232 | | pub(crate) arbitrary_substitution_functions: SeenStatus<'i>, |
233 | | pub(crate) source_map_url: Option<&'i str>, |
234 | | pub(crate) source_url: Option<&'i str>, |
235 | | } |
236 | | |
237 | | struct CachedToken<'i> { |
238 | | token: Token<'i>, |
239 | | start_position: SourcePosition, |
240 | | end_state: ParserState, |
241 | | } |
242 | | |
243 | | #[derive(Copy, Clone, PartialEq, Eq, Debug)] |
244 | | pub(crate) enum BlockType { |
245 | | Parenthesis, |
246 | | SquareBracket, |
247 | | CurlyBracket, |
248 | | } |
249 | | |
250 | | impl BlockType { |
251 | 1.25M | fn closing(token: &Token) -> Option<BlockType> { |
252 | 1.25M | match *token { |
253 | 13.1k | Token::CloseParenthesis => Some(BlockType::Parenthesis), |
254 | 4.45k | Token::CloseSquareBracket => Some(BlockType::SquareBracket), |
255 | 342k | Token::CloseCurlyBracket => Some(BlockType::CurlyBracket), |
256 | 895k | _ => None, |
257 | | } |
258 | 1.25M | } |
259 | | } |
260 | | |
261 | | /// A set of characters, to be used with the `Parser::parse_until*` methods. |
262 | | /// |
263 | | /// The union of two sets can be obtained with the `|` operator. Example: |
264 | | /// |
265 | | /// ```rust,ignore |
266 | | /// input.parse_until_before(Delimiter::CurlyBracketBlock | Delimiter::Semicolon) |
267 | | /// ``` |
268 | | #[derive(Copy, Clone, PartialEq, Eq, Debug)] |
269 | | pub struct Delimiters { |
270 | | bits: u8, |
271 | | } |
272 | | |
273 | | /// `Delimiters` constants. |
274 | | #[allow(non_upper_case_globals, non_snake_case)] |
275 | | pub mod Delimiter { |
276 | | use super::Delimiters; |
277 | | |
278 | | /// The empty delimiter set |
279 | | pub const None: Delimiters = Delimiters { bits: 0 }; |
280 | | /// The delimiter set with only the `{` opening curly bracket |
281 | | pub const CurlyBracketBlock: Delimiters = Delimiters { bits: 1 << 1 }; |
282 | | /// The delimiter set with only the `;` semicolon |
283 | | pub const Semicolon: Delimiters = Delimiters { bits: 1 << 2 }; |
284 | | /// The delimiter set with only the `!` exclamation point |
285 | | pub const Bang: Delimiters = Delimiters { bits: 1 << 3 }; |
286 | | /// The delimiter set with only the `,` comma |
287 | | pub const Comma: Delimiters = Delimiters { bits: 1 << 4 }; |
288 | | } |
289 | | |
290 | | #[allow(non_upper_case_globals, non_snake_case)] |
291 | | mod ClosingDelimiter { |
292 | | use super::Delimiters; |
293 | | |
294 | | pub const CloseCurlyBracket: Delimiters = Delimiters { bits: 1 << 5 }; |
295 | | pub const CloseSquareBracket: Delimiters = Delimiters { bits: 1 << 6 }; |
296 | | pub const CloseParenthesis: Delimiters = Delimiters { bits: 1 << 7 }; |
297 | | } |
298 | | |
299 | | impl BitOr<Delimiters> for Delimiters { |
300 | | type Output = Delimiters; |
301 | | |
302 | | #[inline] |
303 | | fn bitor(self, other: Delimiters) -> Delimiters { |
304 | | Delimiters { |
305 | | bits: self.bits | other.bits, |
306 | | } |
307 | | } |
308 | | } |
309 | | |
310 | | impl Delimiters { |
311 | | #[inline] |
312 | 127M | fn contains(self, other: Delimiters) -> bool { |
313 | 127M | (self.bits & other.bits) != 0 |
314 | 127M | } |
315 | | |
316 | | #[inline] |
317 | 127M | pub(crate) fn from_byte(byte: u8) -> Delimiters { |
318 | | const TABLE: [Delimiters; 256] = { |
319 | | let mut table = [Delimiter::None; 256]; |
320 | | table[b';' as usize] = Delimiter::Semicolon; |
321 | | table[b'!' as usize] = Delimiter::Bang; |
322 | | table[b',' as usize] = Delimiter::Comma; |
323 | | table[b'{' as usize] = Delimiter::CurlyBracketBlock; |
324 | | table[b'}' as usize] = ClosingDelimiter::CloseCurlyBracket; |
325 | | table[b']' as usize] = ClosingDelimiter::CloseSquareBracket; |
326 | | table[b')' as usize] = ClosingDelimiter::CloseParenthesis; |
327 | | table |
328 | | }; |
329 | | |
330 | 127M | TABLE[byte as usize] |
331 | 127M | } |
332 | | } |
333 | | |
334 | | /// Used in some `fn expect_*` methods |
335 | | macro_rules! expect { |
336 | | ($parser: ident, $($branches: tt)+) => { |
337 | | { |
338 | | match *$parser.next()? { |
339 | | $($branches)+ |
340 | | _ => { |
341 | | return Err(BasicParseError::unexpected_token()) |
342 | | } |
343 | | } |
344 | | } |
345 | | } |
346 | | } |
347 | | |
348 | | /// A list of arbitrary substitution functions. Should be lowercase ascii. |
349 | | /// See https://drafts.csswg.org/css-values-5/#arbitrary-substitution |
350 | | pub type ArbitrarySubstitutionFunctions<'a> = &'a [&'static str]; |
351 | | |
352 | | impl<'i> Parser<'i> { |
353 | | /// 75 nested blocks seems reasonable enough. |
354 | | const REASONABLE_NESTED_BLOCK_LIMIT: u8 = 75; |
355 | | |
356 | | /// Create a new parser for the given input. |
357 | | #[inline] |
358 | 16.2k | pub fn new(input: &'i str) -> Self { |
359 | 16.2k | Self { |
360 | 16.2k | input, |
361 | 16.2k | state: ParserState::default(), |
362 | 16.2k | stop_before: Delimiter::None, |
363 | 16.2k | nested_block_limit: Self::REASONABLE_NESTED_BLOCK_LIMIT, |
364 | 16.2k | current_block_depth: 0, |
365 | 16.2k | cached_token: CachedToken { |
366 | 16.2k | token: Token::Semicolon, // Anything would do. |
367 | 16.2k | start_position: SourcePosition(usize::MAX), // No token would match this cache. |
368 | 16.2k | end_state: ParserState::default(), |
369 | 16.2k | }, |
370 | 16.2k | arbitrary_substitution_functions: SeenStatus::DontCare, |
371 | 16.2k | source_map_url: None, |
372 | 16.2k | source_url: None, |
373 | 16.2k | } |
374 | 16.2k | } <cssparser::parser::Parser>::new Line | Count | Source | 358 | 16.2k | pub fn new(input: &'i str) -> Self { | 359 | 16.2k | Self { | 360 | 16.2k | input, | 361 | 16.2k | state: ParserState::default(), | 362 | 16.2k | stop_before: Delimiter::None, | 363 | 16.2k | nested_block_limit: Self::REASONABLE_NESTED_BLOCK_LIMIT, | 364 | 16.2k | current_block_depth: 0, | 365 | 16.2k | cached_token: CachedToken { | 366 | 16.2k | token: Token::Semicolon, // Anything would do. | 367 | 16.2k | start_position: SourcePosition(usize::MAX), // No token would match this cache. | 368 | 16.2k | end_state: ParserState::default(), | 369 | 16.2k | }, | 370 | 16.2k | arbitrary_substitution_functions: SeenStatus::DontCare, | 371 | 16.2k | source_map_url: None, | 372 | 16.2k | source_url: None, | 373 | 16.2k | } | 374 | 16.2k | } |
Unexecuted instantiation: <cssparser::parser::Parser>::new |
375 | | |
376 | | /// Sets a limit for how many nested blocks we're allowed to parse. This is useful to avoid |
377 | | /// running out of stack space. By default, it's set to `REASONABLE_NESTED_BLOCK_LIMIT`, but it |
378 | | /// can be overridden or cleared. A limit of 0 will be equivalent to no limit at all. |
379 | 0 | pub fn set_nested_block_limit(&mut self, limit: u8) { |
380 | 0 | self.nested_block_limit = limit; |
381 | 0 | } |
382 | | |
383 | | /// Check whether the input is exhausted. That is, if `.next()` would return a token. |
384 | | /// |
385 | | /// This ignores whitespace and comments. |
386 | | #[inline] |
387 | 0 | pub fn is_exhausted(&mut self) -> bool { |
388 | 0 | self.expect_exhausted().is_ok() |
389 | 0 | } |
390 | | |
391 | | /// Check whether the input is exhausted. That is, if `.next()` would return a token. |
392 | | /// Return a `Result` so that the `?` operator can be used: `input.expect_exhausted()?` |
393 | | /// |
394 | | /// This ignores whitespace and comments. |
395 | | #[inline] |
396 | 364k | pub fn expect_exhausted(&mut self) -> Result<(), BasicParseError> { |
397 | 364k | let start = self.state(); |
398 | 364k | let result = match self.next() { |
399 | | Err(BasicParseError { |
400 | | kind: BasicParseErrorKind::EndOfInput, |
401 | | .. |
402 | 364k | }) => Ok(()), |
403 | 0 | Err(e) => unreachable!("Unexpected error encountered: {:?}", e), |
404 | 0 | Ok(_) => Err(BasicParseError::unexpected_token()), |
405 | | }; |
406 | 364k | self.reset(&start); |
407 | 364k | result |
408 | 364k | } <cssparser::parser::Parser>::expect_exhausted Line | Count | Source | 396 | 364k | pub fn expect_exhausted(&mut self) -> Result<(), BasicParseError> { | 397 | 364k | let start = self.state(); | 398 | 364k | let result = match self.next() { | 399 | | Err(BasicParseError { | 400 | | kind: BasicParseErrorKind::EndOfInput, | 401 | | .. | 402 | 364k | }) => Ok(()), | 403 | 0 | Err(e) => unreachable!("Unexpected error encountered: {:?}", e), | 404 | 0 | Ok(_) => Err(BasicParseError::unexpected_token()), | 405 | | }; | 406 | 364k | self.reset(&start); | 407 | 364k | result | 408 | 364k | } |
Unexecuted instantiation: <cssparser::parser::Parser>::expect_exhausted |
409 | | |
410 | | /// Create a new unexpected token or EOF ParseError at the current location |
411 | | #[inline] |
412 | | pub fn new_error_for_next_token<E>(&mut self) -> ParseError<E> { |
413 | | match self.next() { |
414 | | Ok(_) => ParseError::unexpected_token(), |
415 | | Err(e) => e.into(), |
416 | | } |
417 | | } |
418 | | |
419 | | /// Return the current internal state of the parser (including position within the input). |
420 | | /// |
421 | | /// This state can later be restored with the `Parser::reset` method. |
422 | | #[inline] |
423 | 364k | pub fn state(&self) -> ParserState { |
424 | 364k | self.state.clone() |
425 | 364k | } <cssparser::parser::Parser>::state Line | Count | Source | 423 | 364k | pub fn state(&self) -> ParserState { | 424 | 364k | self.state.clone() | 425 | 364k | } |
Unexecuted instantiation: <cssparser::parser::Parser>::state |
426 | | |
427 | | /// Like `next_byte`, but returns `None` if the next byte is one of the delimiters this |
428 | | /// parser was told to stop before. |
429 | | #[inline] |
430 | 127M | pub(crate) fn next_byte_before_delimiter(&self) -> Option<u8> { |
431 | 127M | let byte = self.next_byte()?; |
432 | 127M | if self.stop_before.contains(Delimiters::from_byte(byte)) { |
433 | 697k | return None; |
434 | 126M | } |
435 | 126M | Some(byte) |
436 | 127M | } |
437 | | |
438 | | /// Restore the internal state of the parser (including position within the input) |
439 | | /// to what was previously saved by the `Parser::position` method. |
440 | | /// |
441 | | /// Should only be used with `SourcePosition` values from the same `Parser` instance. |
442 | | #[inline] |
443 | 364k | pub fn reset(&mut self, state: &ParserState) { |
444 | 364k | self.state = state.clone(); |
445 | 364k | } <cssparser::parser::Parser>::reset Line | Count | Source | 443 | 364k | pub fn reset(&mut self, state: &ParserState) { | 444 | 364k | self.state = state.clone(); | 445 | 364k | } |
Unexecuted instantiation: <cssparser::parser::Parser>::reset |
446 | | |
447 | | /// The old name of `try_parse`, which requires raw identifiers in the Rust 2018 edition. |
448 | | #[inline] |
449 | | pub fn r#try<F, T, E>(&mut self, thing: F) -> Result<T, E> |
450 | | where |
451 | | F: FnOnce(&mut Parser<'i>) -> Result<T, E>, |
452 | | { |
453 | | self.try_parse(thing) |
454 | | } |
455 | | |
456 | | /// Execute the given closure, passing it the parser. |
457 | | /// If the result (returned unchanged) is `Err`, |
458 | | /// the internal state of the parser (including position within the input) |
459 | | /// is restored to what it was before the call. |
460 | | #[inline] |
461 | | pub fn try_parse<F, T, E>(&mut self, thing: F) -> Result<T, E> |
462 | | where |
463 | | F: FnOnce(&mut Parser<'i>) -> Result<T, E>, |
464 | | { |
465 | | let start = self.state(); |
466 | | let result = thing(self); |
467 | | if result.is_err() { |
468 | | self.reset(&start) |
469 | | } |
470 | | result |
471 | | } |
472 | | |
473 | | /// Return the next token in the input that is neither whitespace or a comment, |
474 | | /// and advance the position accordingly. |
475 | | /// |
476 | | /// After returning a `Function`, `ParenthesisBlock`, |
477 | | /// `CurlyBracketBlock`, or `SquareBracketBlock` token, |
478 | | /// the next call will skip until after the matching `CloseParenthesis`, |
479 | | /// `CloseCurlyBracket`, or `CloseSquareBracket` token. |
480 | | /// |
481 | | /// See the `Parser::parse_nested_block` method to parse the content of functions or blocks. |
482 | | /// |
483 | | /// This only returns a closing token when it is unmatched (and therefore an error). |
484 | | #[allow(clippy::should_implement_trait)] |
485 | 364k | pub fn next(&mut self) -> Result<&Token<'i>, BasicParseError> { |
486 | 364k | self.skip_whitespace(); |
487 | 364k | self.next_including_whitespace_and_comments() |
488 | 364k | } |
489 | | |
490 | | /// Same as `Parser::next`, but does not skip whitespace tokens. |
491 | 92.6M | pub fn next_including_whitespace(&mut self) -> Result<&Token<'i>, BasicParseError> { |
492 | 126M | while let Token::Comment(..) = self.next_including_whitespace_and_comments()? { |
493 | 34.0M | // Keep going |
494 | 34.0M | } |
495 | 92.2M | Ok(&self.cached_token.token) |
496 | 92.6M | } |
497 | | |
498 | | /// Same as `Parser::next`, but does not skip whitespace or comment tokens. |
499 | | /// |
500 | | /// **Note**: This should only be used in contexts like a CSS pre-processor |
501 | | /// where comments are preserved. |
502 | | /// When parsing higher-level values, per the CSS Syntax specification, |
503 | | /// comments should always be ignored between tokens. |
504 | 127M | pub fn next_including_whitespace_and_comments( |
505 | 127M | &mut self, |
506 | 127M | ) -> Result<&Token<'i>, BasicParseError> { |
507 | 127M | self.skip_block_at_start(); |
508 | | |
509 | 127M | if self.next_byte_before_delimiter().is_none() { |
510 | 744k | return Err(BasicParseError::new(BasicParseErrorKind::EndOfInput)); |
511 | 126M | } |
512 | | |
513 | 126M | let token_start_position = self.position(); |
514 | 126M | let using_cached_token = self.cached_token.start_position == token_start_position; |
515 | 126M | if using_cached_token { |
516 | 0 | self.state = self.cached_token.end_state.clone(); |
517 | 126M | } else { |
518 | 126M | let new_token = self.next_unchecked(); |
519 | 126M | self.cached_token = CachedToken { |
520 | 126M | token: new_token, |
521 | 126M | start_position: token_start_position, |
522 | 126M | end_state: self.state.clone(), |
523 | 126M | }; |
524 | 126M | } |
525 | | |
526 | 126M | Ok(&self.cached_token.token) |
527 | 127M | } |
528 | | |
529 | | /// Have the given closure parse something, then check the the input is exhausted. |
530 | | /// The result is overridden to an `Err(..)` if some input remains. |
531 | | /// |
532 | | /// This can help tell e.g. `color: green;` from `color: green 4px;` |
533 | | #[inline] |
534 | 366k | pub fn parse_entirely<F, T, E>(&mut self, parse: F) -> Result<T, ParseError<E>> |
535 | 366k | where |
536 | 366k | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
537 | | { |
538 | 366k | let result = parse(self)?; |
539 | 364k | self.expect_exhausted()?; |
540 | 364k | Ok(result) |
541 | 366k | } |
542 | | |
543 | | /// Parse a list of comma-separated values, all with the same syntax. |
544 | | /// |
545 | | /// The given closure is called repeatedly with a "delimited" parser |
546 | | /// (see the `Parser::parse_until_before` method) so that it can over |
547 | | /// consume the input past a comma at this block/function nesting level. |
548 | | /// |
549 | | /// Successful results are accumulated in a vector. |
550 | | /// |
551 | | /// This method returns an`Err(..)` the first time that a closure call does, |
552 | | /// or if a closure call leaves some input before the next comma or the end |
553 | | /// of the input. |
554 | | #[inline] |
555 | | pub fn parse_comma_separated<F, T, E>(&mut self, parse_one: F) -> Result<Vec<T>, ParseError<E>> |
556 | | where |
557 | | F: FnMut(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
558 | | { |
559 | | self.parse_comma_separated_internal(parse_one, /* ignore_errors = */ false) |
560 | | } |
561 | | |
562 | | /// Like `parse_comma_separated`, but ignores errors on unknown components, |
563 | | /// rather than erroring out in the whole list. |
564 | | /// |
565 | | /// Caller must deal with the fact that the resulting list might be empty, |
566 | | /// if there's no valid component on the list. |
567 | | #[inline] |
568 | | pub fn parse_comma_separated_ignoring_errors<F, T, E>(&mut self, parse_one: F) -> Vec<T> |
569 | | where |
570 | | F: FnMut(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
571 | | { |
572 | | match self.parse_comma_separated_internal(parse_one, /* ignore_errors = */ true) { |
573 | | Ok(values) => values, |
574 | | Err(..) => unreachable!(), |
575 | | } |
576 | | } |
577 | | |
578 | | #[inline] |
579 | | fn parse_comma_separated_internal<F, T, E>( |
580 | | &mut self, |
581 | | mut parse_one: F, |
582 | | ignore_errors: bool, |
583 | | ) -> Result<Vec<T>, ParseError<E>> |
584 | | where |
585 | | F: FnMut(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
586 | | { |
587 | | // Vec grows from 0 to 4 by default on first push(). So allocate with |
588 | | // capacity 1, so in the somewhat common case of only one item we don't |
589 | | // way overallocate. Note that we always push at least one item if |
590 | | // parsing succeeds. |
591 | | let mut values = Vec::with_capacity(1); |
592 | | loop { |
593 | | self.skip_whitespace(); // Unnecessary for correctness, but may help try() in parse_one rewind less. |
594 | | match self.parse_until_before(Delimiter::Comma, &mut parse_one) { |
595 | | Ok(v) => values.push(v), |
596 | | Err(e) if !ignore_errors => return Err(e), |
597 | | Err(_) => {} |
598 | | } |
599 | | match self.next() { |
600 | | Err(_) => return Ok(values), |
601 | | Ok(&Token::Comma) => continue, |
602 | | Ok(_) => unreachable!(), |
603 | | } |
604 | | } |
605 | | } |
606 | | |
607 | | /// Parse the content of a block or function. |
608 | | /// |
609 | | /// This method panics if the last token yielded by this parser |
610 | | /// (from one of the `next*` methods) |
611 | | /// is not a on that marks the start of a block or function: |
612 | | /// a `Function`, `ParenthesisBlock`, `CurlyBracketBlock`, or `SquareBracketBlock`. |
613 | | /// |
614 | | /// The given closure is called with a "delimited" parser |
615 | | /// that stops at the end of the block or function (at the matching closing token). |
616 | | /// |
617 | | /// The result is overridden to an `Err(..)` if the closure leaves some input before that point. |
618 | | #[inline] |
619 | 366k | pub fn parse_nested_block<F, T, E>(&mut self, parse: F) -> Result<T, ParseError<E>> |
620 | 366k | where |
621 | 366k | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
622 | | { |
623 | 366k | parse_nested_block(self, parse) |
624 | 366k | } |
625 | | |
626 | | /// Limit parsing to until a given delimiter or the end of the input. (E.g. |
627 | | /// a semicolon for a property value.) |
628 | | /// |
629 | | /// The given closure is called with a "delimited" parser |
630 | | /// that stops before the first character at this block/function nesting level |
631 | | /// that matches the given set of delimiters, or at the end of the input. |
632 | | /// |
633 | | /// The result is overridden to an `Err(..)` if the closure leaves some input before that point. |
634 | | #[inline] |
635 | | pub fn parse_until_before<F, T, E>( |
636 | | &mut self, |
637 | | delimiters: Delimiters, |
638 | | parse: F, |
639 | | ) -> Result<T, ParseError<E>> |
640 | | where |
641 | | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
642 | | { |
643 | | parse_until_before(self, delimiters, ParseUntilErrorBehavior::Consume, parse) |
644 | | } |
645 | | |
646 | | /// Like `parse_until_before`, but also consume the delimiter token. |
647 | | /// |
648 | | /// This can be useful when you don’t need to know which delimiter it was |
649 | | /// (e.g. if these is only one in the given set) |
650 | | /// or if it was there at all (as opposed to reaching the end of the input). |
651 | | #[inline] |
652 | | pub fn parse_until_after<F, T, E>( |
653 | | &mut self, |
654 | | delimiters: Delimiters, |
655 | | parse: F, |
656 | | ) -> Result<T, ParseError<E>> |
657 | | where |
658 | | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
659 | | { |
660 | | parse_until_after(self, delimiters, ParseUntilErrorBehavior::Consume, parse) |
661 | | } |
662 | | |
663 | | /// Parse a <whitespace-token> and return its value. |
664 | | #[inline] |
665 | | pub fn expect_whitespace(&mut self) -> Result<&'i str, BasicParseError> { |
666 | | match *self.next_including_whitespace()? { |
667 | | Token::WhiteSpace(value) => Ok(value), |
668 | | _ => Err(BasicParseError::unexpected_token()), |
669 | | } |
670 | | } |
671 | | |
672 | | /// Parse a <ident-token> and return the unescaped value. |
673 | | #[inline] |
674 | 0 | pub fn expect_ident(&mut self) -> Result<&CowRcStr<'i>, BasicParseError> { |
675 | 0 | expect! {self, |
676 | 0 | Token::Ident(ref value) => Ok(value), |
677 | | } |
678 | 0 | } |
679 | | |
680 | | /// expect_ident, but clone the CowRcStr |
681 | | #[inline] |
682 | | pub fn expect_ident_cloned(&mut self) -> Result<CowRcStr<'i>, BasicParseError> { |
683 | | self.expect_ident().cloned() |
684 | | } |
685 | | |
686 | | /// Parse a <ident-token> whose unescaped value is an ASCII-insensitive match for the given value. |
687 | | #[inline] |
688 | 0 | pub fn expect_ident_matching(&mut self, expected_value: &str) -> Result<(), BasicParseError> { |
689 | 0 | expect! {self, |
690 | 0 | Token::Ident(ref value) if value.eq_ignore_ascii_case(expected_value) => Ok(()), |
691 | | } |
692 | 0 | } |
693 | | |
694 | | /// Parse a <string-token> and return the unescaped value. |
695 | | #[inline] |
696 | | pub fn expect_string(&mut self) -> Result<&CowRcStr<'i>, BasicParseError> { |
697 | | expect! {self, |
698 | | Token::QuotedString(ref value) => Ok(value), |
699 | | } |
700 | | } |
701 | | |
702 | | /// expect_string, but clone the CowRcStr |
703 | | #[inline] |
704 | | pub fn expect_string_cloned(&mut self) -> Result<CowRcStr<'i>, BasicParseError> { |
705 | | self.expect_string().cloned() |
706 | | } |
707 | | |
708 | | /// Parse either a <ident-token> or a <string-token>, and return the unescaped value. |
709 | | #[inline] |
710 | | pub fn expect_ident_or_string(&mut self) -> Result<&CowRcStr<'i>, BasicParseError> { |
711 | | expect! {self, |
712 | | Token::Ident(ref value) => Ok(value), |
713 | | Token::QuotedString(ref value) => Ok(value), |
714 | | } |
715 | | } |
716 | | |
717 | | /// Parse a <url-token> and return the unescaped value. |
718 | | #[inline] |
719 | | pub fn expect_url(&mut self) -> Result<CowRcStr<'i>, BasicParseError> { |
720 | | expect! {self, |
721 | | Token::UnquotedUrl(ref value) => Ok(value.clone()), |
722 | | Token::Function(ref name) if name.eq_ignore_ascii_case("url") => { |
723 | | self.parse_nested_block(|input| { |
724 | | input.expect_string().map_err(Into::into).cloned() |
725 | | }) |
726 | | .map_err(ParseError::<()>::basic) |
727 | | } |
728 | | } |
729 | | } |
730 | | |
731 | | /// Parse either a <url-token> or a <string-token>, and return the unescaped value. |
732 | | #[inline] |
733 | | pub fn expect_url_or_string(&mut self) -> Result<CowRcStr<'i>, BasicParseError> { |
734 | | expect! {self, |
735 | | Token::UnquotedUrl(ref value) => Ok(value.clone()), |
736 | | Token::QuotedString(ref value) => Ok(value.clone()), |
737 | | Token::Function(ref name) if name.eq_ignore_ascii_case("url") => { |
738 | | self.parse_nested_block(|input| { |
739 | | input.expect_string().map_err(Into::into).cloned() |
740 | | }) |
741 | | .map_err(ParseError::<()>::basic) |
742 | | } |
743 | | } |
744 | | } |
745 | | |
746 | | /// Parse a <number-token> and return the integer value. |
747 | | #[inline] |
748 | | pub fn expect_number(&mut self) -> Result<f32, BasicParseError> { |
749 | | expect! {self, |
750 | | Token::Number { value, .. } => Ok(value), |
751 | | } |
752 | | } |
753 | | |
754 | | /// Parse a <number-token> that does not have a fractional part, and return the integer value. |
755 | | #[inline] |
756 | | pub fn expect_integer(&mut self) -> Result<i32, BasicParseError> { |
757 | | expect! {self, |
758 | | Token::Number { int_value: Some(int_value), .. } => Ok(int_value), |
759 | | } |
760 | | } |
761 | | |
762 | | /// Parse a <percentage-token> and return the value. |
763 | | /// `0%` and `100%` map to `0.0` and `1.0` (not `100.0`), respectively. |
764 | | #[inline] |
765 | | pub fn expect_percentage(&mut self) -> Result<f32, BasicParseError> { |
766 | | expect! {self, |
767 | | Token::Percentage { unit_value, .. } => Ok(unit_value), |
768 | | } |
769 | | } |
770 | | |
771 | | /// Parse a `:` <colon-token>. |
772 | | #[inline] |
773 | 0 | pub fn expect_colon(&mut self) -> Result<(), BasicParseError> { |
774 | 0 | expect! {self, |
775 | 0 | Token::Colon => Ok(()), |
776 | | } |
777 | 0 | } |
778 | | |
779 | | /// Parse a `;` <semicolon-token>. |
780 | | #[inline] |
781 | | pub fn expect_semicolon(&mut self) -> Result<(), BasicParseError> { |
782 | | expect! {self, |
783 | | Token::Semicolon => Ok(()), |
784 | | } |
785 | | } |
786 | | |
787 | | /// Parse a `,` <comma-token>. |
788 | | #[inline] |
789 | | pub fn expect_comma(&mut self) -> Result<(), BasicParseError> { |
790 | | expect! {self, |
791 | | Token::Comma => Ok(()), |
792 | | } |
793 | | } |
794 | | |
795 | | /// Parse a <delim-token> with the given value. |
796 | | #[inline] |
797 | 0 | pub fn expect_delim(&mut self, expected_value: char) -> Result<(), BasicParseError> { |
798 | 0 | expect! {self, |
799 | 0 | Token::Delim(value) if value == expected_value => Ok(()), |
800 | | } |
801 | 0 | } |
802 | | |
803 | | /// Parse a `{ /* ... */ }` curly brackets block. |
804 | | /// |
805 | | /// If the result is `Ok`, you can then call the `Parser::parse_nested_block` method. |
806 | | #[inline] |
807 | | pub fn expect_curly_bracket_block(&mut self) -> Result<(), BasicParseError> { |
808 | | expect! {self, |
809 | | Token::CurlyBracketBlock => Ok(()), |
810 | | } |
811 | | } |
812 | | |
813 | | /// Parse a `[ /* ... */ ]` square brackets block. |
814 | | /// |
815 | | /// If the result is `Ok`, you can then call the `Parser::parse_nested_block` method. |
816 | | #[inline] |
817 | | pub fn expect_square_bracket_block(&mut self) -> Result<(), BasicParseError> { |
818 | | expect! {self, |
819 | | Token::SquareBracketBlock => Ok(()), |
820 | | } |
821 | | } |
822 | | |
823 | | /// Parse a `( /* ... */ )` parenthesis block. |
824 | | /// |
825 | | /// If the result is `Ok`, you can then call the `Parser::parse_nested_block` method. |
826 | | #[inline] |
827 | | pub fn expect_parenthesis_block(&mut self) -> Result<(), BasicParseError> { |
828 | | expect! {self, |
829 | | Token::ParenthesisBlock => Ok(()), |
830 | | } |
831 | | } |
832 | | |
833 | | /// Parse a <function> token and return its name. |
834 | | /// |
835 | | /// If the result is `Ok`, you can then call the `Parser::parse_nested_block` method. |
836 | | #[inline] |
837 | | pub fn expect_function(&mut self) -> Result<&CowRcStr<'i>, BasicParseError> { |
838 | | expect! {self, |
839 | | Token::Function(ref name) => Ok(name), |
840 | | } |
841 | | } |
842 | | |
843 | | /// Parse a <function> token whose name is an ASCII-insensitive match for the given value. |
844 | | /// |
845 | | /// If the result is `Ok`, you can then call the `Parser::parse_nested_block` method. |
846 | | #[inline] |
847 | | pub fn expect_function_matching(&mut self, expected_name: &str) -> Result<(), BasicParseError> { |
848 | | expect! {self, |
849 | | Token::Function(ref name) if name.eq_ignore_ascii_case(expected_name) => Ok(()), |
850 | | } |
851 | | } |
852 | | |
853 | | /// Parse the input until exhaustion and check that it contains no “error” token. |
854 | | /// |
855 | | /// See `Token::is_parse_error`. This also checks nested blocks and functions recursively. |
856 | | #[inline] |
857 | | pub fn expect_no_error_token(&mut self) -> Result<(), BasicParseError> { |
858 | | loop { |
859 | | match self.next_including_whitespace_and_comments() { |
860 | | Ok(&Token::Function(_)) |
861 | | | Ok(&Token::ParenthesisBlock) |
862 | | | Ok(&Token::SquareBracketBlock) |
863 | | | Ok(&Token::CurlyBracketBlock) => self |
864 | | .parse_nested_block(|input| input.expect_no_error_token().map_err(Into::into)) |
865 | | .map_err(ParseError::<()>::basic)?, |
866 | | Ok(t) => { |
867 | | // FIXME: maybe these should be separate variants of |
868 | | // BasicParseError instead? |
869 | | if t.is_parse_error() { |
870 | | return Err(BasicParseError::unexpected_token()); |
871 | | } |
872 | | } |
873 | | Err(_) => return Ok(()), |
874 | | } |
875 | | } |
876 | | } |
877 | | } |
878 | | |
879 | | pub fn parse_until_before<'i, F, T, E>( |
880 | | parser: &mut Parser<'i>, |
881 | | delimiters: Delimiters, |
882 | | error_behavior: ParseUntilErrorBehavior, |
883 | | parse: F, |
884 | | ) -> Result<T, ParseError<E>> |
885 | | where |
886 | | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
887 | | { |
888 | | let old_stop_before = parser.stop_before; |
889 | | let delimiters = parser.stop_before | delimiters; |
890 | | parser.stop_before = delimiters; |
891 | | let result = parser.parse_entirely(parse); |
892 | | parser.stop_before = old_stop_before; |
893 | | if error_behavior == ParseUntilErrorBehavior::Stop && result.is_err() { |
894 | | return result; |
895 | | } |
896 | | parser.skip_block_at_start(); |
897 | | // FIXME: have a special-purpose tokenizer method for this that does less work. |
898 | | while let Some(next_byte) = parser.next_byte() { |
899 | | if delimiters.contains(Delimiters::from_byte(next_byte)) { |
900 | | break; |
901 | | } |
902 | | parser.next_unchecked(); |
903 | | parser.skip_block_at_start(); |
904 | | } |
905 | | result |
906 | | } |
907 | | |
908 | | pub fn parse_until_after<'i, F, T, E>( |
909 | | parser: &mut Parser<'i>, |
910 | | delimiters: Delimiters, |
911 | | error_behavior: ParseUntilErrorBehavior, |
912 | | parse: F, |
913 | | ) -> Result<T, ParseError<E>> |
914 | | where |
915 | | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
916 | | { |
917 | | let result = parse_until_before(parser, delimiters, error_behavior, parse); |
918 | | if error_behavior == ParseUntilErrorBehavior::Stop && result.is_err() { |
919 | | return result; |
920 | | } |
921 | | if let Some(next_byte) = parser.next_byte() { |
922 | | let delimiter = Delimiters::from_byte(next_byte); |
923 | | if !parser.stop_before.contains(delimiter) { |
924 | | debug_assert!(delimiters.contains(delimiter)); |
925 | | // We know this byte is ASCII. |
926 | | parser.advance(1); |
927 | | if next_byte == b'{' { |
928 | | parser.consume_until_end_of_block(BlockType::CurlyBracket); |
929 | | } |
930 | | } |
931 | | } |
932 | | result |
933 | | } |
934 | | |
935 | 366k | pub fn parse_nested_block<'i, F, T, E>( |
936 | 366k | parser: &mut Parser<'i>, |
937 | 366k | parse: F, |
938 | 366k | ) -> Result<T, ParseError<E>> |
939 | 366k | where |
940 | 366k | F: FnOnce(&mut Parser<'i>) -> Result<T, ParseError<E>>, |
941 | | { |
942 | 366k | let block_type = parser.state.at_start_of.take().expect( |
943 | 366k | "\ |
944 | 366k | A nested parser can only be created when a Function, \ |
945 | 366k | ParenthesisBlock, SquareBracketBlock, or CurlyBracketBlock \ |
946 | 366k | token was just consumed.\ |
947 | 366k | ", |
948 | | ); |
949 | 366k | if parser.current_block_depth >= parser.nested_block_limit && parser.nested_block_limit != 0 { |
950 | 7 | return Err(ParseError::from_basic_kind( |
951 | 7 | BasicParseErrorKind::TooManyNestedBlocks, |
952 | 7 | )); |
953 | 366k | } |
954 | | // Fine to use wrapping addition, overflow can only occur without a limit. |
955 | 366k | parser.current_block_depth = parser.current_block_depth.wrapping_add(1); |
956 | | |
957 | 366k | let old_stop_before = parser.stop_before; |
958 | 366k | parser.stop_before = match block_type { |
959 | 339k | BlockType::CurlyBracket => ClosingDelimiter::CloseCurlyBracket, |
960 | 7.45k | BlockType::SquareBracket => ClosingDelimiter::CloseSquareBracket, |
961 | 19.9k | BlockType::Parenthesis => ClosingDelimiter::CloseParenthesis, |
962 | | }; |
963 | 366k | let result = parser.parse_entirely(parse); |
964 | 366k | parser.skip_block_at_start(); |
965 | 366k | parser.consume_until_end_of_block(block_type); |
966 | 366k | parser.stop_before = old_stop_before; |
967 | 366k | parser.current_block_depth = parser.current_block_depth.wrapping_sub(1); |
968 | 366k | result |
969 | 366k | } |
970 | | |
971 | | impl Parser<'_> { |
972 | | /// Consume tokens until the end of a block of the given type that we're at the start of, |
973 | | /// ignoring any `stop_before` delimiters. |
974 | | #[inline(never)] |
975 | | #[cold] |
976 | 366k | pub(crate) fn consume_until_end_of_block(&mut self, block_type: BlockType) { |
977 | 366k | let mut stack = SmallVec::<[BlockType; 16]>::new(); |
978 | 366k | stack.push(block_type); |
979 | | |
980 | | // FIXME: have a special-purpose tokenizer method for this that does less work. |
981 | 2.61M | while !self.is_eof() { |
982 | 2.60M | let token = self.next_unchecked(); |
983 | 2.60M | if let Some(nested_block_type) = self.state.at_start_of.take() { |
984 | 1.34M | stack.push(nested_block_type); |
985 | 1.34M | continue; |
986 | 1.25M | } |
987 | 1.25M | if let Some(b) = BlockType::closing(&token) { |
988 | 359k | if *stack.last().unwrap() == b { |
989 | 357k | stack.pop(); |
990 | 357k | if stack.is_empty() { |
991 | 349k | return; |
992 | 7.97k | } |
993 | 2.52k | } |
994 | 895k | } |
995 | | } |
996 | 366k | } |
997 | | } |