/rust/registry/src/index.crates.io-1949cf8c6b5b557f/ammonia-4.1.2/src/style.rs
Line | Count | Source |
1 | | //! HTML living standard 3.2.6.5 The `style` attribute: |
2 | | //! |
3 | | //! > All HTML elements may have the `style` content attribute set. |
4 | | //! > This is a style attribute as defined by [CSS Style Attributes](CSSATTR). |
5 | | //! |
6 | | //! CSSATTR 3. Syntax and Parsing |
7 | | //! |
8 | | //! The value of the style attribute must match the syntax of the contents of a CSS |
9 | | //! declaration block (excluding the delimiting braces), whose formal grammar is given |
10 | | //! below in the terms and conventions of the CSS core grammar: |
11 | | //! |
12 | | //! ```yacc |
13 | | //! style-attribute |
14 | | //! : S* declaration-list |
15 | | //! ; |
16 | | //! |
17 | | //! declaration-list |
18 | | //! : declaration [ ';' S* declaration-list ]? |
19 | | //! | at-rule declaration-list |
20 | | //! | /* empty */ |
21 | | //! ; |
22 | | //! ``` |
23 | | //! |
24 | | //! > Note that because there is no open brace delimiting the declaration list in |
25 | | //! > the CSS style attribute syntax, a close brace (`}`) in the style attribute's |
26 | | //! > value does not terminate the style data: it is merely an invalid token. |
27 | | //! |
28 | | //! > [...] Although the grammar allows it, no at-rule valid in style attributes is |
29 | | //! > define[d] at the moment. The forward-compatible parsing rules are such that |
30 | | //! > a declaration following an at-rule is *not* ignored |
31 | | //! |
32 | | //! [CSSATTR]: https://w3c.github.io/csswg-drafts/css-style-attr/ |
33 | | use std::collections::HashSet; |
34 | | |
35 | | use cssparser::{BasicParseErrorKind, DeclarationParser, ParseError, ParseErrorKind, Parser, ParserInput, ParserState, ToCss, Token}; |
36 | | |
37 | | |
38 | | |
39 | | /// Filters `style` to only keep the declarations whose property name are listed in |
40 | | /// `names`. Also normalises the style attribute by stripping broken declarations |
41 | | /// and constructs per [CSSATTR] rules. |
42 | 0 | pub fn filter_style_attribute( |
43 | 0 | style: &str, |
44 | 0 | names: &HashSet<&str>, |
45 | 0 | ) -> String { |
46 | | // add room for the trailing semicolon because we lazy |
47 | 0 | let mut out = String::with_capacity(style.len() + 1); |
48 | | |
49 | 0 | let mut input = ParserInput::new(style); |
50 | 0 | let mut p = Parser::new(&mut input); |
51 | | |
52 | | loop { |
53 | 0 | match parse_one_declaration(&mut p, names) { |
54 | 0 | Ok((name, value)) => { |
55 | 0 | if !name.is_empty() { |
56 | 0 | out.push_str(&name); |
57 | 0 | out.push(':'); |
58 | 0 | out.push_str(&value); |
59 | 0 | out.push(';'); |
60 | 0 | } |
61 | | }, |
62 | 0 | Err(e) => match e.kind { |
63 | 0 | ParseErrorKind::Basic(BasicParseErrorKind::EndOfInput) => break, |
64 | 0 | ParseErrorKind::Basic(BasicParseErrorKind::UnexpectedToken(Token::Semicolon)) => (), |
65 | 0 | ParseErrorKind::Basic(BasicParseErrorKind::UnexpectedToken(_)) => { |
66 | 0 | advance(&mut p); |
67 | 0 | }, |
68 | 0 | _ => unreachable!( |
69 | | "parse_one_declaration should only attempt to parse an ident, a colon, \ |
70 | | or a Declaration, so its only errors should be EOF or an unexpected token" |
71 | | ), |
72 | | }, |
73 | | } |
74 | | } |
75 | 0 | if !out.is_empty() { |
76 | 0 | // remove trailing semicolon (?) |
77 | 0 | out.pop(); |
78 | 0 | } |
79 | 0 | out |
80 | 0 | } |
81 | | |
82 | | |
83 | | /// The builtin parse_one_declaration errors on a declaration list, that is not what we want. |
84 | | /// |
85 | | /// Also we don't need the errorneous slice on failure, since we just skip. |
86 | | /// |
87 | | /// Finally, add property filtering directly so we don't need to pay for the |
88 | | /// `DeclarationParser::parse_value` if the property is not whitelisted. If |
89 | | /// a property is filtered out, it gets parsed as `("", "")`. |
90 | 0 | pub fn parse_one_declaration<'i, 't>( |
91 | 0 | input: &mut Parser<'i, 't>, |
92 | 0 | valid_properties: &HashSet<&str>, |
93 | 0 | ) -> Result<(cssparser::CowRcStr<'i>, String), ParseError<'i, ()>> |
94 | | { |
95 | 0 | let name = input.expect_ident()?.clone(); |
96 | 0 | if !valid_properties.contains(&*name) { |
97 | 0 | advance(input); |
98 | 0 | return Ok(("".into(), String::new())); |
99 | 0 | } |
100 | 0 | input.expect_colon()?; |
101 | 0 | Declarations.parse_value(name, input, &input.state()) |
102 | 0 | } |
103 | | |
104 | | |
105 | | struct Declarations; |
106 | | impl <'i> DeclarationParser<'i> for Declarations { |
107 | | type Declaration = (cssparser::CowRcStr<'i>, String); |
108 | | type Error = (); |
109 | | |
110 | 0 | fn parse_value<'t>( |
111 | 0 | &mut self, |
112 | 0 | name: cssparser::CowRcStr<'i>, |
113 | 0 | input: &mut Parser<'i, 't>, |
114 | 0 | _declaration_start: &ParserState, |
115 | 0 | ) -> Result<Self::Declaration, cssparser::ParseError<'i, Self::Error>> { |
116 | 0 | let mut value = String::new(); |
117 | | loop { |
118 | 0 | let t = match input.next() { |
119 | 0 | Err(e) if e.kind == cssparser::BasicParseErrorKind::EndOfInput => { |
120 | 0 | &Token::Semicolon |
121 | | } |
122 | 0 | t => t?, |
123 | | }; |
124 | | use Token::*; |
125 | 0 | match t { |
126 | | Semicolon => { |
127 | 0 | if value.chars().all(char::is_whitespace) { |
128 | 0 | return Ok(("".into(), String::new())); |
129 | 0 | } |
130 | 0 | break |
131 | | } |
132 | | |
133 | | BadString(_) | BadUrl(_) => { |
134 | 0 | let err = cssparser::BasicParseErrorKind::UnexpectedToken(t.clone()); |
135 | 0 | return Err(input.new_error(err)); |
136 | | } |
137 | | |
138 | | Function(_) => { |
139 | 0 | if !value.is_empty() && value.chars().last() != Some(' ') { |
140 | 0 | value.push(' '); |
141 | 0 | } |
142 | 0 | let Ok(_) = t.to_css(&mut value) else { |
143 | 0 | let err = cssparser::BasicParseErrorKind::UnexpectedToken(t.clone()); |
144 | 0 | return Err(input.new_error::<()>(err)); |
145 | | }; |
146 | 0 | input.parse_nested_block(|p| { |
147 | 0 | let mut first = true; |
148 | | loop { |
149 | 0 | match p.next() { |
150 | 0 | Ok(t) => { |
151 | 0 | if t.is_parse_error() { |
152 | 0 | let err = cssparser::BasicParseErrorKind::UnexpectedToken(t.clone()); |
153 | 0 | return Err(p.new_error(err)); |
154 | 0 | } |
155 | 0 | if !first && t != &Comma { |
156 | 0 | value.push(' '); |
157 | 0 | } |
158 | 0 | let Ok(_) = t.to_css(&mut value) else { |
159 | 0 | let err = cssparser::BasicParseErrorKind::UnexpectedToken(t.clone()); |
160 | 0 | return Err(p.new_error::<()>(err)); |
161 | | }; |
162 | 0 | first = false; |
163 | | } |
164 | 0 | Err(e) if e.kind == BasicParseErrorKind::EndOfInput => break Ok(()), |
165 | 0 | Err(e) => return Err(e.into()), |
166 | | } |
167 | | } |
168 | 0 | })?; |
169 | 0 | value.push(')'); |
170 | 0 | continue; |
171 | | } |
172 | | |
173 | 0 | _ => (), |
174 | | } |
175 | 0 | if !value.is_empty() && value.chars().last() != Some(' ') { |
176 | 0 | value.push(' '); |
177 | 0 | } |
178 | 0 | let Ok(_) = t.to_css(&mut value) else { |
179 | 0 | let err = cssparser::BasicParseErrorKind::UnexpectedToken(t.clone()); |
180 | 0 | return Err(input.new_error(err)); |
181 | | }; |
182 | | } |
183 | 0 | if value.chars().all(char::is_whitespace) { |
184 | 0 | Err(input.new_error(cssparser::BasicParseErrorKind::EndOfInput)) |
185 | | } else { |
186 | 0 | Ok((name, value)) |
187 | | } |
188 | 0 | } |
189 | | } |
190 | | |
191 | | // find end of declaration (EOF or semicolon) in order to recover |
192 | 0 | fn advance<'i, 't>(p: &mut Parser<'i, 't>) { |
193 | | loop { |
194 | 0 | match p.next() { |
195 | 0 | Ok(Token::Semicolon) => { return } |
196 | | // cssparser automatically handles paired delimiters, if we encounter a curly |
197 | | // bracket the next token is whatever follows the corresponding closing |
198 | | // bracket, which may be a new declaration |
199 | 0 | Ok(Token::CurlyBracketBlock) => { return }, |
200 | 0 | Err(e) if e.kind == cssparser::BasicParseErrorKind::EndOfInput => { return } |
201 | 0 | _ => () |
202 | | } |
203 | | |
204 | | } |
205 | 0 | } |
206 | | |
207 | | #[cfg(test)] |
208 | | mod tests { |
209 | | use super::filter_style_attribute; |
210 | | use std::{collections::HashSet, sync::LazyLock}; |
211 | | |
212 | | #[test] |
213 | | fn single_declaration() { |
214 | | assert_eq!( |
215 | | filter_style_attribute("font-style: italic", &HashSet::from(["font-style"])), |
216 | | "font-style:italic", |
217 | | ); |
218 | | } |
219 | | |
220 | | #[test] |
221 | | fn terminated_declaration() { |
222 | | assert_eq!( |
223 | | filter_style_attribute("font-style: italic;", &HashSet::from(["font-style"])), |
224 | | "font-style:italic", |
225 | | ); |
226 | | } |
227 | | |
228 | | #[test] |
229 | | fn complex() { |
230 | | assert_eq!( |
231 | | filter_style_attribute( |
232 | | "background: no-repeat center/80% url(\"../img/image.png\");", |
233 | | &HashSet::from(["background"]), |
234 | | ), |
235 | | "background:no-repeat center / 80% url(\"../img/image.png\")", |
236 | | ) |
237 | | } |
238 | | |
239 | | /// forward-compatible parsing rules should just skip the unknown / contextually invalid at-rule |
240 | | #[test] |
241 | | fn at_rule() { |
242 | | assert_eq!( |
243 | | filter_style_attribute( |
244 | | "@unsupported { splines: reticulating } color: green", |
245 | | &HashSet::from(["color", "splines"]), |
246 | | ), |
247 | | "color:green", |
248 | | ); |
249 | | } |
250 | | |
251 | | #[test] |
252 | | fn invalid_at_rules() { |
253 | | assert_eq!( |
254 | | filter_style_attribute("@charset 'utf-8'; color: green", &HashSet::from(["color"])), |
255 | | "color:green", |
256 | | ); |
257 | | assert_eq!( |
258 | | filter_style_attribute("@foo url(https://example.org); color: green", &HashSet::from(["color"])), |
259 | | "color:green", |
260 | | ); |
261 | | assert_eq!( |
262 | | filter_style_attribute("@media screen { color: red }; color: green", &HashSet::from(["color"])), |
263 | | "color:green", |
264 | | ); |
265 | | |
266 | | assert_eq!( |
267 | | filter_style_attribute("@scope (main) { div { color: red } }; color: green", &HashSet::from(["color"])), |
268 | | "color:green", |
269 | | ); |
270 | | } |
271 | | |
272 | | #[test] |
273 | | fn empty_value() { |
274 | | assert_eq!( |
275 | | filter_style_attribute("content: ''", &HashSet::from(["content"])), |
276 | | "content:\"\"", |
277 | | ) |
278 | | } |
279 | | |
280 | | static ALLOWED: LazyLock<HashSet<&str>> = LazyLock::new(|| HashSet::from(["color", "foo"])); |
281 | | #[test] |
282 | | fn multiple() { |
283 | | assert_eq!(filter_style_attribute("foo: 1; color: green", &ALLOWED), "foo:1;color:green"); |
284 | | } |
285 | | |
286 | | /// https://www.w3.org/TR/CSS21/syndata.html#:~:text=malformed%20declarations |
287 | | #[test] |
288 | | fn malformed_declarations() { |
289 | | let h = &HashSet::from(["color"]); |
290 | | for decl in [ |
291 | | "color:green", |
292 | | "color:green; color", |
293 | | "color:green; color:", |
294 | | "color:green; color{;color:maroon}", |
295 | | ] { |
296 | | assert_eq!( |
297 | | filter_style_attribute(decl, h), |
298 | | "color:green", |
299 | | "{}", decl, |
300 | | ); |
301 | | } |
302 | | // should we also keep track of properties and remove duplicates? |
303 | | for decl in [ |
304 | | "color:red; color; color:green", |
305 | | "color:red; color:; color:green", |
306 | | "color:red; color{;color:maroon}; color:green", |
307 | | ] { |
308 | | assert_eq!( |
309 | | filter_style_attribute(decl, h), |
310 | | "color:red;color:green", |
311 | | "{}", decl, |
312 | | ); |
313 | | } |
314 | | } |
315 | | |
316 | | #[ignore = "can't recover from such a BadString (servo/rust-cssparser#393)"] |
317 | | #[test] |
318 | | fn badstring_escaped_newline() { |
319 | | assert_eq!(filter_style_attribute("foo: '\n'; color: green", &ALLOWED), "color:green"); |
320 | | } |
321 | | |
322 | | #[ignore = "can't recover from such a BadString (servo/rust-cssparser#393)"] |
323 | | #[test] |
324 | | fn badstring_literal_newline() { |
325 | | assert_eq!(filter_style_attribute("foo: ' |
326 | | '; color: green", &ALLOWED), "color:green"); |
327 | | } |
328 | | |
329 | | #[test] |
330 | | fn bad_url() { |
331 | | assert_eq!(filter_style_attribute("foo: url(x'y); color: green", &ALLOWED), "color:green"); |
332 | | } |
333 | | } |