Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/cssselect/parser.py: 74%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""
2cssselect.parser
3================
5Tokenizer, parser and parsed objects for CSS selectors.
8:copyright: (c) 2007-2012 Ian Bicking and contributors.
9See AUTHORS for more details.
10:license: BSD, see LICENSE for more details.
12"""
14from __future__ import annotations
16import operator
17import re
18import sys
19from typing import TYPE_CHECKING, Literal, Protocol, TypeAlias, Union, cast, overload
21if TYPE_CHECKING:
22 from collections.abc import Iterable, Iterator, Sequence
24 # typing.Self requires Python 3.11
25 from typing_extensions import Self
28def ascii_lower(string: str) -> str:
29 """Lower-case, but only in the ASCII range."""
30 return string.encode("utf8").lower().decode("utf8")
33class SelectorError(Exception):
34 """Common parent for :class:`SelectorSyntaxError` and
35 :class:`ExpressionError`.
37 You can just use ``except SelectorError:`` when calling
38 :meth:`~GenericTranslator.css_to_xpath` and handle both exceptions types.
40 """
43class SelectorSyntaxError(SelectorError, SyntaxError):
44 """Parsing a selector that does not match the grammar."""
47#### Parsed objects
49Tree: TypeAlias = Union[
50 "Element",
51 "Hash",
52 "Class",
53 "Function",
54 "Pseudo",
55 "Attrib",
56 "Negation",
57 "Relation",
58 "Matching",
59 "SpecificityAdjustment",
60 "CombinedSelector",
61]
62PseudoElement: TypeAlias = Union["FunctionalPseudoElement", str]
65class Selector:
66 """
67 Represents a parsed selector.
69 :meth:`~GenericTranslator.selector_to_xpath` accepts this object,
70 but ignores :attr:`pseudo_element`. It is the user’s responsibility
71 to account for pseudo-elements and reject selectors with unknown
72 or unsupported pseudo-elements.
74 """
76 def __init__(self, tree: Tree, pseudo_element: PseudoElement | None = None) -> None:
77 self.parsed_tree = tree
78 if pseudo_element is not None and not isinstance(
79 pseudo_element, FunctionalPseudoElement
80 ):
81 pseudo_element = ascii_lower(pseudo_element)
82 #: A :class:`FunctionalPseudoElement`,
83 #: or the identifier for the pseudo-element as a string,
84 # or ``None``.
85 #:
86 #: +-------------------------+----------------+--------------------------------+
87 #: | | Selector | Pseudo-element |
88 #: +=========================+================+================================+
89 #: | CSS3 syntax | ``a::before`` | ``'before'`` |
90 #: +-------------------------+----------------+--------------------------------+
91 #: | Older syntax | ``a:before`` | ``'before'`` |
92 #: +-------------------------+----------------+--------------------------------+
93 #: | From the Lists3_ draft, | ``li::marker`` | ``'marker'`` |
94 #: | not in Selectors3 | | |
95 #: +-------------------------+----------------+--------------------------------+
96 #: | Invalid pseudo-class | ``li:marker`` | ``None`` |
97 #: +-------------------------+----------------+--------------------------------+
98 #: | Functional | ``a::foo(2)`` | ``FunctionalPseudoElement(…)`` |
99 #: +-------------------------+----------------+--------------------------------+
100 #:
101 #: .. _Lists3: http://www.w3.org/TR/2011/WD-css3-lists-20110524/#marker-pseudoelement
102 self.pseudo_element = pseudo_element
104 def __repr__(self) -> str:
105 if isinstance(self.pseudo_element, FunctionalPseudoElement):
106 pseudo_element = repr(self.pseudo_element)
107 elif self.pseudo_element:
108 pseudo_element = f"::{self.pseudo_element}"
109 else:
110 pseudo_element = ""
111 return f"{self.__class__.__name__}[{self.parsed_tree!r}{pseudo_element}]"
113 def canonical(self) -> str:
114 """Return a CSS representation for this selector (a string)"""
115 if isinstance(self.pseudo_element, FunctionalPseudoElement):
116 pseudo_element = f"::{self.pseudo_element.canonical()}"
117 elif self.pseudo_element:
118 pseudo_element = f"::{_serialize_ident(self.pseudo_element)}"
119 else:
120 pseudo_element = ""
121 res = f"{self.parsed_tree.canonical()}{pseudo_element}"
122 # Strip a redundant universal selector from e.g. "*.foo" (but not
123 # from e.g. "* > foo").
124 if len(res) > 1 and res[0] == "*" and res[1] in "#.[:":
125 res = res[1:]
126 return res
128 def specificity(self) -> tuple[int, int, int]:
129 """Return the specificity_ of this selector as a tuple of 3 integers.
131 .. _specificity: http://www.w3.org/TR/selectors/#specificity
133 """
134 a, b, c = self.parsed_tree.specificity()
135 if self.pseudo_element:
136 c += 1
137 return a, b, c
140class Class:
141 """
142 Represents selector.class_name
143 """
145 def __init__(self, selector: Tree, class_name: str) -> None:
146 self.selector = selector
147 self.class_name = class_name
149 def __repr__(self) -> str:
150 return f"{self.__class__.__name__}[{self.selector!r}.{self.class_name}]"
152 def canonical(self) -> str:
153 return f"{self.selector.canonical()}.{_serialize_ident(self.class_name)}"
155 def specificity(self) -> tuple[int, int, int]:
156 a, b, c = self.selector.specificity()
157 b += 1
158 return a, b, c
161class FunctionalPseudoElement:
162 """
163 Represents selector::name(arguments)
165 .. attribute:: name
167 The name (identifier) of the pseudo-element, as a string.
169 .. attribute:: arguments
171 The arguments of the pseudo-element, as a list of tokens.
173 **Note:** tokens are not part of the public API,
174 and may change between cssselect versions.
175 Use at your own risks.
177 """
179 def __init__(self, name: str, arguments: Sequence[Token]):
180 self.name = ascii_lower(name)
181 self.arguments = arguments
183 def __repr__(self) -> str:
184 token_values = [token.value for token in self.arguments]
185 return f"{self.__class__.__name__}[::{self.name}({token_values!r})]"
187 def argument_types(self) -> list[str]:
188 return [token.type for token in self.arguments]
190 def canonical(self) -> str:
191 args = "".join(token.css() for token in self.arguments)
192 return f"{_serialize_ident(self.name)}({args})"
195class Function:
196 """
197 Represents selector:name(expr)
198 """
200 def __init__(self, selector: Tree, name: str, arguments: Sequence[Token]) -> None:
201 self.selector = selector
202 self.name = ascii_lower(name)
203 self.arguments = arguments
205 def __repr__(self) -> str:
206 token_values = [token.value for token in self.arguments]
207 return f"{self.__class__.__name__}[{self.selector!r}:{self.name}({token_values!r})]"
209 def argument_types(self) -> list[str]:
210 return [token.type for token in self.arguments]
212 def canonical(self) -> str:
213 args = "".join(token.css() for token in self.arguments)
214 return f"{self.selector.canonical()}:{_serialize_ident(self.name)}({args})"
216 def specificity(self) -> tuple[int, int, int]:
217 a, b, c = self.selector.specificity()
218 b += 1
219 return a, b, c
222class Pseudo:
223 """
224 Represents selector:ident
225 """
227 def __init__(self, selector: Tree, ident: str) -> None:
228 self.selector = selector
229 self.ident = ascii_lower(ident)
231 def __repr__(self) -> str:
232 return f"{self.__class__.__name__}[{self.selector!r}:{self.ident}]"
234 def canonical(self) -> str:
235 return f"{self.selector.canonical()}:{_serialize_ident(self.ident)}"
237 def specificity(self) -> tuple[int, int, int]:
238 a, b, c = self.selector.specificity()
239 b += 1
240 return a, b, c
243class Negation:
244 """
245 Represents selector:not(subselector)
246 """
248 def __init__(self, selector: Tree, subselector: Tree) -> None:
249 self.selector = selector
250 self.subselector = subselector
252 def __repr__(self) -> str:
253 return f"{self.__class__.__name__}[{self.selector!r}:not({self.subselector!r})]"
255 def canonical(self) -> str:
256 subsel = self.subselector.canonical()
257 # Strip a redundant universal selector from e.g. "*.foo" (but not
258 # from e.g. "* > foo").
259 if len(subsel) > 1 and subsel[0] == "*" and subsel[1] in "#.[:":
260 subsel = subsel[1:]
261 return f"{self.selector.canonical()}:not({subsel})"
263 def specificity(self) -> tuple[int, int, int]:
264 a1, b1, c1 = self.selector.specificity()
265 a2, b2, c2 = self.subselector.specificity()
266 return a1 + a2, b1 + b2, c1 + c2
269class Relation:
270 """
271 Represents selector:has(subselector)
272 """
274 def __init__(self, selector: Tree, combinator: Token, subselector: Selector):
275 self.selector = selector
276 self.combinator = combinator
277 self.subselector = subselector
279 def _combinator_prefix(self) -> str:
280 # The descendant combinator is implicit in :has() arguments.
281 if self.combinator.value == " ":
282 return ""
283 return f"{self.combinator.value} "
285 def __repr__(self) -> str:
286 return (
287 f"{self.__class__.__name__}[{self.selector!r}"
288 f":has({self._combinator_prefix()}{self.subselector!r})]"
289 )
291 def canonical(self) -> str:
292 subsel = self.subselector.canonical()
293 if len(subsel) > 1:
294 subsel = subsel.lstrip("*")
295 return f"{self.selector.canonical()}:has({self._combinator_prefix()}{subsel})"
297 def specificity(self) -> tuple[int, int, int]:
298 a1, b1, c1 = self.selector.specificity()
299 a2, b2, c2 = self.subselector.specificity()
300 return a1 + a2, b1 + b2, c1 + c2
303class Matching:
304 """
305 Represents selector:is(selector_list)
306 """
308 def __init__(self, selector: Tree, selector_list: Iterable[Tree]):
309 self.selector = selector
310 self.selector_list = selector_list
312 def __repr__(self) -> str:
313 args_str = ", ".join(repr(s) for s in self.selector_list)
314 return f"{self.__class__.__name__}[{self.selector!r}:is({args_str})]"
316 def canonical(self) -> str:
317 selector_arguments = []
318 for s in self.selector_list:
319 selarg = s.canonical()
320 if len(selarg) > 1:
321 selarg = selarg.lstrip("*")
322 selector_arguments.append(selarg)
323 args_str = ", ".join(selector_arguments)
324 return f"{self.selector.canonical()}:is({args_str})"
326 def specificity(self) -> tuple[int, int, int]:
327 a1, b1, c1 = self.selector.specificity()
328 a2, b2, c2 = max(x.specificity() for x in self.selector_list)
329 return a1 + a2, b1 + b2, c1 + c2
332class SpecificityAdjustment:
333 """
334 Represents selector:where(selector_list)
335 Same as selector:is(selector_list), but its specificity is always 0
336 """
338 def __init__(self, selector: Tree, selector_list: list[Tree]):
339 self.selector = selector
340 self.selector_list = selector_list
342 def __repr__(self) -> str:
343 args_str = ", ".join(repr(s) for s in self.selector_list)
344 return f"{self.__class__.__name__}[{self.selector!r}:where({args_str})]"
346 def canonical(self) -> str:
347 selector_arguments = []
348 for s in self.selector_list:
349 selarg = s.canonical()
350 if len(selarg) > 1:
351 selarg = selarg.lstrip("*")
352 selector_arguments.append(selarg)
353 args_str = ", ".join(selector_arguments)
354 return f"{self.selector.canonical()}:where({args_str})"
356 def specificity(self) -> tuple[int, int, int]:
357 # :where() itself contributes no specificity, but the compound
358 # selector it applies to does.
359 return self.selector.specificity()
362class Attrib:
363 """
364 Represents selector[namespace|attrib operator value flag]
366 *flag* is ``'i'`` for a case-insensitive value match, ``'s'`` for a
367 case-sensitive one, and `None` when the selector sets neither.
368 """
370 @overload
371 def __init__(
372 self,
373 selector: Tree,
374 namespace: str | None,
375 attrib: str,
376 operator: Literal["exists"],
377 value: None,
378 flag: None = None,
379 ) -> None: ...
381 @overload
382 def __init__(
383 self,
384 selector: Tree,
385 namespace: str | None,
386 attrib: str,
387 operator: str,
388 value: Token,
389 flag: str | None = None,
390 ) -> None: ...
392 def __init__(
393 self,
394 selector: Tree,
395 namespace: str | None,
396 attrib: str,
397 operator: str,
398 value: Token | None,
399 flag: str | None = None,
400 ) -> None:
401 self.selector = selector
402 self.namespace = namespace
403 self.attrib = attrib
404 self.operator = operator
405 self.value = value
406 self.flag = flag
408 def __repr__(self) -> str:
409 attrib = f"{self.namespace}|{self.attrib}" if self.namespace else self.attrib
410 if self.operator == "exists":
411 return f"{self.__class__.__name__}[{self.selector!r}[{attrib}]]"
412 assert self.value is not None
413 flag = f" {self.flag}" if self.flag else ""
414 return f"{self.__class__.__name__}[{self.selector!r}[{attrib} {self.operator} {self.value.value!r}{flag}]]"
416 def canonical(self) -> str:
417 attrib = _serialize_ident(self.attrib)
418 if self.namespace:
419 attrib = f"{_serialize_ident(self.namespace)}|{attrib}"
421 if self.operator == "exists":
422 op = attrib
423 else:
424 assert self.value is not None
425 flag = f" {self.flag}" if self.flag else ""
426 op = f"{attrib}{self.operator}{self.value.css()}{flag}"
428 return f"{self.selector.canonical()}[{op}]"
430 def specificity(self) -> tuple[int, int, int]:
431 a, b, c = self.selector.specificity()
432 b += 1
433 return a, b, c
436class Element:
437 """
438 Represents namespace|element
440 `None` is for the universal selector '*'
442 """
444 def __init__(
445 self, namespace: str | None = None, element: str | None = None
446 ) -> None:
447 self.namespace = namespace
448 self.element = element
450 def __repr__(self) -> str:
451 return f"{self.__class__.__name__}[{self.canonical()}]"
453 def canonical(self) -> str:
454 element = _serialize_ident(self.element) if self.element else "*"
455 if self.namespace:
456 element = f"{_serialize_ident(self.namespace)}|{element}"
457 return element
459 def specificity(self) -> tuple[int, int, int]:
460 if self.element:
461 return 0, 0, 1
462 return 0, 0, 0
465class Hash:
466 """
467 Represents selector#id
468 """
470 def __init__(self, selector: Tree, id: str) -> None: # noqa: A002
471 self.selector = selector
472 self.id = id
474 def __repr__(self) -> str:
475 return f"{self.__class__.__name__}[{self.selector!r}#{self.id}]"
477 def canonical(self) -> str:
478 return f"{self.selector.canonical()}#{_serialize_ident(self.id)}"
480 def specificity(self) -> tuple[int, int, int]:
481 a, b, c = self.selector.specificity()
482 a += 1
483 return a, b, c
486class CombinedSelector:
487 def __init__(self, selector: Tree, combinator: str, subselector: Tree) -> None:
488 assert selector is not None
489 self.selector = selector
490 self.combinator = combinator
491 self.subselector = subselector
493 def __repr__(self) -> str:
494 comb = "<followed>" if self.combinator == " " else self.combinator
495 return (
496 f"{self.__class__.__name__}[{self.selector!r} {comb} {self.subselector!r}]"
497 )
499 def canonical(self) -> str:
500 subsel = self.subselector.canonical()
501 if len(subsel) > 1:
502 subsel = subsel.lstrip("*")
503 combinator = " " if self.combinator == " " else f" {self.combinator} "
504 return f"{self.selector.canonical()}{combinator}{subsel}"
506 def specificity(self) -> tuple[int, int, int]:
507 a1, b1, c1 = self.selector.specificity()
508 a2, b2, c2 = self.subselector.specificity()
509 return a1 + a2, b1 + b2, c1 + c2
512#### Parser
514# foo
515_el_re = re.compile(r"^[ \t\r\n\f]*([a-zA-Z]+)[ \t\r\n\f]*$")
517# foo#bar or #bar
518_id_re = re.compile(r"^[ \t\r\n\f]*([a-zA-Z]*)#([a-zA-Z0-9_-]+)[ \t\r\n\f]*$")
520# foo.bar or .bar
521_class_re = re.compile(
522 r"^[ \t\r\n\f]*([a-zA-Z]*)\.([a-zA-Z][a-zA-Z0-9_-]*)[ \t\r\n\f]*$"
523)
526def parse(css: str) -> list[Selector]:
527 """Parse a CSS *group of selectors*.
529 If you don't care about pseudo-elements or selector specificity,
530 you can skip this and use :meth:`~GenericTranslator.css_to_xpath`.
532 :param css:
533 A *group of selectors* as a string.
534 :raises:
535 :class:`SelectorSyntaxError` on invalid selectors.
536 :returns:
537 A list of parsed :class:`Selector` objects, one for each
538 selector in the comma-separated group.
540 """
541 # Fast path for simple cases
542 match = _el_re.match(css)
543 if match:
544 return [Selector(Element(element=match.group(1)))]
545 match = _id_re.match(css)
546 if match is not None:
547 return [Selector(Hash(Element(element=match.group(1) or None), match.group(2)))]
548 match = _class_re.match(css)
549 if match is not None:
550 return [
551 Selector(Class(Element(element=match.group(1) or None), match.group(2)))
552 ]
554 stream = TokenStream(tokenize(css))
555 stream.source = css
556 return list(parse_selector_group(stream))
559# except SelectorSyntaxError:
560# e = sys.exc_info()[1]
561# message = "%s at %s -> %r" % (
562# e, stream.used, stream.peek())
563# e.msg = message
564# e.args = tuple([message])
565# raise
568def parse_selector_group(stream: TokenStream) -> Iterator[Selector]:
569 stream.skip_whitespace()
570 while 1:
571 yield Selector(*parse_selector(stream))
572 if stream.peek() == ("DELIM", ","):
573 stream.next()
574 stream.skip_whitespace()
575 else:
576 break
579def parse_selector(stream: TokenStream) -> tuple[Tree, PseudoElement | None]:
580 result, pseudo_element = parse_simple_selector(stream)
581 while 1:
582 stream.skip_whitespace()
583 peek = stream.peek()
584 if peek in (("EOF", None), ("DELIM", ",")):
585 break
586 if pseudo_element:
587 raise SelectorSyntaxError(
588 f"Got pseudo-element ::{pseudo_element} not at the end of a selector"
589 )
590 if peek.is_delim("+", ">", "~"):
591 # A combinator
592 combinator = cast("str", stream.next().value)
593 stream.skip_whitespace()
594 else:
595 # By exclusion, the last parse_simple_selector() ended
596 # at peek == ' '
597 combinator = " "
598 next_selector, pseudo_element = parse_simple_selector(stream)
599 result = CombinedSelector(result, combinator, next_selector)
600 return result, pseudo_element
603def parse_simple_selector(
604 stream: TokenStream,
605 inside_negation: bool = False,
606 inside_selector_list: bool = False,
607) -> tuple[Tree, PseudoElement | None]:
608 stream.skip_whitespace()
609 selector_start = len(stream.used)
610 peek = stream.peek()
611 if peek.type == "IDENT" or peek == ("DELIM", "*"):
612 if peek.type == "IDENT":
613 namespace = stream.next().value
614 else:
615 stream.next()
616 namespace = None
617 if stream.peek() == ("DELIM", "|"):
618 stream.next()
619 element = stream.next_ident_or_star()
620 else:
621 element = namespace
622 namespace = None
623 else:
624 element = namespace = None
625 result: Tree = Element(namespace, element)
626 pseudo_element: PseudoElement | None = None
627 while 1:
628 peek = stream.peek()
629 if (
630 peek.type in ("S", "EOF")
631 or peek.is_delim(",", "+", ">", "~")
632 or (inside_negation and peek == ("DELIM", ")"))
633 ):
634 break
635 if pseudo_element:
636 raise SelectorSyntaxError(
637 f"Got pseudo-element ::{pseudo_element} not at the end of a selector"
638 )
639 if peek.type == "HASH":
640 result = Hash(result, cast("str", stream.next().value))
641 elif peek == ("DELIM", "."):
642 stream.next()
643 result = Class(result, stream.next_ident())
644 elif peek == ("DELIM", "|"):
645 # The explicit "no namespace" syntax, e.g. |div: only valid at
646 # the very start of a simple selector.
647 if len(stream.used) != selector_start:
648 raise SelectorSyntaxError(f"Expected selector, got {peek}")
649 stream.next()
650 result = Element(None, stream.next_ident_or_star())
651 elif peek == ("DELIM", "["):
652 stream.next()
653 result = parse_attrib(result, stream)
654 elif peek == ("DELIM", ":"):
655 stream.next()
656 if stream.peek() == ("DELIM", ":"):
657 stream.next()
658 pseudo_element = stream.next_ident()
659 if stream.peek() == ("DELIM", "("):
660 stream.next()
661 pseudo_element = FunctionalPseudoElement(
662 pseudo_element, parse_arguments(stream)
663 )
664 continue
665 ident = stream.next_ident()
666 if ident.lower() in ("first-line", "first-letter", "before", "after"):
667 # Special case: CSS 2.1 pseudo-elements can have a single ':'
668 # Any new pseudo-element must have two.
669 pseudo_element = str(ident)
670 continue
671 if stream.peek() != ("DELIM", "("):
672 result = Pseudo(result, ident)
673 if result.ident == "scope":
674 # :scope is only supported at the start of a selector,
675 # i.e. never in :is()/:where()/:matches() arguments
676 # (where a preceding comma separates arguments, not
677 # selectors), and otherwise only when the tokens
678 # preceding its compound selector are the start of the
679 # input or a comma.
680 preceding = stream.used[:selector_start]
681 while preceding and preceding[-1].type == "S":
682 preceding = preceding[:-1]
683 if inside_selector_list or (
684 preceding and not preceding[-1].is_delim(",")
685 ):
686 raise SelectorSyntaxError(
687 'Got pseudo-class ":scope" not at the start of a selector'
688 )
689 continue
690 stream.next()
691 stream.skip_whitespace()
692 if ident.lower() == "not":
693 if inside_selector_list:
694 raise SelectorSyntaxError(
695 ":not() is not supported inside :is(), :where() and :matches()"
696 )
697 if inside_negation:
698 raise SelectorSyntaxError("Got nested :not()")
699 argument, argument_pseudo_element = parse_simple_selector(
700 stream, inside_negation=True
701 )
702 while 1:
703 # Whitespace before the closing parenthesis is not a
704 # descendant combinator.
705 stream.skip_whitespace()
706 peek = stream.peek()
707 if argument_pseudo_element:
708 raise SelectorSyntaxError(
709 f"Got pseudo-element ::{argument_pseudo_element} inside :not() at {peek.pos}"
710 )
711 if peek == ("DELIM", ")"):
712 stream.next()
713 break
714 if peek.is_delim("+", ">", "~"):
715 argument_combinator = cast("str", stream.next().value)
716 stream.skip_whitespace()
717 elif peek.type == "EOF" or peek.is_delim(","):
718 # A selector list is not supported in :not().
719 raise SelectorSyntaxError(f"Expected ')', got {peek}")
720 else:
721 argument_combinator = " "
722 next_selector, argument_pseudo_element = parse_simple_selector(
723 stream, inside_negation=True
724 )
725 argument = CombinedSelector(
726 argument, argument_combinator, next_selector
727 )
728 result = Negation(result, argument)
729 elif ident.lower() == "has":
730 combinator, arguments = parse_relative_selector(stream)
731 result = Relation(result, combinator, arguments)
733 elif ident.lower() in ("matches", "is"):
734 selectors = parse_simple_selector_arguments(stream)
735 result = Matching(result, selectors)
736 elif ident.lower() == "where":
737 selectors = parse_simple_selector_arguments(stream)
738 result = SpecificityAdjustment(result, selectors)
739 else:
740 result = Function(result, ident, parse_arguments(stream))
741 else:
742 raise SelectorSyntaxError(f"Expected selector, got {peek}")
743 if len(stream.used) == selector_start:
744 raise SelectorSyntaxError(f"Expected selector, got {stream.peek()}")
745 return result, pseudo_element
748def parse_arguments(stream: TokenStream) -> list[Token]: # noqa: RET503
749 arguments: list[Token] = []
750 while 1:
751 stream.skip_whitespace()
752 next_ = stream.next()
753 if next_.type in ("IDENT", "STRING", "NUMBER") or next_ in [
754 ("DELIM", "+"),
755 ("DELIM", "-"),
756 ]:
757 arguments.append(next_)
758 elif next_ == ("DELIM", ")"):
759 return arguments
760 else:
761 raise SelectorSyntaxError(f"Expected an argument, got {next_}")
764def parse_relative_selector(stream: TokenStream) -> tuple[Token, Selector]:
765 stream.skip_whitespace()
766 subselector_tokens: list[Token] = []
767 next_ = stream.next()
769 if next_.is_delim("+", ">", "~"):
770 combinator = next_
771 stream.skip_whitespace()
772 next_ = stream.next()
773 else:
774 combinator = Token("DELIM", " ", pos=0)
776 seen_whitespace = False
777 while 1:
778 if next_.type == "S":
779 # Whitespace is valid before the closing parenthesis; anywhere
780 # else it would be a descendant combinator, which is not
781 # supported in :has() arguments.
782 seen_whitespace = True
783 elif next_.type == "IDENT" or next_.is_delim(".", "*"):
784 if seen_whitespace:
785 raise SelectorSyntaxError(f"Expected an argument, got {next_}")
786 subselector_tokens.append(next_)
787 elif next_.is_delim(")"):
788 break
789 else:
790 raise SelectorSyntaxError(f"Expected an argument, got {next_}")
791 next_ = stream.next()
793 # Reparse the collected tokens instead of their concatenated source
794 # text, so that escaped identifiers are preserved.
795 subselector_tokens.append(EOFToken(next_.pos))
796 result, _ = parse_simple_selector(TokenStream(subselector_tokens))
797 return combinator, Selector(result)
800def parse_simple_selector_arguments(stream: TokenStream) -> list[Tree]:
801 arguments = []
802 while 1:
803 result, pseudo_element = parse_simple_selector(
804 stream, inside_negation=True, inside_selector_list=True
805 )
806 if pseudo_element:
807 raise SelectorSyntaxError(
808 f"Got pseudo-element ::{pseudo_element} inside function"
809 )
810 stream.skip_whitespace()
811 next_ = stream.next()
812 if next_ == ("DELIM", ","):
813 stream.skip_whitespace()
814 arguments.append(result)
815 elif next_ == ("DELIM", ")"):
816 arguments.append(result)
817 break
818 else:
819 raise SelectorSyntaxError(f"Expected an argument, got {next_}")
820 return arguments
823def parse_attrib(selector: Tree, stream: TokenStream) -> Attrib:
824 stream.skip_whitespace()
825 attrib = stream.next_ident_or_star()
826 if attrib is None and stream.peek() != ("DELIM", "|"):
827 raise SelectorSyntaxError(f"Expected '|', got {stream.peek()}")
828 namespace: str | None
829 op: str | None
830 if stream.peek() == ("DELIM", "|"):
831 stream.next()
832 if stream.peek() == ("DELIM", "="):
833 namespace = None
834 stream.next()
835 op = "|="
836 else:
837 namespace = attrib
838 attrib = stream.next_ident()
839 op = None
840 else:
841 namespace = op = None
842 if op is None:
843 stream.skip_whitespace()
844 next_ = stream.next()
845 if next_ == ("DELIM", "]"):
846 return Attrib(selector, namespace, cast("str", attrib), "exists", None)
847 if next_ == ("DELIM", "="):
848 op = "="
849 elif next_.is_delim("^", "$", "*", "~", "|", "!") and (
850 stream.peek() == ("DELIM", "=")
851 ):
852 op = cast("str", next_.value) + "="
853 stream.next()
854 else:
855 raise SelectorSyntaxError(f"Operator expected, got {next_}")
856 stream.skip_whitespace()
857 value = stream.next()
858 if value.type not in ("IDENT", "STRING"):
859 raise SelectorSyntaxError(f"Expected string or ident, got {value}")
860 stream.skip_whitespace()
861 next_ = stream.next()
862 flag = None
863 if next_.type == "IDENT":
864 flag = ascii_lower(cast("str", next_.value))
865 if flag not in ("i", "s"):
866 raise SelectorSyntaxError(f"Expected ']', got {next_}")
867 stream.skip_whitespace()
868 next_ = stream.next()
869 if next_ != ("DELIM", "]"):
870 raise SelectorSyntaxError(f"Expected ']', got {next_}")
871 return Attrib(selector, namespace, cast("str", attrib), op, value, flag)
874def parse_series(tokens: Iterable[Token]) -> tuple[int, int]:
875 """Parses the arguments for :nth-child() and friends."""
876 for token in tokens:
877 if token.type == "STRING":
878 raise ValueError("String tokens not allowed in series.")
879 # The An+B microsyntax is ASCII-case-insensitive: 2N+1, EVEN, Odd...
880 s = ascii_lower("".join(cast("str", token.value) for token in tokens).strip())
881 if s == "odd":
882 return 2, 1
883 if s == "even":
884 return 2, 0
885 if s == "n":
886 return 1, 0
887 if "n" not in s:
888 # Just b
889 return 0, int(s)
890 a, b = s.split("n", 1)
891 a_as_int: int
892 if not a:
893 a_as_int = 1
894 elif a in {"-", "+"}:
895 a_as_int = int(a + "1")
896 else:
897 a_as_int = int(a)
898 b_as_int = int(b) if b else 0
899 return a_as_int, b_as_int
902#### Token objects
905class Token(tuple[str, str | None]): # noqa: SLOT001
906 @overload
907 def __new__(
908 cls,
909 type_: Literal["IDENT", "HASH", "STRING", "S", "DELIM", "NUMBER"],
910 value: str,
911 pos: int,
912 ) -> Self: ...
914 @overload
915 def __new__(cls, type_: Literal["EOF"], value: None, pos: int) -> Self: ...
917 def __new__(cls, type_: str, value: str | None, pos: int) -> Self:
918 obj = tuple.__new__(cls, (type_, value))
919 obj.pos = pos
920 return obj
922 def __repr__(self) -> str:
923 return f"<{self.type} '{self.value}' at {self.pos}>"
925 def is_delim(self, *values: str) -> bool:
926 return self.type == "DELIM" and self.value in values
928 pos: int
930 @property
931 def type(self) -> str:
932 return self[0]
934 @property
935 def value(self) -> str | None:
936 return self[1]
938 def css(self) -> str:
939 if self.type == "STRING":
940 # Escape as CSS (repr() would use Python escapes, which mean
941 # something else in CSS, e.g. '\n' is just the letter 'n').
942 escaped = cast("str", self.value).replace("\\", "\\\\").replace("'", "\\'")
943 escaped = _sub_string_control_char(_replace_string_control_char, escaped)
944 return f"'{escaped}'"
945 if self.type == "IDENT":
946 return _serialize_ident(cast("str", self.value))
947 return cast("str", self.value)
950class EOFToken(Token):
951 def __new__(cls, pos: int) -> Self:
952 return Token.__new__(cls, "EOF", None, pos)
954 def __repr__(self) -> str:
955 return f"<{self.type} at {self.pos}>"
958#### Tokenizer
961class TokenMacros:
962 unicode_escape = r"\\([0-9a-f]{1,6})(?:\r\n|[ \n\r\t\f])?"
963 escape = unicode_escape + r"|\\[^\n\r\f0-9a-f]"
964 string_escape = r"\\(?:\n|\r\n|\r|\f)|" + escape
965 nonascii = r"[^\0-\177]"
966 nmchar = f"[_a-z0-9-]|{escape}|{nonascii}"
967 nmstart = f"[_a-z]|{escape}|{nonascii}"
970class MatchFunc(Protocol):
971 def __call__(
972 self, string: str, pos: int = ..., endpos: int = ...
973 ) -> re.Match[str] | None: ...
976def _compile(pattern: str) -> MatchFunc:
977 return re.compile(pattern % vars(TokenMacros), re.IGNORECASE).match
980_match_whitespace = _compile(r"[ \t\r\n\f]+")
981_match_number = _compile(r"[+-]?(?:[0-9]*\.[0-9]+|[0-9]+)")
982_match_hash = _compile("#(?:%(nmchar)s)+")
983_match_ident = _compile("-?(?:%(nmstart)s)(?:%(nmchar)s)*")
984_match_string_by_quote = {
985 "'": _compile(r"([^\n\r\f\\']|%(string_escape)s)*"),
986 '"': _compile(r'([^\n\r\f\\"]|%(string_escape)s)*'),
987}
989_sub_simple_escape = re.compile(r"\\(.)").sub
990_sub_unicode_escape = re.compile(TokenMacros.unicode_escape, re.IGNORECASE).sub
991_sub_newline_escape = re.compile(r"\\(?:\n|\r\n|\r|\f)").sub
992_sub_string_control_char = re.compile(r"[\x00-\x1f\x7f]").sub
994# CSS Syntax Level 3, §3.3: fold a raw U+0000 or surrogate code point to U+FFFD.
995_sub_invalid_input_char = re.compile("[\x00\ud800-\udfff]").sub
997# Same as r'\1', but faster on CPython
998_replace_simple = operator.methodcaller("group", 1)
1001def _replace_unicode(match: re.Match[str]) -> str:
1002 codepoint = int(match.group(1), 16)
1003 if codepoint == 0 or codepoint > sys.maxunicode or 0xD800 <= codepoint <= 0xDFFF:
1004 codepoint = 0xFFFD
1005 return chr(codepoint)
1008def _replace_string_control_char(match: re.Match[str]) -> str:
1009 # The trailing space ends the escape sequence, in case the next
1010 # character is a hexadecimal digit.
1011 return f"\\{ord(match.group()):x} "
1014def unescape_ident(value: str) -> str:
1015 value = _sub_unicode_escape(_replace_unicode, value)
1016 return _sub_simple_escape(_replace_simple, value)
1019def _serialize_ident(value: str) -> str:
1020 """Serialize a string as a CSS identifier, escaping special characters.
1022 Implements the CSSOM "serialize an identifier" algorithm:
1023 https://drafts.csswg.org/cssom/#serialize-an-identifier
1024 """
1025 result = []
1026 for i, char in enumerate(value):
1027 code = ord(char)
1028 serialized = char
1029 if code == 0:
1030 serialized = "\N{REPLACEMENT CHARACTER}"
1031 elif code <= 0x1F or code == 0x7F:
1032 serialized = f"\\{code:x} "
1033 elif "0" <= char <= "9":
1034 if i == 0 or (i == 1 and value[0] == "-"):
1035 # An identifier cannot start with a digit
1036 # (or a '-' followed by a digit).
1037 serialized = f"\\{code:x} "
1038 elif char == "-":
1039 if len(value) == 1 or (i == 0 and value[1] == "-"):
1040 # CSSOM leaves a leading "--" unescaped (such identifiers
1041 # are valid since CSS Syntax 3), but the tokenizer only
1042 # implements the CSS 2.1 identifier grammar and would not
1043 # be able to parse the result, so escape the first "-".
1044 serialized = "\\-"
1045 elif not (
1046 code >= 0x80 or char == "_" or "a" <= char <= "z" or "A" <= char <= "Z"
1047 ):
1048 serialized = f"\\{char}"
1049 result.append(serialized)
1050 return "".join(result)
1053def tokenize(s: str) -> Iterator[Token]:
1054 # Preprocess the input stream (§3.3); the substitution is length-preserving.
1055 s = _sub_invalid_input_char("\N{REPLACEMENT CHARACTER}", s)
1056 pos = 0
1057 len_s = len(s)
1058 while pos < len_s:
1059 match = _match_whitespace(s, pos=pos)
1060 if match:
1061 yield Token("S", " ", pos)
1062 pos = match.end()
1063 continue
1065 match = _match_ident(s, pos=pos)
1066 if match:
1067 value = unescape_ident(match.group())
1068 yield Token("IDENT", value, pos)
1069 pos = match.end()
1070 continue
1072 match = _match_hash(s, pos=pos)
1073 if match:
1074 value = unescape_ident(match.group()[1:])
1075 yield Token("HASH", value, pos)
1076 pos = match.end()
1077 continue
1079 quote = s[pos]
1080 if quote in _match_string_by_quote:
1081 match = _match_string_by_quote[quote](s, pos=pos + 1)
1082 assert match, "Should have found at least an empty match"
1083 end_pos = match.end()
1084 if end_pos == len_s:
1085 raise SelectorSyntaxError(f"Unclosed string at {pos}")
1086 if s[end_pos] != quote:
1087 raise SelectorSyntaxError(f"Invalid string at {pos}")
1088 value = _sub_simple_escape(
1089 _replace_simple,
1090 _sub_unicode_escape(
1091 _replace_unicode, _sub_newline_escape("", match.group())
1092 ),
1093 )
1094 yield Token("STRING", value, pos)
1095 pos = end_pos + 1
1096 continue
1098 match = _match_number(s, pos=pos)
1099 if match:
1100 value = match.group()
1101 yield Token("NUMBER", value, pos)
1102 pos = match.end()
1103 continue
1105 pos2 = pos + 2
1106 if s[pos:pos2] == "/*":
1107 pos = s.find("*/", pos2)
1108 if pos == -1:
1109 pos = len_s
1110 else:
1111 pos += 2
1112 continue
1114 yield Token("DELIM", s[pos], pos)
1115 pos += 1
1117 assert pos == len_s
1118 yield EOFToken(pos)
1121class TokenStream:
1122 def __init__(self, tokens: Iterable[Token], source: str | None = None) -> None:
1123 self.used: list[Token] = []
1124 self.tokens = iter(tokens)
1125 self.source = source
1126 self.peeked: Token | None = None
1127 self._peeking = False
1128 self.next_token = self.tokens.__next__
1130 def next(self) -> Token:
1131 if self._peeking:
1132 self._peeking = False
1133 assert self.peeked is not None
1134 self.used.append(self.peeked)
1135 return self.peeked
1136 next_ = self.next_token()
1137 self.used.append(next_)
1138 return next_
1140 def peek(self) -> Token:
1141 if not self._peeking:
1142 self.peeked = self.next_token()
1143 self._peeking = True
1144 assert self.peeked is not None
1145 return self.peeked
1147 def next_ident(self) -> str:
1148 next_ = self.next()
1149 if next_.type != "IDENT":
1150 raise SelectorSyntaxError(f"Expected ident, got {next_}")
1151 return cast("str", next_.value)
1153 def next_ident_or_star(self) -> str | None:
1154 next_ = self.next()
1155 if next_.type == "IDENT":
1156 return next_.value
1157 if next_ == ("DELIM", "*"):
1158 return None
1159 raise SelectorSyntaxError(f"Expected ident or '*', got {next_}")
1161 def skip_whitespace(self) -> None:
1162 # A comment between two whitespace runs yields two consecutive
1163 # whitespace tokens, so a single check is not enough.
1164 while self.peek().type == "S":
1165 self.next()