Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/cssselect/parser.py: 74%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

672 statements  

1""" 

2cssselect.parser 

3================ 

4 

5Tokenizer, parser and parsed objects for CSS selectors. 

6 

7 

8:copyright: (c) 2007-2012 Ian Bicking and contributors. 

9See AUTHORS for more details. 

10:license: BSD, see LICENSE for more details. 

11 

12""" 

13 

14from __future__ import annotations 

15 

16import operator 

17import re 

18import sys 

19from typing import TYPE_CHECKING, Literal, Protocol, TypeAlias, Union, cast, overload 

20 

21if TYPE_CHECKING: 

22 from collections.abc import Iterable, Iterator, Sequence 

23 

24 # typing.Self requires Python 3.11 

25 from typing_extensions import Self 

26 

27 

28def ascii_lower(string: str) -> str: 

29 """Lower-case, but only in the ASCII range.""" 

30 return string.encode("utf8").lower().decode("utf8") 

31 

32 

33class SelectorError(Exception): 

34 """Common parent for :class:`SelectorSyntaxError` and 

35 :class:`ExpressionError`. 

36 

37 You can just use ``except SelectorError:`` when calling 

38 :meth:`~GenericTranslator.css_to_xpath` and handle both exceptions types. 

39 

40 """ 

41 

42 

43class SelectorSyntaxError(SelectorError, SyntaxError): 

44 """Parsing a selector that does not match the grammar.""" 

45 

46 

47#### Parsed objects 

48 

49Tree: TypeAlias = Union[ 

50 "Element", 

51 "Hash", 

52 "Class", 

53 "Function", 

54 "Pseudo", 

55 "Attrib", 

56 "Negation", 

57 "Relation", 

58 "Matching", 

59 "SpecificityAdjustment", 

60 "CombinedSelector", 

61] 

62PseudoElement: TypeAlias = Union["FunctionalPseudoElement", str] 

63 

64 

65class Selector: 

66 """ 

67 Represents a parsed selector. 

68 

69 :meth:`~GenericTranslator.selector_to_xpath` accepts this object, 

70 but ignores :attr:`pseudo_element`. It is the user’s responsibility 

71 to account for pseudo-elements and reject selectors with unknown 

72 or unsupported pseudo-elements. 

73 

74 """ 

75 

76 def __init__(self, tree: Tree, pseudo_element: PseudoElement | None = None) -> None: 

77 self.parsed_tree = tree 

78 if pseudo_element is not None and not isinstance( 

79 pseudo_element, FunctionalPseudoElement 

80 ): 

81 pseudo_element = ascii_lower(pseudo_element) 

82 #: A :class:`FunctionalPseudoElement`, 

83 #: or the identifier for the pseudo-element as a string, 

84 # or ``None``. 

85 #: 

86 #: +-------------------------+----------------+--------------------------------+ 

87 #: | | Selector | Pseudo-element | 

88 #: +=========================+================+================================+ 

89 #: | CSS3 syntax | ``a::before`` | ``'before'`` | 

90 #: +-------------------------+----------------+--------------------------------+ 

91 #: | Older syntax | ``a:before`` | ``'before'`` | 

92 #: +-------------------------+----------------+--------------------------------+ 

93 #: | From the Lists3_ draft, | ``li::marker`` | ``'marker'`` | 

94 #: | not in Selectors3 | | | 

95 #: +-------------------------+----------------+--------------------------------+ 

96 #: | Invalid pseudo-class | ``li:marker`` | ``None`` | 

97 #: +-------------------------+----------------+--------------------------------+ 

98 #: | Functional | ``a::foo(2)`` | ``FunctionalPseudoElement(…)`` | 

99 #: +-------------------------+----------------+--------------------------------+ 

100 #: 

101 #: .. _Lists3: http://www.w3.org/TR/2011/WD-css3-lists-20110524/#marker-pseudoelement 

102 self.pseudo_element = pseudo_element 

103 

104 def __repr__(self) -> str: 

105 if isinstance(self.pseudo_element, FunctionalPseudoElement): 

106 pseudo_element = repr(self.pseudo_element) 

107 elif self.pseudo_element: 

108 pseudo_element = f"::{self.pseudo_element}" 

109 else: 

110 pseudo_element = "" 

111 return f"{self.__class__.__name__}[{self.parsed_tree!r}{pseudo_element}]" 

112 

113 def canonical(self) -> str: 

114 """Return a CSS representation for this selector (a string)""" 

115 if isinstance(self.pseudo_element, FunctionalPseudoElement): 

116 pseudo_element = f"::{self.pseudo_element.canonical()}" 

117 elif self.pseudo_element: 

118 pseudo_element = f"::{_serialize_ident(self.pseudo_element)}" 

119 else: 

120 pseudo_element = "" 

121 res = f"{self.parsed_tree.canonical()}{pseudo_element}" 

122 # Strip a redundant universal selector from e.g. "*.foo" (but not 

123 # from e.g. "* > foo"). 

124 if len(res) > 1 and res[0] == "*" and res[1] in "#.[:": 

125 res = res[1:] 

126 return res 

127 

128 def specificity(self) -> tuple[int, int, int]: 

129 """Return the specificity_ of this selector as a tuple of 3 integers. 

130 

131 .. _specificity: http://www.w3.org/TR/selectors/#specificity 

132 

133 """ 

134 a, b, c = self.parsed_tree.specificity() 

135 if self.pseudo_element: 

136 c += 1 

137 return a, b, c 

138 

139 

140class Class: 

141 """ 

142 Represents selector.class_name 

143 """ 

144 

145 def __init__(self, selector: Tree, class_name: str) -> None: 

146 self.selector = selector 

147 self.class_name = class_name 

148 

149 def __repr__(self) -> str: 

150 return f"{self.__class__.__name__}[{self.selector!r}.{self.class_name}]" 

151 

152 def canonical(self) -> str: 

153 return f"{self.selector.canonical()}.{_serialize_ident(self.class_name)}" 

154 

155 def specificity(self) -> tuple[int, int, int]: 

156 a, b, c = self.selector.specificity() 

157 b += 1 

158 return a, b, c 

159 

160 

161class FunctionalPseudoElement: 

162 """ 

163 Represents selector::name(arguments) 

164 

165 .. attribute:: name 

166 

167 The name (identifier) of the pseudo-element, as a string. 

168 

169 .. attribute:: arguments 

170 

171 The arguments of the pseudo-element, as a list of tokens. 

172 

173 **Note:** tokens are not part of the public API, 

174 and may change between cssselect versions. 

175 Use at your own risks. 

176 

177 """ 

178 

179 def __init__(self, name: str, arguments: Sequence[Token]): 

180 self.name = ascii_lower(name) 

181 self.arguments = arguments 

182 

183 def __repr__(self) -> str: 

184 token_values = [token.value for token in self.arguments] 

185 return f"{self.__class__.__name__}[::{self.name}({token_values!r})]" 

186 

187 def argument_types(self) -> list[str]: 

188 return [token.type for token in self.arguments] 

189 

190 def canonical(self) -> str: 

191 args = "".join(token.css() for token in self.arguments) 

192 return f"{_serialize_ident(self.name)}({args})" 

193 

194 

195class Function: 

196 """ 

197 Represents selector:name(expr) 

198 """ 

199 

200 def __init__(self, selector: Tree, name: str, arguments: Sequence[Token]) -> None: 

201 self.selector = selector 

202 self.name = ascii_lower(name) 

203 self.arguments = arguments 

204 

205 def __repr__(self) -> str: 

206 token_values = [token.value for token in self.arguments] 

207 return f"{self.__class__.__name__}[{self.selector!r}:{self.name}({token_values!r})]" 

208 

209 def argument_types(self) -> list[str]: 

210 return [token.type for token in self.arguments] 

211 

212 def canonical(self) -> str: 

213 args = "".join(token.css() for token in self.arguments) 

214 return f"{self.selector.canonical()}:{_serialize_ident(self.name)}({args})" 

215 

216 def specificity(self) -> tuple[int, int, int]: 

217 a, b, c = self.selector.specificity() 

218 b += 1 

219 return a, b, c 

220 

221 

222class Pseudo: 

223 """ 

224 Represents selector:ident 

225 """ 

226 

227 def __init__(self, selector: Tree, ident: str) -> None: 

228 self.selector = selector 

229 self.ident = ascii_lower(ident) 

230 

231 def __repr__(self) -> str: 

232 return f"{self.__class__.__name__}[{self.selector!r}:{self.ident}]" 

233 

234 def canonical(self) -> str: 

235 return f"{self.selector.canonical()}:{_serialize_ident(self.ident)}" 

236 

237 def specificity(self) -> tuple[int, int, int]: 

238 a, b, c = self.selector.specificity() 

239 b += 1 

240 return a, b, c 

241 

242 

243class Negation: 

244 """ 

245 Represents selector:not(subselector) 

246 """ 

247 

248 def __init__(self, selector: Tree, subselector: Tree) -> None: 

249 self.selector = selector 

250 self.subselector = subselector 

251 

252 def __repr__(self) -> str: 

253 return f"{self.__class__.__name__}[{self.selector!r}:not({self.subselector!r})]" 

254 

255 def canonical(self) -> str: 

256 subsel = self.subselector.canonical() 

257 # Strip a redundant universal selector from e.g. "*.foo" (but not 

258 # from e.g. "* > foo"). 

259 if len(subsel) > 1 and subsel[0] == "*" and subsel[1] in "#.[:": 

260 subsel = subsel[1:] 

261 return f"{self.selector.canonical()}:not({subsel})" 

262 

263 def specificity(self) -> tuple[int, int, int]: 

264 a1, b1, c1 = self.selector.specificity() 

265 a2, b2, c2 = self.subselector.specificity() 

266 return a1 + a2, b1 + b2, c1 + c2 

267 

268 

269class Relation: 

270 """ 

271 Represents selector:has(subselector) 

272 """ 

273 

274 def __init__(self, selector: Tree, combinator: Token, subselector: Selector): 

275 self.selector = selector 

276 self.combinator = combinator 

277 self.subselector = subselector 

278 

279 def _combinator_prefix(self) -> str: 

280 # The descendant combinator is implicit in :has() arguments. 

281 if self.combinator.value == " ": 

282 return "" 

283 return f"{self.combinator.value} " 

284 

285 def __repr__(self) -> str: 

286 return ( 

287 f"{self.__class__.__name__}[{self.selector!r}" 

288 f":has({self._combinator_prefix()}{self.subselector!r})]" 

289 ) 

290 

291 def canonical(self) -> str: 

292 subsel = self.subselector.canonical() 

293 if len(subsel) > 1: 

294 subsel = subsel.lstrip("*") 

295 return f"{self.selector.canonical()}:has({self._combinator_prefix()}{subsel})" 

296 

297 def specificity(self) -> tuple[int, int, int]: 

298 a1, b1, c1 = self.selector.specificity() 

299 a2, b2, c2 = self.subselector.specificity() 

300 return a1 + a2, b1 + b2, c1 + c2 

301 

302 

303class Matching: 

304 """ 

305 Represents selector:is(selector_list) 

306 """ 

307 

308 def __init__(self, selector: Tree, selector_list: Iterable[Tree]): 

309 self.selector = selector 

310 self.selector_list = selector_list 

311 

312 def __repr__(self) -> str: 

313 args_str = ", ".join(repr(s) for s in self.selector_list) 

314 return f"{self.__class__.__name__}[{self.selector!r}:is({args_str})]" 

315 

316 def canonical(self) -> str: 

317 selector_arguments = [] 

318 for s in self.selector_list: 

319 selarg = s.canonical() 

320 if len(selarg) > 1: 

321 selarg = selarg.lstrip("*") 

322 selector_arguments.append(selarg) 

323 args_str = ", ".join(selector_arguments) 

324 return f"{self.selector.canonical()}:is({args_str})" 

325 

326 def specificity(self) -> tuple[int, int, int]: 

327 a1, b1, c1 = self.selector.specificity() 

328 a2, b2, c2 = max(x.specificity() for x in self.selector_list) 

329 return a1 + a2, b1 + b2, c1 + c2 

330 

331 

332class SpecificityAdjustment: 

333 """ 

334 Represents selector:where(selector_list) 

335 Same as selector:is(selector_list), but its specificity is always 0 

336 """ 

337 

338 def __init__(self, selector: Tree, selector_list: list[Tree]): 

339 self.selector = selector 

340 self.selector_list = selector_list 

341 

342 def __repr__(self) -> str: 

343 args_str = ", ".join(repr(s) for s in self.selector_list) 

344 return f"{self.__class__.__name__}[{self.selector!r}:where({args_str})]" 

345 

346 def canonical(self) -> str: 

347 selector_arguments = [] 

348 for s in self.selector_list: 

349 selarg = s.canonical() 

350 if len(selarg) > 1: 

351 selarg = selarg.lstrip("*") 

352 selector_arguments.append(selarg) 

353 args_str = ", ".join(selector_arguments) 

354 return f"{self.selector.canonical()}:where({args_str})" 

355 

356 def specificity(self) -> tuple[int, int, int]: 

357 # :where() itself contributes no specificity, but the compound 

358 # selector it applies to does. 

359 return self.selector.specificity() 

360 

361 

362class Attrib: 

363 """ 

364 Represents selector[namespace|attrib operator value flag] 

365 

366 *flag* is ``'i'`` for a case-insensitive value match, ``'s'`` for a 

367 case-sensitive one, and `None` when the selector sets neither. 

368 """ 

369 

370 @overload 

371 def __init__( 

372 self, 

373 selector: Tree, 

374 namespace: str | None, 

375 attrib: str, 

376 operator: Literal["exists"], 

377 value: None, 

378 flag: None = None, 

379 ) -> None: ... 

380 

381 @overload 

382 def __init__( 

383 self, 

384 selector: Tree, 

385 namespace: str | None, 

386 attrib: str, 

387 operator: str, 

388 value: Token, 

389 flag: str | None = None, 

390 ) -> None: ... 

391 

392 def __init__( 

393 self, 

394 selector: Tree, 

395 namespace: str | None, 

396 attrib: str, 

397 operator: str, 

398 value: Token | None, 

399 flag: str | None = None, 

400 ) -> None: 

401 self.selector = selector 

402 self.namespace = namespace 

403 self.attrib = attrib 

404 self.operator = operator 

405 self.value = value 

406 self.flag = flag 

407 

408 def __repr__(self) -> str: 

409 attrib = f"{self.namespace}|{self.attrib}" if self.namespace else self.attrib 

410 if self.operator == "exists": 

411 return f"{self.__class__.__name__}[{self.selector!r}[{attrib}]]" 

412 assert self.value is not None 

413 flag = f" {self.flag}" if self.flag else "" 

414 return f"{self.__class__.__name__}[{self.selector!r}[{attrib} {self.operator} {self.value.value!r}{flag}]]" 

415 

416 def canonical(self) -> str: 

417 attrib = _serialize_ident(self.attrib) 

418 if self.namespace: 

419 attrib = f"{_serialize_ident(self.namespace)}|{attrib}" 

420 

421 if self.operator == "exists": 

422 op = attrib 

423 else: 

424 assert self.value is not None 

425 flag = f" {self.flag}" if self.flag else "" 

426 op = f"{attrib}{self.operator}{self.value.css()}{flag}" 

427 

428 return f"{self.selector.canonical()}[{op}]" 

429 

430 def specificity(self) -> tuple[int, int, int]: 

431 a, b, c = self.selector.specificity() 

432 b += 1 

433 return a, b, c 

434 

435 

436class Element: 

437 """ 

438 Represents namespace|element 

439 

440 `None` is for the universal selector '*' 

441 

442 """ 

443 

444 def __init__( 

445 self, namespace: str | None = None, element: str | None = None 

446 ) -> None: 

447 self.namespace = namespace 

448 self.element = element 

449 

450 def __repr__(self) -> str: 

451 return f"{self.__class__.__name__}[{self.canonical()}]" 

452 

453 def canonical(self) -> str: 

454 element = _serialize_ident(self.element) if self.element else "*" 

455 if self.namespace: 

456 element = f"{_serialize_ident(self.namespace)}|{element}" 

457 return element 

458 

459 def specificity(self) -> tuple[int, int, int]: 

460 if self.element: 

461 return 0, 0, 1 

462 return 0, 0, 0 

463 

464 

465class Hash: 

466 """ 

467 Represents selector#id 

468 """ 

469 

470 def __init__(self, selector: Tree, id: str) -> None: # noqa: A002 

471 self.selector = selector 

472 self.id = id 

473 

474 def __repr__(self) -> str: 

475 return f"{self.__class__.__name__}[{self.selector!r}#{self.id}]" 

476 

477 def canonical(self) -> str: 

478 return f"{self.selector.canonical()}#{_serialize_ident(self.id)}" 

479 

480 def specificity(self) -> tuple[int, int, int]: 

481 a, b, c = self.selector.specificity() 

482 a += 1 

483 return a, b, c 

484 

485 

486class CombinedSelector: 

487 def __init__(self, selector: Tree, combinator: str, subselector: Tree) -> None: 

488 assert selector is not None 

489 self.selector = selector 

490 self.combinator = combinator 

491 self.subselector = subselector 

492 

493 def __repr__(self) -> str: 

494 comb = "<followed>" if self.combinator == " " else self.combinator 

495 return ( 

496 f"{self.__class__.__name__}[{self.selector!r} {comb} {self.subselector!r}]" 

497 ) 

498 

499 def canonical(self) -> str: 

500 subsel = self.subselector.canonical() 

501 if len(subsel) > 1: 

502 subsel = subsel.lstrip("*") 

503 combinator = " " if self.combinator == " " else f" {self.combinator} " 

504 return f"{self.selector.canonical()}{combinator}{subsel}" 

505 

506 def specificity(self) -> tuple[int, int, int]: 

507 a1, b1, c1 = self.selector.specificity() 

508 a2, b2, c2 = self.subselector.specificity() 

509 return a1 + a2, b1 + b2, c1 + c2 

510 

511 

512#### Parser 

513 

514# foo 

515_el_re = re.compile(r"^[ \t\r\n\f]*([a-zA-Z]+)[ \t\r\n\f]*$") 

516 

517# foo#bar or #bar 

518_id_re = re.compile(r"^[ \t\r\n\f]*([a-zA-Z]*)#([a-zA-Z0-9_-]+)[ \t\r\n\f]*$") 

519 

520# foo.bar or .bar 

521_class_re = re.compile( 

522 r"^[ \t\r\n\f]*([a-zA-Z]*)\.([a-zA-Z][a-zA-Z0-9_-]*)[ \t\r\n\f]*$" 

523) 

524 

525 

526def parse(css: str) -> list[Selector]: 

527 """Parse a CSS *group of selectors*. 

528 

529 If you don't care about pseudo-elements or selector specificity, 

530 you can skip this and use :meth:`~GenericTranslator.css_to_xpath`. 

531 

532 :param css: 

533 A *group of selectors* as a string. 

534 :raises: 

535 :class:`SelectorSyntaxError` on invalid selectors. 

536 :returns: 

537 A list of parsed :class:`Selector` objects, one for each 

538 selector in the comma-separated group. 

539 

540 """ 

541 # Fast path for simple cases 

542 match = _el_re.match(css) 

543 if match: 

544 return [Selector(Element(element=match.group(1)))] 

545 match = _id_re.match(css) 

546 if match is not None: 

547 return [Selector(Hash(Element(element=match.group(1) or None), match.group(2)))] 

548 match = _class_re.match(css) 

549 if match is not None: 

550 return [ 

551 Selector(Class(Element(element=match.group(1) or None), match.group(2))) 

552 ] 

553 

554 stream = TokenStream(tokenize(css)) 

555 stream.source = css 

556 return list(parse_selector_group(stream)) 

557 

558 

559# except SelectorSyntaxError: 

560# e = sys.exc_info()[1] 

561# message = "%s at %s -> %r" % ( 

562# e, stream.used, stream.peek()) 

563# e.msg = message 

564# e.args = tuple([message]) 

565# raise 

566 

567 

568def parse_selector_group(stream: TokenStream) -> Iterator[Selector]: 

569 stream.skip_whitespace() 

570 while 1: 

571 yield Selector(*parse_selector(stream)) 

572 if stream.peek() == ("DELIM", ","): 

573 stream.next() 

574 stream.skip_whitespace() 

575 else: 

576 break 

577 

578 

579def parse_selector(stream: TokenStream) -> tuple[Tree, PseudoElement | None]: 

580 result, pseudo_element = parse_simple_selector(stream) 

581 while 1: 

582 stream.skip_whitespace() 

583 peek = stream.peek() 

584 if peek in (("EOF", None), ("DELIM", ",")): 

585 break 

586 if pseudo_element: 

587 raise SelectorSyntaxError( 

588 f"Got pseudo-element ::{pseudo_element} not at the end of a selector" 

589 ) 

590 if peek.is_delim("+", ">", "~"): 

591 # A combinator 

592 combinator = cast("str", stream.next().value) 

593 stream.skip_whitespace() 

594 else: 

595 # By exclusion, the last parse_simple_selector() ended 

596 # at peek == ' ' 

597 combinator = " " 

598 next_selector, pseudo_element = parse_simple_selector(stream) 

599 result = CombinedSelector(result, combinator, next_selector) 

600 return result, pseudo_element 

601 

602 

603def parse_simple_selector( 

604 stream: TokenStream, 

605 inside_negation: bool = False, 

606 inside_selector_list: bool = False, 

607) -> tuple[Tree, PseudoElement | None]: 

608 stream.skip_whitespace() 

609 selector_start = len(stream.used) 

610 peek = stream.peek() 

611 if peek.type == "IDENT" or peek == ("DELIM", "*"): 

612 if peek.type == "IDENT": 

613 namespace = stream.next().value 

614 else: 

615 stream.next() 

616 namespace = None 

617 if stream.peek() == ("DELIM", "|"): 

618 stream.next() 

619 element = stream.next_ident_or_star() 

620 else: 

621 element = namespace 

622 namespace = None 

623 else: 

624 element = namespace = None 

625 result: Tree = Element(namespace, element) 

626 pseudo_element: PseudoElement | None = None 

627 while 1: 

628 peek = stream.peek() 

629 if ( 

630 peek.type in ("S", "EOF") 

631 or peek.is_delim(",", "+", ">", "~") 

632 or (inside_negation and peek == ("DELIM", ")")) 

633 ): 

634 break 

635 if pseudo_element: 

636 raise SelectorSyntaxError( 

637 f"Got pseudo-element ::{pseudo_element} not at the end of a selector" 

638 ) 

639 if peek.type == "HASH": 

640 result = Hash(result, cast("str", stream.next().value)) 

641 elif peek == ("DELIM", "."): 

642 stream.next() 

643 result = Class(result, stream.next_ident()) 

644 elif peek == ("DELIM", "|"): 

645 # The explicit "no namespace" syntax, e.g. |div: only valid at 

646 # the very start of a simple selector. 

647 if len(stream.used) != selector_start: 

648 raise SelectorSyntaxError(f"Expected selector, got {peek}") 

649 stream.next() 

650 result = Element(None, stream.next_ident_or_star()) 

651 elif peek == ("DELIM", "["): 

652 stream.next() 

653 result = parse_attrib(result, stream) 

654 elif peek == ("DELIM", ":"): 

655 stream.next() 

656 if stream.peek() == ("DELIM", ":"): 

657 stream.next() 

658 pseudo_element = stream.next_ident() 

659 if stream.peek() == ("DELIM", "("): 

660 stream.next() 

661 pseudo_element = FunctionalPseudoElement( 

662 pseudo_element, parse_arguments(stream) 

663 ) 

664 continue 

665 ident = stream.next_ident() 

666 if ident.lower() in ("first-line", "first-letter", "before", "after"): 

667 # Special case: CSS 2.1 pseudo-elements can have a single ':' 

668 # Any new pseudo-element must have two. 

669 pseudo_element = str(ident) 

670 continue 

671 if stream.peek() != ("DELIM", "("): 

672 result = Pseudo(result, ident) 

673 if result.ident == "scope": 

674 # :scope is only supported at the start of a selector, 

675 # i.e. never in :is()/:where()/:matches() arguments 

676 # (where a preceding comma separates arguments, not 

677 # selectors), and otherwise only when the tokens 

678 # preceding its compound selector are the start of the 

679 # input or a comma. 

680 preceding = stream.used[:selector_start] 

681 while preceding and preceding[-1].type == "S": 

682 preceding = preceding[:-1] 

683 if inside_selector_list or ( 

684 preceding and not preceding[-1].is_delim(",") 

685 ): 

686 raise SelectorSyntaxError( 

687 'Got pseudo-class ":scope" not at the start of a selector' 

688 ) 

689 continue 

690 stream.next() 

691 stream.skip_whitespace() 

692 if ident.lower() == "not": 

693 if inside_selector_list: 

694 raise SelectorSyntaxError( 

695 ":not() is not supported inside :is(), :where() and :matches()" 

696 ) 

697 if inside_negation: 

698 raise SelectorSyntaxError("Got nested :not()") 

699 argument, argument_pseudo_element = parse_simple_selector( 

700 stream, inside_negation=True 

701 ) 

702 while 1: 

703 # Whitespace before the closing parenthesis is not a 

704 # descendant combinator. 

705 stream.skip_whitespace() 

706 peek = stream.peek() 

707 if argument_pseudo_element: 

708 raise SelectorSyntaxError( 

709 f"Got pseudo-element ::{argument_pseudo_element} inside :not() at {peek.pos}" 

710 ) 

711 if peek == ("DELIM", ")"): 

712 stream.next() 

713 break 

714 if peek.is_delim("+", ">", "~"): 

715 argument_combinator = cast("str", stream.next().value) 

716 stream.skip_whitespace() 

717 elif peek.type == "EOF" or peek.is_delim(","): 

718 # A selector list is not supported in :not(). 

719 raise SelectorSyntaxError(f"Expected ')', got {peek}") 

720 else: 

721 argument_combinator = " " 

722 next_selector, argument_pseudo_element = parse_simple_selector( 

723 stream, inside_negation=True 

724 ) 

725 argument = CombinedSelector( 

726 argument, argument_combinator, next_selector 

727 ) 

728 result = Negation(result, argument) 

729 elif ident.lower() == "has": 

730 combinator, arguments = parse_relative_selector(stream) 

731 result = Relation(result, combinator, arguments) 

732 

733 elif ident.lower() in ("matches", "is"): 

734 selectors = parse_simple_selector_arguments(stream) 

735 result = Matching(result, selectors) 

736 elif ident.lower() == "where": 

737 selectors = parse_simple_selector_arguments(stream) 

738 result = SpecificityAdjustment(result, selectors) 

739 else: 

740 result = Function(result, ident, parse_arguments(stream)) 

741 else: 

742 raise SelectorSyntaxError(f"Expected selector, got {peek}") 

743 if len(stream.used) == selector_start: 

744 raise SelectorSyntaxError(f"Expected selector, got {stream.peek()}") 

745 return result, pseudo_element 

746 

747 

748def parse_arguments(stream: TokenStream) -> list[Token]: # noqa: RET503 

749 arguments: list[Token] = [] 

750 while 1: 

751 stream.skip_whitespace() 

752 next_ = stream.next() 

753 if next_.type in ("IDENT", "STRING", "NUMBER") or next_ in [ 

754 ("DELIM", "+"), 

755 ("DELIM", "-"), 

756 ]: 

757 arguments.append(next_) 

758 elif next_ == ("DELIM", ")"): 

759 return arguments 

760 else: 

761 raise SelectorSyntaxError(f"Expected an argument, got {next_}") 

762 

763 

764def parse_relative_selector(stream: TokenStream) -> tuple[Token, Selector]: 

765 stream.skip_whitespace() 

766 subselector_tokens: list[Token] = [] 

767 next_ = stream.next() 

768 

769 if next_.is_delim("+", ">", "~"): 

770 combinator = next_ 

771 stream.skip_whitespace() 

772 next_ = stream.next() 

773 else: 

774 combinator = Token("DELIM", " ", pos=0) 

775 

776 seen_whitespace = False 

777 while 1: 

778 if next_.type == "S": 

779 # Whitespace is valid before the closing parenthesis; anywhere 

780 # else it would be a descendant combinator, which is not 

781 # supported in :has() arguments. 

782 seen_whitespace = True 

783 elif next_.type == "IDENT" or next_.is_delim(".", "*"): 

784 if seen_whitespace: 

785 raise SelectorSyntaxError(f"Expected an argument, got {next_}") 

786 subselector_tokens.append(next_) 

787 elif next_.is_delim(")"): 

788 break 

789 else: 

790 raise SelectorSyntaxError(f"Expected an argument, got {next_}") 

791 next_ = stream.next() 

792 

793 # Reparse the collected tokens instead of their concatenated source 

794 # text, so that escaped identifiers are preserved. 

795 subselector_tokens.append(EOFToken(next_.pos)) 

796 result, _ = parse_simple_selector(TokenStream(subselector_tokens)) 

797 return combinator, Selector(result) 

798 

799 

800def parse_simple_selector_arguments(stream: TokenStream) -> list[Tree]: 

801 arguments = [] 

802 while 1: 

803 result, pseudo_element = parse_simple_selector( 

804 stream, inside_negation=True, inside_selector_list=True 

805 ) 

806 if pseudo_element: 

807 raise SelectorSyntaxError( 

808 f"Got pseudo-element ::{pseudo_element} inside function" 

809 ) 

810 stream.skip_whitespace() 

811 next_ = stream.next() 

812 if next_ == ("DELIM", ","): 

813 stream.skip_whitespace() 

814 arguments.append(result) 

815 elif next_ == ("DELIM", ")"): 

816 arguments.append(result) 

817 break 

818 else: 

819 raise SelectorSyntaxError(f"Expected an argument, got {next_}") 

820 return arguments 

821 

822 

823def parse_attrib(selector: Tree, stream: TokenStream) -> Attrib: 

824 stream.skip_whitespace() 

825 attrib = stream.next_ident_or_star() 

826 if attrib is None and stream.peek() != ("DELIM", "|"): 

827 raise SelectorSyntaxError(f"Expected '|', got {stream.peek()}") 

828 namespace: str | None 

829 op: str | None 

830 if stream.peek() == ("DELIM", "|"): 

831 stream.next() 

832 if stream.peek() == ("DELIM", "="): 

833 namespace = None 

834 stream.next() 

835 op = "|=" 

836 else: 

837 namespace = attrib 

838 attrib = stream.next_ident() 

839 op = None 

840 else: 

841 namespace = op = None 

842 if op is None: 

843 stream.skip_whitespace() 

844 next_ = stream.next() 

845 if next_ == ("DELIM", "]"): 

846 return Attrib(selector, namespace, cast("str", attrib), "exists", None) 

847 if next_ == ("DELIM", "="): 

848 op = "=" 

849 elif next_.is_delim("^", "$", "*", "~", "|", "!") and ( 

850 stream.peek() == ("DELIM", "=") 

851 ): 

852 op = cast("str", next_.value) + "=" 

853 stream.next() 

854 else: 

855 raise SelectorSyntaxError(f"Operator expected, got {next_}") 

856 stream.skip_whitespace() 

857 value = stream.next() 

858 if value.type not in ("IDENT", "STRING"): 

859 raise SelectorSyntaxError(f"Expected string or ident, got {value}") 

860 stream.skip_whitespace() 

861 next_ = stream.next() 

862 flag = None 

863 if next_.type == "IDENT": 

864 flag = ascii_lower(cast("str", next_.value)) 

865 if flag not in ("i", "s"): 

866 raise SelectorSyntaxError(f"Expected ']', got {next_}") 

867 stream.skip_whitespace() 

868 next_ = stream.next() 

869 if next_ != ("DELIM", "]"): 

870 raise SelectorSyntaxError(f"Expected ']', got {next_}") 

871 return Attrib(selector, namespace, cast("str", attrib), op, value, flag) 

872 

873 

874def parse_series(tokens: Iterable[Token]) -> tuple[int, int]: 

875 """Parses the arguments for :nth-child() and friends.""" 

876 for token in tokens: 

877 if token.type == "STRING": 

878 raise ValueError("String tokens not allowed in series.") 

879 # The An+B microsyntax is ASCII-case-insensitive: 2N+1, EVEN, Odd... 

880 s = ascii_lower("".join(cast("str", token.value) for token in tokens).strip()) 

881 if s == "odd": 

882 return 2, 1 

883 if s == "even": 

884 return 2, 0 

885 if s == "n": 

886 return 1, 0 

887 if "n" not in s: 

888 # Just b 

889 return 0, int(s) 

890 a, b = s.split("n", 1) 

891 a_as_int: int 

892 if not a: 

893 a_as_int = 1 

894 elif a in {"-", "+"}: 

895 a_as_int = int(a + "1") 

896 else: 

897 a_as_int = int(a) 

898 b_as_int = int(b) if b else 0 

899 return a_as_int, b_as_int 

900 

901 

902#### Token objects 

903 

904 

905class Token(tuple[str, str | None]): # noqa: SLOT001 

906 @overload 

907 def __new__( 

908 cls, 

909 type_: Literal["IDENT", "HASH", "STRING", "S", "DELIM", "NUMBER"], 

910 value: str, 

911 pos: int, 

912 ) -> Self: ... 

913 

914 @overload 

915 def __new__(cls, type_: Literal["EOF"], value: None, pos: int) -> Self: ... 

916 

917 def __new__(cls, type_: str, value: str | None, pos: int) -> Self: 

918 obj = tuple.__new__(cls, (type_, value)) 

919 obj.pos = pos 

920 return obj 

921 

922 def __repr__(self) -> str: 

923 return f"<{self.type} '{self.value}' at {self.pos}>" 

924 

925 def is_delim(self, *values: str) -> bool: 

926 return self.type == "DELIM" and self.value in values 

927 

928 pos: int 

929 

930 @property 

931 def type(self) -> str: 

932 return self[0] 

933 

934 @property 

935 def value(self) -> str | None: 

936 return self[1] 

937 

938 def css(self) -> str: 

939 if self.type == "STRING": 

940 # Escape as CSS (repr() would use Python escapes, which mean 

941 # something else in CSS, e.g. '\n' is just the letter 'n'). 

942 escaped = cast("str", self.value).replace("\\", "\\\\").replace("'", "\\'") 

943 escaped = _sub_string_control_char(_replace_string_control_char, escaped) 

944 return f"'{escaped}'" 

945 if self.type == "IDENT": 

946 return _serialize_ident(cast("str", self.value)) 

947 return cast("str", self.value) 

948 

949 

950class EOFToken(Token): 

951 def __new__(cls, pos: int) -> Self: 

952 return Token.__new__(cls, "EOF", None, pos) 

953 

954 def __repr__(self) -> str: 

955 return f"<{self.type} at {self.pos}>" 

956 

957 

958#### Tokenizer 

959 

960 

961class TokenMacros: 

962 unicode_escape = r"\\([0-9a-f]{1,6})(?:\r\n|[ \n\r\t\f])?" 

963 escape = unicode_escape + r"|\\[^\n\r\f0-9a-f]" 

964 string_escape = r"\\(?:\n|\r\n|\r|\f)|" + escape 

965 nonascii = r"[^\0-\177]" 

966 nmchar = f"[_a-z0-9-]|{escape}|{nonascii}" 

967 nmstart = f"[_a-z]|{escape}|{nonascii}" 

968 

969 

970class MatchFunc(Protocol): 

971 def __call__( 

972 self, string: str, pos: int = ..., endpos: int = ... 

973 ) -> re.Match[str] | None: ... 

974 

975 

976def _compile(pattern: str) -> MatchFunc: 

977 return re.compile(pattern % vars(TokenMacros), re.IGNORECASE).match 

978 

979 

980_match_whitespace = _compile(r"[ \t\r\n\f]+") 

981_match_number = _compile(r"[+-]?(?:[0-9]*\.[0-9]+|[0-9]+)") 

982_match_hash = _compile("#(?:%(nmchar)s)+") 

983_match_ident = _compile("-?(?:%(nmstart)s)(?:%(nmchar)s)*") 

984_match_string_by_quote = { 

985 "'": _compile(r"([^\n\r\f\\']|%(string_escape)s)*"), 

986 '"': _compile(r'([^\n\r\f\\"]|%(string_escape)s)*'), 

987} 

988 

989_sub_simple_escape = re.compile(r"\\(.)").sub 

990_sub_unicode_escape = re.compile(TokenMacros.unicode_escape, re.IGNORECASE).sub 

991_sub_newline_escape = re.compile(r"\\(?:\n|\r\n|\r|\f)").sub 

992_sub_string_control_char = re.compile(r"[\x00-\x1f\x7f]").sub 

993 

994# CSS Syntax Level 3, §3.3: fold a raw U+0000 or surrogate code point to U+FFFD. 

995_sub_invalid_input_char = re.compile("[\x00\ud800-\udfff]").sub 

996 

997# Same as r'\1', but faster on CPython 

998_replace_simple = operator.methodcaller("group", 1) 

999 

1000 

1001def _replace_unicode(match: re.Match[str]) -> str: 

1002 codepoint = int(match.group(1), 16) 

1003 if codepoint == 0 or codepoint > sys.maxunicode or 0xD800 <= codepoint <= 0xDFFF: 

1004 codepoint = 0xFFFD 

1005 return chr(codepoint) 

1006 

1007 

1008def _replace_string_control_char(match: re.Match[str]) -> str: 

1009 # The trailing space ends the escape sequence, in case the next 

1010 # character is a hexadecimal digit. 

1011 return f"\\{ord(match.group()):x} " 

1012 

1013 

1014def unescape_ident(value: str) -> str: 

1015 value = _sub_unicode_escape(_replace_unicode, value) 

1016 return _sub_simple_escape(_replace_simple, value) 

1017 

1018 

1019def _serialize_ident(value: str) -> str: 

1020 """Serialize a string as a CSS identifier, escaping special characters. 

1021 

1022 Implements the CSSOM "serialize an identifier" algorithm: 

1023 https://drafts.csswg.org/cssom/#serialize-an-identifier 

1024 """ 

1025 result = [] 

1026 for i, char in enumerate(value): 

1027 code = ord(char) 

1028 serialized = char 

1029 if code == 0: 

1030 serialized = "\N{REPLACEMENT CHARACTER}" 

1031 elif code <= 0x1F or code == 0x7F: 

1032 serialized = f"\\{code:x} " 

1033 elif "0" <= char <= "9": 

1034 if i == 0 or (i == 1 and value[0] == "-"): 

1035 # An identifier cannot start with a digit 

1036 # (or a '-' followed by a digit). 

1037 serialized = f"\\{code:x} " 

1038 elif char == "-": 

1039 if len(value) == 1 or (i == 0 and value[1] == "-"): 

1040 # CSSOM leaves a leading "--" unescaped (such identifiers 

1041 # are valid since CSS Syntax 3), but the tokenizer only 

1042 # implements the CSS 2.1 identifier grammar and would not 

1043 # be able to parse the result, so escape the first "-". 

1044 serialized = "\\-" 

1045 elif not ( 

1046 code >= 0x80 or char == "_" or "a" <= char <= "z" or "A" <= char <= "Z" 

1047 ): 

1048 serialized = f"\\{char}" 

1049 result.append(serialized) 

1050 return "".join(result) 

1051 

1052 

1053def tokenize(s: str) -> Iterator[Token]: 

1054 # Preprocess the input stream (§3.3); the substitution is length-preserving. 

1055 s = _sub_invalid_input_char("\N{REPLACEMENT CHARACTER}", s) 

1056 pos = 0 

1057 len_s = len(s) 

1058 while pos < len_s: 

1059 match = _match_whitespace(s, pos=pos) 

1060 if match: 

1061 yield Token("S", " ", pos) 

1062 pos = match.end() 

1063 continue 

1064 

1065 match = _match_ident(s, pos=pos) 

1066 if match: 

1067 value = unescape_ident(match.group()) 

1068 yield Token("IDENT", value, pos) 

1069 pos = match.end() 

1070 continue 

1071 

1072 match = _match_hash(s, pos=pos) 

1073 if match: 

1074 value = unescape_ident(match.group()[1:]) 

1075 yield Token("HASH", value, pos) 

1076 pos = match.end() 

1077 continue 

1078 

1079 quote = s[pos] 

1080 if quote in _match_string_by_quote: 

1081 match = _match_string_by_quote[quote](s, pos=pos + 1) 

1082 assert match, "Should have found at least an empty match" 

1083 end_pos = match.end() 

1084 if end_pos == len_s: 

1085 raise SelectorSyntaxError(f"Unclosed string at {pos}") 

1086 if s[end_pos] != quote: 

1087 raise SelectorSyntaxError(f"Invalid string at {pos}") 

1088 value = _sub_simple_escape( 

1089 _replace_simple, 

1090 _sub_unicode_escape( 

1091 _replace_unicode, _sub_newline_escape("", match.group()) 

1092 ), 

1093 ) 

1094 yield Token("STRING", value, pos) 

1095 pos = end_pos + 1 

1096 continue 

1097 

1098 match = _match_number(s, pos=pos) 

1099 if match: 

1100 value = match.group() 

1101 yield Token("NUMBER", value, pos) 

1102 pos = match.end() 

1103 continue 

1104 

1105 pos2 = pos + 2 

1106 if s[pos:pos2] == "/*": 

1107 pos = s.find("*/", pos2) 

1108 if pos == -1: 

1109 pos = len_s 

1110 else: 

1111 pos += 2 

1112 continue 

1113 

1114 yield Token("DELIM", s[pos], pos) 

1115 pos += 1 

1116 

1117 assert pos == len_s 

1118 yield EOFToken(pos) 

1119 

1120 

1121class TokenStream: 

1122 def __init__(self, tokens: Iterable[Token], source: str | None = None) -> None: 

1123 self.used: list[Token] = [] 

1124 self.tokens = iter(tokens) 

1125 self.source = source 

1126 self.peeked: Token | None = None 

1127 self._peeking = False 

1128 self.next_token = self.tokens.__next__ 

1129 

1130 def next(self) -> Token: 

1131 if self._peeking: 

1132 self._peeking = False 

1133 assert self.peeked is not None 

1134 self.used.append(self.peeked) 

1135 return self.peeked 

1136 next_ = self.next_token() 

1137 self.used.append(next_) 

1138 return next_ 

1139 

1140 def peek(self) -> Token: 

1141 if not self._peeking: 

1142 self.peeked = self.next_token() 

1143 self._peeking = True 

1144 assert self.peeked is not None 

1145 return self.peeked 

1146 

1147 def next_ident(self) -> str: 

1148 next_ = self.next() 

1149 if next_.type != "IDENT": 

1150 raise SelectorSyntaxError(f"Expected ident, got {next_}") 

1151 return cast("str", next_.value) 

1152 

1153 def next_ident_or_star(self) -> str | None: 

1154 next_ = self.next() 

1155 if next_.type == "IDENT": 

1156 return next_.value 

1157 if next_ == ("DELIM", "*"): 

1158 return None 

1159 raise SelectorSyntaxError(f"Expected ident or '*', got {next_}") 

1160 

1161 def skip_whitespace(self) -> None: 

1162 # A comment between two whitespace runs yields two consecutive 

1163 # whitespace tokens, so a single check is not enough. 

1164 while self.peek().type == "S": 

1165 self.next()