Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/cssselect/xpath.py: 77%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

355 statements  

1""" 

2cssselect.xpath 

3=============== 

4 

5Translation of parsed CSS selectors to XPath expressions. 

6 

7 

8:copyright: (c) 2007-2012 Ian Bicking and contributors. 

9See AUTHORS for more details. 

10:license: BSD, see LICENSE for more details. 

11 

12""" 

13 

14from __future__ import annotations 

15 

16import re 

17from string import ascii_lowercase, ascii_uppercase 

18from typing import TYPE_CHECKING, cast 

19 

20from cssselect.parser import ( 

21 Attrib, 

22 Class, 

23 CombinedSelector, 

24 Element, 

25 Function, 

26 Hash, 

27 Matching, 

28 Negation, 

29 Pseudo, 

30 PseudoElement, 

31 Relation, 

32 Selector, 

33 SelectorError, 

34 SpecificityAdjustment, 

35 Tree, 

36 ascii_lower, 

37 parse, 

38 parse_series, 

39) 

40 

41if TYPE_CHECKING: 

42 from collections.abc import Callable, Iterable 

43 

44 # typing.Self requires Python 3.11 

45 from typing_extensions import Self 

46 

47 

48class ExpressionError(SelectorError, RuntimeError): 

49 """Unknown or unsupported selector (eg. pseudo-class).""" 

50 

51 

52#### XPath Helpers 

53 

54 

55class XPathExpr: 

56 def __init__( 

57 self, 

58 path: str = "", 

59 element: str = "*", 

60 condition: str = "", 

61 star_prefix: bool = False, 

62 ) -> None: 

63 self.path = path 

64 self.element = element 

65 self.condition = condition 

66 

67 def __str__(self) -> str: 

68 path = str(self.path) + str(self.element) 

69 if self.condition: 

70 path += f"[{self.condition}]" 

71 return path 

72 

73 def __repr__(self) -> str: 

74 return f"{self.__class__.__name__}[{self}]" 

75 

76 def add_condition(self, condition: str, conjuction: str = "and") -> Self: 

77 if self.condition: 

78 self.condition = f"({self.condition}) {conjuction} ({condition})" 

79 else: 

80 self.condition = condition 

81 return self 

82 

83 def add_name_test(self) -> None: 

84 if self.element == "*": 

85 # We weren't doing a test anyway 

86 return 

87 prefix, colon, local = self.element.partition(":") 

88 if is_safe_name(prefix) and (not colon or local == "*" or is_safe_name(local)): 

89 # A node test, not a name() comparison: name() returns the 

90 # qualified name as written in the document, which would bypass 

91 # the XPath prefix mapping for a prefixed name like "ns:f" or 

92 # "ns:*" (from the CSS "ns|f" or "ns|*"), and would match 

93 # elements in a default namespace for an unprefixed name (which 

94 # the node test emitted for a bare "f" selector does not). 

95 self.add_condition(f"self::{self.element}") 

96 else: 

97 self.add_condition( 

98 f"name() = {GenericTranslator.xpath_literal(self.element)}" 

99 ) 

100 self.element = "*" 

101 

102 def add_star_prefix(self) -> None: 

103 """ 

104 Append '*/' to the path to keep the context constrained 

105 to a single parent. 

106 """ 

107 self.path += "*/" 

108 

109 def join(self, combiner: str, other: XPathExpr) -> Self: 

110 path = str(self) + combiner 

111 # Any "star prefix" is redundant when joining. 

112 if other.path != "*/": 

113 path += other.path 

114 self.path = path 

115 self.element = other.element 

116 self.condition = other.condition 

117 return self 

118 

119 

120split_at_single_quotes = re.compile("('+)").split 

121 

122# The spec is actually more permissive than that, but don’t bother. 

123# This is just for the fast path. 

124# http://www.w3.org/TR/REC-xml/#NT-NameStartChar 

125is_safe_name = re.compile("^[a-zA-Z_][a-zA-Z0-9_.-]*$").match 

126 

127# Test that the string is not empty and does not contain whitespace 

128is_non_whitespace = re.compile(r"^[^ \t\r\n\f]+$").match 

129 

130 

131#### Translation 

132 

133 

134class GenericTranslator: 

135 """ 

136 Translator for "generic" XML documents. 

137 

138 Everything is case-sensitive, no assumption is made on the meaning 

139 of element names and attribute names. 

140 

141 """ 

142 

143 #### 

144 #### HERE BE DRAGONS 

145 #### 

146 #### You are welcome to hook into this to change some behavior, 

147 #### but do so at your own risks. 

148 #### Until it has received a lot more work and review, 

149 #### I reserve the right to change this API in backward-incompatible ways 

150 #### with any minor version of cssselect. 

151 #### See https://github.com/scrapy/cssselect/pull/22 

152 #### -- Simon Sapin. 

153 #### 

154 

155 combinator_mapping = { 

156 " ": "descendant", 

157 ">": "child", 

158 "+": "direct_adjacent", 

159 "~": "indirect_adjacent", 

160 } 

161 

162 # Used to match a combinator against the context node, in :not(). 

163 _reverse_combinator_mapping = { 

164 " ": "ancestor::*", 

165 ">": "parent::*", 

166 "+": "preceding-sibling::*[1]", 

167 "~": "preceding-sibling::*", 

168 } 

169 

170 attribute_operator_mapping = { 

171 "exists": "exists", 

172 "=": "equals", 

173 "~=": "includes", 

174 "|=": "dashmatch", 

175 "^=": "prefixmatch", 

176 "$=": "suffixmatch", 

177 "*=": "substringmatch", 

178 "!=": "different", # not part of Selectors Level 3, but widely supported 

179 } 

180 

181 #: The attribute used for ID selectors depends on the document language: 

182 #: http://www.w3.org/TR/selectors/#id-selectors 

183 id_attribute = "id" 

184 

185 #: The attribute used for ``:lang()`` depends on the document language: 

186 #: http://www.w3.org/TR/selectors/#lang-pseudo 

187 lang_attribute = "xml:lang" 

188 

189 #: The case sensitivity of document language element names, 

190 #: attribute names, and attribute values in selectors depends 

191 #: on the document language. 

192 #: http://www.w3.org/TR/selectors/#casesens 

193 #: 

194 #: When a document language defines one of these as case-insensitive, 

195 #: cssselect assumes that the document parser makes the parsed values 

196 #: lower-case. Making the selector lower-case too makes the comparaison 

197 #: case-insensitive. 

198 #: 

199 #: In HTML, element names and attributes names (but not attribute values) 

200 #: are case-insensitive. All of lxml.html, html5lib, BeautifulSoup4 

201 #: and HTMLParser make them lower-case in their parse result, so 

202 #: the assumption holds. 

203 lower_case_element_names = False 

204 lower_case_attribute_names = False 

205 lower_case_attribute_values = False 

206 

207 # class used to represent and xpath expression 

208 xpathexpr_cls = XPathExpr 

209 

210 def css_to_xpath(self, css: str, prefix: str = "descendant-or-self::") -> str: 

211 """Translate a *group of selectors* to XPath. 

212 

213 Pseudo-elements are not supported here since XPath only knows 

214 about "real" elements. 

215 

216 :param css: 

217 A *group of selectors* as a string. 

218 :param prefix: 

219 This string is prepended to the XPath expression for each selector. 

220 The default makes selectors scoped to the context node’s subtree. 

221 :raises: 

222 :class:`~cssselect.SelectorSyntaxError` on invalid selectors, 

223 :class:`ExpressionError` on unknown/unsupported selectors, 

224 including pseudo-elements. 

225 :returns: 

226 The equivalent XPath 1.0 expression as a string. 

227 

228 """ 

229 return " | ".join( 

230 self.selector_to_xpath(selector, prefix, translate_pseudo_elements=True) 

231 for selector in parse(css) 

232 ) 

233 

234 def selector_to_xpath( 

235 self, 

236 selector: Selector, 

237 prefix: str = "descendant-or-self::", 

238 translate_pseudo_elements: bool = False, 

239 ) -> str: 

240 """Translate a parsed selector to XPath. 

241 

242 

243 :param selector: 

244 A parsed :class:`Selector` object. 

245 :param prefix: 

246 This string is prepended to the resulting XPath expression. 

247 The default makes selectors scoped to the context node’s subtree. 

248 :param translate_pseudo_elements: 

249 Unless this is set to ``True`` (as :meth:`css_to_xpath` does), 

250 the :attr:`~Selector.pseudo_element` attribute of the selector 

251 is ignored. 

252 It is the caller's responsibility to reject selectors 

253 with pseudo-elements, or to account for them somehow. 

254 :raises: 

255 :class:`ExpressionError` on unknown/unsupported selectors. 

256 :returns: 

257 The equivalent XPath 1.0 expression as a string. 

258 

259 """ 

260 tree = getattr(selector, "parsed_tree", None) 

261 if not tree: 

262 raise TypeError(f"Expected a parsed selector, got {selector!r}") 

263 xpath = self.xpath(tree) 

264 assert isinstance(xpath, self.xpathexpr_cls) # help debug a missing 'return' 

265 if translate_pseudo_elements and selector.pseudo_element: 

266 xpath = self.xpath_pseudo_element(xpath, selector.pseudo_element) 

267 return (prefix or "") + str(xpath) 

268 

269 def xpath_pseudo_element( 

270 self, xpath: XPathExpr, pseudo_element: PseudoElement 

271 ) -> XPathExpr: 

272 """Translate a pseudo-element. 

273 

274 Defaults to not supporting pseudo-elements at all, 

275 but can be overridden by sub-classes. 

276 

277 """ 

278 raise ExpressionError("Pseudo-elements are not supported.") 

279 

280 @staticmethod 

281 def xpath_literal(s: str) -> str: 

282 s = str(s) 

283 if "'" not in s: 

284 s = f"'{s}'" 

285 elif '"' not in s: 

286 s = f'"{s}"' 

287 else: 

288 parts_quoted = [ 

289 f'"{part}"' if "'" in part else f"'{part}'" 

290 for part in split_at_single_quotes(s) 

291 if part 

292 ] 

293 s = "concat({})".format(",".join(parts_quoted)) 

294 return s 

295 

296 def xpath(self, parsed_selector: Tree) -> XPathExpr: 

297 """Translate any parsed selector object.""" 

298 type_name = type(parsed_selector).__name__ 

299 method = cast( 

300 "Callable[[Tree], XPathExpr] | None", 

301 getattr(self, f"xpath_{type_name.lower()}", None), 

302 ) 

303 if method is None: 

304 raise ExpressionError(f"{type_name} is not supported.") 

305 return method(parsed_selector) 

306 

307 # Dispatched by parsed object type 

308 

309 def xpath_combinedselector(self, combined: CombinedSelector) -> XPathExpr: 

310 """Translate a combined selector.""" 

311 combinator = self.combinator_mapping[combined.combinator] 

312 method = cast( 

313 "Callable[[XPathExpr, XPathExpr], XPathExpr]", 

314 getattr(self, f"xpath_{combinator}_combinator"), 

315 ) 

316 return method(self.xpath(combined.selector), self.xpath(combined.subselector)) 

317 

318 def xpath_negation(self, negation: Negation) -> XPathExpr: 

319 xpath = self.xpath(negation.selector) 

320 condition = self._xpath_match_condition(negation.subselector) 

321 if condition is None: 

322 # The argument matches every element, so :not() matches none. 

323 return xpath.add_condition("0") 

324 return xpath.add_condition(f"not({condition})") 

325 

326 def _xpath_match_condition(self, selector: Tree) -> str | None: 

327 """Return a condition that holds for the elements matching *selector*, 

328 or None if that is every element. 

329 

330 Unlike xpath(), which walks a selector from left to right, this 

331 matches the whole selector against the context node, using reverse 

332 axes for combinators. 

333 """ 

334 if isinstance(selector, CombinedSelector): 

335 axis = self._reverse_combinator_mapping[selector.combinator] 

336 left = self._xpath_match_condition(selector.selector) 

337 if left is not None: 

338 axis = f"{axis}[{left}]" 

339 right = self._xpath_match_condition(selector.subselector) 

340 return axis if right is None else f"{right} and {axis}" 

341 sub_xpath = self.xpath(selector) 

342 sub_xpath.add_name_test() 

343 return sub_xpath.condition or None 

344 

345 def xpath_relation(self, relation: Relation) -> XPathExpr: 

346 xpath = self.xpath(relation.selector) 

347 combinator = relation.combinator 

348 subselector = relation.subselector 

349 right = self.xpath(subselector.parsed_tree) 

350 method = cast( 

351 "Callable[[XPathExpr, XPathExpr], XPathExpr]", 

352 getattr( 

353 self, 

354 f"xpath_relation_{self.combinator_mapping[cast('str', combinator.value)]}_combinator", 

355 ), 

356 ) 

357 return method(xpath, right) 

358 

359 def xpath_matching(self, matching: Matching) -> XPathExpr: 

360 return self._xpath_add_selector_list_condition( 

361 self.xpath(matching.selector), matching.selector_list 

362 ) 

363 

364 def xpath_specificityadjustment(self, matching: SpecificityAdjustment) -> XPathExpr: 

365 return self._xpath_add_selector_list_condition( 

366 self.xpath(matching.selector), matching.selector_list 

367 ) 

368 

369 def _xpath_add_selector_list_condition( 

370 self, xpath: XPathExpr, selector_list: Iterable[Tree] 

371 ) -> XPathExpr: 

372 """Add a condition matching any selector of the list 

373 (for :is() and :where()).""" 

374 condition = "" 

375 for e in (self.xpath(selector) for selector in selector_list): 

376 if e.path: 

377 # Only a combined selector (e.g. "a b") translates to a path, 

378 # which cannot be embedded into a predicate of the outer 

379 # expression. The parser rejects combinators in these arguments, 

380 # so this is only reachable through a hand-built Matching or 

381 # SpecificityAdjustment node. 

382 raise ExpressionError( 

383 "Combined selectors are not supported inside " 

384 ":is(), :where() and :matches()" 

385 ) 

386 e.add_name_test() 

387 if not e.condition: 

388 # This argument matches any element, so the whole selector 

389 # list does too: it adds no condition. 

390 return xpath 

391 condition = ( 

392 f"({condition}) or ({e.condition})" if condition else e.condition 

393 ) 

394 return xpath.add_condition(condition) 

395 

396 def xpath_function(self, function: Function) -> XPathExpr: 

397 """Translate a functional pseudo-class.""" 

398 method_name = "xpath_{}_function".format(function.name.replace("-", "_")) 

399 method = cast( 

400 "Callable[[XPathExpr, Function], XPathExpr] | None", 

401 getattr(self, method_name, None), 

402 ) 

403 if not method: 

404 raise ExpressionError(f"The pseudo-class :{function.name}() is unknown") 

405 return method(self.xpath(function.selector), function) 

406 

407 def xpath_pseudo(self, pseudo: Pseudo) -> XPathExpr: 

408 """Translate a pseudo-class.""" 

409 method_name = "xpath_{}_pseudo".format(pseudo.ident.replace("-", "_")) 

410 method = cast( 

411 "Callable[[XPathExpr], XPathExpr] | None", 

412 getattr(self, method_name, None), 

413 ) 

414 if not method: 

415 raise ExpressionError(f"The pseudo-class :{pseudo.ident} is unknown") 

416 return method(self.xpath(pseudo.selector)) 

417 

418 def xpath_attrib(self, selector: Attrib) -> XPathExpr: 

419 """Translate an attribute selector.""" 

420 operator = self.attribute_operator_mapping[selector.operator] 

421 method = cast( 

422 "Callable[[XPathExpr, str, str | None], XPathExpr]", 

423 getattr(self, f"xpath_attrib_{operator}"), 

424 ) 

425 if self.lower_case_attribute_names: 

426 name = selector.attrib.lower() 

427 else: 

428 name = selector.attrib 

429 safe = is_safe_name(name) 

430 if selector.namespace: 

431 name = f"{selector.namespace}:{name}" 

432 safe = safe and is_safe_name(selector.namespace) 

433 if safe: 

434 attrib = "@" + name 

435 else: 

436 attrib = f"attribute::*[name() = {self.xpath_literal(name)}]" 

437 if selector.value is None: 

438 value = None 

439 elif self.lower_case_attribute_values: 

440 value = cast("str", selector.value.value).lower() 

441 else: 

442 value = selector.value.value 

443 if selector.flag == "i" and value: 

444 # ASCII-lowering both sides is what the specification defines a 

445 # case-insensitive match as, and all XPath 1.0 can express. 

446 attrib = f"translate({attrib}, {self.xpath_literal(ascii_uppercase)}, {self.xpath_literal(ascii_lowercase)})" 

447 value = ascii_lower(value) 

448 return method(self.xpath(selector.selector), attrib, value) 

449 

450 def xpath_class(self, class_selector: Class) -> XPathExpr: 

451 """Translate a class selector.""" 

452 # .foo is defined as [class~=foo] in the spec. 

453 xpath = self.xpath(class_selector.selector) 

454 return self.xpath_attrib_includes(xpath, "@class", class_selector.class_name) 

455 

456 def xpath_hash(self, id_selector: Hash) -> XPathExpr: 

457 """Translate an ID selector.""" 

458 xpath = self.xpath(id_selector.selector) 

459 return self.xpath_attrib_equals(xpath, "@id", id_selector.id) 

460 

461 def xpath_element(self, selector: Element) -> XPathExpr: 

462 """Translate a type or universal selector.""" 

463 element = selector.element 

464 if not element: 

465 element = "*" 

466 safe = True 

467 else: 

468 safe = bool(is_safe_name(element)) 

469 if self.lower_case_element_names: 

470 element = element.lower() 

471 if selector.namespace: 

472 # Namespace prefixes are case-sensitive. 

473 # http://www.w3.org/TR/css3-namespace/#prefixes 

474 element = f"{selector.namespace}:{element}" 

475 safe = safe and bool(is_safe_name(selector.namespace)) 

476 xpath = self.xpathexpr_cls(element=element) 

477 if not safe: 

478 # Not usable as an XPath name test (e.g. an escaped identifier 

479 # like di\a0 v): compare the serialized name instead. Done here 

480 # rather than through add_name_test(), which would mistake a ":" 

481 # inside such a name for a namespace prefix separator. 

482 xpath.add_condition(f"name() = {self.xpath_literal(element)}") 

483 xpath.element = "*" 

484 return xpath 

485 

486 # CombinedSelector: dispatch by combinator 

487 

488 def xpath_descendant_combinator( 

489 self, left: XPathExpr, right: XPathExpr 

490 ) -> XPathExpr: 

491 """right is a child, grand-child or further descendant of left""" 

492 return left.join("/descendant-or-self::*/", right) 

493 

494 def xpath_child_combinator(self, left: XPathExpr, right: XPathExpr) -> XPathExpr: 

495 """right is an immediate child of left""" 

496 return left.join("/", right) 

497 

498 def xpath_direct_adjacent_combinator( 

499 self, left: XPathExpr, right: XPathExpr 

500 ) -> XPathExpr: 

501 """right is a sibling immediately after left""" 

502 xpath = left.join("/following-sibling::", right) 

503 xpath.add_name_test() 

504 return xpath.add_condition("position() = 1") 

505 

506 def xpath_indirect_adjacent_combinator( 

507 self, left: XPathExpr, right: XPathExpr 

508 ) -> XPathExpr: 

509 """right is a sibling after left, immediately or not""" 

510 return left.join("/following-sibling::", right) 

511 

512 # The relative selector is kept in `condition` (instead of being folded 

513 # into `path`/`element`) so that `element` stays a plain element name: 

514 # later steps such as :first-of-type or :not() read and rewrite it. 

515 

516 def xpath_relation_descendant_combinator( 

517 self, left: XPathExpr, right: XPathExpr 

518 ) -> XPathExpr: 

519 """right is a child, grand-child or further descendant of left; select left""" 

520 return left.add_condition(f"descendant::{right}") 

521 

522 def xpath_relation_child_combinator( 

523 self, left: XPathExpr, right: XPathExpr 

524 ) -> XPathExpr: 

525 """right is an immediate child of left; select left""" 

526 return left.add_condition(f"./{right}") 

527 

528 def xpath_relation_direct_adjacent_combinator( 

529 self, left: XPathExpr, right: XPathExpr 

530 ) -> XPathExpr: 

531 """right is a sibling immediately after left; select left""" 

532 right.add_name_test() 

533 right.add_condition("position() = 1") 

534 return left.add_condition(f"following-sibling::{right}") 

535 

536 def xpath_relation_indirect_adjacent_combinator( 

537 self, left: XPathExpr, right: XPathExpr 

538 ) -> XPathExpr: 

539 """right is a sibling after left, immediately or not; select left""" 

540 return left.add_condition(f"following-sibling::{right}") 

541 

542 # Function: dispatch by function/pseudo-class name 

543 

544 def xpath_nth_child_function( 

545 self, 

546 xpath: XPathExpr, 

547 function: Function, 

548 last: bool = False, 

549 add_name_test: bool = True, 

550 ) -> XPathExpr: 

551 try: 

552 a, b = parse_series(function.arguments) 

553 except ValueError as ex: 

554 raise ExpressionError(f"Invalid series: '{function.arguments!r}'") from ex 

555 

556 # From https://www.w3.org/TR/css3-selectors/#structural-pseudos: 

557 # 

558 # :nth-child(an+b) 

559 # an+b-1 siblings before 

560 # 

561 # :nth-last-child(an+b) 

562 # an+b-1 siblings after 

563 # 

564 # :nth-of-type(an+b) 

565 # an+b-1 siblings with the same expanded element name before 

566 # 

567 # :nth-last-of-type(an+b) 

568 # an+b-1 siblings with the same expanded element name after 

569 # 

570 # So, 

571 # for :nth-child and :nth-of-type 

572 # 

573 # count(preceding-sibling::<nodetest>) = an+b-1 

574 # 

575 # for :nth-last-child and :nth-last-of-type 

576 # 

577 # count(following-sibling::<nodetest>) = an+b-1 

578 # 

579 # therefore, 

580 # count(...) - (b-1) ≡ 0 (mod a) 

581 # 

582 # if a == 0: 

583 # ~~~~~~~~~~ 

584 # count(...) = b-1 

585 # 

586 # if a < 0: 

587 # ~~~~~~~~~ 

588 # count(...) - b +1 <= 0 

589 # -> count(...) <= b-1 

590 # 

591 # if a > 0: 

592 # ~~~~~~~~~ 

593 # count(...) - b +1 >= 0 

594 # -> count(...) >= b-1 

595 

596 # work with b-1 instead 

597 b_min_1 = b - 1 

598 

599 # early-exit condition 1: 

600 # ~~~~~~~~~~~~~~~~~~~~~~~ 

601 # for a == 1, nth-*(an+b) means n+b-1 siblings before/after, 

602 # and since n ∈ {0, 1, 2, ...}, if b-1<=0, 

603 # there is always an "n" matching any number of siblings (maybe none) 

604 if a == 1 and b_min_1 <= 0: 

605 return xpath 

606 

607 # early-exit condition 2: 

608 # ~~~~~~~~~~~~~~~~~~~~~~~ 

609 # an+b-1 siblings with a<0 and (b-1)<0 is not possible 

610 if a < 0 and b_min_1 < 0: 

611 return xpath.add_condition("0") 

612 

613 # `add_name_test` boolean is inverted and somewhat counter-intuitive: 

614 # 

615 # nth_of_type() calls nth_child(add_name_test=False) 

616 nodetest = "*" if add_name_test else f"{xpath.element}" 

617 

618 # count siblings before or after the element 

619 if not last: 

620 siblings_count = f"count(preceding-sibling::{nodetest})" 

621 else: 

622 siblings_count = f"count(following-sibling::{nodetest})" 

623 

624 # special case of fixed position: nth-*(0n+b) 

625 # if a == 0: 

626 # ~~~~~~~~~~ 

627 # count(***-sibling::***) = b-1 

628 if a == 0: 

629 return xpath.add_condition(f"{siblings_count} = {b_min_1}") 

630 

631 expressions = [] 

632 

633 if a > 0: 

634 # siblings count, an+b-1, is always >= 0, 

635 # so if a>0, and (b-1)<=0, an "n" exists to satisfy this, 

636 # therefore, the predicate is only interesting if (b-1)>0 

637 if b_min_1 > 0: 

638 expressions.append(f"{siblings_count} >= {b_min_1}") 

639 else: 

640 # if a<0, and (b-1)<0, no "n" satisfies this, 

641 # this is tested above as an early exit condition 

642 # otherwise, 

643 expressions.append(f"{siblings_count} <= {b_min_1}") 

644 

645 # operations modulo 1 or -1 are simpler, one only needs to verify: 

646 # 

647 # - either: 

648 # count(***-sibling::***) - (b-1) = n = 0, 1, 2, 3, etc., 

649 # i.e. count(***-sibling::***) >= (b-1) 

650 # 

651 # - or: 

652 # count(***-sibling::***) - (b-1) = -n = 0, -1, -2, -3, etc., 

653 # i.e. count(***-sibling::***) <= (b-1) 

654 # we just did above. 

655 # 

656 if abs(a) != 1: 

657 # count(***-sibling::***) - (b-1) ≡ 0 (mod a) 

658 left = siblings_count 

659 

660 # apply "modulo a" on 2nd term, -(b-1), 

661 # to simplify things like "(... +6) % -3", 

662 # and also make it positive with |a| 

663 b_neg = (-b_min_1) % abs(a) 

664 

665 if b_neg != 0: 

666 left = f"({left} +{b_neg})" 

667 

668 expressions.append(f"{left} mod {a} = 0") 

669 

670 template = "(%s)" if len(expressions) > 1 else "%s" 

671 xpath.add_condition( 

672 " and ".join(template % expression for expression in expressions) 

673 ) 

674 return xpath 

675 

676 def xpath_nth_last_child_function( 

677 self, xpath: XPathExpr, function: Function 

678 ) -> XPathExpr: 

679 return self.xpath_nth_child_function(xpath, function, last=True) 

680 

681 @staticmethod 

682 def _check_of_type_element(xpath: XPathExpr, pseudo: str) -> None: 

683 """Raise an exception if an -of-type pseudo-class can't be used with 

684 the given element. 

685 

686 For "*" and namespace wildcards like "ns:*" the type of the element is 

687 not known, and counting same-type siblings cannot be expressed as an 

688 XPath 1.0 node test. 

689 """ 

690 element = xpath.element 

691 if element == "*" or element.endswith(":*"): 

692 css_element = element.replace(":", "|") 

693 raise ExpressionError(f"{css_element}:{pseudo} is not implemented") 

694 

695 def xpath_nth_of_type_function( 

696 self, xpath: XPathExpr, function: Function 

697 ) -> XPathExpr: 

698 self._check_of_type_element(xpath, "nth-of-type()") 

699 return self.xpath_nth_child_function(xpath, function, add_name_test=False) 

700 

701 def xpath_nth_last_of_type_function( 

702 self, xpath: XPathExpr, function: Function 

703 ) -> XPathExpr: 

704 self._check_of_type_element(xpath, "nth-last-of-type()") 

705 return self.xpath_nth_child_function( 

706 xpath, function, last=True, add_name_test=False 

707 ) 

708 

709 def xpath_contains_function( 

710 self, xpath: XPathExpr, function: Function 

711 ) -> XPathExpr: 

712 # Defined there, removed in later drafts: 

713 # http://www.w3.org/TR/2001/CR-css3-selectors-20011113/#content-selectors 

714 if function.argument_types() not in (["STRING"], ["IDENT"]): 

715 raise ExpressionError( 

716 f"Expected a single string or ident for :contains(), got {function.arguments!r}" 

717 ) 

718 value = cast("str", function.arguments[0].value) 

719 return xpath.add_condition(f"contains(., {self.xpath_literal(value)})") 

720 

721 def xpath_lang_function(self, xpath: XPathExpr, function: Function) -> XPathExpr: 

722 if function.argument_types() not in (["STRING"], ["IDENT"]): 

723 raise ExpressionError( 

724 f"Expected a single string or ident for :lang(), got {function.arguments!r}" 

725 ) 

726 value = cast("str", function.arguments[0].value) 

727 return xpath.add_condition(f"lang({self.xpath_literal(value)})") 

728 

729 # Pseudo: dispatch by pseudo-class name 

730 

731 def xpath_root_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

732 return xpath.add_condition("not(parent::*)") 

733 

734 # CSS immediate children (CSS ":scope > div" to XPath "child::div" or "./div") 

735 # Works only at the start of a selector 

736 # Needed to get immediate children of a processed selector in Scrapy 

737 # for product in response.css('.product'): 

738 # description = product.css(':scope > div::text').get() 

739 def xpath_scope_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

740 xpath.add_name_test() 

741 return xpath.add_condition("position() = 1") 

742 

743 def xpath_first_child_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

744 return xpath.add_condition("count(preceding-sibling::*) = 0") 

745 

746 def xpath_last_child_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

747 return xpath.add_condition("count(following-sibling::*) = 0") 

748 

749 def xpath_first_of_type_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

750 self._check_of_type_element(xpath, "first-of-type") 

751 return xpath.add_condition(f"count(preceding-sibling::{xpath.element}) = 0") 

752 

753 def xpath_last_of_type_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

754 self._check_of_type_element(xpath, "last-of-type") 

755 return xpath.add_condition(f"count(following-sibling::{xpath.element}) = 0") 

756 

757 def xpath_only_child_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

758 # Count siblings, not the parent's children: the root element has 

759 # no parent, but it has no siblings either, so it must match. 

760 return xpath.add_condition( 

761 "count(preceding-sibling::*) = 0 and count(following-sibling::*) = 0" 

762 ) 

763 

764 def xpath_only_of_type_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

765 self._check_of_type_element(xpath, "only-of-type") 

766 return xpath.add_condition( 

767 f"count(preceding-sibling::{xpath.element}) = 0 " 

768 f"and count(following-sibling::{xpath.element}) = 0" 

769 ) 

770 

771 def xpath_empty_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

772 return xpath.add_condition("not(*) and not(string-length())") 

773 

774 def pseudo_never_matches(self, xpath: XPathExpr) -> XPathExpr: 

775 """Common implementation for pseudo-classes that never match.""" 

776 return xpath.add_condition("0") 

777 

778 xpath_link_pseudo = pseudo_never_matches 

779 xpath_visited_pseudo = pseudo_never_matches 

780 xpath_hover_pseudo = pseudo_never_matches 

781 xpath_active_pseudo = pseudo_never_matches 

782 xpath_focus_pseudo = pseudo_never_matches 

783 xpath_target_pseudo = pseudo_never_matches 

784 xpath_enabled_pseudo = pseudo_never_matches 

785 xpath_disabled_pseudo = pseudo_never_matches 

786 xpath_checked_pseudo = pseudo_never_matches 

787 

788 # Attrib: dispatch by attribute operator 

789 

790 def xpath_attrib_exists( 

791 self, xpath: XPathExpr, name: str, value: str | None 

792 ) -> XPathExpr: 

793 assert not value 

794 xpath.add_condition(name) 

795 return xpath 

796 

797 def xpath_attrib_equals( 

798 self, xpath: XPathExpr, name: str, value: str | None 

799 ) -> XPathExpr: 

800 assert value is not None 

801 xpath.add_condition(f"{name} = {self.xpath_literal(value)}") 

802 return xpath 

803 

804 def xpath_attrib_different( 

805 self, xpath: XPathExpr, name: str, value: str | None 

806 ) -> XPathExpr: 

807 assert value is not None 

808 # FIXME: this seems like a weird hack... 

809 if value: 

810 xpath.add_condition(f"not({name}) or {name} != {self.xpath_literal(value)}") 

811 else: 

812 xpath.add_condition(f"{name} != {self.xpath_literal(value)}") 

813 return xpath 

814 

815 def xpath_attrib_includes( 

816 self, xpath: XPathExpr, name: str, value: str | None 

817 ) -> XPathExpr: 

818 if value and is_non_whitespace(value): 

819 arg = self.xpath_literal(" " + value + " ") 

820 xpath.add_condition( 

821 f"{name} and contains(concat(' ', normalize-space({name}), ' '), {arg})" 

822 ) 

823 else: 

824 xpath.add_condition("0") 

825 return xpath 

826 

827 def xpath_attrib_dashmatch( 

828 self, xpath: XPathExpr, name: str, value: str | None 

829 ) -> XPathExpr: 

830 assert value is not None 

831 arg = self.xpath_literal(value) 

832 arg_dash = self.xpath_literal(value + "-") 

833 # Weird, but true... 

834 xpath.add_condition( 

835 f"{name} and ({name} = {arg} or starts-with({name}, {arg_dash}))" 

836 ) 

837 return xpath 

838 

839 def xpath_attrib_prefixmatch( 

840 self, xpath: XPathExpr, name: str, value: str | None 

841 ) -> XPathExpr: 

842 if value: 

843 xpath.add_condition( 

844 f"{name} and starts-with({name}, {self.xpath_literal(value)})" 

845 ) 

846 else: 

847 xpath.add_condition("0") 

848 return xpath 

849 

850 def xpath_attrib_suffixmatch( 

851 self, xpath: XPathExpr, name: str, value: str | None 

852 ) -> XPathExpr: 

853 if value: 

854 # Oddly there is a starts-with in XPath 1.0, but not ends-with 

855 xpath.add_condition( 

856 f"{name} and substring({name}, string-length({name})-{len(value) - 1}) = {self.xpath_literal(value)}" 

857 ) 

858 else: 

859 xpath.add_condition("0") 

860 return xpath 

861 

862 def xpath_attrib_substringmatch( 

863 self, xpath: XPathExpr, name: str, value: str | None 

864 ) -> XPathExpr: 

865 if value: 

866 # Attribute selectors are case sensitive 

867 xpath.add_condition( 

868 f"{name} and contains({name}, {self.xpath_literal(value)})" 

869 ) 

870 else: 

871 xpath.add_condition("0") 

872 return xpath 

873 

874 

875class HTMLTranslator(GenericTranslator): 

876 """ 

877 Translator for (X)HTML documents. 

878 

879 Has a more useful implementation of some pseudo-classes based on 

880 HTML-specific element names and attribute names, as described in 

881 the `HTML5 specification`_. It assumes no-quirks mode. 

882 The API is the same as :class:`GenericTranslator`. 

883 

884 .. _HTML5 specification: http://www.w3.org/TR/html5/links.html#selectors 

885 

886 :param xhtml: 

887 If false (the default), element names and attribute names 

888 are case-insensitive. 

889 

890 """ 

891 

892 lang_attribute = "lang" 

893 

894 def __init__(self, xhtml: bool = False) -> None: 

895 self.xhtml = xhtml # Might be useful for sub-classes? 

896 if not xhtml: 

897 # See their definition in GenericTranslator. 

898 self.lower_case_element_names = True 

899 self.lower_case_attribute_names = True 

900 

901 def xpath_checked_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

902 # FIXME: is this really all the elements? 

903 return xpath.add_condition( 

904 "(@selected and name(.) = 'option') or " 

905 "(@checked " 

906 "and (name(.) = 'input' or name(.) = 'command')" 

907 "and (@type = 'checkbox' or @type = 'radio'))" 

908 ) 

909 

910 def xpath_lang_function(self, xpath: XPathExpr, function: Function) -> XPathExpr: 

911 if function.argument_types() not in (["STRING"], ["IDENT"]): 

912 raise ExpressionError( 

913 f"Expected a single string or ident for :lang(), got {function.arguments!r}" 

914 ) 

915 value = function.arguments[0].value 

916 assert value 

917 arg = self.xpath_literal(value.lower() + "-") 

918 return xpath.add_condition( 

919 "ancestor-or-self::*[@lang][1][starts-with(concat(" 

920 # XPath 1.0 has no lower-case function... 

921 f"translate(@{self.lang_attribute}, 'ABCDEFGHIJKLMNOPQRSTUVWXYZ', " 

922 "'abcdefghijklmnopqrstuvwxyz'), " 

923 f"'-'), {arg})]" 

924 ) 

925 

926 def xpath_link_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

927 return xpath.add_condition( 

928 "@href and (name(.) = 'a' or name(.) = 'link' or name(.) = 'area')" 

929 ) 

930 

931 # Links are never visited, the implementation for :visited is the same 

932 # as in GenericTranslator 

933 

934 def xpath_disabled_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

935 # http://www.w3.org/TR/html5/section-index.html#attributes-1 

936 return xpath.add_condition( 

937 """ 

938 ( 

939 @disabled and 

940 ( 

941 (name(.) = 'input' and @type != 'hidden') or 

942 name(.) = 'button' or 

943 name(.) = 'select' or 

944 name(.) = 'textarea' or 

945 name(.) = 'command' or 

946 name(.) = 'fieldset' or 

947 name(.) = 'optgroup' or 

948 name(.) = 'option' 

949 ) 

950 ) or ( 

951 ( 

952 (name(.) = 'input' and @type != 'hidden') or 

953 name(.) = 'button' or 

954 name(.) = 'select' or 

955 name(.) = 'textarea' 

956 ) 

957 and ancestor::fieldset[@disabled] 

958 ) 

959 """ 

960 ) 

961 # FIXME: in the second half, add "and is not a descendant of that 

962 # fieldset element's first legend element child, if any." 

963 

964 def xpath_enabled_pseudo(self, xpath: XPathExpr) -> XPathExpr: 

965 # http://www.w3.org/TR/html5/section-index.html#attributes-1 

966 return xpath.add_condition( 

967 """ 

968 ( 

969 @href and ( 

970 name(.) = 'a' or 

971 name(.) = 'link' or 

972 name(.) = 'area' 

973 ) 

974 ) or ( 

975 ( 

976 name(.) = 'command' or 

977 name(.) = 'fieldset' or 

978 name(.) = 'optgroup' 

979 ) 

980 and not(@disabled) 

981 ) or ( 

982 ( 

983 (name(.) = 'input' and @type != 'hidden') or 

984 name(.) = 'button' or 

985 name(.) = 'select' or 

986 name(.) = 'textarea' or 

987 name(.) = 'keygen' 

988 ) 

989 and not (@disabled or ancestor::fieldset[@disabled]) 

990 ) or ( 

991 name(.) = 'option' and not( 

992 @disabled or ancestor::optgroup[@disabled] 

993 ) 

994 ) 

995 """ 

996 ) 

997 # FIXME: ... or "li elements that are children of menu elements, 

998 # and that have a child element that defines a command, if the first 

999 # such element's Disabled State facet is false (not disabled)". 

1000 # FIXME: after ancestor::fieldset[@disabled], add "and is not a 

1001 # descendant of that fieldset element's first legend element child, 

1002 # if any."