Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/soupsieve/css_match.py: 17%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

970 statements  

1"""CSS matcher.""" 

2from __future__ import annotations 

3from datetime import datetime 

4from collections.abc import Hashable 

5from . import util 

6import re 

7from . import css_types as ct 

8import unicodedata 

9import bs4 

10from typing import Iterator, Iterable, Any, Callable, Sequence, Any, overload, Literal, cast # noqa: F401, F811 

11 

12# Empty tag pattern (whitespace okay) 

13RE_NOT_EMPTY = re.compile('[^ \t\r\n\f]') 

14 

15RE_NOT_WS = re.compile('[^ \t\r\n\f]+') 

16 

17# Relationships 

18REL_PARENT = ' ' 

19REL_CLOSE_PARENT = '>' 

20REL_SIBLING = '~' 

21REL_CLOSE_SIBLING = '+' 

22 

23# Relationships for :has() (forward looking) 

24REL_HAS_PARENT = ': ' 

25REL_HAS_CLOSE_PARENT = ':>' 

26REL_HAS_SIBLING = ':~' 

27REL_HAS_CLOSE_SIBLING = ':+' 

28 

29NS_XHTML = 'http://www.w3.org/1999/xhtml' 

30NS_XML = 'http://www.w3.org/XML/1998/namespace' 

31 

32DIR_FLAGS = ct.SEL_DIR_LTR | ct.SEL_DIR_RTL 

33RANGES = ct.SEL_IN_RANGE | ct.SEL_OUT_OF_RANGE 

34 

35DIR_MAP = { 

36 'ltr': ct.SEL_DIR_LTR, 

37 'rtl': ct.SEL_DIR_RTL, 

38 'auto': 0 

39} 

40 

41RE_NUM = re.compile(r"^(?P<value>-?(?:[0-9]{1,}(\.[0-9]+)?|\.[0-9]+))$") 

42RE_TIME = re.compile(r'^(?P<hour>[0-9]{2}):(?P<minutes>[0-9]{2})$') 

43RE_MONTH = re.compile(r'^(?P<year>[0-9]{4,})-(?P<month>[0-9]{2})$') 

44RE_WEEK = re.compile(r'^(?P<year>[0-9]{4,})-W(?P<week>[0-9]{2})$') 

45RE_DATE = re.compile(r'^(?P<year>[0-9]{4,})-(?P<month>[0-9]{2})-(?P<day>[0-9]{2})$') 

46RE_DATETIME = re.compile( 

47 r'^(?P<year>[0-9]{4,})-(?P<month>[0-9]{2})-(?P<day>[0-9]{2})T(?P<hour>[0-9]{2}):(?P<minutes>[0-9]{2})$' 

48) 

49RE_WILD_STRIP = re.compile(r'(?:(?:-\*-)(?:\*(?:-|$))*|-\*$)') 

50 

51MONTHS_30 = (4, 6, 9, 11) # April, June, September, and November 

52FEB = 2 

53SHORT_MONTH = 30 

54LONG_MONTH = 31 

55FEB_MONTH = 28 

56FEB_LEAP_MONTH = 29 

57DAYS_IN_WEEK = 7 

58 

59 

60def within(target: bs4.Tag, parent: bs4.Tag | bs4.BeautifulSoup, start: int, end: int | None = None) -> bool: 

61 """Check if target is within data.""" 

62 

63 contents = parent.contents 

64 return any(contents[i] is target for i in range(start, end if end is not None else len(contents))) 

65 

66 

67class _DocumentNav: 

68 """Navigate a Beautiful Soup document.""" 

69 

70 @classmethod 

71 def assert_valid_input(cls, tag: Any) -> None: 

72 """Check if valid input tag or document.""" 

73 

74 # Fail on unexpected types. 

75 if not cls.is_tag(tag): 

76 raise TypeError(f"Expected a BeautifulSoup 'Tag', but instead received type {type(tag)}") 

77 

78 @staticmethod 

79 def is_doc(obj: bs4.element.PageElement | None) -> bool: 

80 """Is `BeautifulSoup` object.""" 

81 return isinstance(obj, bs4.BeautifulSoup) 

82 

83 @staticmethod 

84 def is_tag(obj: bs4.element.PageElement | None) -> bool: 

85 """Is tag.""" 

86 return isinstance(obj, bs4.Tag) 

87 

88 @staticmethod 

89 def is_declaration(obj: bs4.element.PageElement | None) -> bool: # pragma: no cover 

90 """Is declaration.""" 

91 return isinstance(obj, bs4.Declaration) 

92 

93 @staticmethod 

94 def is_cdata(obj: bs4.element.PageElement | None) -> bool: 

95 """Is CDATA.""" 

96 return isinstance(obj, bs4.CData) 

97 

98 @staticmethod 

99 def is_processing_instruction(obj: bs4.element.PageElement | None) -> bool: # pragma: no cover 

100 """Is processing instruction.""" 

101 return isinstance(obj, bs4.ProcessingInstruction) 

102 

103 @staticmethod 

104 def is_navigable_string(obj: bs4.element.PageElement | None) -> bool: 

105 """Is navigable string.""" 

106 return isinstance(obj, bs4.element.NavigableString) 

107 

108 @staticmethod 

109 def is_special_string(obj: bs4.element.PageElement | None) -> bool: 

110 """Is special string.""" 

111 return isinstance(obj, (bs4.Comment, bs4.Declaration, bs4.CData, bs4.ProcessingInstruction, bs4.Doctype)) 

112 

113 @classmethod 

114 def is_content_string(cls, obj: bs4.element.PageElement | None) -> bool: 

115 """Check if node is content string.""" 

116 

117 return cls.is_navigable_string(obj) and not cls.is_special_string(obj) 

118 

119 @staticmethod 

120 def is_xml_tree(el: bs4.Tag | None) -> bool: 

121 """Check if element (or document) is from a XML tree.""" 

122 

123 return el is not None and bool(el._is_xml) 

124 

125 def is_iframe(self, el: bs4.Tag | None) -> bool: 

126 """Check if element is an `iframe`.""" 

127 

128 if el is None: # pragma: no cover 

129 return False 

130 

131 return bool( 

132 ((el.name if self.is_xml_tree(el) else util.lower(el.name)) == 'iframe') and 

133 self.is_html_tag(el) # type: ignore[attr-defined] 

134 ) 

135 

136 def is_root(self, el: bs4.Tag) -> bool: 

137 """ 

138 Return whether element is a root element. 

139 

140 We check that the element is the root of the tree (which we have already pre-calculated), 

141 and we check if it is the root element under an `iframe`. 

142 """ 

143 

144 root = self.root and self.root is el # type: ignore[attr-defined] 

145 if not root: 

146 parent = self.get_parent(el) 

147 root = parent is not None and self.is_html and self.is_iframe(parent) # type: ignore[attr-defined] 

148 return root 

149 

150 def get_contents(self, el: bs4.Tag | None, no_iframe: bool = False) -> Iterator[bs4.element.PageElement]: 

151 """Get contents or contents in reverse.""" 

152 

153 if el is not None: 

154 if not no_iframe or not self.is_iframe(el): 

155 yield from el.contents 

156 

157 def get_tag_children( 

158 self, 

159 el: bs4.Tag | None, 

160 start: int | None = None, 

161 reverse: bool = False, 

162 no_iframe: bool = False 

163 ) -> Iterator[bs4.Tag]: 

164 """Get tag children.""" 

165 

166 return self.get_children(el, start, reverse, True, no_iframe) 

167 

168 @overload 

169 def get_children( 

170 self, 

171 el: bs4.Tag | None, 

172 start: int | None = None, 

173 reverse: bool = False, 

174 tags: Literal[True] = ..., 

175 no_iframe: bool = False 

176 ) -> Iterator[bs4.Tag]: 

177 ... 

178 

179 @overload 

180 def get_children( 

181 self, 

182 el: bs4.Tag | None, 

183 start: int | None = None, 

184 reverse: bool = False, 

185 tags: Literal[False] = ..., 

186 no_iframe: bool = False 

187 ) -> Iterator[bs4.element.PageElement]: 

188 ... 

189 

190 def get_children( 

191 self, 

192 el: bs4.Tag | None, 

193 start: int | None = None, 

194 reverse: bool = False, 

195 tags: Literal[True] | Literal[False] = False, 

196 no_iframe: bool = False 

197 ) -> Iterator[bs4.element.PageElement]: 

198 """Get children.""" 

199 

200 if el is not None and (not no_iframe or not self.is_iframe(el)): 

201 last = len(el.contents) - 1 

202 if start is None: 

203 index = last if reverse else 0 

204 else: 

205 index = start 

206 end = -1 if reverse else last + 1 

207 incr = -1 if reverse else 1 

208 

209 if 0 <= index <= last: 

210 for i in range(index, end, incr): 

211 node = el.contents[i] 

212 if not tags or self.is_tag(node): 

213 yield node 

214 

215 def get_tag_descendants( 

216 self, 

217 el: bs4.Tag | None, 

218 no_iframe: bool = False 

219 ) -> Iterator[bs4.Tag]: 

220 """Specifically get tag descendants.""" 

221 

222 yield from self.get_descendants(el, tags=True, no_iframe=no_iframe) # type: ignore[misc] 

223 

224 def get_descendants( 

225 self, 

226 el: bs4.Tag | None, 

227 tags: bool = False, 

228 no_iframe: bool = False 

229 ) -> Iterator[bs4.element.PageElement]: 

230 """Get descendants.""" 

231 

232 if el is not None and (not no_iframe or not self.is_iframe(el)): 

233 next_good = None 

234 for child in el.descendants: 

235 

236 if next_good is not None: 

237 if child is not next_good: 

238 continue 

239 next_good = None 

240 

241 if isinstance(child, bs4.Tag): 

242 if no_iframe and self.is_iframe(child): 

243 if child.next_sibling is not None: 

244 next_good = child.next_sibling 

245 else: 

246 last_child = child # type: bs4.element.PageElement 

247 while isinstance(last_child, bs4.Tag) and last_child.contents: 

248 last_child = last_child.contents[-1] 

249 next_good = last_child.next_element 

250 yield child 

251 if next_good is None: 

252 break 

253 # Coverage isn't seeing this even though it's executed 

254 continue # pragma: no cover 

255 yield child 

256 

257 elif not tags: 

258 yield child 

259 

260 def get_parent(self, el: bs4.Tag | None, no_iframe: bool = False) -> bs4.Tag | None: 

261 """Get parent.""" 

262 

263 parent = el.parent if el is not None else None 

264 if no_iframe and parent is not None and self.is_iframe(parent): # pragma: no cover 

265 parent = None 

266 return parent 

267 

268 @staticmethod 

269 def get_tag_name(el: bs4.Tag | None) -> str | None: 

270 """Get tag.""" 

271 

272 return el.name if el is not None else None 

273 

274 @staticmethod 

275 def get_prefix_name(el: bs4.Tag) -> str | None: 

276 """Get prefix.""" 

277 

278 return el.prefix 

279 

280 @staticmethod 

281 def get_uri(el: bs4.Tag | None) -> str | None: 

282 """Get namespace `URI`.""" 

283 

284 return el.namespace if el is not None else None 

285 

286 @classmethod 

287 def get_next_tag(cls, el: bs4.Tag) -> bs4.Tag | None: 

288 """Get next sibling tag.""" 

289 

290 return cls.get_next(el, tags=True) # type: ignore[return-value] 

291 

292 @classmethod 

293 def get_next(cls, el: bs4.Tag, tags: bool = False) -> bs4.element.PageElement | None: 

294 """Get next sibling tag.""" 

295 

296 sibling = el.next_sibling 

297 while tags and not isinstance(sibling, bs4.Tag) and sibling is not None: 

298 sibling = sibling.next_sibling 

299 

300 if tags and not isinstance(sibling, bs4.Tag): 

301 sibling = None 

302 

303 return sibling 

304 

305 @classmethod 

306 def get_previous_tag(cls, el: bs4.Tag, tags: bool = True) -> bs4.Tag | None: 

307 """Get previous sibling tag.""" 

308 

309 return cls.get_previous(el, True) # type: ignore[return-value] 

310 

311 @classmethod 

312 def get_previous(cls, el: bs4.Tag, tags: bool = False) -> bs4.element.PageElement | None: 

313 """Get previous sibling tag.""" 

314 

315 sibling = el.previous_sibling 

316 while tags and not isinstance(sibling, bs4.Tag) and sibling is not None: 

317 sibling = sibling.previous_sibling 

318 

319 if tags and not isinstance(sibling, bs4.Tag): 

320 sibling = None 

321 

322 return sibling 

323 

324 @staticmethod 

325 def has_html_ns(el: bs4.Tag | None) -> bool: 

326 """ 

327 Check if element has an HTML namespace. 

328 

329 This is a bit different than whether a element is treated as having an HTML namespace, 

330 like we do in the case of `is_html_tag`. 

331 """ 

332 

333 ns = getattr(el, 'namespace') if el is not None else None # noqa: B009 

334 return bool(ns and ns == NS_XHTML) 

335 

336 @staticmethod 

337 def split_namespace(el: bs4.Tag | None, attr_name: str) -> tuple[str | None, str | None]: 

338 """Return namespace and attribute name without the prefix.""" 

339 

340 if el is None: # pragma: no cover 

341 return None, None 

342 

343 return getattr(attr_name, 'namespace', None), getattr(attr_name, 'name', None) 

344 

345 @classmethod 

346 def get_attribute_by_name( 

347 cls, 

348 el: bs4.Tag, 

349 name: str, 

350 default: str | Sequence[str] | None = None 

351 ) -> str | Sequence[str] | None: 

352 """Get attribute by name.""" 

353 

354 value = default 

355 if el._is_xml: 

356 if name in el.attrs: 

357 v = el.attrs[name] 

358 value = '' if v is None else v 

359 else: 

360 for k, v in el.attrs.items(): 

361 if util.lower(k) == name: 

362 value = '' if v is None else v 

363 break 

364 return value 

365 

366 @classmethod 

367 def iter_attributes(cls, el: bs4.Tag | None) -> Iterator[tuple[str, str | Sequence[str] | None]]: 

368 """Iterate attributes.""" 

369 

370 if el is not None: 

371 for k, v in el.attrs.items(): 

372 yield k, '' if v is None else v 

373 

374 @classmethod 

375 def get_classes(cls, el: bs4.Tag) -> Sequence[str]: 

376 """Get classes.""" 

377 

378 classes = cls.get_attribute_by_name(el, 'class', []) 

379 if isinstance(classes, str): 

380 classes = RE_NOT_WS.findall(classes) 

381 return cast(Sequence[str], classes) 

382 

383 def get_text(self, el: bs4.Tag, no_iframe: bool = False) -> str: 

384 """Get text.""" 

385 

386 return ''.join( 

387 [ 

388 node for node in self.get_descendants(el, no_iframe=no_iframe) # type: ignore[misc] 

389 if self.is_content_string(node) 

390 ] 

391 ) 

392 

393 def get_own_text(self, el: bs4.Tag, no_iframe: bool = False) -> list[str]: 

394 """Get Own Text.""" 

395 

396 return [ 

397 node for node in self.get_contents(el, no_iframe=no_iframe) if self.is_content_string(node) # type: ignore[misc] 

398 ] 

399 

400 

401class Inputs: 

402 """Class for parsing and validating input items.""" 

403 

404 @staticmethod 

405 def validate_day(year: int, month: int, day: int) -> bool: 

406 """Validate day.""" 

407 

408 max_days = LONG_MONTH 

409 if month == FEB: 

410 max_days = FEB_LEAP_MONTH if ((year % 4 == 0) and (year % 100 != 0)) or (year % 400 == 0) else FEB_MONTH 

411 elif month in MONTHS_30: 

412 max_days = SHORT_MONTH 

413 return 1 <= day <= max_days 

414 

415 @staticmethod 

416 def validate_week(year: int, week: int) -> bool: 

417 """Validate week.""" 

418 

419 # Validate an ISO week number for `year`. 

420 # 

421 # Per ISO 8601 rules, the last ISO week of a year is the week 

422 # containing Dec 28. Using Dec 28 guarantees we obtain the 

423 # correct ISO week-number for the final week of `year`, even in 

424 # years where Dec 31 falls in ISO week 01 of the following year. 

425 # 

426 # Example: if Dec 31 is a Thursday the year's last ISO week will 

427 # be week 53; if Dec 31 is a Monday and that week is counted as 

428 # week 1 of the next year, Dec 28 still belongs to the final 

429 # week of the current ISO year and yields the correct max week. 

430 max_week = datetime(year, 12, 28).isocalendar()[1] 

431 return 1 <= week <= max_week 

432 

433 @staticmethod 

434 def validate_month(month: int) -> bool: 

435 """Validate month.""" 

436 

437 return 1 <= month <= 12 

438 

439 @staticmethod 

440 def validate_year(year: int) -> bool: 

441 """Validate year.""" 

442 

443 return 1 <= year 

444 

445 @staticmethod 

446 def validate_hour(hour: int) -> bool: 

447 """Validate hour.""" 

448 

449 return 0 <= hour <= 23 

450 

451 @staticmethod 

452 def validate_minutes(minutes: int) -> bool: 

453 """Validate minutes.""" 

454 

455 return 0 <= minutes <= 59 

456 

457 @classmethod 

458 def parse_value(cls, itype: str, value: str | None) -> tuple[float, ...] | None: 

459 """Parse the input value.""" 

460 

461 parsed = None # type: tuple[float, ...] | None 

462 if value is None: 

463 return value 

464 if itype == "date": 

465 m = RE_DATE.match(value) 

466 if m: 

467 year = int(m.group('year'), 10) 

468 month = int(m.group('month'), 10) 

469 day = int(m.group('day'), 10) 

470 if cls.validate_year(year) and cls.validate_month(month) and cls.validate_day(year, month, day): 

471 parsed = (year, month, day) 

472 elif itype == "month": 

473 m = RE_MONTH.match(value) 

474 if m: 

475 year = int(m.group('year'), 10) 

476 month = int(m.group('month'), 10) 

477 if cls.validate_year(year) and cls.validate_month(month): 

478 parsed = (year, month) 

479 elif itype == "week": 

480 m = RE_WEEK.match(value) 

481 if m: 

482 year = int(m.group('year'), 10) 

483 week = int(m.group('week'), 10) 

484 if cls.validate_year(year) and cls.validate_week(year, week): 

485 parsed = (year, week) 

486 elif itype == "time": 

487 m = RE_TIME.match(value) 

488 if m: 

489 hour = int(m.group('hour'), 10) 

490 minutes = int(m.group('minutes'), 10) 

491 if cls.validate_hour(hour) and cls.validate_minutes(minutes): 

492 parsed = (hour, minutes) 

493 elif itype == "datetime-local": 

494 m = RE_DATETIME.match(value) 

495 if m: 

496 year = int(m.group('year'), 10) 

497 month = int(m.group('month'), 10) 

498 day = int(m.group('day'), 10) 

499 hour = int(m.group('hour'), 10) 

500 minutes = int(m.group('minutes'), 10) 

501 if ( 

502 cls.validate_year(year) and cls.validate_month(month) and cls.validate_day(year, month, day) and 

503 cls.validate_hour(hour) and cls.validate_minutes(minutes) 

504 ): 

505 parsed = (year, month, day, hour, minutes) 

506 elif itype in ("number", "range"): 

507 m = RE_NUM.match(value) 

508 if m: 

509 parsed = (float(m.group('value')),) 

510 return parsed 

511 

512 

513class CSSMatch(_DocumentNav): 

514 """Perform CSS matching.""" 

515 

516 def __init__( 

517 self, 

518 selectors: ct.SelectorList, 

519 scope: bs4.Tag | None, 

520 namespaces: ct.Namespaces | None, 

521 flags: int 

522 ) -> None: 

523 """Initialize.""" 

524 

525 self.assert_valid_input(scope) 

526 self.tag = scope 

527 self.cached_meta_lang = [] # type: list[tuple[str, str]] 

528 self.cached_default_forms = [] # type: list[tuple[bs4.Tag, bs4.Tag]] 

529 self.cached_indeterminate_forms = [] # type: list[tuple[bs4.Tag, str, bool]] 

530 self.selectors = selectors 

531 self.namespaces = {} if namespaces is None else namespaces # type: ct.Namespaces | dict[str, str] 

532 self.flags = flags 

533 self.enable_cache = not bool(self.flags & util.NOCACHE) 

534 self.iframe_restrict = False 

535 self.nth_cache: dict[Hashable, dict[Hashable, list[int]]] = {} 

536 self.sib_cache: dict[Hashable, dict[Hashable, int]] = {} 

537 

538 # Find the root element for the whole tree 

539 doc = scope 

540 parent = self.get_parent(doc) 

541 while parent: 

542 doc = parent 

543 parent = self.get_parent(doc) 

544 root = None # type: bs4.Tag | None 

545 if not self.is_doc(doc): 

546 root = doc 

547 else: 

548 for child in self.get_tag_children(doc): 

549 root = child 

550 break 

551 

552 self.root = root 

553 self.scope = scope if scope is not doc else root 

554 self.has_html_namespace = self.has_html_ns(root) 

555 

556 # A document can be both XML and HTML (XHTML) 

557 self.is_xml = self.is_xml_tree(doc) 

558 self.is_html = not self.is_xml or self.has_html_namespace 

559 

560 def reset(self) -> None: # pragma: no cover 

561 """Reset.""" 

562 

563 self.nth_cache.clear() 

564 self.sib_cache.clear() 

565 

566 def supports_namespaces(self) -> bool: 

567 """Check if namespaces are supported in the HTML type.""" 

568 

569 return self.is_xml or self.has_html_namespace 

570 

571 def get_tag_ns(self, el: bs4.Tag | None) -> str: 

572 """Get tag namespace.""" 

573 

574 namespace = '' 

575 if el is None: # pragma: no cover 

576 return namespace 

577 

578 if self.supports_namespaces(): 

579 ns = self.get_uri(el) 

580 if ns: 

581 namespace = ns 

582 else: 

583 namespace = NS_XHTML 

584 return namespace 

585 

586 def is_html_tag(self, el: bs4.Tag | None) -> bool: 

587 """Check if tag is in HTML namespace.""" 

588 

589 return self.get_tag_ns(el) == NS_XHTML 

590 

591 def get_tag(self, el: bs4.Tag | None) -> str | None: 

592 """Get tag.""" 

593 

594 name = self.get_tag_name(el) 

595 return util.lower(name) if name is not None and not self.is_xml else name 

596 

597 def get_prefix(self, el: bs4.Tag) -> str | None: 

598 """Get prefix.""" 

599 

600 prefix = self.get_prefix_name(el) 

601 return util.lower(prefix) if prefix is not None and not self.is_xml else prefix 

602 

603 def find_bidi(self, el: bs4.Tag) -> int | None: 

604 """Get directionality from element text.""" 

605 

606 for node in self.get_children(el): 

607 

608 # Analyze child text nodes 

609 if self.is_tag(node): 

610 

611 # Avoid analyzing certain elements specified in the specification. 

612 direction = DIR_MAP.get(util.lower(self.get_attribute_by_name(node, 'dir', '')), None) 

613 name = self.get_tag(node) 

614 if ( 

615 (name and name in ('bdi', 'script', 'style', 'textarea', 'iframe')) or 

616 not self.is_html_tag(node) or 

617 direction is not None 

618 ): 

619 continue # pragma: no cover 

620 

621 # Check directionality of this node's text 

622 value = self.find_bidi(node) 

623 if value is not None: 

624 return value 

625 

626 # Direction could not be determined 

627 continue # pragma: no cover 

628 

629 # Skip `doctype` comments, etc. 

630 if self.is_special_string(node): 

631 continue 

632 

633 # Analyze text nodes for directionality. 

634 for c in cast('bs4.element.NavigableString', node): 

635 bidi = unicodedata.bidirectional(c) 

636 if bidi in ('AL', 'R', 'L'): 

637 return ct.SEL_DIR_LTR if bidi == 'L' else ct.SEL_DIR_RTL 

638 return None 

639 

640 def extended_language_filter(self, lang_range: str, lang_tag: str) -> bool: 

641 """Filter the language tags.""" 

642 

643 match = True 

644 lang_range = RE_WILD_STRIP.sub('-', lang_range).lower() 

645 ranges = lang_range.split('-') 

646 subtags = lang_tag.lower().split('-') 

647 length = len(ranges) 

648 slength = len(subtags) 

649 rindex = 0 

650 sindex = 0 

651 r = ranges[rindex] 

652 s = subtags[sindex] 

653 

654 # Empty specified language should match unspecified language attributes 

655 if length == 1 and slength == 1 and not r and r == s: 

656 return True 

657 

658 # Primary tag needs to match 

659 if (r != '*' and r != s) or (r == '*' and slength == 1 and not s): 

660 match = False 

661 

662 rindex += 1 

663 sindex += 1 

664 

665 # Match until we run out of ranges 

666 while match and rindex < length: 

667 r = ranges[rindex] 

668 try: 

669 s = subtags[sindex] 

670 except IndexError: 

671 # Ran out of subtags, 

672 # but we still have ranges 

673 match = False 

674 continue 

675 

676 # Empty range 

677 if not r: 

678 match = False 

679 continue 

680 

681 # Matched range 

682 elif s == r: 

683 rindex += 1 

684 

685 # Implicit wildcard cannot match 

686 # singletons 

687 elif len(s) == 1: 

688 match = False 

689 continue 

690 

691 # Implicitly matched, so grab next subtag 

692 sindex += 1 

693 

694 return match 

695 

696 def match_attribute_name( 

697 self, 

698 el: bs4.Tag, 

699 attr: str, 

700 prefix: str | None 

701 ) -> str | Sequence[str] | None: 

702 """Match attribute name and return value if it exists.""" 

703 

704 value = None 

705 if self.supports_namespaces(): 

706 value = None 

707 # If we have not defined namespaces, we can't very well find them, so don't bother trying. 

708 if prefix: 

709 ns = self.namespaces.get(prefix) 

710 if ns is None and prefix != '*': 

711 return None 

712 else: 

713 ns = None 

714 

715 for k, v in self.iter_attributes(el): 

716 

717 # Get attribute parts 

718 namespace, name = self.split_namespace(el, k) 

719 

720 # Can't match a prefix attribute as we haven't specified one to match 

721 # Try to match it normally as a whole `p:a` as selector may be trying `p\:a`. 

722 if ns is None: 

723 if (self.is_xml and attr == k) or (not self.is_xml and util.lower(attr) == util.lower(k)): 

724 value = v 

725 break 

726 # Coverage is not finding this even though it is executed. 

727 # Adding a print statement before this (and erasing coverage) causes coverage to find the line. 

728 # Ignore the false positive message. 

729 continue # pragma: no cover 

730 

731 # We can't match our desired prefix attribute as the attribute doesn't have a prefix 

732 if namespace is None or (ns != namespace and prefix != '*'): 

733 continue 

734 

735 # The attribute doesn't match. 

736 if (util.lower(attr) != util.lower(name)) if not self.is_xml else (attr != name): 

737 continue 

738 

739 value = v 

740 break 

741 else: 

742 for k, v in self.iter_attributes(el): 

743 if util.lower(attr) != util.lower(k): 

744 continue 

745 value = v 

746 break 

747 return value 

748 

749 def match_namespace(self, el: bs4.Tag, tag: ct.SelectorTag) -> bool: 

750 """Match the namespace of the element.""" 

751 

752 match = True 

753 namespace = self.get_tag_ns(el) 

754 default_namespace = self.namespaces.get('') 

755 tag_ns = '' if tag.prefix is None else self.namespaces.get(tag.prefix) 

756 # We must match the default namespace if one is not provided 

757 if tag.prefix is None and (default_namespace is not None and namespace != default_namespace): 

758 match = False 

759 # If we specified `|tag`, we must not have a namespace. 

760 elif (tag.prefix is not None and tag.prefix == '' and namespace): 

761 match = False 

762 # Verify prefix matches 

763 elif ( 

764 tag.prefix and 

765 tag.prefix != '*' and (tag_ns is None or namespace != tag_ns) 

766 ): 

767 match = False 

768 return match 

769 

770 def match_attributes(self, el: bs4.Tag, attributes: tuple[ct.SelectorAttribute, ...]) -> bool: 

771 """Match attributes.""" 

772 

773 match = True 

774 if attributes: 

775 for a in attributes: 

776 temp = self.match_attribute_name(el, a.attribute, a.prefix) 

777 pattern = a.xml_type_pattern if self.is_xml and a.xml_type_pattern else a.pattern 

778 if temp is None: 

779 match = False 

780 break 

781 value = temp if isinstance(temp, str) else ' '.join(temp) 

782 if pattern is None: 

783 continue 

784 elif pattern.match(value) is None: 

785 match = False 

786 break 

787 return match 

788 

789 def match_tagname(self, el: bs4.Tag, tag: ct.SelectorTag) -> bool: 

790 """Match tag name.""" 

791 

792 name = (util.lower(tag.name) if not self.is_xml and tag.name is not None else tag.name) 

793 return not ( 

794 name is not None and 

795 name not in (self.get_tag(el), '*') 

796 ) 

797 

798 def match_tag(self, el: bs4.Tag, tag: ct.SelectorTag | None) -> bool: 

799 """Match the tag.""" 

800 

801 match = True 

802 if tag is not None: 

803 # Verify namespace 

804 if not self.match_tagname(el, tag): 

805 match = False 

806 if match and not self.match_namespace(el, tag): 

807 match = False 

808 return match 

809 

810 def match_general_sibling(self, el: bs4.Tag, relation: ct.SelectorList) -> bool: 

811 """Match general sibling combinator.""" 

812 

813 found = False 

814 

815 if relation[0] is ct.Null: # pragma: no cover 

816 return found 

817 

818 pkey: tuple[str | None, int] | None = None 

819 key: tuple[ct.SelectorList, int] | None = None 

820 

821 # Setup the cache by the parent if present 

822 parent = self.get_parent(el) 

823 if parent is None: # pragma: no cover 

824 return found 

825 

826 if parent: 

827 pkey = (parent.name, id(parent)) 

828 

829 # Initialize the cache if necessary 

830 if pkey not in self.sib_cache: 

831 self.sib_cache[pkey] = {} 

832 

833 # Check the cache to see if we already know where the first sibling is, 

834 # and if we do, check if we are on the correct side of it. 

835 # If we've previously searched and found no sibling, there is no sibling. 

836 # Lastly, if this is our first time, setup the cache. 

837 reverse = relation[0].rel_type == REL_HAS_SIBLING 

838 start = len(parent) - 1 if reverse else 0 

839 if pkey: 

840 key = (relation, id(relation)) 

841 if key in self.sib_cache[pkey]: 

842 index = self.sib_cache[pkey][key] 

843 if index >= 0: 

844 a, b = (index, start) if reverse else (start, index) 

845 return not within(el, parent, a, b) 

846 else: 

847 return False 

848 self.sib_cache[pkey][key] = start 

849 

850 # Start at the furthest endpoint and walk back towards the element looking for siblings. 

851 # The current element counts as a sibling, but will not cause a match. 

852 passed = False 

853 incr = -1 if reverse else 1 

854 for child in self.get_children(parent, start=start, reverse=reverse): 

855 start += incr 

856 if not isinstance(child, bs4.Tag): 

857 continue 

858 

859 # Flag that we are passing the element. 

860 # Any siblings we find aren't valid for this element. 

861 if child is el: 

862 passed = True 

863 

864 # We found the furthest sibling. 

865 if self.match_selectors(child, relation): 

866 found = True 

867 break 

868 

869 # Cache the index of the sibling or mark as there being no siblings. 

870 if pkey and key: 

871 self.sib_cache[pkey][key] = start if found else -1 

872 

873 # If we passed the current element and then found a sibling, it doesn't count as a match. 

874 if passed: 

875 found = False 

876 

877 return found 

878 

879 def match_past_relations(self, el: bs4.Tag, relation: ct.SelectorList) -> bool: 

880 """Match past relationship.""" 

881 

882 found = False 

883 # I don't think this can ever happen, but it makes `mypy` happy 

884 if relation[0] is ct.Null: # pragma: no cover 

885 return found 

886 

887 if relation[0].rel_type == REL_PARENT: 

888 parent: bs4.Tag | None = el 

889 while not found and parent and (parent := self.get_parent(parent, no_iframe=self.iframe_restrict)): 

890 found = parent is not None and self.match_selectors(parent, relation) 

891 elif relation[0].rel_type == REL_CLOSE_PARENT: 

892 parent = self.get_parent(el, no_iframe=self.iframe_restrict) 

893 found = parent is not None and self.match_selectors(parent, relation) 

894 elif relation[0].rel_type == REL_SIBLING: 

895 if self.enable_cache: 

896 found = self.match_general_sibling(el, relation) 

897 else: 

898 sibling: bs4.Tag | None = el 

899 while not found and sibling and (sibling := self.get_previous_tag(sibling)): 

900 found = sibling is not None and self.match_selectors(sibling, relation) 

901 elif relation[0].rel_type == REL_CLOSE_SIBLING: 

902 sibling = self.get_previous_tag(el) 

903 found = sibling is not None and self.match_selectors(sibling, relation) 

904 return found 

905 

906 def match_future_child(self, parent: bs4.Tag, relation: ct.SelectorList, recursive: bool = False) -> bool: 

907 """Match future child.""" 

908 

909 match = False 

910 if recursive: 

911 children = self.get_tag_descendants # type: Callable[..., Iterator[bs4.Tag]] 

912 else: 

913 children = self.get_tag_children 

914 for child in children(parent, no_iframe=self.iframe_restrict): 

915 if self.match_selectors(child, relation): 

916 match = True 

917 break 

918 return match 

919 

920 def match_future_relations(self, el: bs4.Tag, relation: ct.SelectorList) -> bool: 

921 """Match future relationship.""" 

922 

923 found = False 

924 # I don't think this can ever happen, but it makes `mypy` happy 

925 if relation[0] is ct.Null: # pragma: no cover 

926 return found 

927 

928 if relation[0].rel_type == REL_HAS_PARENT: 

929 found = self.match_future_child(el, relation, True) 

930 elif relation[0].rel_type == REL_HAS_CLOSE_PARENT: 

931 found = self.match_future_child(el, relation) 

932 elif relation[0].rel_type == REL_HAS_SIBLING: 

933 if self.enable_cache: 

934 found = self.match_general_sibling(el, relation) 

935 else: 

936 sibling: bs4.Tag | None = el 

937 while not found and sibling and (sibling := self.get_next_tag(sibling)): 

938 found = self.match_selectors(sibling, relation) 

939 elif relation[0].rel_type == REL_HAS_CLOSE_SIBLING: 

940 sibling = self.get_next_tag(el) 

941 found = sibling is not None and self.match_selectors(sibling, relation) 

942 return found 

943 

944 def match_relations(self, el: bs4.Tag, relation: ct.SelectorList) -> bool: 

945 """Match relationship to other elements.""" 

946 

947 found = False 

948 

949 if relation[0] is ct.Null or relation[0].rel_type is None: 

950 return found 

951 

952 if relation[0].rel_type.startswith(':'): 

953 found = self.match_future_relations(el, relation) 

954 else: 

955 found = self.match_past_relations(el, relation) 

956 

957 return found 

958 

959 def match_id(self, el: bs4.Tag, ids: tuple[str, ...]) -> bool: 

960 """Match element's ID.""" 

961 

962 found = True 

963 for i in ids: 

964 if i != self.get_attribute_by_name(el, 'id', ''): 

965 found = False 

966 break 

967 return found 

968 

969 def match_classes(self, el: bs4.Tag, classes: tuple[str, ...]) -> bool: 

970 """Match element's classes.""" 

971 

972 current_classes = self.get_classes(el) 

973 found = True 

974 for c in classes: 

975 if c not in current_classes: 

976 found = False 

977 break 

978 return found 

979 

980 def match_root(self, el: bs4.Tag) -> bool: 

981 """Match element as root.""" 

982 

983 is_root = self.is_root(el) 

984 if is_root: 

985 sibling = self.get_previous(el) # type: Any 

986 while is_root and sibling is not None: 

987 if ( 

988 self.is_tag(sibling) or (self.is_content_string(sibling) and sibling.strip()) or 

989 self.is_cdata(sibling) 

990 ): 

991 is_root = False 

992 else: 

993 sibling = self.get_previous(sibling) 

994 if is_root: 

995 sibling = self.get_next(el) 

996 while is_root and sibling is not None: 

997 if ( 

998 self.is_tag(sibling) or (self.is_content_string(sibling) and sibling.strip()) or 

999 self.is_cdata(sibling) 

1000 ): 

1001 is_root = False 

1002 else: 

1003 sibling = self.get_next(sibling) 

1004 return is_root 

1005 

1006 def match_scope(self, el: bs4.Tag) -> bool: 

1007 """Match element as scope.""" 

1008 

1009 return self.scope is el 

1010 

1011 def match_nth_tag_type(self, el: bs4.Tag, child: bs4.Tag) -> bool: 

1012 """Match tag type for `nth` matches.""" 

1013 

1014 return ( 

1015 (self.get_tag(child) == self.get_tag(el)) and 

1016 (self.get_tag_ns(child) == self.get_tag_ns(el)) 

1017 ) 

1018 

1019 def match_nth(self, el: bs4.Tag, nth: tuple[ct.SelectorNth, ...]) -> bool: 

1020 """Match `nth` elements.""" 

1021 

1022 # `nth` selectors are evaluated against siblings under the same parent. 

1023 parent = self.get_parent(el) # type: bs4.Tag | None 

1024 pkey: tuple[str | None, int] | None = None 

1025 key: tuple[ct.SelectorNth, int, str | None, str | None] | None = None 

1026 start = rindex = 0 

1027 incr = rincr = 0 

1028 

1029 # Setup the cache by the parent, if parent a parent is present 

1030 if self.enable_cache and parent: 

1031 pkey = (parent.name, id(parent)) 

1032 

1033 # Initialize the cache if necessary 

1034 if pkey not in self.nth_cache: 

1035 self.nth_cache[pkey] = {} 

1036 

1037 # Test element against the `nth` selectors. 

1038 matched = True 

1039 for n in nth: 

1040 matched = False 

1041 last = n.last 

1042 key = None 

1043 

1044 # Prepare the child iterator and get the starting, real index and the relative index 

1045 if pkey and parent: 

1046 # Get last info from the cache 

1047 key = (n, id(n), self.get_tag(el), self.get_tag_ns(el)) if n.of_type else (n, id(n), None, None) 

1048 valid = False 

1049 if key in self.nth_cache[pkey]: 

1050 start, rindex = self.nth_cache[pkey][key] 

1051 if within(el, parent, start): 

1052 last = False 

1053 rincr = -1 if n.last else 1 

1054 valid = True 

1055 

1056 # Start/overwrite the cache if the cache was empty or invalid 

1057 if not valid: 

1058 start, rindex = len(parent) - 1 if last else 0, 0 

1059 self.nth_cache[pkey][key] = [start, rindex] 

1060 rincr = 1 

1061 

1062 incr = 1 if not last else -1 

1063 children = self.get_children(parent, start=start, reverse=last) 

1064 

1065 # Non-cached handling of parented element 

1066 elif parent: 

1067 rindex = 0 

1068 start = len(parent) - 1 if last else 0 

1069 rincr = incr = 1 

1070 children = self.get_children(parent, start=start, reverse=last) 

1071 

1072 # No parent, just evaluate the element against the selectors 

1073 else: 

1074 start = rindex = 0 

1075 rincr = incr = 1 

1076 children = iter([el]) 

1077 

1078 # Find index of element compared to its siblings and check the index conditions 

1079 child: bs4.Tag 

1080 for child in children: 

1081 start += incr 

1082 

1083 # We only care about tags 

1084 if not self.is_tag(child): 

1085 continue 

1086 

1087 # Handle `of S` in `nth-child` and handle `of-type` 

1088 if ( 

1089 (n.selectors and not self.match_selectors(child, n.selectors)) or 

1090 (n.of_type and not self.match_nth_tag_type(el, child)) 

1091 ): 

1092 if child is el: 

1093 break 

1094 continue 

1095 

1096 # Test the relative index against the `nth` requirement. 

1097 rindex += rincr 

1098 if child is el: 

1099 if n.a != 0: 

1100 v = (rindex - n.b) / n.a 

1101 matched = v.is_integer() and v >= 0 

1102 else: 

1103 matched = rindex == n.b and n.b >= 1 

1104 break 

1105 

1106 # "Last index" selectors evaluate first from the bottom and then evaluate 

1107 # from the first found element top-down. Start will be incremented in the 

1108 # wrong direction, so increment it and step over the current index. 

1109 if last: 

1110 start += 2 

1111 

1112 # Update the cache 

1113 if pkey and key: 

1114 self.nth_cache[pkey][key] = [start, rindex] 

1115 

1116 # If we failed to match any `nth` selectors, quit. 

1117 if not matched: 

1118 break 

1119 

1120 return matched 

1121 

1122 def match_empty(self, el: bs4.Tag) -> bool: 

1123 """Check if element is empty (if requested).""" 

1124 

1125 is_empty = True 

1126 for child in self.get_children(el): 

1127 if self.is_tag(child): 

1128 is_empty = False 

1129 break 

1130 elif self.is_content_string(child) and RE_NOT_EMPTY.search(child): # type: ignore[call-overload] 

1131 is_empty = False 

1132 break 

1133 return is_empty 

1134 

1135 def match_subselectors(self, el: bs4.Tag, selectors: tuple[ct.SelectorList, ...]) -> bool: 

1136 """Match selectors.""" 

1137 

1138 match = True 

1139 for sel in selectors: 

1140 if not self.match_selectors(el, sel): 

1141 match = False 

1142 return match 

1143 

1144 def match_contains(self, el: bs4.Tag, contains: tuple[ct.SelectorContains, ...]) -> bool: 

1145 """Match element if it contains text.""" 

1146 

1147 match = True 

1148 content = None # type: str | Sequence[str] | None 

1149 for contain_list in contains: 

1150 if content is None: 

1151 if contain_list.own: 

1152 content = self.get_own_text(el, no_iframe=self.is_html) 

1153 else: 

1154 content = self.get_text(el, no_iframe=self.is_html) 

1155 found = False 

1156 for text in contain_list.text: 

1157 if contain_list.own: 

1158 for c in content: 

1159 if text in c: 

1160 found = True 

1161 break 

1162 if found: 

1163 break 

1164 else: 

1165 if text in content: 

1166 found = True 

1167 break 

1168 if not found: 

1169 match = False 

1170 return match 

1171 

1172 def match_default(self, el: bs4.Tag) -> bool: 

1173 """Match default.""" 

1174 

1175 match = False 

1176 

1177 # Find this input's form 

1178 form = None # type: bs4.Tag | None 

1179 parent = self.get_parent(el, no_iframe=True) 

1180 while parent and form is None: 

1181 if self.get_tag(parent) == 'form' and self.is_html_tag(parent): 

1182 form = parent 

1183 else: 

1184 parent = self.get_parent(parent, no_iframe=True) 

1185 

1186 if form is not None: 

1187 # Look in form cache to see if we've already located its default button 

1188 found_form = False 

1189 for f, t in self.cached_default_forms: 

1190 if f is form: 

1191 found_form = True 

1192 if t is el: 

1193 match = True 

1194 break 

1195 

1196 # We didn't have the form cached, so look for its default button 

1197 if not found_form: 

1198 for child in self.get_tag_descendants(form, no_iframe=True): 

1199 name = self.get_tag(child) 

1200 # Can't do nested forms (haven't figured out why we never hit this) 

1201 if name == 'form': # pragma: no cover 

1202 break 

1203 if name in ('input', 'button'): 

1204 v = self.get_attribute_by_name(child, 'type', '') 

1205 if v and util.lower(v) == 'submit': 

1206 self.cached_default_forms.append((form, child)) 

1207 if el is child: 

1208 match = True 

1209 break 

1210 return match 

1211 

1212 def match_indeterminate(self, el: bs4.Tag) -> bool: 

1213 """Match default.""" 

1214 

1215 match = False 

1216 name = cast(str, self.get_attribute_by_name(el, 'name')) 

1217 

1218 def get_parent_form(el: bs4.Tag) -> bs4.Tag | None: 

1219 """Find this input's form.""" 

1220 form = None 

1221 parent = self.get_parent(el, no_iframe=True) 

1222 while form is None: 

1223 if self.get_tag(parent) == 'form' and self.is_html_tag(parent): 

1224 form = parent 

1225 break 

1226 last_parent = parent 

1227 parent = self.get_parent(parent, no_iframe=True) 

1228 if parent is None: 

1229 form = last_parent 

1230 break 

1231 return form 

1232 

1233 form = get_parent_form(el) 

1234 

1235 # Look in form cache to see if we've already evaluated that its fellow radio buttons are indeterminate 

1236 if form is not None: 

1237 found_form = False 

1238 for f, n, i in self.cached_indeterminate_forms: 

1239 if f is form and n == name: 

1240 found_form = True 

1241 if i is True: 

1242 match = True 

1243 break 

1244 

1245 # We didn't have the form cached, so validate that the radio button is indeterminate 

1246 if not found_form: 

1247 checked = False 

1248 for child in self.get_tag_descendants(form, no_iframe=True): 

1249 if child is el: 

1250 continue 

1251 tag_name = self.get_tag(child) 

1252 if tag_name == 'input': 

1253 is_radio = False 

1254 check = False 

1255 has_name = False 

1256 for k, v in self.iter_attributes(child): 

1257 if util.lower(k) == 'type' and util.lower(v) == 'radio': 

1258 is_radio = True 

1259 elif util.lower(k) == 'name' and v == name: 

1260 has_name = True 

1261 elif util.lower(k) == 'checked': 

1262 check = True 

1263 if is_radio and check and has_name and get_parent_form(child) is form: 

1264 checked = True 

1265 break 

1266 if checked: 

1267 break 

1268 if not checked: 

1269 match = True 

1270 self.cached_indeterminate_forms.append((form, name, match)) 

1271 

1272 return match 

1273 

1274 def match_lang(self, el: bs4.Tag, langs: tuple[ct.SelectorLang, ...]) -> bool: 

1275 """Match languages.""" 

1276 

1277 match = False 

1278 has_ns = self.supports_namespaces() 

1279 root = self.root 

1280 has_html_namespace = self.has_html_namespace 

1281 

1282 # Walk parents looking for `lang` (HTML) or `xml:lang` XML property. 

1283 parent = el # type: bs4.Tag | None 

1284 found_lang = None 

1285 last = None 

1286 while not found_lang: 

1287 has_html_ns = self.has_html_ns(parent) 

1288 for k, v in self.iter_attributes(parent): 

1289 attr_ns, attr = self.split_namespace(parent, k) 

1290 if ( 

1291 ((not has_ns or has_html_ns) and (util.lower(k) if not self.is_xml else k) == 'lang') or 

1292 ( 

1293 has_ns and not has_html_ns and attr_ns == NS_XML and 

1294 (util.lower(attr) if not self.is_xml and attr is not None else attr) == 'lang' 

1295 ) 

1296 ): 

1297 found_lang = v 

1298 break 

1299 last = parent 

1300 parent = self.get_parent(parent, no_iframe=self.is_html) 

1301 

1302 if parent is None: 

1303 root = last 

1304 has_html_namespace = self.has_html_ns(root) 

1305 parent = last 

1306 break 

1307 

1308 # Use cached meta language. 

1309 if found_lang is None and self.cached_meta_lang: 

1310 for cache in self.cached_meta_lang: 

1311 if root is not None and cast(str, root) is cache[0]: 

1312 found_lang = cache[1] 

1313 

1314 # If we couldn't find a language, and the document is HTML, look to meta to determine language. 

1315 if found_lang is None and (not self.is_xml or (has_html_namespace and root and root.name == 'html')): 

1316 # Find head 

1317 found = False 

1318 for tag in ('html', 'head'): 

1319 found = False 

1320 for child in self.get_tag_children(parent, no_iframe=self.is_html): 

1321 if self.get_tag(child) == tag and self.is_html_tag(child): 

1322 found = True 

1323 parent = child 

1324 break 

1325 if not found: # pragma: no cover 

1326 break 

1327 

1328 # Search meta tags 

1329 if found and parent is not None: 

1330 for child2 in parent: 

1331 if isinstance(child2, bs4.Tag) and self.get_tag(child2) == 'meta' and self.is_html_tag(parent): 

1332 c_lang = False 

1333 content = None 

1334 for k, v in self.iter_attributes(child2): 

1335 if util.lower(k) == 'http-equiv' and util.lower(v) == 'content-language': 

1336 c_lang = True 

1337 if util.lower(k) == 'content': 

1338 content = v 

1339 if c_lang and content: 

1340 found_lang = content 

1341 self.cached_meta_lang.append((cast(str, root), cast(str, found_lang))) 

1342 break 

1343 if found_lang is not None: 

1344 break 

1345 if found_lang is None: 

1346 self.cached_meta_lang.append((cast(str, root), '')) 

1347 

1348 # If we determined a language, compare. 

1349 if found_lang is not None: 

1350 for patterns in langs: 

1351 match = False 

1352 for pattern in patterns: 

1353 if self.extended_language_filter(pattern, cast(str, found_lang)): 

1354 match = True 

1355 if not match: 

1356 break 

1357 

1358 return match 

1359 

1360 def match_dir(self, el: bs4.Tag | None, directionality: int) -> bool: 

1361 """Check directionality.""" 

1362 

1363 # If we have to match both left and right, we can't match either. 

1364 if directionality & ct.SEL_DIR_LTR and directionality & ct.SEL_DIR_RTL: 

1365 return False 

1366 

1367 if el is None or not self.is_html_tag(el): 

1368 return False 

1369 

1370 # Element has defined direction of left to right or right to left 

1371 direction = DIR_MAP.get(util.lower(self.get_attribute_by_name(el, 'dir', '')), None) 

1372 if direction not in (None, 0): 

1373 return direction == directionality 

1374 

1375 # Element is the document element (the root) and no direction assigned, assume left to right. 

1376 is_root = self.is_root(el) 

1377 if is_root and direction is None: 

1378 return ct.SEL_DIR_LTR == directionality 

1379 

1380 # If `input[type=telephone]` and no direction is assigned, assume left to right. 

1381 name = self.get_tag(el) 

1382 is_input = name == 'input' 

1383 is_textarea = name == 'textarea' 

1384 is_bdi = name == 'bdi' 

1385 itype = util.lower(self.get_attribute_by_name(el, 'type', '')) if is_input else '' 

1386 if is_input and itype == 'tel' and direction is None: 

1387 return ct.SEL_DIR_LTR == directionality 

1388 

1389 # Auto handling for text inputs 

1390 if ((is_input and itype in ('text', 'search', 'tel', 'url', 'email')) or is_textarea) and direction == 0: 

1391 if is_textarea: 

1392 value = ''.join(node for node in self.get_contents(el, no_iframe=True) if self.is_content_string(node)) # type: ignore[misc] 

1393 else: 

1394 value = cast(str, self.get_attribute_by_name(el, 'value', '')) 

1395 if value: 

1396 for c in value: 

1397 bidi = unicodedata.bidirectional(c) 

1398 if bidi in ('AL', 'R', 'L'): 

1399 direction = ct.SEL_DIR_LTR if bidi == 'L' else ct.SEL_DIR_RTL 

1400 return direction == directionality 

1401 # Assume left to right 

1402 return ct.SEL_DIR_LTR == directionality 

1403 elif is_root: 

1404 return ct.SEL_DIR_LTR == directionality 

1405 return self.match_dir(self.get_parent(el, no_iframe=True), directionality) 

1406 

1407 # Auto handling for `bdi` and other non text inputs. 

1408 if (is_bdi and direction is None) or direction == 0: 

1409 direction = self.find_bidi(el) 

1410 if direction is not None: 

1411 return direction == directionality 

1412 elif is_root: 

1413 return ct.SEL_DIR_LTR == directionality 

1414 return self.match_dir(self.get_parent(el, no_iframe=True), directionality) 

1415 

1416 # Match parents direction 

1417 return self.match_dir(self.get_parent(el, no_iframe=True), directionality) 

1418 

1419 def match_range(self, el: bs4.Tag, condition: int) -> bool: 

1420 """ 

1421 Match range. 

1422 

1423 Behavior is modeled after what we see in browsers. Browsers seem to evaluate 

1424 if the value is out of range, and if not, it is in range. So a missing value 

1425 will not evaluate out of range; therefore, value is in range. Personally, I 

1426 feel like this should evaluate as neither in or out of range. 

1427 """ 

1428 

1429 out_of_range = False 

1430 

1431 itype = util.lower(self.get_attribute_by_name(el, 'type')) 

1432 mn = Inputs.parse_value(itype, cast(str, self.get_attribute_by_name(el, 'min', None))) 

1433 mx = Inputs.parse_value(itype, cast(str, self.get_attribute_by_name(el, 'max', None))) 

1434 

1435 # There is no valid min or max, so we cannot evaluate a range 

1436 if mn is None and mx is None: 

1437 return False 

1438 

1439 value = Inputs.parse_value(itype, cast(str, self.get_attribute_by_name(el, 'value', None))) 

1440 if value is not None: 

1441 if itype in ("date", "datetime-local", "month", "week", "number", "range"): 

1442 if mn is not None and value < mn: 

1443 out_of_range = True 

1444 if not out_of_range and mx is not None and value > mx: 

1445 out_of_range = True 

1446 elif itype == "time": 

1447 if mn is not None and mx is not None and mn > mx: 

1448 # Time is periodic, so this is a reversed/discontinuous range 

1449 if value < mn and value > mx: 

1450 out_of_range = True 

1451 else: 

1452 if mn is not None and value < mn: 

1453 out_of_range = True 

1454 if not out_of_range and mx is not None and value > mx: 

1455 out_of_range = True 

1456 

1457 return not out_of_range if condition & ct.SEL_IN_RANGE else out_of_range 

1458 

1459 def match_defined(self, el: bs4.Tag) -> bool: 

1460 """ 

1461 Match defined. 

1462 

1463 `:defined` is related to custom elements in a browser. 

1464 

1465 - If the document is XML (not XHTML), all tags will match. 

1466 - Tags that are not custom (don't have a hyphen) are marked defined. 

1467 - If the tag has a prefix (without or without a namespace), it will not match. 

1468 

1469 This is of course requires the parser to provide us with the proper prefix and namespace info, 

1470 if it doesn't, there is nothing we can do. 

1471 """ 

1472 

1473 name = self.get_tag(el) 

1474 return ( 

1475 name is not None and ( 

1476 name.find('-') == -1 or 

1477 name.find(':') != -1 or 

1478 self.get_prefix(el) is not None 

1479 ) 

1480 ) 

1481 

1482 def match_placeholder_shown(self, el: bs4.Tag) -> bool: 

1483 """ 

1484 Match placeholder shown according to HTML spec. 

1485 

1486 - text area should be checked if they have content. A single newline does not count as content. 

1487 

1488 """ 

1489 

1490 match = False 

1491 content = self.get_text(el) 

1492 if content in ('', '\n'): 

1493 match = True 

1494 

1495 return match 

1496 

1497 def match_selectors(self, el: bs4.Tag, selectors: ct.SelectorList) -> bool: 

1498 """Check if element matches one of the selectors.""" 

1499 

1500 match = False 

1501 is_not = selectors.is_not 

1502 is_html = selectors.is_html 

1503 

1504 # Internal selector lists that use the HTML flag, will automatically get the `html` namespace. 

1505 if is_html: 

1506 namespaces = self.namespaces 

1507 iframe_restrict = self.iframe_restrict 

1508 self.namespaces = {'html': NS_XHTML} 

1509 self.iframe_restrict = True 

1510 

1511 if not is_html or self.is_html: 

1512 for selector in selectors: 

1513 match = is_not 

1514 # We have a un-matchable situation (like `:focus` as you can focus an element in this environment) 

1515 if selector is ct.Null: 

1516 continue 

1517 # Verify tag matches 

1518 if not self.match_tag(el, selector.tag): 

1519 continue 

1520 # Verify tag is defined 

1521 if selector.flags & ct.SEL_DEFINED and not self.match_defined(el): 

1522 continue 

1523 # Verify element is root 

1524 if selector.flags & ct.SEL_ROOT and not self.match_root(el): 

1525 continue 

1526 # Verify element is scope 

1527 if selector.flags & ct.SEL_SCOPE and not self.match_scope(el): 

1528 continue 

1529 # Verify element has placeholder shown 

1530 if selector.flags & ct.SEL_PLACEHOLDER_SHOWN and not self.match_placeholder_shown(el): 

1531 continue 

1532 # Verify `nth` matches 

1533 if selector.nth and not self.match_nth(el, selector.nth): 

1534 continue 

1535 if selector.flags & ct.SEL_EMPTY and not self.match_empty(el): 

1536 continue 

1537 # Verify id matches 

1538 if selector.ids and not self.match_id(el, selector.ids): 

1539 continue 

1540 # Verify classes match 

1541 if selector.classes and not self.match_classes(el, selector.classes): 

1542 continue 

1543 # Verify attribute(s) match 

1544 if not self.match_attributes(el, selector.attributes): 

1545 continue 

1546 # Verify ranges 

1547 if selector.flags & RANGES and not self.match_range(el, selector.flags & RANGES): 

1548 continue 

1549 # Verify language patterns 

1550 if selector.lang and not self.match_lang(el, selector.lang): 

1551 continue 

1552 # Verify pseudo selector patterns 

1553 if selector.selectors and not self.match_subselectors(el, selector.selectors): 

1554 continue 

1555 # Verify relationship selectors 

1556 if selector.relation and not self.match_relations(el, selector.relation): 

1557 continue 

1558 # Validate that the current default selector match corresponds to the first submit button in the form 

1559 if selector.flags & ct.SEL_DEFAULT and not self.match_default(el): 

1560 continue 

1561 # Validate that the unset radio button is among radio buttons with the same name in a form that are 

1562 # also not set. 

1563 if selector.flags & ct.SEL_INDETERMINATE and not self.match_indeterminate(el): 

1564 continue 

1565 # Validate element directionality 

1566 if selector.flags & DIR_FLAGS and not self.match_dir(el, selector.flags & DIR_FLAGS): 

1567 continue 

1568 # Validate that the tag contains the specified text. 

1569 if selector.contains and not self.match_contains(el, selector.contains): 

1570 continue 

1571 match = not is_not 

1572 break 

1573 

1574 # Restore actual namespaces being used for external selector lists 

1575 if is_html: 

1576 self.namespaces = namespaces 

1577 self.iframe_restrict = iframe_restrict 

1578 

1579 return match 

1580 

1581 def select(self, limit: int = 0) -> Iterator[bs4.Tag]: 

1582 """Match all tags under the targeted tag.""" 

1583 

1584 lim = None if limit < 1 else limit 

1585 

1586 for child in self.get_tag_descendants(self.tag): 

1587 if self.match(child): 

1588 yield child 

1589 if lim is not None: 

1590 lim -= 1 

1591 if lim < 1: 

1592 break 

1593 

1594 def closest(self) -> bs4.Tag | None: 

1595 """Match closest ancestor.""" 

1596 

1597 current = self.tag # type: bs4.Tag | None 

1598 closest = None 

1599 while closest is None and current is not None: 

1600 if self.match(current): 

1601 closest = current 

1602 else: 

1603 current = self.get_parent(current) 

1604 return closest 

1605 

1606 def filter(self) -> list[bs4.Tag]: # noqa A001 

1607 """Filter tag's children.""" 

1608 

1609 return [ 

1610 tag for tag in self.get_contents(self.tag) 

1611 if isinstance(tag, bs4.Tag) and self.match(tag) 

1612 ] 

1613 

1614 def match(self, el: bs4.Tag) -> bool: 

1615 """Match.""" 

1616 

1617 return not self.is_doc(el) and self.is_tag(el) and self.match_selectors(el, self.selectors) 

1618 

1619 

1620class SoupSieve(ct.Immutable): 

1621 """Compiled Soup Sieve selector matching object.""" 

1622 

1623 pattern: str 

1624 selectors: ct.SelectorList 

1625 namespaces: ct.Namespaces | None 

1626 custom: dict[str, str] 

1627 flags: int 

1628 

1629 __slots__ = ("pattern", "selectors", "namespaces", "custom", "flags", "_hash") 

1630 

1631 def __init__( 

1632 self, 

1633 pattern: str, 

1634 selectors: ct.SelectorList, 

1635 namespaces: ct.Namespaces | None, 

1636 custom: ct.CustomSelectors | None, 

1637 flags: int 

1638 ): 

1639 """Initialize.""" 

1640 

1641 super().__init__( 

1642 pattern=pattern, 

1643 selectors=selectors, 

1644 namespaces=namespaces, 

1645 custom=custom, 

1646 flags=flags 

1647 ) 

1648 

1649 def match(self, tag: bs4.Tag) -> bool: 

1650 """Match.""" 

1651 

1652 return CSSMatch(self.selectors, tag, self.namespaces, self.flags).match(tag) 

1653 

1654 def closest(self, tag: bs4.Tag) -> bs4.Tag | None: 

1655 """Match closest ancestor.""" 

1656 

1657 return CSSMatch(self.selectors, tag, self.namespaces, self.flags).closest() 

1658 

1659 def filter(self, iterable: Iterable[bs4.Tag]) -> list[bs4.Tag]: # noqa A001 

1660 """ 

1661 Filter. 

1662 

1663 `CSSMatch` can cache certain searches for tags of the same document, 

1664 so if we are given a tag, all tags are from the same document, 

1665 and we can take advantage of the optimization. 

1666 

1667 Any other kind of iterable could have tags from different documents or detached tags, 

1668 so for those, we use a new `CSSMatch` for each item in the iterable. 

1669 """ 

1670 

1671 if isinstance(iterable, bs4.Tag): 

1672 return CSSMatch(self.selectors, iterable, self.namespaces, self.flags).filter() 

1673 else: 

1674 # There is no guarantee that elements are from the same document, evaluate them separately. 

1675 return [node for node in iterable if not CSSMatch.is_navigable_string(node) and self.match(node)] 

1676 

1677 def select_one(self, tag: bs4.Tag) -> bs4.Tag | None: 

1678 """Select a single tag.""" 

1679 

1680 tags = self.select(tag, limit=1) 

1681 return tags[0] if tags else None 

1682 

1683 def select(self, tag: bs4.Tag, limit: int = 0) -> list[bs4.Tag]: 

1684 """Select the specified tags.""" 

1685 

1686 return list(self.iselect(tag, limit)) 

1687 

1688 def iselect(self, tag: bs4.Tag, limit: int = 0) -> Iterator[bs4.Tag]: 

1689 """Iterate the specified tags.""" 

1690 

1691 yield from CSSMatch(self.selectors, tag, self.namespaces, self.flags).select(limit) 

1692 

1693 def __repr__(self) -> str: # pragma: no cover 

1694 """Representation.""" 

1695 

1696 return ( 

1697 f"SoupSieve(pattern={self.pattern!r}, namespaces={self.namespaces!r}, " 

1698 f"custom={self.custom!r}, flags={self.flags!r})" 

1699 ) 

1700 

1701 __str__ = __repr__ 

1702 

1703 

1704ct.pickle_register(SoupSieve)