Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/soupsieve/css_match.py: 17%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""CSS matcher."""
2from __future__ import annotations
3from datetime import datetime
4from collections.abc import Hashable
5from . import util
6import re
7from . import css_types as ct
8import unicodedata
9import bs4
10from typing import Iterator, Iterable, Any, Callable, Sequence, Any, overload, Literal, cast # noqa: F401, F811
12# Empty tag pattern (whitespace okay)
13RE_NOT_EMPTY = re.compile('[^ \t\r\n\f]')
15RE_NOT_WS = re.compile('[^ \t\r\n\f]+')
17# Relationships
18REL_PARENT = ' '
19REL_CLOSE_PARENT = '>'
20REL_SIBLING = '~'
21REL_CLOSE_SIBLING = '+'
23# Relationships for :has() (forward looking)
24REL_HAS_PARENT = ': '
25REL_HAS_CLOSE_PARENT = ':>'
26REL_HAS_SIBLING = ':~'
27REL_HAS_CLOSE_SIBLING = ':+'
29NS_XHTML = 'http://www.w3.org/1999/xhtml'
30NS_XML = 'http://www.w3.org/XML/1998/namespace'
32DIR_FLAGS = ct.SEL_DIR_LTR | ct.SEL_DIR_RTL
33RANGES = ct.SEL_IN_RANGE | ct.SEL_OUT_OF_RANGE
35DIR_MAP = {
36 'ltr': ct.SEL_DIR_LTR,
37 'rtl': ct.SEL_DIR_RTL,
38 'auto': 0
39}
41RE_NUM = re.compile(r"^(?P<value>-?(?:[0-9]{1,}(\.[0-9]+)?|\.[0-9]+))$")
42RE_TIME = re.compile(r'^(?P<hour>[0-9]{2}):(?P<minutes>[0-9]{2})$')
43RE_MONTH = re.compile(r'^(?P<year>[0-9]{4,})-(?P<month>[0-9]{2})$')
44RE_WEEK = re.compile(r'^(?P<year>[0-9]{4,})-W(?P<week>[0-9]{2})$')
45RE_DATE = re.compile(r'^(?P<year>[0-9]{4,})-(?P<month>[0-9]{2})-(?P<day>[0-9]{2})$')
46RE_DATETIME = re.compile(
47 r'^(?P<year>[0-9]{4,})-(?P<month>[0-9]{2})-(?P<day>[0-9]{2})T(?P<hour>[0-9]{2}):(?P<minutes>[0-9]{2})$'
48)
49RE_WILD_STRIP = re.compile(r'(?:(?:-\*-)(?:\*(?:-|$))*|-\*$)')
51MONTHS_30 = (4, 6, 9, 11) # April, June, September, and November
52FEB = 2
53SHORT_MONTH = 30
54LONG_MONTH = 31
55FEB_MONTH = 28
56FEB_LEAP_MONTH = 29
57DAYS_IN_WEEK = 7
60def within(target: bs4.Tag, parent: bs4.Tag | bs4.BeautifulSoup, start: int, end: int | None = None) -> bool:
61 """Check if target is within data."""
63 contents = parent.contents
64 return any(contents[i] is target for i in range(start, end if end is not None else len(contents)))
67class _DocumentNav:
68 """Navigate a Beautiful Soup document."""
70 @classmethod
71 def assert_valid_input(cls, tag: Any) -> None:
72 """Check if valid input tag or document."""
74 # Fail on unexpected types.
75 if not cls.is_tag(tag):
76 raise TypeError(f"Expected a BeautifulSoup 'Tag', but instead received type {type(tag)}")
78 @staticmethod
79 def is_doc(obj: bs4.element.PageElement | None) -> bool:
80 """Is `BeautifulSoup` object."""
81 return isinstance(obj, bs4.BeautifulSoup)
83 @staticmethod
84 def is_tag(obj: bs4.element.PageElement | None) -> bool:
85 """Is tag."""
86 return isinstance(obj, bs4.Tag)
88 @staticmethod
89 def is_declaration(obj: bs4.element.PageElement | None) -> bool: # pragma: no cover
90 """Is declaration."""
91 return isinstance(obj, bs4.Declaration)
93 @staticmethod
94 def is_cdata(obj: bs4.element.PageElement | None) -> bool:
95 """Is CDATA."""
96 return isinstance(obj, bs4.CData)
98 @staticmethod
99 def is_processing_instruction(obj: bs4.element.PageElement | None) -> bool: # pragma: no cover
100 """Is processing instruction."""
101 return isinstance(obj, bs4.ProcessingInstruction)
103 @staticmethod
104 def is_navigable_string(obj: bs4.element.PageElement | None) -> bool:
105 """Is navigable string."""
106 return isinstance(obj, bs4.element.NavigableString)
108 @staticmethod
109 def is_special_string(obj: bs4.element.PageElement | None) -> bool:
110 """Is special string."""
111 return isinstance(obj, (bs4.Comment, bs4.Declaration, bs4.CData, bs4.ProcessingInstruction, bs4.Doctype))
113 @classmethod
114 def is_content_string(cls, obj: bs4.element.PageElement | None) -> bool:
115 """Check if node is content string."""
117 return cls.is_navigable_string(obj) and not cls.is_special_string(obj)
119 @staticmethod
120 def is_xml_tree(el: bs4.Tag | None) -> bool:
121 """Check if element (or document) is from a XML tree."""
123 return el is not None and bool(el._is_xml)
125 def is_iframe(self, el: bs4.Tag | None) -> bool:
126 """Check if element is an `iframe`."""
128 if el is None: # pragma: no cover
129 return False
131 return bool(
132 ((el.name if self.is_xml_tree(el) else util.lower(el.name)) == 'iframe') and
133 self.is_html_tag(el) # type: ignore[attr-defined]
134 )
136 def is_root(self, el: bs4.Tag) -> bool:
137 """
138 Return whether element is a root element.
140 We check that the element is the root of the tree (which we have already pre-calculated),
141 and we check if it is the root element under an `iframe`.
142 """
144 root = self.root and self.root is el # type: ignore[attr-defined]
145 if not root:
146 parent = self.get_parent(el)
147 root = parent is not None and self.is_html and self.is_iframe(parent) # type: ignore[attr-defined]
148 return root
150 def get_contents(self, el: bs4.Tag | None, no_iframe: bool = False) -> Iterator[bs4.element.PageElement]:
151 """Get contents or contents in reverse."""
153 if el is not None:
154 if not no_iframe or not self.is_iframe(el):
155 yield from el.contents
157 def get_tag_children(
158 self,
159 el: bs4.Tag | None,
160 start: int | None = None,
161 reverse: bool = False,
162 no_iframe: bool = False
163 ) -> Iterator[bs4.Tag]:
164 """Get tag children."""
166 return self.get_children(el, start, reverse, True, no_iframe)
168 @overload
169 def get_children(
170 self,
171 el: bs4.Tag | None,
172 start: int | None = None,
173 reverse: bool = False,
174 tags: Literal[True] = ...,
175 no_iframe: bool = False
176 ) -> Iterator[bs4.Tag]:
177 ...
179 @overload
180 def get_children(
181 self,
182 el: bs4.Tag | None,
183 start: int | None = None,
184 reverse: bool = False,
185 tags: Literal[False] = ...,
186 no_iframe: bool = False
187 ) -> Iterator[bs4.element.PageElement]:
188 ...
190 def get_children(
191 self,
192 el: bs4.Tag | None,
193 start: int | None = None,
194 reverse: bool = False,
195 tags: Literal[True] | Literal[False] = False,
196 no_iframe: bool = False
197 ) -> Iterator[bs4.element.PageElement]:
198 """Get children."""
200 if el is not None and (not no_iframe or not self.is_iframe(el)):
201 last = len(el.contents) - 1
202 if start is None:
203 index = last if reverse else 0
204 else:
205 index = start
206 end = -1 if reverse else last + 1
207 incr = -1 if reverse else 1
209 if 0 <= index <= last:
210 for i in range(index, end, incr):
211 node = el.contents[i]
212 if not tags or self.is_tag(node):
213 yield node
215 def get_tag_descendants(
216 self,
217 el: bs4.Tag | None,
218 no_iframe: bool = False
219 ) -> Iterator[bs4.Tag]:
220 """Specifically get tag descendants."""
222 yield from self.get_descendants(el, tags=True, no_iframe=no_iframe) # type: ignore[misc]
224 def get_descendants(
225 self,
226 el: bs4.Tag | None,
227 tags: bool = False,
228 no_iframe: bool = False
229 ) -> Iterator[bs4.element.PageElement]:
230 """Get descendants."""
232 if el is not None and (not no_iframe or not self.is_iframe(el)):
233 next_good = None
234 for child in el.descendants:
236 if next_good is not None:
237 if child is not next_good:
238 continue
239 next_good = None
241 if isinstance(child, bs4.Tag):
242 if no_iframe and self.is_iframe(child):
243 if child.next_sibling is not None:
244 next_good = child.next_sibling
245 else:
246 last_child = child # type: bs4.element.PageElement
247 while isinstance(last_child, bs4.Tag) and last_child.contents:
248 last_child = last_child.contents[-1]
249 next_good = last_child.next_element
250 yield child
251 if next_good is None:
252 break
253 # Coverage isn't seeing this even though it's executed
254 continue # pragma: no cover
255 yield child
257 elif not tags:
258 yield child
260 def get_parent(self, el: bs4.Tag | None, no_iframe: bool = False) -> bs4.Tag | None:
261 """Get parent."""
263 parent = el.parent if el is not None else None
264 if no_iframe and parent is not None and self.is_iframe(parent): # pragma: no cover
265 parent = None
266 return parent
268 @staticmethod
269 def get_tag_name(el: bs4.Tag | None) -> str | None:
270 """Get tag."""
272 return el.name if el is not None else None
274 @staticmethod
275 def get_prefix_name(el: bs4.Tag) -> str | None:
276 """Get prefix."""
278 return el.prefix
280 @staticmethod
281 def get_uri(el: bs4.Tag | None) -> str | None:
282 """Get namespace `URI`."""
284 return el.namespace if el is not None else None
286 @classmethod
287 def get_next_tag(cls, el: bs4.Tag) -> bs4.Tag | None:
288 """Get next sibling tag."""
290 return cls.get_next(el, tags=True) # type: ignore[return-value]
292 @classmethod
293 def get_next(cls, el: bs4.Tag, tags: bool = False) -> bs4.element.PageElement | None:
294 """Get next sibling tag."""
296 sibling = el.next_sibling
297 while tags and not isinstance(sibling, bs4.Tag) and sibling is not None:
298 sibling = sibling.next_sibling
300 if tags and not isinstance(sibling, bs4.Tag):
301 sibling = None
303 return sibling
305 @classmethod
306 def get_previous_tag(cls, el: bs4.Tag, tags: bool = True) -> bs4.Tag | None:
307 """Get previous sibling tag."""
309 return cls.get_previous(el, True) # type: ignore[return-value]
311 @classmethod
312 def get_previous(cls, el: bs4.Tag, tags: bool = False) -> bs4.element.PageElement | None:
313 """Get previous sibling tag."""
315 sibling = el.previous_sibling
316 while tags and not isinstance(sibling, bs4.Tag) and sibling is not None:
317 sibling = sibling.previous_sibling
319 if tags and not isinstance(sibling, bs4.Tag):
320 sibling = None
322 return sibling
324 @staticmethod
325 def has_html_ns(el: bs4.Tag | None) -> bool:
326 """
327 Check if element has an HTML namespace.
329 This is a bit different than whether a element is treated as having an HTML namespace,
330 like we do in the case of `is_html_tag`.
331 """
333 ns = getattr(el, 'namespace') if el is not None else None # noqa: B009
334 return bool(ns and ns == NS_XHTML)
336 @staticmethod
337 def split_namespace(el: bs4.Tag | None, attr_name: str) -> tuple[str | None, str | None]:
338 """Return namespace and attribute name without the prefix."""
340 if el is None: # pragma: no cover
341 return None, None
343 return getattr(attr_name, 'namespace', None), getattr(attr_name, 'name', None)
345 @classmethod
346 def get_attribute_by_name(
347 cls,
348 el: bs4.Tag,
349 name: str,
350 default: str | Sequence[str] | None = None
351 ) -> str | Sequence[str] | None:
352 """Get attribute by name."""
354 value = default
355 if el._is_xml:
356 if name in el.attrs:
357 v = el.attrs[name]
358 value = '' if v is None else v
359 else:
360 for k, v in el.attrs.items():
361 if util.lower(k) == name:
362 value = '' if v is None else v
363 break
364 return value
366 @classmethod
367 def iter_attributes(cls, el: bs4.Tag | None) -> Iterator[tuple[str, str | Sequence[str] | None]]:
368 """Iterate attributes."""
370 if el is not None:
371 for k, v in el.attrs.items():
372 yield k, '' if v is None else v
374 @classmethod
375 def get_classes(cls, el: bs4.Tag) -> Sequence[str]:
376 """Get classes."""
378 classes = cls.get_attribute_by_name(el, 'class', [])
379 if isinstance(classes, str):
380 classes = RE_NOT_WS.findall(classes)
381 return cast(Sequence[str], classes)
383 def get_text(self, el: bs4.Tag, no_iframe: bool = False) -> str:
384 """Get text."""
386 return ''.join(
387 [
388 node for node in self.get_descendants(el, no_iframe=no_iframe) # type: ignore[misc]
389 if self.is_content_string(node)
390 ]
391 )
393 def get_own_text(self, el: bs4.Tag, no_iframe: bool = False) -> list[str]:
394 """Get Own Text."""
396 return [
397 node for node in self.get_contents(el, no_iframe=no_iframe) if self.is_content_string(node) # type: ignore[misc]
398 ]
401class Inputs:
402 """Class for parsing and validating input items."""
404 @staticmethod
405 def validate_day(year: int, month: int, day: int) -> bool:
406 """Validate day."""
408 max_days = LONG_MONTH
409 if month == FEB:
410 max_days = FEB_LEAP_MONTH if ((year % 4 == 0) and (year % 100 != 0)) or (year % 400 == 0) else FEB_MONTH
411 elif month in MONTHS_30:
412 max_days = SHORT_MONTH
413 return 1 <= day <= max_days
415 @staticmethod
416 def validate_week(year: int, week: int) -> bool:
417 """Validate week."""
419 # Validate an ISO week number for `year`.
420 #
421 # Per ISO 8601 rules, the last ISO week of a year is the week
422 # containing Dec 28. Using Dec 28 guarantees we obtain the
423 # correct ISO week-number for the final week of `year`, even in
424 # years where Dec 31 falls in ISO week 01 of the following year.
425 #
426 # Example: if Dec 31 is a Thursday the year's last ISO week will
427 # be week 53; if Dec 31 is a Monday and that week is counted as
428 # week 1 of the next year, Dec 28 still belongs to the final
429 # week of the current ISO year and yields the correct max week.
430 max_week = datetime(year, 12, 28).isocalendar()[1]
431 return 1 <= week <= max_week
433 @staticmethod
434 def validate_month(month: int) -> bool:
435 """Validate month."""
437 return 1 <= month <= 12
439 @staticmethod
440 def validate_year(year: int) -> bool:
441 """Validate year."""
443 return 1 <= year
445 @staticmethod
446 def validate_hour(hour: int) -> bool:
447 """Validate hour."""
449 return 0 <= hour <= 23
451 @staticmethod
452 def validate_minutes(minutes: int) -> bool:
453 """Validate minutes."""
455 return 0 <= minutes <= 59
457 @classmethod
458 def parse_value(cls, itype: str, value: str | None) -> tuple[float, ...] | None:
459 """Parse the input value."""
461 parsed = None # type: tuple[float, ...] | None
462 if value is None:
463 return value
464 if itype == "date":
465 m = RE_DATE.match(value)
466 if m:
467 year = int(m.group('year'), 10)
468 month = int(m.group('month'), 10)
469 day = int(m.group('day'), 10)
470 if cls.validate_year(year) and cls.validate_month(month) and cls.validate_day(year, month, day):
471 parsed = (year, month, day)
472 elif itype == "month":
473 m = RE_MONTH.match(value)
474 if m:
475 year = int(m.group('year'), 10)
476 month = int(m.group('month'), 10)
477 if cls.validate_year(year) and cls.validate_month(month):
478 parsed = (year, month)
479 elif itype == "week":
480 m = RE_WEEK.match(value)
481 if m:
482 year = int(m.group('year'), 10)
483 week = int(m.group('week'), 10)
484 if cls.validate_year(year) and cls.validate_week(year, week):
485 parsed = (year, week)
486 elif itype == "time":
487 m = RE_TIME.match(value)
488 if m:
489 hour = int(m.group('hour'), 10)
490 minutes = int(m.group('minutes'), 10)
491 if cls.validate_hour(hour) and cls.validate_minutes(minutes):
492 parsed = (hour, minutes)
493 elif itype == "datetime-local":
494 m = RE_DATETIME.match(value)
495 if m:
496 year = int(m.group('year'), 10)
497 month = int(m.group('month'), 10)
498 day = int(m.group('day'), 10)
499 hour = int(m.group('hour'), 10)
500 minutes = int(m.group('minutes'), 10)
501 if (
502 cls.validate_year(year) and cls.validate_month(month) and cls.validate_day(year, month, day) and
503 cls.validate_hour(hour) and cls.validate_minutes(minutes)
504 ):
505 parsed = (year, month, day, hour, minutes)
506 elif itype in ("number", "range"):
507 m = RE_NUM.match(value)
508 if m:
509 parsed = (float(m.group('value')),)
510 return parsed
513class CSSMatch(_DocumentNav):
514 """Perform CSS matching."""
516 def __init__(
517 self,
518 selectors: ct.SelectorList,
519 scope: bs4.Tag | None,
520 namespaces: ct.Namespaces | None,
521 flags: int
522 ) -> None:
523 """Initialize."""
525 self.assert_valid_input(scope)
526 self.tag = scope
527 self.cached_meta_lang = [] # type: list[tuple[str, str]]
528 self.cached_default_forms = [] # type: list[tuple[bs4.Tag, bs4.Tag]]
529 self.cached_indeterminate_forms = [] # type: list[tuple[bs4.Tag, str, bool]]
530 self.selectors = selectors
531 self.namespaces = {} if namespaces is None else namespaces # type: ct.Namespaces | dict[str, str]
532 self.flags = flags
533 self.enable_cache = not bool(self.flags & util.NOCACHE)
534 self.iframe_restrict = False
535 self.nth_cache: dict[Hashable, dict[Hashable, list[int]]] = {}
536 self.sib_cache: dict[Hashable, dict[Hashable, int]] = {}
538 # Find the root element for the whole tree
539 doc = scope
540 parent = self.get_parent(doc)
541 while parent:
542 doc = parent
543 parent = self.get_parent(doc)
544 root = None # type: bs4.Tag | None
545 if not self.is_doc(doc):
546 root = doc
547 else:
548 for child in self.get_tag_children(doc):
549 root = child
550 break
552 self.root = root
553 self.scope = scope if scope is not doc else root
554 self.has_html_namespace = self.has_html_ns(root)
556 # A document can be both XML and HTML (XHTML)
557 self.is_xml = self.is_xml_tree(doc)
558 self.is_html = not self.is_xml or self.has_html_namespace
560 def reset(self) -> None: # pragma: no cover
561 """Reset."""
563 self.nth_cache.clear()
564 self.sib_cache.clear()
566 def supports_namespaces(self) -> bool:
567 """Check if namespaces are supported in the HTML type."""
569 return self.is_xml or self.has_html_namespace
571 def get_tag_ns(self, el: bs4.Tag | None) -> str:
572 """Get tag namespace."""
574 namespace = ''
575 if el is None: # pragma: no cover
576 return namespace
578 if self.supports_namespaces():
579 ns = self.get_uri(el)
580 if ns:
581 namespace = ns
582 else:
583 namespace = NS_XHTML
584 return namespace
586 def is_html_tag(self, el: bs4.Tag | None) -> bool:
587 """Check if tag is in HTML namespace."""
589 return self.get_tag_ns(el) == NS_XHTML
591 def get_tag(self, el: bs4.Tag | None) -> str | None:
592 """Get tag."""
594 name = self.get_tag_name(el)
595 return util.lower(name) if name is not None and not self.is_xml else name
597 def get_prefix(self, el: bs4.Tag) -> str | None:
598 """Get prefix."""
600 prefix = self.get_prefix_name(el)
601 return util.lower(prefix) if prefix is not None and not self.is_xml else prefix
603 def find_bidi(self, el: bs4.Tag) -> int | None:
604 """Get directionality from element text."""
606 for node in self.get_children(el):
608 # Analyze child text nodes
609 if self.is_tag(node):
611 # Avoid analyzing certain elements specified in the specification.
612 direction = DIR_MAP.get(util.lower(self.get_attribute_by_name(node, 'dir', '')), None)
613 name = self.get_tag(node)
614 if (
615 (name and name in ('bdi', 'script', 'style', 'textarea', 'iframe')) or
616 not self.is_html_tag(node) or
617 direction is not None
618 ):
619 continue # pragma: no cover
621 # Check directionality of this node's text
622 value = self.find_bidi(node)
623 if value is not None:
624 return value
626 # Direction could not be determined
627 continue # pragma: no cover
629 # Skip `doctype` comments, etc.
630 if self.is_special_string(node):
631 continue
633 # Analyze text nodes for directionality.
634 for c in cast('bs4.element.NavigableString', node):
635 bidi = unicodedata.bidirectional(c)
636 if bidi in ('AL', 'R', 'L'):
637 return ct.SEL_DIR_LTR if bidi == 'L' else ct.SEL_DIR_RTL
638 return None
640 def extended_language_filter(self, lang_range: str, lang_tag: str) -> bool:
641 """Filter the language tags."""
643 match = True
644 lang_range = RE_WILD_STRIP.sub('-', lang_range).lower()
645 ranges = lang_range.split('-')
646 subtags = lang_tag.lower().split('-')
647 length = len(ranges)
648 slength = len(subtags)
649 rindex = 0
650 sindex = 0
651 r = ranges[rindex]
652 s = subtags[sindex]
654 # Empty specified language should match unspecified language attributes
655 if length == 1 and slength == 1 and not r and r == s:
656 return True
658 # Primary tag needs to match
659 if (r != '*' and r != s) or (r == '*' and slength == 1 and not s):
660 match = False
662 rindex += 1
663 sindex += 1
665 # Match until we run out of ranges
666 while match and rindex < length:
667 r = ranges[rindex]
668 try:
669 s = subtags[sindex]
670 except IndexError:
671 # Ran out of subtags,
672 # but we still have ranges
673 match = False
674 continue
676 # Empty range
677 if not r:
678 match = False
679 continue
681 # Matched range
682 elif s == r:
683 rindex += 1
685 # Implicit wildcard cannot match
686 # singletons
687 elif len(s) == 1:
688 match = False
689 continue
691 # Implicitly matched, so grab next subtag
692 sindex += 1
694 return match
696 def match_attribute_name(
697 self,
698 el: bs4.Tag,
699 attr: str,
700 prefix: str | None
701 ) -> str | Sequence[str] | None:
702 """Match attribute name and return value if it exists."""
704 value = None
705 if self.supports_namespaces():
706 value = None
707 # If we have not defined namespaces, we can't very well find them, so don't bother trying.
708 if prefix:
709 ns = self.namespaces.get(prefix)
710 if ns is None and prefix != '*':
711 return None
712 else:
713 ns = None
715 for k, v in self.iter_attributes(el):
717 # Get attribute parts
718 namespace, name = self.split_namespace(el, k)
720 # Can't match a prefix attribute as we haven't specified one to match
721 # Try to match it normally as a whole `p:a` as selector may be trying `p\:a`.
722 if ns is None:
723 if (self.is_xml and attr == k) or (not self.is_xml and util.lower(attr) == util.lower(k)):
724 value = v
725 break
726 # Coverage is not finding this even though it is executed.
727 # Adding a print statement before this (and erasing coverage) causes coverage to find the line.
728 # Ignore the false positive message.
729 continue # pragma: no cover
731 # We can't match our desired prefix attribute as the attribute doesn't have a prefix
732 if namespace is None or (ns != namespace and prefix != '*'):
733 continue
735 # The attribute doesn't match.
736 if (util.lower(attr) != util.lower(name)) if not self.is_xml else (attr != name):
737 continue
739 value = v
740 break
741 else:
742 for k, v in self.iter_attributes(el):
743 if util.lower(attr) != util.lower(k):
744 continue
745 value = v
746 break
747 return value
749 def match_namespace(self, el: bs4.Tag, tag: ct.SelectorTag) -> bool:
750 """Match the namespace of the element."""
752 match = True
753 namespace = self.get_tag_ns(el)
754 default_namespace = self.namespaces.get('')
755 tag_ns = '' if tag.prefix is None else self.namespaces.get(tag.prefix)
756 # We must match the default namespace if one is not provided
757 if tag.prefix is None and (default_namespace is not None and namespace != default_namespace):
758 match = False
759 # If we specified `|tag`, we must not have a namespace.
760 elif (tag.prefix is not None and tag.prefix == '' and namespace):
761 match = False
762 # Verify prefix matches
763 elif (
764 tag.prefix and
765 tag.prefix != '*' and (tag_ns is None or namespace != tag_ns)
766 ):
767 match = False
768 return match
770 def match_attributes(self, el: bs4.Tag, attributes: tuple[ct.SelectorAttribute, ...]) -> bool:
771 """Match attributes."""
773 match = True
774 if attributes:
775 for a in attributes:
776 temp = self.match_attribute_name(el, a.attribute, a.prefix)
777 pattern = a.xml_type_pattern if self.is_xml and a.xml_type_pattern else a.pattern
778 if temp is None:
779 match = False
780 break
781 value = temp if isinstance(temp, str) else ' '.join(temp)
782 if pattern is None:
783 continue
784 elif pattern.match(value) is None:
785 match = False
786 break
787 return match
789 def match_tagname(self, el: bs4.Tag, tag: ct.SelectorTag) -> bool:
790 """Match tag name."""
792 name = (util.lower(tag.name) if not self.is_xml and tag.name is not None else tag.name)
793 return not (
794 name is not None and
795 name not in (self.get_tag(el), '*')
796 )
798 def match_tag(self, el: bs4.Tag, tag: ct.SelectorTag | None) -> bool:
799 """Match the tag."""
801 match = True
802 if tag is not None:
803 # Verify namespace
804 if not self.match_tagname(el, tag):
805 match = False
806 if match and not self.match_namespace(el, tag):
807 match = False
808 return match
810 def match_general_sibling(self, el: bs4.Tag, relation: ct.SelectorList) -> bool:
811 """Match general sibling combinator."""
813 found = False
815 if relation[0] is ct.Null: # pragma: no cover
816 return found
818 pkey: tuple[str | None, int] | None = None
819 key: tuple[ct.SelectorList, int] | None = None
821 # Setup the cache by the parent if present
822 parent = self.get_parent(el)
823 if parent is None: # pragma: no cover
824 return found
826 if parent:
827 pkey = (parent.name, id(parent))
829 # Initialize the cache if necessary
830 if pkey not in self.sib_cache:
831 self.sib_cache[pkey] = {}
833 # Check the cache to see if we already know where the first sibling is,
834 # and if we do, check if we are on the correct side of it.
835 # If we've previously searched and found no sibling, there is no sibling.
836 # Lastly, if this is our first time, setup the cache.
837 reverse = relation[0].rel_type == REL_HAS_SIBLING
838 start = len(parent) - 1 if reverse else 0
839 if pkey:
840 key = (relation, id(relation))
841 if key in self.sib_cache[pkey]:
842 index = self.sib_cache[pkey][key]
843 if index >= 0:
844 a, b = (index, start) if reverse else (start, index)
845 return not within(el, parent, a, b)
846 else:
847 return False
848 self.sib_cache[pkey][key] = start
850 # Start at the furthest endpoint and walk back towards the element looking for siblings.
851 # The current element counts as a sibling, but will not cause a match.
852 passed = False
853 incr = -1 if reverse else 1
854 for child in self.get_children(parent, start=start, reverse=reverse):
855 start += incr
856 if not isinstance(child, bs4.Tag):
857 continue
859 # Flag that we are passing the element.
860 # Any siblings we find aren't valid for this element.
861 if child is el:
862 passed = True
864 # We found the furthest sibling.
865 if self.match_selectors(child, relation):
866 found = True
867 break
869 # Cache the index of the sibling or mark as there being no siblings.
870 if pkey and key:
871 self.sib_cache[pkey][key] = start if found else -1
873 # If we passed the current element and then found a sibling, it doesn't count as a match.
874 if passed:
875 found = False
877 return found
879 def match_past_relations(self, el: bs4.Tag, relation: ct.SelectorList) -> bool:
880 """Match past relationship."""
882 found = False
883 # I don't think this can ever happen, but it makes `mypy` happy
884 if relation[0] is ct.Null: # pragma: no cover
885 return found
887 if relation[0].rel_type == REL_PARENT:
888 parent: bs4.Tag | None = el
889 while not found and parent and (parent := self.get_parent(parent, no_iframe=self.iframe_restrict)):
890 found = parent is not None and self.match_selectors(parent, relation)
891 elif relation[0].rel_type == REL_CLOSE_PARENT:
892 parent = self.get_parent(el, no_iframe=self.iframe_restrict)
893 found = parent is not None and self.match_selectors(parent, relation)
894 elif relation[0].rel_type == REL_SIBLING:
895 if self.enable_cache:
896 found = self.match_general_sibling(el, relation)
897 else:
898 sibling: bs4.Tag | None = el
899 while not found and sibling and (sibling := self.get_previous_tag(sibling)):
900 found = sibling is not None and self.match_selectors(sibling, relation)
901 elif relation[0].rel_type == REL_CLOSE_SIBLING:
902 sibling = self.get_previous_tag(el)
903 found = sibling is not None and self.match_selectors(sibling, relation)
904 return found
906 def match_future_child(self, parent: bs4.Tag, relation: ct.SelectorList, recursive: bool = False) -> bool:
907 """Match future child."""
909 match = False
910 if recursive:
911 children = self.get_tag_descendants # type: Callable[..., Iterator[bs4.Tag]]
912 else:
913 children = self.get_tag_children
914 for child in children(parent, no_iframe=self.iframe_restrict):
915 if self.match_selectors(child, relation):
916 match = True
917 break
918 return match
920 def match_future_relations(self, el: bs4.Tag, relation: ct.SelectorList) -> bool:
921 """Match future relationship."""
923 found = False
924 # I don't think this can ever happen, but it makes `mypy` happy
925 if relation[0] is ct.Null: # pragma: no cover
926 return found
928 if relation[0].rel_type == REL_HAS_PARENT:
929 found = self.match_future_child(el, relation, True)
930 elif relation[0].rel_type == REL_HAS_CLOSE_PARENT:
931 found = self.match_future_child(el, relation)
932 elif relation[0].rel_type == REL_HAS_SIBLING:
933 if self.enable_cache:
934 found = self.match_general_sibling(el, relation)
935 else:
936 sibling: bs4.Tag | None = el
937 while not found and sibling and (sibling := self.get_next_tag(sibling)):
938 found = self.match_selectors(sibling, relation)
939 elif relation[0].rel_type == REL_HAS_CLOSE_SIBLING:
940 sibling = self.get_next_tag(el)
941 found = sibling is not None and self.match_selectors(sibling, relation)
942 return found
944 def match_relations(self, el: bs4.Tag, relation: ct.SelectorList) -> bool:
945 """Match relationship to other elements."""
947 found = False
949 if relation[0] is ct.Null or relation[0].rel_type is None:
950 return found
952 if relation[0].rel_type.startswith(':'):
953 found = self.match_future_relations(el, relation)
954 else:
955 found = self.match_past_relations(el, relation)
957 return found
959 def match_id(self, el: bs4.Tag, ids: tuple[str, ...]) -> bool:
960 """Match element's ID."""
962 found = True
963 for i in ids:
964 if i != self.get_attribute_by_name(el, 'id', ''):
965 found = False
966 break
967 return found
969 def match_classes(self, el: bs4.Tag, classes: tuple[str, ...]) -> bool:
970 """Match element's classes."""
972 current_classes = self.get_classes(el)
973 found = True
974 for c in classes:
975 if c not in current_classes:
976 found = False
977 break
978 return found
980 def match_root(self, el: bs4.Tag) -> bool:
981 """Match element as root."""
983 is_root = self.is_root(el)
984 if is_root:
985 sibling = self.get_previous(el) # type: Any
986 while is_root and sibling is not None:
987 if (
988 self.is_tag(sibling) or (self.is_content_string(sibling) and sibling.strip()) or
989 self.is_cdata(sibling)
990 ):
991 is_root = False
992 else:
993 sibling = self.get_previous(sibling)
994 if is_root:
995 sibling = self.get_next(el)
996 while is_root and sibling is not None:
997 if (
998 self.is_tag(sibling) or (self.is_content_string(sibling) and sibling.strip()) or
999 self.is_cdata(sibling)
1000 ):
1001 is_root = False
1002 else:
1003 sibling = self.get_next(sibling)
1004 return is_root
1006 def match_scope(self, el: bs4.Tag) -> bool:
1007 """Match element as scope."""
1009 return self.scope is el
1011 def match_nth_tag_type(self, el: bs4.Tag, child: bs4.Tag) -> bool:
1012 """Match tag type for `nth` matches."""
1014 return (
1015 (self.get_tag(child) == self.get_tag(el)) and
1016 (self.get_tag_ns(child) == self.get_tag_ns(el))
1017 )
1019 def match_nth(self, el: bs4.Tag, nth: tuple[ct.SelectorNth, ...]) -> bool:
1020 """Match `nth` elements."""
1022 # `nth` selectors are evaluated against siblings under the same parent.
1023 parent = self.get_parent(el) # type: bs4.Tag | None
1024 pkey: tuple[str | None, int] | None = None
1025 key: tuple[ct.SelectorNth, int, str | None, str | None] | None = None
1026 start = rindex = 0
1027 incr = rincr = 0
1029 # Setup the cache by the parent, if parent a parent is present
1030 if self.enable_cache and parent:
1031 pkey = (parent.name, id(parent))
1033 # Initialize the cache if necessary
1034 if pkey not in self.nth_cache:
1035 self.nth_cache[pkey] = {}
1037 # Test element against the `nth` selectors.
1038 matched = True
1039 for n in nth:
1040 matched = False
1041 last = n.last
1042 key = None
1044 # Prepare the child iterator and get the starting, real index and the relative index
1045 if pkey and parent:
1046 # Get last info from the cache
1047 key = (n, id(n), self.get_tag(el), self.get_tag_ns(el)) if n.of_type else (n, id(n), None, None)
1048 valid = False
1049 if key in self.nth_cache[pkey]:
1050 start, rindex = self.nth_cache[pkey][key]
1051 if within(el, parent, start):
1052 last = False
1053 rincr = -1 if n.last else 1
1054 valid = True
1056 # Start/overwrite the cache if the cache was empty or invalid
1057 if not valid:
1058 start, rindex = len(parent) - 1 if last else 0, 0
1059 self.nth_cache[pkey][key] = [start, rindex]
1060 rincr = 1
1062 incr = 1 if not last else -1
1063 children = self.get_children(parent, start=start, reverse=last)
1065 # Non-cached handling of parented element
1066 elif parent:
1067 rindex = 0
1068 start = len(parent) - 1 if last else 0
1069 rincr = incr = 1
1070 children = self.get_children(parent, start=start, reverse=last)
1072 # No parent, just evaluate the element against the selectors
1073 else:
1074 start = rindex = 0
1075 rincr = incr = 1
1076 children = iter([el])
1078 # Find index of element compared to its siblings and check the index conditions
1079 child: bs4.Tag
1080 for child in children:
1081 start += incr
1083 # We only care about tags
1084 if not self.is_tag(child):
1085 continue
1087 # Handle `of S` in `nth-child` and handle `of-type`
1088 if (
1089 (n.selectors and not self.match_selectors(child, n.selectors)) or
1090 (n.of_type and not self.match_nth_tag_type(el, child))
1091 ):
1092 if child is el:
1093 break
1094 continue
1096 # Test the relative index against the `nth` requirement.
1097 rindex += rincr
1098 if child is el:
1099 if n.a != 0:
1100 v = (rindex - n.b) / n.a
1101 matched = v.is_integer() and v >= 0
1102 else:
1103 matched = rindex == n.b and n.b >= 1
1104 break
1106 # "Last index" selectors evaluate first from the bottom and then evaluate
1107 # from the first found element top-down. Start will be incremented in the
1108 # wrong direction, so increment it and step over the current index.
1109 if last:
1110 start += 2
1112 # Update the cache
1113 if pkey and key:
1114 self.nth_cache[pkey][key] = [start, rindex]
1116 # If we failed to match any `nth` selectors, quit.
1117 if not matched:
1118 break
1120 return matched
1122 def match_empty(self, el: bs4.Tag) -> bool:
1123 """Check if element is empty (if requested)."""
1125 is_empty = True
1126 for child in self.get_children(el):
1127 if self.is_tag(child):
1128 is_empty = False
1129 break
1130 elif self.is_content_string(child) and RE_NOT_EMPTY.search(child): # type: ignore[call-overload]
1131 is_empty = False
1132 break
1133 return is_empty
1135 def match_subselectors(self, el: bs4.Tag, selectors: tuple[ct.SelectorList, ...]) -> bool:
1136 """Match selectors."""
1138 match = True
1139 for sel in selectors:
1140 if not self.match_selectors(el, sel):
1141 match = False
1142 return match
1144 def match_contains(self, el: bs4.Tag, contains: tuple[ct.SelectorContains, ...]) -> bool:
1145 """Match element if it contains text."""
1147 match = True
1148 content = None # type: str | Sequence[str] | None
1149 for contain_list in contains:
1150 if content is None:
1151 if contain_list.own:
1152 content = self.get_own_text(el, no_iframe=self.is_html)
1153 else:
1154 content = self.get_text(el, no_iframe=self.is_html)
1155 found = False
1156 for text in contain_list.text:
1157 if contain_list.own:
1158 for c in content:
1159 if text in c:
1160 found = True
1161 break
1162 if found:
1163 break
1164 else:
1165 if text in content:
1166 found = True
1167 break
1168 if not found:
1169 match = False
1170 return match
1172 def match_default(self, el: bs4.Tag) -> bool:
1173 """Match default."""
1175 match = False
1177 # Find this input's form
1178 form = None # type: bs4.Tag | None
1179 parent = self.get_parent(el, no_iframe=True)
1180 while parent and form is None:
1181 if self.get_tag(parent) == 'form' and self.is_html_tag(parent):
1182 form = parent
1183 else:
1184 parent = self.get_parent(parent, no_iframe=True)
1186 if form is not None:
1187 # Look in form cache to see if we've already located its default button
1188 found_form = False
1189 for f, t in self.cached_default_forms:
1190 if f is form:
1191 found_form = True
1192 if t is el:
1193 match = True
1194 break
1196 # We didn't have the form cached, so look for its default button
1197 if not found_form:
1198 for child in self.get_tag_descendants(form, no_iframe=True):
1199 name = self.get_tag(child)
1200 # Can't do nested forms (haven't figured out why we never hit this)
1201 if name == 'form': # pragma: no cover
1202 break
1203 if name in ('input', 'button'):
1204 v = self.get_attribute_by_name(child, 'type', '')
1205 if v and util.lower(v) == 'submit':
1206 self.cached_default_forms.append((form, child))
1207 if el is child:
1208 match = True
1209 break
1210 return match
1212 def match_indeterminate(self, el: bs4.Tag) -> bool:
1213 """Match default."""
1215 match = False
1216 name = cast(str, self.get_attribute_by_name(el, 'name'))
1218 def get_parent_form(el: bs4.Tag) -> bs4.Tag | None:
1219 """Find this input's form."""
1220 form = None
1221 parent = self.get_parent(el, no_iframe=True)
1222 while form is None:
1223 if self.get_tag(parent) == 'form' and self.is_html_tag(parent):
1224 form = parent
1225 break
1226 last_parent = parent
1227 parent = self.get_parent(parent, no_iframe=True)
1228 if parent is None:
1229 form = last_parent
1230 break
1231 return form
1233 form = get_parent_form(el)
1235 # Look in form cache to see if we've already evaluated that its fellow radio buttons are indeterminate
1236 if form is not None:
1237 found_form = False
1238 for f, n, i in self.cached_indeterminate_forms:
1239 if f is form and n == name:
1240 found_form = True
1241 if i is True:
1242 match = True
1243 break
1245 # We didn't have the form cached, so validate that the radio button is indeterminate
1246 if not found_form:
1247 checked = False
1248 for child in self.get_tag_descendants(form, no_iframe=True):
1249 if child is el:
1250 continue
1251 tag_name = self.get_tag(child)
1252 if tag_name == 'input':
1253 is_radio = False
1254 check = False
1255 has_name = False
1256 for k, v in self.iter_attributes(child):
1257 if util.lower(k) == 'type' and util.lower(v) == 'radio':
1258 is_radio = True
1259 elif util.lower(k) == 'name' and v == name:
1260 has_name = True
1261 elif util.lower(k) == 'checked':
1262 check = True
1263 if is_radio and check and has_name and get_parent_form(child) is form:
1264 checked = True
1265 break
1266 if checked:
1267 break
1268 if not checked:
1269 match = True
1270 self.cached_indeterminate_forms.append((form, name, match))
1272 return match
1274 def match_lang(self, el: bs4.Tag, langs: tuple[ct.SelectorLang, ...]) -> bool:
1275 """Match languages."""
1277 match = False
1278 has_ns = self.supports_namespaces()
1279 root = self.root
1280 has_html_namespace = self.has_html_namespace
1282 # Walk parents looking for `lang` (HTML) or `xml:lang` XML property.
1283 parent = el # type: bs4.Tag | None
1284 found_lang = None
1285 last = None
1286 while not found_lang:
1287 has_html_ns = self.has_html_ns(parent)
1288 for k, v in self.iter_attributes(parent):
1289 attr_ns, attr = self.split_namespace(parent, k)
1290 if (
1291 ((not has_ns or has_html_ns) and (util.lower(k) if not self.is_xml else k) == 'lang') or
1292 (
1293 has_ns and not has_html_ns and attr_ns == NS_XML and
1294 (util.lower(attr) if not self.is_xml and attr is not None else attr) == 'lang'
1295 )
1296 ):
1297 found_lang = v
1298 break
1299 last = parent
1300 parent = self.get_parent(parent, no_iframe=self.is_html)
1302 if parent is None:
1303 root = last
1304 has_html_namespace = self.has_html_ns(root)
1305 parent = last
1306 break
1308 # Use cached meta language.
1309 if found_lang is None and self.cached_meta_lang:
1310 for cache in self.cached_meta_lang:
1311 if root is not None and cast(str, root) is cache[0]:
1312 found_lang = cache[1]
1314 # If we couldn't find a language, and the document is HTML, look to meta to determine language.
1315 if found_lang is None and (not self.is_xml or (has_html_namespace and root and root.name == 'html')):
1316 # Find head
1317 found = False
1318 for tag in ('html', 'head'):
1319 found = False
1320 for child in self.get_tag_children(parent, no_iframe=self.is_html):
1321 if self.get_tag(child) == tag and self.is_html_tag(child):
1322 found = True
1323 parent = child
1324 break
1325 if not found: # pragma: no cover
1326 break
1328 # Search meta tags
1329 if found and parent is not None:
1330 for child2 in parent:
1331 if isinstance(child2, bs4.Tag) and self.get_tag(child2) == 'meta' and self.is_html_tag(parent):
1332 c_lang = False
1333 content = None
1334 for k, v in self.iter_attributes(child2):
1335 if util.lower(k) == 'http-equiv' and util.lower(v) == 'content-language':
1336 c_lang = True
1337 if util.lower(k) == 'content':
1338 content = v
1339 if c_lang and content:
1340 found_lang = content
1341 self.cached_meta_lang.append((cast(str, root), cast(str, found_lang)))
1342 break
1343 if found_lang is not None:
1344 break
1345 if found_lang is None:
1346 self.cached_meta_lang.append((cast(str, root), ''))
1348 # If we determined a language, compare.
1349 if found_lang is not None:
1350 for patterns in langs:
1351 match = False
1352 for pattern in patterns:
1353 if self.extended_language_filter(pattern, cast(str, found_lang)):
1354 match = True
1355 if not match:
1356 break
1358 return match
1360 def match_dir(self, el: bs4.Tag | None, directionality: int) -> bool:
1361 """Check directionality."""
1363 # If we have to match both left and right, we can't match either.
1364 if directionality & ct.SEL_DIR_LTR and directionality & ct.SEL_DIR_RTL:
1365 return False
1367 if el is None or not self.is_html_tag(el):
1368 return False
1370 # Element has defined direction of left to right or right to left
1371 direction = DIR_MAP.get(util.lower(self.get_attribute_by_name(el, 'dir', '')), None)
1372 if direction not in (None, 0):
1373 return direction == directionality
1375 # Element is the document element (the root) and no direction assigned, assume left to right.
1376 is_root = self.is_root(el)
1377 if is_root and direction is None:
1378 return ct.SEL_DIR_LTR == directionality
1380 # If `input[type=telephone]` and no direction is assigned, assume left to right.
1381 name = self.get_tag(el)
1382 is_input = name == 'input'
1383 is_textarea = name == 'textarea'
1384 is_bdi = name == 'bdi'
1385 itype = util.lower(self.get_attribute_by_name(el, 'type', '')) if is_input else ''
1386 if is_input and itype == 'tel' and direction is None:
1387 return ct.SEL_DIR_LTR == directionality
1389 # Auto handling for text inputs
1390 if ((is_input and itype in ('text', 'search', 'tel', 'url', 'email')) or is_textarea) and direction == 0:
1391 if is_textarea:
1392 value = ''.join(node for node in self.get_contents(el, no_iframe=True) if self.is_content_string(node)) # type: ignore[misc]
1393 else:
1394 value = cast(str, self.get_attribute_by_name(el, 'value', ''))
1395 if value:
1396 for c in value:
1397 bidi = unicodedata.bidirectional(c)
1398 if bidi in ('AL', 'R', 'L'):
1399 direction = ct.SEL_DIR_LTR if bidi == 'L' else ct.SEL_DIR_RTL
1400 return direction == directionality
1401 # Assume left to right
1402 return ct.SEL_DIR_LTR == directionality
1403 elif is_root:
1404 return ct.SEL_DIR_LTR == directionality
1405 return self.match_dir(self.get_parent(el, no_iframe=True), directionality)
1407 # Auto handling for `bdi` and other non text inputs.
1408 if (is_bdi and direction is None) or direction == 0:
1409 direction = self.find_bidi(el)
1410 if direction is not None:
1411 return direction == directionality
1412 elif is_root:
1413 return ct.SEL_DIR_LTR == directionality
1414 return self.match_dir(self.get_parent(el, no_iframe=True), directionality)
1416 # Match parents direction
1417 return self.match_dir(self.get_parent(el, no_iframe=True), directionality)
1419 def match_range(self, el: bs4.Tag, condition: int) -> bool:
1420 """
1421 Match range.
1423 Behavior is modeled after what we see in browsers. Browsers seem to evaluate
1424 if the value is out of range, and if not, it is in range. So a missing value
1425 will not evaluate out of range; therefore, value is in range. Personally, I
1426 feel like this should evaluate as neither in or out of range.
1427 """
1429 out_of_range = False
1431 itype = util.lower(self.get_attribute_by_name(el, 'type'))
1432 mn = Inputs.parse_value(itype, cast(str, self.get_attribute_by_name(el, 'min', None)))
1433 mx = Inputs.parse_value(itype, cast(str, self.get_attribute_by_name(el, 'max', None)))
1435 # There is no valid min or max, so we cannot evaluate a range
1436 if mn is None and mx is None:
1437 return False
1439 value = Inputs.parse_value(itype, cast(str, self.get_attribute_by_name(el, 'value', None)))
1440 if value is not None:
1441 if itype in ("date", "datetime-local", "month", "week", "number", "range"):
1442 if mn is not None and value < mn:
1443 out_of_range = True
1444 if not out_of_range and mx is not None and value > mx:
1445 out_of_range = True
1446 elif itype == "time":
1447 if mn is not None and mx is not None and mn > mx:
1448 # Time is periodic, so this is a reversed/discontinuous range
1449 if value < mn and value > mx:
1450 out_of_range = True
1451 else:
1452 if mn is not None and value < mn:
1453 out_of_range = True
1454 if not out_of_range and mx is not None and value > mx:
1455 out_of_range = True
1457 return not out_of_range if condition & ct.SEL_IN_RANGE else out_of_range
1459 def match_defined(self, el: bs4.Tag) -> bool:
1460 """
1461 Match defined.
1463 `:defined` is related to custom elements in a browser.
1465 - If the document is XML (not XHTML), all tags will match.
1466 - Tags that are not custom (don't have a hyphen) are marked defined.
1467 - If the tag has a prefix (without or without a namespace), it will not match.
1469 This is of course requires the parser to provide us with the proper prefix and namespace info,
1470 if it doesn't, there is nothing we can do.
1471 """
1473 name = self.get_tag(el)
1474 return (
1475 name is not None and (
1476 name.find('-') == -1 or
1477 name.find(':') != -1 or
1478 self.get_prefix(el) is not None
1479 )
1480 )
1482 def match_placeholder_shown(self, el: bs4.Tag) -> bool:
1483 """
1484 Match placeholder shown according to HTML spec.
1486 - text area should be checked if they have content. A single newline does not count as content.
1488 """
1490 match = False
1491 content = self.get_text(el)
1492 if content in ('', '\n'):
1493 match = True
1495 return match
1497 def match_selectors(self, el: bs4.Tag, selectors: ct.SelectorList) -> bool:
1498 """Check if element matches one of the selectors."""
1500 match = False
1501 is_not = selectors.is_not
1502 is_html = selectors.is_html
1504 # Internal selector lists that use the HTML flag, will automatically get the `html` namespace.
1505 if is_html:
1506 namespaces = self.namespaces
1507 iframe_restrict = self.iframe_restrict
1508 self.namespaces = {'html': NS_XHTML}
1509 self.iframe_restrict = True
1511 if not is_html or self.is_html:
1512 for selector in selectors:
1513 match = is_not
1514 # We have a un-matchable situation (like `:focus` as you can focus an element in this environment)
1515 if selector is ct.Null:
1516 continue
1517 # Verify tag matches
1518 if not self.match_tag(el, selector.tag):
1519 continue
1520 # Verify tag is defined
1521 if selector.flags & ct.SEL_DEFINED and not self.match_defined(el):
1522 continue
1523 # Verify element is root
1524 if selector.flags & ct.SEL_ROOT and not self.match_root(el):
1525 continue
1526 # Verify element is scope
1527 if selector.flags & ct.SEL_SCOPE and not self.match_scope(el):
1528 continue
1529 # Verify element has placeholder shown
1530 if selector.flags & ct.SEL_PLACEHOLDER_SHOWN and not self.match_placeholder_shown(el):
1531 continue
1532 # Verify `nth` matches
1533 if selector.nth and not self.match_nth(el, selector.nth):
1534 continue
1535 if selector.flags & ct.SEL_EMPTY and not self.match_empty(el):
1536 continue
1537 # Verify id matches
1538 if selector.ids and not self.match_id(el, selector.ids):
1539 continue
1540 # Verify classes match
1541 if selector.classes and not self.match_classes(el, selector.classes):
1542 continue
1543 # Verify attribute(s) match
1544 if not self.match_attributes(el, selector.attributes):
1545 continue
1546 # Verify ranges
1547 if selector.flags & RANGES and not self.match_range(el, selector.flags & RANGES):
1548 continue
1549 # Verify language patterns
1550 if selector.lang and not self.match_lang(el, selector.lang):
1551 continue
1552 # Verify pseudo selector patterns
1553 if selector.selectors and not self.match_subselectors(el, selector.selectors):
1554 continue
1555 # Verify relationship selectors
1556 if selector.relation and not self.match_relations(el, selector.relation):
1557 continue
1558 # Validate that the current default selector match corresponds to the first submit button in the form
1559 if selector.flags & ct.SEL_DEFAULT and not self.match_default(el):
1560 continue
1561 # Validate that the unset radio button is among radio buttons with the same name in a form that are
1562 # also not set.
1563 if selector.flags & ct.SEL_INDETERMINATE and not self.match_indeterminate(el):
1564 continue
1565 # Validate element directionality
1566 if selector.flags & DIR_FLAGS and not self.match_dir(el, selector.flags & DIR_FLAGS):
1567 continue
1568 # Validate that the tag contains the specified text.
1569 if selector.contains and not self.match_contains(el, selector.contains):
1570 continue
1571 match = not is_not
1572 break
1574 # Restore actual namespaces being used for external selector lists
1575 if is_html:
1576 self.namespaces = namespaces
1577 self.iframe_restrict = iframe_restrict
1579 return match
1581 def select(self, limit: int = 0) -> Iterator[bs4.Tag]:
1582 """Match all tags under the targeted tag."""
1584 lim = None if limit < 1 else limit
1586 for child in self.get_tag_descendants(self.tag):
1587 if self.match(child):
1588 yield child
1589 if lim is not None:
1590 lim -= 1
1591 if lim < 1:
1592 break
1594 def closest(self) -> bs4.Tag | None:
1595 """Match closest ancestor."""
1597 current = self.tag # type: bs4.Tag | None
1598 closest = None
1599 while closest is None and current is not None:
1600 if self.match(current):
1601 closest = current
1602 else:
1603 current = self.get_parent(current)
1604 return closest
1606 def filter(self) -> list[bs4.Tag]: # noqa A001
1607 """Filter tag's children."""
1609 return [
1610 tag for tag in self.get_contents(self.tag)
1611 if isinstance(tag, bs4.Tag) and self.match(tag)
1612 ]
1614 def match(self, el: bs4.Tag) -> bool:
1615 """Match."""
1617 return not self.is_doc(el) and self.is_tag(el) and self.match_selectors(el, self.selectors)
1620class SoupSieve(ct.Immutable):
1621 """Compiled Soup Sieve selector matching object."""
1623 pattern: str
1624 selectors: ct.SelectorList
1625 namespaces: ct.Namespaces | None
1626 custom: dict[str, str]
1627 flags: int
1629 __slots__ = ("pattern", "selectors", "namespaces", "custom", "flags", "_hash")
1631 def __init__(
1632 self,
1633 pattern: str,
1634 selectors: ct.SelectorList,
1635 namespaces: ct.Namespaces | None,
1636 custom: ct.CustomSelectors | None,
1637 flags: int
1638 ):
1639 """Initialize."""
1641 super().__init__(
1642 pattern=pattern,
1643 selectors=selectors,
1644 namespaces=namespaces,
1645 custom=custom,
1646 flags=flags
1647 )
1649 def match(self, tag: bs4.Tag) -> bool:
1650 """Match."""
1652 return CSSMatch(self.selectors, tag, self.namespaces, self.flags).match(tag)
1654 def closest(self, tag: bs4.Tag) -> bs4.Tag | None:
1655 """Match closest ancestor."""
1657 return CSSMatch(self.selectors, tag, self.namespaces, self.flags).closest()
1659 def filter(self, iterable: Iterable[bs4.Tag]) -> list[bs4.Tag]: # noqa A001
1660 """
1661 Filter.
1663 `CSSMatch` can cache certain searches for tags of the same document,
1664 so if we are given a tag, all tags are from the same document,
1665 and we can take advantage of the optimization.
1667 Any other kind of iterable could have tags from different documents or detached tags,
1668 so for those, we use a new `CSSMatch` for each item in the iterable.
1669 """
1671 if isinstance(iterable, bs4.Tag):
1672 return CSSMatch(self.selectors, iterable, self.namespaces, self.flags).filter()
1673 else:
1674 # There is no guarantee that elements are from the same document, evaluate them separately.
1675 return [node for node in iterable if not CSSMatch.is_navigable_string(node) and self.match(node)]
1677 def select_one(self, tag: bs4.Tag) -> bs4.Tag | None:
1678 """Select a single tag."""
1680 tags = self.select(tag, limit=1)
1681 return tags[0] if tags else None
1683 def select(self, tag: bs4.Tag, limit: int = 0) -> list[bs4.Tag]:
1684 """Select the specified tags."""
1686 return list(self.iselect(tag, limit))
1688 def iselect(self, tag: bs4.Tag, limit: int = 0) -> Iterator[bs4.Tag]:
1689 """Iterate the specified tags."""
1691 yield from CSSMatch(self.selectors, tag, self.namespaces, self.flags).select(limit)
1693 def __repr__(self) -> str: # pragma: no cover
1694 """Representation."""
1696 return (
1697 f"SoupSieve(pattern={self.pattern!r}, namespaces={self.namespaces!r}, "
1698 f"custom={self.custom!r}, flags={self.flags!r})"
1699 )
1701 __str__ = __repr__
1704ct.pickle_register(SoupSieve)