Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/markdown/util.py: 75%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1# Python Markdown
3# A Python implementation of John Gruber's Markdown.
5# Documentation: https://python-markdown.github.io/
6# GitHub: https://github.com/Python-Markdown/markdown/
7# PyPI: https://pypi.org/project/Markdown/
9# Started by Manfred Stienstra (http://www.dwerg.net/).
10# Maintained for a few years by Yuri Takhteyev (http://www.freewisdom.org).
11# Currently maintained by Waylan Limberg (https://github.com/waylan),
12# Dmitry Shachnev (https://github.com/mitya57) and Isaac Muse (https://github.com/facelessuser).
14# Copyright 2007-2023 The Python Markdown Project (v. 1.7 and later)
15# Copyright 2004, 2005, 2006 Yuri Takhteyev (v. 0.2-1.6b)
16# Copyright 2004 Manfred Stienstra (the original version)
18# License: BSD (see LICENSE.md for details).
20"""
21This module contains various contacts, classes and functions which get referenced and used
22throughout the code base.
23"""
25from __future__ import annotations
27import re
28import sys
29import warnings
30from functools import wraps, lru_cache
31from itertools import count
32from typing import TYPE_CHECKING, Generic, Iterator, NamedTuple, TypeVar, TypedDict, overload
34if TYPE_CHECKING: # pragma: no cover
35 from markdown import Markdown
36 import xml.etree.ElementTree as etree
38_T = TypeVar('_T')
41"""
42Constants you might want to modify
43-----------------------------------------------------------------------------
44"""
47BLOCK_LEVEL_ELEMENTS: list[str] = [
48 # Elements which are invalid to wrap in a `<p>` tag.
49 # See https://w3c.github.io/html/grouping-content.html#the-p-element
50 'address', 'article', 'aside', 'blockquote', 'details', 'div', 'dl',
51 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2', 'h3',
52 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'main', 'menu', 'nav', 'ol',
53 'p', 'pre', 'section', 'table', 'ul',
54 # Other elements which Markdown should not be mucking up the contents of.
55 'canvas', 'colgroup', 'dd', 'body', 'dt', 'group', 'html', 'iframe', 'li', 'legend',
56 'math', 'map', 'noscript', 'output', 'object', 'option', 'progress', 'script',
57 'style', 'summary', 'tbody', 'td', 'textarea', 'tfoot', 'th', 'thead', 'tr', 'video',
58 'center'
59]
60"""
61List of HTML tags which get treated as block-level elements. Same as the `block_level_elements`
62attribute of the [`Markdown`][markdown.Markdown] class. Generally one should use the
63attribute on the class. This remains for compatibility with older extensions.
64"""
66# Placeholders
67STX = '\u0002'
68""" "Start of Text" marker for placeholder templates. """
69ETX = '\u0003'
70""" "End of Text" marker for placeholder templates. """
71INLINE_PLACEHOLDER_PREFIX = STX+"klzzwxh:"
72""" Prefix for inline placeholder template. """
73INLINE_PLACEHOLDER = INLINE_PLACEHOLDER_PREFIX + "%s" + ETX
74""" Placeholder template for stashed inline text. """
75INLINE_PLACEHOLDER_RE = re.compile(INLINE_PLACEHOLDER % r'([0-9]+)')
76""" Regular Expression which matches inline placeholders. """
77AMP_SUBSTITUTE = STX+"amp"+ETX
78""" Placeholder template for HTML entities. """
79HTML_PLACEHOLDER = STX + "wzxhzdk:%s" + ETX
80""" Placeholder template for raw HTML. """
81HTML_PLACEHOLDER_RE = re.compile(HTML_PLACEHOLDER % r'([0-9]+)')
82""" Regular expression which matches HTML placeholders. """
83TAG_PLACEHOLDER = STX + "hzzhzkh:%s" + ETX
84""" Placeholder template for tags. """
87# Constants you probably do not need to change
88# -----------------------------------------------------------------------------
90RTL_BIDI_RANGES = (
91 ('\u0590', '\u07FF'),
92 # Hebrew (0590-05FF), Arabic (0600-06FF),
93 # Syriac (0700-074F), Arabic supplement (0750-077F),
94 # Thaana (0780-07BF), Nko (07C0-07FF).
95 ('\u2D30', '\u2D7F') # Tifinagh
96)
99# AUXILIARY GLOBAL FUNCTIONS
100# =============================================================================
103@lru_cache(maxsize=None)
104def get_installed_extensions():
105 """ Return all entry_points in the `markdown.extensions` group. """
106 from importlib import metadata
107 # Only load extension entry_points once.
108 return metadata.entry_points(group='markdown.extensions')
111def deprecated(message: str, stacklevel: int = 2):
112 """
113 Raise a [`DeprecationWarning`][] when wrapped function/method is called.
115 Usage:
117 ```python
118 @deprecated("This method will be removed in version X; use Y instead.")
119 def some_method():
120 pass
121 ```
122 """
123 def wrapper(func):
124 @wraps(func)
125 def deprecated_func(*args, **kwargs):
126 warnings.warn(
127 f"'{func.__name__}' is deprecated. {message}",
128 category=DeprecationWarning,
129 stacklevel=stacklevel
130 )
131 return func(*args, **kwargs)
132 return deprecated_func
133 return wrapper
136def parseBoolValue(value: str | None, fail_on_errors: bool = True, preserve_none: bool = False) -> bool | None:
137 """Parses a string representing a boolean value. If parsing was successful,
138 returns `True` or `False`. If `preserve_none=True`, returns `True`, `False`,
139 or `None`. If parsing was not successful, raises `ValueError`, or, if
140 `fail_on_errors=False`, returns `None`."""
141 if not isinstance(value, str):
142 if preserve_none and value is None:
143 return value
144 return bool(value)
145 elif preserve_none and value.lower() == 'none':
146 return None
147 elif value.lower() in ('true', 'yes', 'y', 'on', '1'):
148 return True
149 elif value.lower() in ('false', 'no', 'n', 'off', '0', 'none'):
150 return False
151 elif fail_on_errors:
152 raise ValueError('Cannot parse bool value: %r' % value)
155def code_escape(text: str) -> str:
156 """HTML escape a string of code."""
157 if "&" in text:
158 text = text.replace("&", "&")
159 if "<" in text:
160 text = text.replace("<", "<")
161 if ">" in text:
162 text = text.replace(">", ">")
163 return text
166def _get_stack_depth(size: int = 2) -> int:
167 """Get current stack depth, performantly.
168 """
169 frame = sys._getframe(size)
171 for size in count(size):
172 frame = frame.f_back
173 if not frame:
174 return size
177def nearing_recursion_limit() -> bool:
178 """Return true if current stack depth is within 100 of maximum limit."""
179 return sys.getrecursionlimit() - _get_stack_depth() < 100
182# MISC AUXILIARY CLASSES
183# =============================================================================
186class AtomicString(str):
187 """A string which should not be further processed."""
188 pass
191class Processor:
192 """ The base class for all processors.
194 Attributes:
195 Processor.md: The `Markdown` instance passed in an initialization.
197 Arguments:
198 md: The `Markdown` instance this processor is a part of.
200 """
201 def __init__(self, md: Markdown | None = None):
202 self.md = md
205if TYPE_CHECKING: # pragma: no cover
206 class TagData(TypedDict):
207 tag: str
208 attrs: dict[str, str]
209 left_index: int
210 right_index: int
213class HtmlStash:
214 """
215 This class is used for stashing HTML objects that we extract
216 in the beginning and replace with place-holders.
217 """
219 def __init__(self):
220 """ Create an `HtmlStash`. """
221 self.html_counter = 0 # for counting inline html segments
222 self.rawHtmlBlocks: list[str | etree.Element] = []
223 self.tag_counter = 0
224 self.tag_data: list[TagData] = [] # list of dictionaries in the order tags appear
226 def store(self, html: str | etree.Element) -> str:
227 """
228 Saves an HTML segment for later reinsertion. Returns a
229 placeholder string that needs to be inserted into the
230 document.
232 Keyword arguments:
233 html: An html segment.
235 Returns:
236 A placeholder string.
238 """
239 self.rawHtmlBlocks.append(html)
240 placeholder = self.get_placeholder(self.html_counter)
241 self.html_counter += 1
242 return placeholder
244 def reset(self) -> None:
245 """ Clear the stash. """
246 self.html_counter = 0
247 self.rawHtmlBlocks = []
249 def get_placeholder(self, key: int) -> str:
250 return HTML_PLACEHOLDER % key
252 def store_tag(self, tag: str, attrs: dict[str, str], left_index: int, right_index: int) -> str:
253 """Store tag data and return a placeholder."""
254 self.tag_data.append({'tag': tag, 'attrs': attrs,
255 'left_index': left_index,
256 'right_index': right_index})
257 placeholder = TAG_PLACEHOLDER % str(self.tag_counter)
258 self.tag_counter += 1 # equal to the tag's index in `self.tag_data`
259 return placeholder
262# Used internally by `Registry` for each item in its sorted list.
263# Provides an easier to read API when editing the code later.
264# For example, `item.name` is more clear than `item[0]`.
265class _PriorityItem(NamedTuple):
266 name: str
267 priority: float
270class Registry(Generic[_T]):
271 """
272 A priority sorted registry.
274 A `Registry` instance provides two public methods to alter the data of the
275 registry: `register` and `deregister`. Use `register` to add items and
276 `deregister` to remove items. See each method for specifics.
278 When registering an item, a "name" and a "priority" must be provided. All
279 items are automatically sorted by "priority" from highest to lowest. The
280 "name" is used to remove ("deregister") and get items.
282 A `Registry` instance it like a list (which maintains order) when reading
283 data. You may iterate over the items, get an item and get a count (length)
284 of all items. You may also check that the registry contains an item.
286 When getting an item you may use either the index of the item or the
287 string-based "name". For example:
289 registry = Registry()
290 registry.register(SomeItem(), 'itemname', 20)
291 # Get the item by index
292 item = registry[0]
293 # Get the item by name
294 item = registry['itemname']
296 When checking that the registry contains an item, you may use either the
297 string-based "name", or a reference to the actual item. For example:
299 someitem = SomeItem()
300 registry.register(someitem, 'itemname', 20)
301 # Contains the name
302 assert 'itemname' in registry
303 # Contains the item instance
304 assert someitem in registry
306 The method `get_index_for_name` is also available to obtain the index of
307 an item using that item's assigned "name".
308 """
310 def __init__(self):
311 self._data: dict[str, _T] = {}
312 self._priority: list[_PriorityItem] = []
313 self._is_sorted = False
315 def __contains__(self, item: str | _T) -> bool:
316 if isinstance(item, str):
317 # Check if an item exists by this name.
318 return item in self._data.keys()
319 # Check if this instance exists.
320 return item in self._data.values()
322 def __iter__(self) -> Iterator[_T]:
323 self._sort()
324 return iter([self._data[k] for k, p in self._priority])
326 @overload
327 def __getitem__(self, key: str | int) -> _T: # pragma: no cover
328 ...
330 @overload
331 def __getitem__(self, key: slice) -> Registry[_T]: # pragma: no cover
332 ...
334 def __getitem__(self, key: str | int | slice) -> _T | Registry[_T]:
335 self._sort()
336 if isinstance(key, slice):
337 data: Registry[_T] = Registry()
338 for k, p in self._priority[key]:
339 data.register(self._data[k], k, p)
340 return data
341 if isinstance(key, int):
342 return self._data[self._priority[key].name]
343 return self._data[key]
345 def __len__(self) -> int:
346 return len(self._priority)
348 def __repr__(self):
349 return '<{}({})>'.format(self.__class__.__name__, list(self))
351 def get_index_for_name(self, name: str) -> int:
352 """
353 Return the index of the given name.
354 """
355 if name in self:
356 self._sort()
357 return self._priority.index(
358 [x for x in self._priority if x.name == name][0]
359 )
360 raise ValueError('No item named "{}" exists.'.format(name))
362 def register(self, item: _T, name: str, priority: float) -> None:
363 """
364 Add an item to the registry with the given name and priority.
366 Arguments:
367 item: The item being registered.
368 name: A string used to reference the item.
369 priority: An integer or float used to sort against all items.
371 If an item is registered with a "name" which already exists, the
372 existing item is replaced with the new item. Treat carefully as the
373 old item is lost with no way to recover it. The new item will be
374 sorted according to its priority and will **not** retain the position
375 of the old item.
377 Items assigned a higher number are given a higher priority. In other
378 words, items assigned a higher number are sorted to be before those
379 assigned a lower number.
380 """
381 if name in self:
382 # Remove existing item of same name first
383 self.deregister(name)
384 self._is_sorted = False
385 self._data[name] = item
386 self._priority.append(_PriorityItem(name, priority))
388 def deregister(self, name: str, strict: bool = True) -> None:
389 """
390 Remove an item from the registry.
392 Set `strict=False` to fail silently. Otherwise a [`ValueError`][] is raised for an unknown `name`.
393 """
394 try:
395 index = self.get_index_for_name(name)
396 del self._priority[index]
397 del self._data[name]
398 except ValueError:
399 if strict:
400 raise
402 def _sort(self) -> None:
403 """
404 Sort the registry by priority from highest to lowest.
406 This method is called internally and should never be explicitly called.
407 """
408 if not self._is_sorted:
409 self._priority.sort(key=lambda item: item.priority, reverse=True)
410 self._is_sorted = True