Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/markdown/util.py: 75%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

153 statements  

1# Python Markdown 

2 

3# A Python implementation of John Gruber's Markdown. 

4 

5# Documentation: https://python-markdown.github.io/ 

6# GitHub: https://github.com/Python-Markdown/markdown/ 

7# PyPI: https://pypi.org/project/Markdown/ 

8 

9# Started by Manfred Stienstra (http://www.dwerg.net/). 

10# Maintained for a few years by Yuri Takhteyev (http://www.freewisdom.org). 

11# Currently maintained by Waylan Limberg (https://github.com/waylan), 

12# Dmitry Shachnev (https://github.com/mitya57) and Isaac Muse (https://github.com/facelessuser). 

13 

14# Copyright 2007-2023 The Python Markdown Project (v. 1.7 and later) 

15# Copyright 2004, 2005, 2006 Yuri Takhteyev (v. 0.2-1.6b) 

16# Copyright 2004 Manfred Stienstra (the original version) 

17 

18# License: BSD (see LICENSE.md for details). 

19 

20""" 

21This module contains various contacts, classes and functions which get referenced and used 

22throughout the code base. 

23""" 

24 

25from __future__ import annotations 

26 

27import re 

28import sys 

29import warnings 

30from functools import wraps, lru_cache 

31from itertools import count 

32from typing import TYPE_CHECKING, Generic, Iterator, NamedTuple, TypeVar, TypedDict, overload 

33 

34if TYPE_CHECKING: # pragma: no cover 

35 from markdown import Markdown 

36 import xml.etree.ElementTree as etree 

37 

38_T = TypeVar('_T') 

39 

40 

41""" 

42Constants you might want to modify 

43----------------------------------------------------------------------------- 

44""" 

45 

46 

47BLOCK_LEVEL_ELEMENTS: list[str] = [ 

48 # Elements which are invalid to wrap in a `<p>` tag. 

49 # See https://w3c.github.io/html/grouping-content.html#the-p-element 

50 'address', 'article', 'aside', 'blockquote', 'details', 'div', 'dl', 

51 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2', 'h3', 

52 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'main', 'menu', 'nav', 'ol', 

53 'p', 'pre', 'section', 'table', 'ul', 

54 # Other elements which Markdown should not be mucking up the contents of. 

55 'canvas', 'colgroup', 'dd', 'body', 'dt', 'group', 'html', 'iframe', 'li', 'legend', 

56 'math', 'map', 'noscript', 'output', 'object', 'option', 'progress', 'script', 

57 'style', 'summary', 'tbody', 'td', 'textarea', 'tfoot', 'th', 'thead', 'tr', 'video', 

58 'center' 

59] 

60""" 

61List of HTML tags which get treated as block-level elements. Same as the `block_level_elements` 

62attribute of the [`Markdown`][markdown.Markdown] class. Generally one should use the 

63attribute on the class. This remains for compatibility with older extensions. 

64""" 

65 

66# Placeholders 

67STX = '\u0002' 

68""" "Start of Text" marker for placeholder templates. """ 

69ETX = '\u0003' 

70""" "End of Text" marker for placeholder templates. """ 

71INLINE_PLACEHOLDER_PREFIX = STX+"klzzwxh:" 

72""" Prefix for inline placeholder template. """ 

73INLINE_PLACEHOLDER = INLINE_PLACEHOLDER_PREFIX + "%s" + ETX 

74""" Placeholder template for stashed inline text. """ 

75INLINE_PLACEHOLDER_RE = re.compile(INLINE_PLACEHOLDER % r'([0-9]+)') 

76""" Regular Expression which matches inline placeholders. """ 

77AMP_SUBSTITUTE = STX+"amp"+ETX 

78""" Placeholder template for HTML entities. """ 

79HTML_PLACEHOLDER = STX + "wzxhzdk:%s" + ETX 

80""" Placeholder template for raw HTML. """ 

81HTML_PLACEHOLDER_RE = re.compile(HTML_PLACEHOLDER % r'([0-9]+)') 

82""" Regular expression which matches HTML placeholders. """ 

83TAG_PLACEHOLDER = STX + "hzzhzkh:%s" + ETX 

84""" Placeholder template for tags. """ 

85 

86 

87# Constants you probably do not need to change 

88# ----------------------------------------------------------------------------- 

89 

90RTL_BIDI_RANGES = ( 

91 ('\u0590', '\u07FF'), 

92 # Hebrew (0590-05FF), Arabic (0600-06FF), 

93 # Syriac (0700-074F), Arabic supplement (0750-077F), 

94 # Thaana (0780-07BF), Nko (07C0-07FF). 

95 ('\u2D30', '\u2D7F') # Tifinagh 

96) 

97 

98 

99# AUXILIARY GLOBAL FUNCTIONS 

100# ============================================================================= 

101 

102 

103@lru_cache(maxsize=None) 

104def get_installed_extensions(): 

105 """ Return all entry_points in the `markdown.extensions` group. """ 

106 from importlib import metadata 

107 # Only load extension entry_points once. 

108 return metadata.entry_points(group='markdown.extensions') 

109 

110 

111def deprecated(message: str, stacklevel: int = 2): 

112 """ 

113 Raise a [`DeprecationWarning`][] when wrapped function/method is called. 

114 

115 Usage: 

116 

117 ```python 

118 @deprecated("This method will be removed in version X; use Y instead.") 

119 def some_method(): 

120 pass 

121 ``` 

122 """ 

123 def wrapper(func): 

124 @wraps(func) 

125 def deprecated_func(*args, **kwargs): 

126 warnings.warn( 

127 f"'{func.__name__}' is deprecated. {message}", 

128 category=DeprecationWarning, 

129 stacklevel=stacklevel 

130 ) 

131 return func(*args, **kwargs) 

132 return deprecated_func 

133 return wrapper 

134 

135 

136def parseBoolValue(value: str | None, fail_on_errors: bool = True, preserve_none: bool = False) -> bool | None: 

137 """Parses a string representing a boolean value. If parsing was successful, 

138 returns `True` or `False`. If `preserve_none=True`, returns `True`, `False`, 

139 or `None`. If parsing was not successful, raises `ValueError`, or, if 

140 `fail_on_errors=False`, returns `None`.""" 

141 if not isinstance(value, str): 

142 if preserve_none and value is None: 

143 return value 

144 return bool(value) 

145 elif preserve_none and value.lower() == 'none': 

146 return None 

147 elif value.lower() in ('true', 'yes', 'y', 'on', '1'): 

148 return True 

149 elif value.lower() in ('false', 'no', 'n', 'off', '0', 'none'): 

150 return False 

151 elif fail_on_errors: 

152 raise ValueError('Cannot parse bool value: %r' % value) 

153 

154 

155def code_escape(text: str) -> str: 

156 """HTML escape a string of code.""" 

157 if "&" in text: 

158 text = text.replace("&", "&amp;") 

159 if "<" in text: 

160 text = text.replace("<", "&lt;") 

161 if ">" in text: 

162 text = text.replace(">", "&gt;") 

163 return text 

164 

165 

166def _get_stack_depth(size: int = 2) -> int: 

167 """Get current stack depth, performantly. 

168 """ 

169 frame = sys._getframe(size) 

170 

171 for size in count(size): 

172 frame = frame.f_back 

173 if not frame: 

174 return size 

175 

176 

177def nearing_recursion_limit() -> bool: 

178 """Return true if current stack depth is within 100 of maximum limit.""" 

179 return sys.getrecursionlimit() - _get_stack_depth() < 100 

180 

181 

182# MISC AUXILIARY CLASSES 

183# ============================================================================= 

184 

185 

186class AtomicString(str): 

187 """A string which should not be further processed.""" 

188 pass 

189 

190 

191class Processor: 

192 """ The base class for all processors. 

193 

194 Attributes: 

195 Processor.md: The `Markdown` instance passed in an initialization. 

196 

197 Arguments: 

198 md: The `Markdown` instance this processor is a part of. 

199 

200 """ 

201 def __init__(self, md: Markdown | None = None): 

202 self.md = md 

203 

204 

205if TYPE_CHECKING: # pragma: no cover 

206 class TagData(TypedDict): 

207 tag: str 

208 attrs: dict[str, str] 

209 left_index: int 

210 right_index: int 

211 

212 

213class HtmlStash: 

214 """ 

215 This class is used for stashing HTML objects that we extract 

216 in the beginning and replace with place-holders. 

217 """ 

218 

219 def __init__(self): 

220 """ Create an `HtmlStash`. """ 

221 self.html_counter = 0 # for counting inline html segments 

222 self.rawHtmlBlocks: list[str | etree.Element] = [] 

223 self.tag_counter = 0 

224 self.tag_data: list[TagData] = [] # list of dictionaries in the order tags appear 

225 

226 def store(self, html: str | etree.Element) -> str: 

227 """ 

228 Saves an HTML segment for later reinsertion. Returns a 

229 placeholder string that needs to be inserted into the 

230 document. 

231 

232 Keyword arguments: 

233 html: An html segment. 

234 

235 Returns: 

236 A placeholder string. 

237 

238 """ 

239 self.rawHtmlBlocks.append(html) 

240 placeholder = self.get_placeholder(self.html_counter) 

241 self.html_counter += 1 

242 return placeholder 

243 

244 def reset(self) -> None: 

245 """ Clear the stash. """ 

246 self.html_counter = 0 

247 self.rawHtmlBlocks = [] 

248 

249 def get_placeholder(self, key: int) -> str: 

250 return HTML_PLACEHOLDER % key 

251 

252 def store_tag(self, tag: str, attrs: dict[str, str], left_index: int, right_index: int) -> str: 

253 """Store tag data and return a placeholder.""" 

254 self.tag_data.append({'tag': tag, 'attrs': attrs, 

255 'left_index': left_index, 

256 'right_index': right_index}) 

257 placeholder = TAG_PLACEHOLDER % str(self.tag_counter) 

258 self.tag_counter += 1 # equal to the tag's index in `self.tag_data` 

259 return placeholder 

260 

261 

262# Used internally by `Registry` for each item in its sorted list. 

263# Provides an easier to read API when editing the code later. 

264# For example, `item.name` is more clear than `item[0]`. 

265class _PriorityItem(NamedTuple): 

266 name: str 

267 priority: float 

268 

269 

270class Registry(Generic[_T]): 

271 """ 

272 A priority sorted registry. 

273 

274 A `Registry` instance provides two public methods to alter the data of the 

275 registry: `register` and `deregister`. Use `register` to add items and 

276 `deregister` to remove items. See each method for specifics. 

277 

278 When registering an item, a "name" and a "priority" must be provided. All 

279 items are automatically sorted by "priority" from highest to lowest. The 

280 "name" is used to remove ("deregister") and get items. 

281 

282 A `Registry` instance it like a list (which maintains order) when reading 

283 data. You may iterate over the items, get an item and get a count (length) 

284 of all items. You may also check that the registry contains an item. 

285 

286 When getting an item you may use either the index of the item or the 

287 string-based "name". For example: 

288 

289 registry = Registry() 

290 registry.register(SomeItem(), 'itemname', 20) 

291 # Get the item by index 

292 item = registry[0] 

293 # Get the item by name 

294 item = registry['itemname'] 

295 

296 When checking that the registry contains an item, you may use either the 

297 string-based "name", or a reference to the actual item. For example: 

298 

299 someitem = SomeItem() 

300 registry.register(someitem, 'itemname', 20) 

301 # Contains the name 

302 assert 'itemname' in registry 

303 # Contains the item instance 

304 assert someitem in registry 

305 

306 The method `get_index_for_name` is also available to obtain the index of 

307 an item using that item's assigned "name". 

308 """ 

309 

310 def __init__(self): 

311 self._data: dict[str, _T] = {} 

312 self._priority: list[_PriorityItem] = [] 

313 self._is_sorted = False 

314 

315 def __contains__(self, item: str | _T) -> bool: 

316 if isinstance(item, str): 

317 # Check if an item exists by this name. 

318 return item in self._data.keys() 

319 # Check if this instance exists. 

320 return item in self._data.values() 

321 

322 def __iter__(self) -> Iterator[_T]: 

323 self._sort() 

324 return iter([self._data[k] for k, p in self._priority]) 

325 

326 @overload 

327 def __getitem__(self, key: str | int) -> _T: # pragma: no cover 

328 ... 

329 

330 @overload 

331 def __getitem__(self, key: slice) -> Registry[_T]: # pragma: no cover 

332 ... 

333 

334 def __getitem__(self, key: str | int | slice) -> _T | Registry[_T]: 

335 self._sort() 

336 if isinstance(key, slice): 

337 data: Registry[_T] = Registry() 

338 for k, p in self._priority[key]: 

339 data.register(self._data[k], k, p) 

340 return data 

341 if isinstance(key, int): 

342 return self._data[self._priority[key].name] 

343 return self._data[key] 

344 

345 def __len__(self) -> int: 

346 return len(self._priority) 

347 

348 def __repr__(self): 

349 return '<{}({})>'.format(self.__class__.__name__, list(self)) 

350 

351 def get_index_for_name(self, name: str) -> int: 

352 """ 

353 Return the index of the given name. 

354 """ 

355 if name in self: 

356 self._sort() 

357 return self._priority.index( 

358 [x for x in self._priority if x.name == name][0] 

359 ) 

360 raise ValueError('No item named "{}" exists.'.format(name)) 

361 

362 def register(self, item: _T, name: str, priority: float) -> None: 

363 """ 

364 Add an item to the registry with the given name and priority. 

365 

366 Arguments: 

367 item: The item being registered. 

368 name: A string used to reference the item. 

369 priority: An integer or float used to sort against all items. 

370 

371 If an item is registered with a "name" which already exists, the 

372 existing item is replaced with the new item. Treat carefully as the 

373 old item is lost with no way to recover it. The new item will be 

374 sorted according to its priority and will **not** retain the position 

375 of the old item. 

376 

377 Items assigned a higher number are given a higher priority. In other 

378 words, items assigned a higher number are sorted to be before those 

379 assigned a lower number. 

380 """ 

381 if name in self: 

382 # Remove existing item of same name first 

383 self.deregister(name) 

384 self._is_sorted = False 

385 self._data[name] = item 

386 self._priority.append(_PriorityItem(name, priority)) 

387 

388 def deregister(self, name: str, strict: bool = True) -> None: 

389 """ 

390 Remove an item from the registry. 

391 

392 Set `strict=False` to fail silently. Otherwise a [`ValueError`][] is raised for an unknown `name`. 

393 """ 

394 try: 

395 index = self.get_index_for_name(name) 

396 del self._priority[index] 

397 del self._data[name] 

398 except ValueError: 

399 if strict: 

400 raise 

401 

402 def _sort(self) -> None: 

403 """ 

404 Sort the registry by priority from highest to lowest. 

405 

406 This method is called internally and should never be explicitly called. 

407 """ 

408 if not self._is_sorted: 

409 self._priority.sort(key=lambda item: item.priority, reverse=True) 

410 self._is_sorted = True