Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/markdown/inlinepatterns.py: 99%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

439 statements  

1# Python Markdown 

2 

3# A Python implementation of John Gruber's Markdown. 

4 

5# Documentation: https://python-markdown.github.io/ 

6# GitHub: https://github.com/Python-Markdown/markdown/ 

7# PyPI: https://pypi.org/project/Markdown/ 

8 

9# Started by Manfred Stienstra (http://www.dwerg.net/). 

10# Maintained for a few years by Yuri Takhteyev (http://www.freewisdom.org). 

11# Currently maintained by Waylan Limberg (https://github.com/waylan), 

12# Dmitry Shachnev (https://github.com/mitya57) and Isaac Muse (https://github.com/facelessuser). 

13 

14# Copyright 2007-2023 The Python Markdown Project (v. 1.7 and later) 

15# Copyright 2004, 2005, 2006 Yuri Takhteyev (v. 0.2-1.6b) 

16# Copyright 2004 Manfred Stienstra (the original version) 

17 

18# License: BSD (see LICENSE.md for details). 

19 

20""" 

21In version 3.0, a new, more flexible inline processor was added, [`markdown.inlinepatterns.InlineProcessor`][]. The 

22original inline patterns, which inherit from [`markdown.inlinepatterns.Pattern`][] or one of its children are still 

23supported, though users are encouraged to migrate. 

24 

25The new `InlineProcessor` provides two major enhancements to `Patterns`: 

26 

271. Inline Processors no longer need to match the entire block, so regular expressions no longer need to start with 

28 `r'^(.*?)'` and end with `r'(.*?)%'`. This runs faster. The returned [`Match`][re.Match] object will only contain 

29 what is explicitly matched in the pattern, and extension pattern groups now start with `m.group(1)`. 

30 

312. The `handleMatch` method now takes an additional input called `data`, which is the entire block under analysis, 

32 not just what is matched with the specified pattern. The method now returns the element *and* the indexes relative 

33 to `data` that the return element is replacing (usually `m.start(0)` and `m.end(0)`). If the boundaries are 

34 returned as `None`, it is assumed that the match did not take place, and nothing will be altered in `data`. 

35 

36 This allows handling of more complex constructs than regular expressions can handle, e.g., matching nested 

37 brackets, and explicit control of the span "consumed" by the processor. 

38 

39""" 

40 

41from __future__ import annotations 

42 

43from . import util 

44from typing import TYPE_CHECKING, Any, Collection, NamedTuple 

45import re 

46import xml.etree.ElementTree as etree 

47from html import entities 

48 

49if TYPE_CHECKING: # pragma: no cover 

50 from markdown import Markdown 

51 

52 

53def build_inlinepatterns(md: Markdown, **kwargs: Any) -> util.Registry[InlineProcessor]: 

54 """ 

55 Build the default set of inline patterns for Markdown. 

56 

57 The order in which processors and/or patterns are applied is very important - e.g. if we first replace 

58 `http://.../` links with `<a>` tags and _then_ try to replace inline HTML, we would end up with a mess. So, we 

59 apply the expressions in the following order: 

60 

61 * backticks and escaped characters have to be handled before everything else so that we can preempt any markdown 

62 patterns by escaping them; 

63 

64 * then we handle the various types of links (auto-links must be handled before inline HTML); 

65 

66 * then we handle inline HTML. At this point we will simply replace all inline HTML strings with a placeholder 

67 and add the actual HTML to a stash; 

68 

69 * finally we apply strong, emphasis, etc. 

70 

71 """ 

72 inlinePatterns = util.Registry() 

73 inlinePatterns.register(BacktickInlineProcessor(BACKTICK_RE), 'backtick', 190) 

74 inlinePatterns.register(EscapeInlineProcessor(ESCAPE_RE, md), 'escape', 180) 

75 inlinePatterns.register(ReferenceInlineProcessor(REFERENCE_RE, md), 'reference', 170) 

76 inlinePatterns.register(LinkInlineProcessor(LINK_RE, md), 'link', 160) 

77 inlinePatterns.register(ImageInlineProcessor(IMAGE_LINK_RE, md), 'image_link', 150) 

78 inlinePatterns.register( 

79 ImageReferenceInlineProcessor(IMAGE_REFERENCE_RE, md), 'image_reference', 140 

80 ) 

81 inlinePatterns.register( 

82 ShortReferenceInlineProcessor(REFERENCE_RE, md), 'short_reference', 130 

83 ) 

84 inlinePatterns.register( 

85 ShortImageReferenceInlineProcessor(IMAGE_REFERENCE_RE, md), 'short_image_ref', 125 

86 ) 

87 inlinePatterns.register(AutolinkInlineProcessor(AUTOLINK_RE, md), 'autolink', 120) 

88 inlinePatterns.register(AutomailInlineProcessor(AUTOMAIL_RE, md), 'automail', 110) 

89 inlinePatterns.register(SubstituteTagInlineProcessor(LINE_BREAK_RE, 'br'), 'linebreak', 100) 

90 inlinePatterns.register(HtmlInlineProcessor(HTML_RE, md), 'html', 90) 

91 inlinePatterns.register(HtmlInlineProcessor(ENTITY_RE, md), 'entity', 80) 

92 inlinePatterns.register(SimpleTextInlineProcessor(NOT_STRONG_RE), 'not_strong', 70) 

93 inlinePatterns.register(AsteriskProcessor(r'\*'), 'em_strong', 60) 

94 inlinePatterns.register(UnderscoreProcessor(r'_'), 'em_strong2', 50) 

95 return inlinePatterns 

96 

97 

98# The actual regular expressions for patterns 

99# ----------------------------------------------------------------------------- 

100 

101NOIMG = r'(?<!\!)' 

102""" Match not an image. Partial regular expression which matches if not preceded by `!`. """ 

103 

104BACKTICK_RE = r'(?:(?<!\\)((?:\\{2})+)(?=`+)|(?<!\\)`)' 

105""" Match backtick quoted string (`` `e=f()` `` or ``` ``e=f("`")`` ```). """ 

106 

107ESCAPE_RE = r'\\(.)' 

108""" Match a backslash escaped character (`\\<` or `\\*`). """ 

109 

110EMPHASIS_RE = r'(\*)([^\*]+)\1' 

111""" Match emphasis with an asterisk (`*emphasis*`). """ 

112 

113STRONG_RE = r'(\*{2})(.+?)\1' 

114""" Match strong with an asterisk (`**strong**`). """ 

115 

116SMART_STRONG_RE = r'(?<!\w)(_{2})(?!_)(.+?)(?<!_)\1(?!\w)' 

117""" Match strong with underscore while ignoring middle word underscores (`__smart__strong__`). """ 

118 

119SMART_EMPHASIS_RE = r'(?<!\w)(_)(?!_)(.+?)(?<!_)\1(?!\w)' 

120""" Match emphasis with underscore while ignoring middle word underscores (`_smart_emphasis_`). """ 

121 

122SMART_STRONG_EM_RE = r'(?<!\w)(\_)\1(?!\1)(.+?)(?<!\w)\1(?!\1)(.+?)\1{3}(?!\w)' 

123""" Match strong emphasis with underscores (`__strong _em__`). """ 

124 

125EM_STRONG_RE = r'(\*)\1{2}(.+?)\1(.*?)\1{2}' 

126""" Match emphasis strong with asterisk (`***strongem***` or `***em*strong**`). """ 

127 

128EM_STRONG2_RE = r'(_)\1{2}(.+?)\1(.*?)\1{2}' 

129""" Match emphasis strong with underscores (`___emstrong___` or `___em_strong__`). """ 

130 

131STRONG_EM_RE = r'(\*)\1{2}(.+?)\1{2}(.*?)\1' 

132""" Match strong emphasis with asterisk (`***strong**em*`). """ 

133 

134STRONG_EM2_RE = r'(_)\1{2}(.+?)\1{2}(.*?)\1' 

135""" Match strong emphasis with underscores (`___strong__em_`). """ 

136 

137STRONG_EM3_RE = r'(\*)\1(?!\1)([^*]+?)\1(?!\1)(.+?)\1{3}' 

138""" Match strong emphasis with asterisk (`**strong*em***`). """ 

139 

140LINK_RE = NOIMG + r'\[' 

141""" Match start of in-line link (`[text](url)` or `[text](<url>)` or `[text](url "title")`). """ 

142 

143IMAGE_LINK_RE = r'\!\[' 

144""" Match start of in-line image link (`![alttxt](url)` or `![alttxt](<url>)`). """ 

145 

146REFERENCE_RE = LINK_RE 

147""" Match start of reference link (`[Label][3]`). """ 

148 

149IMAGE_REFERENCE_RE = IMAGE_LINK_RE 

150""" Match start of image reference (`![alt text][2]`). """ 

151 

152NOT_STRONG_RE = r'((^|(?<=\s))(\*{1,3}|_{1,3})(?=\s|$))' 

153""" Match a stand-alone `*` or `_`. """ 

154 

155AUTOLINK_RE = r'<((?:[Ff]|[Hh][Tt])[Tt][Pp][Ss]?://[^<>]*)>' 

156""" Match an automatic link (`<http://www.example.com>`). """ 

157 

158AUTOMAIL_RE = r'<([^<> !]+@[^@<> ]+)>' 

159""" Match an automatic email link (`<me@example.com>`). """ 

160 

161HTML_RE = ( 

162 r'(<(\/?[a-zA-Z][^<>@ ]*( [^<>]*)?|' # Tag 

163 r'!--(?:(?!<!--|-->).)*--|' # Comment 

164 r'[?](?:(?!<[?]|[?]>).)*[?]|' # Processing instruction 

165 r'!\[CDATA\[(?:(?!<!\[CDATA\[|\]\]>).)*\]\]' # `CDATA` 

166 ')>)' 

167) 

168""" Match an HTML tag (`<...>`). """ 

169 

170ENTITY_RE = r'(&(?:\#[0-9]+|\#x[0-9a-fA-F]+|[a-zA-Z0-9]+);)' 

171""" Match an HTML entity (`&#38;` (decimal) or `&#x26;` (hex) or `&amp;` (named)). """ 

172 

173LINE_BREAK_RE = r' \n' 

174""" Match two spaces at end of line. """ 

175 

176 

177def dequote(string: str) -> str: 

178 """Remove quotes from around a string.""" 

179 if ((string.startswith('"') and string.endswith('"')) or 

180 (string.startswith("'") and string.endswith("'"))): 

181 return string[1:-1] 

182 else: 

183 return string 

184 

185 

186class EmStrongItem(NamedTuple): 

187 """Emphasis/strong pattern item.""" 

188 pattern: re.Pattern[str] 

189 builder: str 

190 tags: str 

191 

192 

193# The pattern classes 

194# ----------------------------------------------------------------------------- 

195 

196 

197class Pattern: # pragma: no cover 

198 """ 

199 Base class that inline patterns subclass. 

200 

201 Inline patterns are handled by means of `Pattern` subclasses, one per regular expression. 

202 Each pattern object uses a single regular expression and must support the following methods: 

203 [`getCompiledRegExp`][markdown.inlinepatterns.Pattern.getCompiledRegExp] and 

204 [`handleMatch`][markdown.inlinepatterns.Pattern.handleMatch]. 

205 

206 All the regular expressions used by `Pattern` subclasses must capture the whole block. For this 

207 reason, they all start with `^(.*)` and end with `(.*)!`. When passing a regular expression on 

208 class initialization, the `^(.*)` and `(.*)!` are added automatically and the regular expression 

209 is pre-compiled. 

210 

211 It is strongly suggested that the newer style [`markdown.inlinepatterns.InlineProcessor`][] that 

212 use a more efficient and flexible search approach be used instead. However, the older style 

213 `Pattern` remains for backward compatibility with many existing third-party extensions. 

214 

215 """ 

216 

217 ANCESTOR_EXCLUDES: Collection[str] = tuple() 

218 """ 

219 A collection of elements which are undesirable ancestors. The processor will be skipped if it 

220 would cause the content to be a descendant of one of the listed tag names. 

221 """ 

222 

223 compiled_re: re.Pattern[str] 

224 md: Markdown | None 

225 

226 def __init__(self, pattern: str, md: Markdown | None = None): 

227 """ 

228 Create an instant of an inline pattern. 

229 

230 Arguments: 

231 pattern: A regular expression that matches a pattern. 

232 md: An optional pointer to the instance of `markdown.Markdown` and is available as 

233 `self.md` on the class instance. 

234 

235 

236 """ 

237 self.pattern = pattern 

238 self.compiled_re = re.compile(r"^(.*?)%s(.*)$" % pattern, 

239 re.DOTALL | re.UNICODE) 

240 

241 self.md = md 

242 

243 def getCompiledRegExp(self) -> re.Pattern: 

244 """ Return a compiled regular expression. """ 

245 return self.compiled_re 

246 

247 def handleMatch(self, m: re.Match[str]) -> etree.Element | str: 

248 """Return a ElementTree element from the given match. 

249 

250 Subclasses should override this method. 

251 

252 Arguments: 

253 m: A match object containing a match of the pattern. 

254 

255 Returns: An ElementTree Element object. 

256 

257 """ 

258 pass # pragma: no cover 

259 

260 def type(self) -> str: 

261 """ Return class name, to define pattern type """ 

262 return self.__class__.__name__ 

263 

264 def unescape(self, text: str) -> str: 

265 """ Return unescaped text given text with an inline placeholder. """ 

266 try: 

267 stash = self.md.treeprocessors['inline'].stashed_nodes 

268 except KeyError: # pragma: no cover 

269 return text 

270 

271 def get_stash(m): 

272 id = m.group(1) 

273 if id in stash: 

274 value = stash.get(id) 

275 if isinstance(value, str): 

276 return value 

277 else: 

278 # An `etree` Element - return text content only 

279 return ''.join(value.itertext()) 

280 return util.INLINE_PLACEHOLDER_RE.sub(get_stash, text) 

281 

282 

283class InlineProcessor(Pattern): 

284 """ 

285 Base class that inline processors subclass. 

286 

287 This is the newer style inline processor that uses a more 

288 efficient and flexible search approach. 

289 

290 """ 

291 

292 def __init__(self, pattern: str, md: Markdown | None = None): 

293 """ 

294 Create an instant of an inline processor. 

295 

296 Arguments: 

297 pattern: A regular expression that matches a pattern. 

298 md: An optional pointer to the instance of `markdown.Markdown` and is available as 

299 `self.md` on the class instance. 

300 

301 """ 

302 self.pattern = pattern 

303 self.compiled_re = re.compile(pattern, re.DOTALL | re.UNICODE) 

304 

305 # API for Markdown to pass `safe_mode` into instance 

306 self.safe_mode = False 

307 self.md = md 

308 

309 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | str | None, int | None, int | None]: 

310 """Return a ElementTree element from the given match and the 

311 start and end index of the matched text. 

312 

313 If `start` and/or `end` are returned as `None`, it will be 

314 assumed that the processor did not find a valid region of text. 

315 

316 Subclasses should override this method. 

317 

318 Arguments: 

319 m: A re match object containing a match of the pattern. 

320 data: The buffer currently under analysis. 

321 

322 Returns: 

323 el: The ElementTree element, text or None. 

324 start: The start of the region that has been matched or None. 

325 end: The end of the region that has been matched or None. 

326 

327 """ 

328 pass # pragma: no cover 

329 

330 

331class SimpleTextPattern(Pattern): # pragma: no cover 

332 """ Return a simple text of `group(2)` of a Pattern. """ 

333 def handleMatch(self, m: re.Match[str]) -> str: 

334 """ Return string content of `group(2)` of a matching pattern. """ 

335 return m.group(2) 

336 

337 

338class SimpleTextInlineProcessor(InlineProcessor): 

339 """ Return a simple text of `group(1)` of a Pattern. """ 

340 def handleMatch(self, m: re.Match[str], data: str) -> tuple[str, int, int]: 

341 """ Return string content of `group(1)` of a matching pattern. """ 

342 return m.group(1), m.start(0), m.end(0) 

343 

344 

345class EscapeInlineProcessor(InlineProcessor): 

346 """ Return an escaped character. """ 

347 

348 def handleMatch(self, m: re.Match[str], data: str) -> tuple[str | None, int, int]: 

349 """ 

350 If the character matched by `group(1)` of a pattern is in [`ESCAPED_CHARS`][markdown.Markdown.ESCAPED_CHARS] 

351 then return the integer representing the character's Unicode code point (as returned by [`ord`][]) wrapped 

352 in [`util.STX`][markdown.util.STX] and [`util.ETX`][markdown.util.ETX]. 

353 

354 If the matched character is not in [`ESCAPED_CHARS`][markdown.Markdown.ESCAPED_CHARS], then return `None`. 

355 """ 

356 

357 char = m.group(1) 

358 if char in self.md.ESCAPED_CHARS: 

359 return '{}{}{}'.format(util.STX, ord(char), util.ETX), m.start(0), m.end(0) 

360 else: 

361 return None, m.start(0), m.end(0) 

362 

363 

364class SimpleTagPattern(Pattern): # pragma: no cover 

365 """ 

366 Return element of type `tag` with a text attribute of `group(3)` 

367 of a Pattern. 

368 

369 """ 

370 def __init__(self, pattern: str, tag: str): 

371 """ 

372 Create an instant of an simple tag pattern. 

373 

374 Arguments: 

375 pattern: A regular expression that matches a pattern. 

376 tag: Tag of element. 

377 

378 """ 

379 Pattern.__init__(self, pattern) 

380 self.tag = tag 

381 """ The tag of the rendered element. """ 

382 

383 def handleMatch(self, m: re.Match[str]) -> etree.Element: 

384 """ 

385 Return [`Element`][xml.etree.ElementTree.Element] of type `tag` with the string in `group(3)` of a 

386 matching pattern as the Element's text. 

387 """ 

388 el = etree.Element(self.tag) 

389 el.text = m.group(3) 

390 return el 

391 

392 

393class SimpleTagInlineProcessor(InlineProcessor): 

394 """ 

395 Return element of type `tag` with a text attribute of `group(2)` 

396 of a Pattern. 

397 

398 """ 

399 def __init__(self, pattern: str, tag: str): 

400 """ 

401 Create an instant of an simple tag processor. 

402 

403 Arguments: 

404 pattern: A regular expression that matches a pattern. 

405 tag: Tag of element. 

406 

407 """ 

408 InlineProcessor.__init__(self, pattern) 

409 self.tag = tag 

410 """ The tag of the rendered element. """ 

411 

412 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: # pragma: no cover 

413 """ 

414 Return [`Element`][xml.etree.ElementTree.Element] of type `tag` with the string in `group(2)` of a 

415 matching pattern as the Element's text. 

416 """ 

417 el = etree.Element(self.tag) 

418 el.text = m.group(2) 

419 return el, m.start(0), m.end(0) 

420 

421 

422class SubstituteTagPattern(SimpleTagPattern): # pragma: no cover 

423 """ Return an element of type `tag` with no children. """ 

424 def handleMatch(self, m: re.Match[str]) -> etree.Element: 

425 """ Return empty [`Element`][xml.etree.ElementTree.Element] of type `tag`. """ 

426 return etree.Element(self.tag) 

427 

428 

429class SubstituteTagInlineProcessor(SimpleTagInlineProcessor): 

430 """ Return an element of type `tag` with no children. """ 

431 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: 

432 """ Return empty [`Element`][xml.etree.ElementTree.Element] of type `tag`. """ 

433 return etree.Element(self.tag), m.start(0), m.end(0) 

434 

435 

436class BacktickInlineProcessor(InlineProcessor): 

437 """ Return a `<code>` element containing the escaped matching text. """ 

438 

439 def __init__(self, pattern: str): 

440 InlineProcessor.__init__(self, pattern) 

441 self.ESCAPED_BSLASH = '{}{}{}'.format(util.STX, ord('\\'), util.ETX) 

442 self.tag = 'code' 

443 """ The tag of the rendered element. """ 

444 

445 def find_code_spans(self, start: int, text: str) -> tuple[int, int] | None: 

446 """Find code spans.""" 

447 

448 last = len(text) 

449 

450 # Get the maximum starting ticks 

451 max_ticks = 0 

452 while start < last and text[start] == '`': 

453 max_ticks += 1 

454 start += 1 

455 

456 if not max_ticks: # pragma: no cover 

457 # This is not ever expected to happen. 

458 return None 

459 

460 longest_span = 0 

461 end = 0 

462 

463 # Find an ending span of backticks that matches our opening 

464 i = start 

465 while i < last: 

466 span_length = 0 

467 while i < last and text[i] == '`': 

468 span_length += 1 

469 i += 1 

470 if not span_length: 

471 i += 1 

472 continue 

473 

474 # Did we find the end? 

475 if max_ticks == span_length: 

476 return start, i - span_length 

477 

478 # Track the longest span of backticks we find as a fallback. 

479 if span_length > longest_span: 

480 longest_span = span_length 

481 end = i 

482 

483 # Since we didn't find an exact matching start and end, 

484 # adjust start to match the largest end we could calculate. 

485 if longest_span: 

486 return start - (max_ticks - longest_span), end - longest_span 

487 

488 # We could not find a suitable pairing 

489 return None 

490 

491 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | str | None, int | None, int | None]: 

492 """ 

493 If the match contains `group(3)` of a pattern, then return a `code` 

494 [`Element`][xml.etree.ElementTree.Element] which contains HTML escaped text (with 

495 [`code_escape`][markdown.util.code_escape]) as an [`AtomicString`][markdown.util.AtomicString]. 

496 

497 If the match contains `group(1)` then return the text of `group(1)` as backslash escaped. 

498 

499 """ 

500 if m.group(1): 

501 return m.group(1).replace('\\\\', self.ESCAPED_BSLASH), m.start(0), m.end(0) 

502 

503 begin = m.start(0) 

504 result = self.find_code_spans(begin, data) 

505 if result is not None: 

506 start, end = result 

507 el = etree.Element(self.tag) 

508 el.text = util.AtomicString(util.code_escape(data[start:end].strip())) 

509 return el, begin, result[1] + (start - begin) 

510 return None, None, None 

511 

512 

513class DoubleTagPattern(SimpleTagPattern): # pragma: no cover 

514 """Return a ElementTree element nested in tag2 nested in tag1. 

515 

516 Useful for strong emphasis etc. 

517 

518 """ 

519 def handleMatch(self, m: re.Match[str]) -> etree.Element: 

520 """ 

521 Return [`Element`][xml.etree.ElementTree.Element] in following format: 

522 `<tag1><tag2>group(3)</tag2>group(4)</tag2>` where `group(4)` is optional. 

523 

524 """ 

525 tag1, tag2 = self.tag.split(",") 

526 el1 = etree.Element(tag1) 

527 el2 = etree.SubElement(el1, tag2) 

528 el2.text = m.group(3) 

529 if len(m.groups()) == 5: 

530 el2.tail = m.group(4) 

531 return el1 

532 

533 

534class DoubleTagInlineProcessor(SimpleTagInlineProcessor): 

535 """Return a ElementTree element nested in tag2 nested in tag1. 

536 

537 Useful for strong emphasis etc. 

538 

539 """ 

540 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: # pragma: no cover 

541 """ 

542 Return [`Element`][xml.etree.ElementTree.Element] in following format: 

543 `<tag1><tag2>group(2)</tag2>group(3)</tag2>` where `group(3)` is optional. 

544 

545 """ 

546 tag1, tag2 = self.tag.split(",") 

547 el1 = etree.Element(tag1) 

548 el2 = etree.SubElement(el1, tag2) 

549 el2.text = m.group(2) 

550 if len(m.groups()) == 3: 

551 el2.tail = m.group(3) 

552 return el1, m.start(0), m.end(0) 

553 

554 

555class HtmlInlineProcessor(InlineProcessor): 

556 """ Store raw inline html and return a placeholder. """ 

557 def handleMatch(self, m: re.Match[str], data: str) -> tuple[str, int, int]: 

558 """ Store the text of `group(1)` of a pattern and return a placeholder string. """ 

559 rawhtml = self.backslash_unescape(self.unescape(m.group(1))) 

560 place_holder = self.md.htmlStash.store(rawhtml) 

561 return place_holder, m.start(0), m.end(0) 

562 

563 def unescape(self, text: str) -> str: 

564 """ Return unescaped text given text with an inline placeholder. """ 

565 try: 

566 stash = self.md.treeprocessors['inline'].stashed_nodes 

567 except KeyError: # pragma: no cover 

568 return text 

569 

570 def get_stash(m: re.Match[str]) -> str: 

571 id = m.group(1) 

572 value = stash.get(id) 

573 if value is not None: 

574 try: 

575 # Ensure we don't have a placeholder inside a placeholder 

576 return self.unescape(self.md.serializer(value)) 

577 except Exception: 

578 return r'\%s' % value 

579 

580 return util.INLINE_PLACEHOLDER_RE.sub(get_stash, text) 

581 

582 def backslash_unescape(self, text: str) -> str: 

583 """ Return text with backslash escapes undone (backslashes are restored). """ 

584 try: 

585 RE = self.md.treeprocessors['unescape'].RE 

586 except KeyError: # pragma: no cover 

587 return text 

588 

589 def _unescape(m: re.Match[str]) -> str: 

590 return chr(int(m.group(1))) 

591 

592 return RE.sub(_unescape, text) 

593 

594 

595class AsteriskProcessor(InlineProcessor): 

596 """Emphasis processor for handling strong and em matches inside asterisks.""" 

597 

598 PATTERNS = [ 

599 EmStrongItem(re.compile(EM_STRONG_RE, re.DOTALL | re.UNICODE), 'double', 'strong,em'), 

600 EmStrongItem(re.compile(STRONG_EM_RE, re.DOTALL | re.UNICODE), 'double', 'em,strong'), 

601 EmStrongItem(re.compile(STRONG_EM3_RE, re.DOTALL | re.UNICODE), 'double2', 'strong,em'), 

602 EmStrongItem(re.compile(STRONG_RE, re.DOTALL | re.UNICODE), 'single', 'strong'), 

603 EmStrongItem(re.compile(EMPHASIS_RE, re.DOTALL | re.UNICODE), 'single', 'em') 

604 ] 

605 """ The various strong and emphasis patterns handled by this processor. """ 

606 

607 def build_single(self, m: re.Match[str], tag: str, idx: int) -> etree.Element: 

608 """Return single tag.""" 

609 el1 = etree.Element(tag) 

610 text = m.group(2) 

611 self.parse_sub_patterns(text, el1, None, idx) 

612 return el1 

613 

614 def build_double(self, m: re.Match[str], tags: str, idx: int) -> etree.Element: 

615 """Return double tag.""" 

616 

617 tag1, tag2 = tags.split(",") 

618 el1 = etree.Element(tag1) 

619 el2 = etree.Element(tag2) 

620 text = m.group(2) 

621 self.parse_sub_patterns(text, el2, None, idx) 

622 el1.append(el2) 

623 if len(m.groups()) == 3: 

624 text = m.group(3) 

625 self.parse_sub_patterns(text, el1, el2, idx) 

626 return el1 

627 

628 def build_double2(self, m: re.Match[str], tags: str, idx: int) -> etree.Element: 

629 """Return double tags (variant 2): `<strong>text <em>text</em></strong>`.""" 

630 

631 tag1, tag2 = tags.split(",") 

632 el1 = etree.Element(tag1) 

633 el2 = etree.Element(tag2) 

634 text = m.group(2) 

635 self.parse_sub_patterns(text, el1, None, idx) 

636 text = m.group(3) 

637 el1.append(el2) 

638 self.parse_sub_patterns(text, el2, None, idx) 

639 return el1 

640 

641 def parse_sub_patterns( 

642 self, data: str, parent: etree.Element, last: etree.Element | None, idx: int 

643 ) -> None: 

644 """ 

645 Parses sub patterns. 

646 

647 `data`: text to evaluate. 

648 

649 `parent`: Parent to attach text and sub elements to. 

650 

651 `last`: Last appended child to parent. Can also be None if parent has no children. 

652 

653 `idx`: Current pattern index that was used to evaluate the parent. 

654 """ 

655 

656 offset = 0 

657 pos = 0 

658 

659 length = len(data) 

660 while pos < length: 

661 # Find the start of potential emphasis or strong tokens 

662 if self.compiled_re.match(data, pos): 

663 matched = False 

664 # See if the we can match an emphasis/strong pattern 

665 for index, item in enumerate(self.PATTERNS): 

666 # Only evaluate patterns that are after what was used on the parent 

667 if index <= idx: 

668 continue 

669 m = item.pattern.match(data, pos) 

670 if m: 

671 # Append child nodes to parent 

672 # Text nodes should be appended to the last 

673 # child if present, and if not, it should 

674 # be added as the parent's text node. 

675 text = data[offset:m.start(0)] 

676 if text: 

677 if last is not None: 

678 last.tail = text 

679 else: 

680 parent.text = text 

681 el = self.build_element(m, item.builder, item.tags, index) 

682 parent.append(el) 

683 last = el 

684 # Move our position past the matched hunk 

685 offset = pos = m.end(0) 

686 matched = True 

687 if not matched: 

688 # We matched nothing, move on to the next character 

689 pos += 1 

690 else: 

691 # Increment position as no potential emphasis start was found. 

692 pos += 1 

693 

694 # Append any leftover text as a text node. 

695 text = data[offset:] 

696 if text: 

697 if last is not None: 

698 last.tail = text 

699 else: 

700 parent.text = text 

701 

702 def build_element(self, m: re.Match[str], builder: str, tags: str, index: int) -> etree.Element: 

703 """Element builder.""" 

704 

705 if builder == 'double2': 

706 return self.build_double2(m, tags, index) 

707 elif builder == 'double': 

708 return self.build_double(m, tags, index) 

709 else: 

710 return self.build_single(m, tags, index) 

711 

712 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]: 

713 """Parse patterns.""" 

714 

715 el = None 

716 start = None 

717 end = None 

718 

719 for index, item in enumerate(self.PATTERNS): 

720 m1 = item.pattern.match(data, m.start(0)) 

721 if m1: 

722 start = m1.start(0) 

723 end = m1.end(0) 

724 el = self.build_element(m1, item.builder, item.tags, index) 

725 break 

726 return el, start, end 

727 

728 

729class UnderscoreProcessor(AsteriskProcessor): 

730 """Emphasis processor for handling strong and em matches inside underscores.""" 

731 

732 PATTERNS = [ 

733 EmStrongItem(re.compile(EM_STRONG2_RE, re.DOTALL | re.UNICODE), 'double', 'strong,em'), 

734 EmStrongItem(re.compile(STRONG_EM2_RE, re.DOTALL | re.UNICODE), 'double', 'em,strong'), 

735 EmStrongItem(re.compile(SMART_STRONG_EM_RE, re.DOTALL | re.UNICODE), 'double2', 'strong,em'), 

736 EmStrongItem(re.compile(SMART_STRONG_RE, re.DOTALL | re.UNICODE), 'single', 'strong'), 

737 EmStrongItem(re.compile(SMART_EMPHASIS_RE, re.DOTALL | re.UNICODE), 'single', 'em') 

738 ] 

739 """ The various strong and emphasis patterns handled by this processor. """ 

740 

741 

742class LinkInlineProcessor(InlineProcessor): 

743 """ Return a link element from the given match. """ 

744 RE_LINK = re.compile(r'''\(\s*(?:(<[^<>]*>)\s*(?:('[^']*'|"[^"]*")\s*)?\))?''', re.DOTALL | re.UNICODE) 

745 RE_TITLE_CLEAN = re.compile(r'\s') 

746 

747 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]: 

748 """ Return an `a` [`Element`][xml.etree.ElementTree.Element] or `(None, None, None)`. """ 

749 text, index, handled = self.getText(data, m.end(0)) 

750 

751 if not handled: 

752 return None, None, None 

753 

754 href, title, index, handled = self.getLink(data, index) 

755 if not handled: 

756 return None, None, None 

757 

758 el = etree.Element("a") 

759 el.text = text 

760 

761 el.set("href", href) 

762 

763 if title is not None: 

764 el.set("title", title) 

765 

766 return el, m.start(0), index 

767 

768 def getLink(self, data: str, index: int) -> tuple[str, str | None, int, bool]: 

769 """Parse data between `()` of `[Text]()` allowing recursive `()`. """ 

770 

771 href = '' 

772 title: str | None = None 

773 handled = False 

774 

775 m = self.RE_LINK.match(data, pos=index) 

776 if m and m.group(1): 

777 # Matches [Text](<link> "title") 

778 href = m.group(1)[1:-1].strip() 

779 if m.group(2): 

780 title = m.group(2)[1:-1] 

781 index = m.end(0) 

782 handled = True 

783 elif m: 

784 # Track bracket nesting and index in string 

785 bracket_count = 1 

786 backtrack_count = 1 

787 start_index = m.end() 

788 index = start_index 

789 last_bracket = -1 

790 

791 # Primary (first found) quote tracking. 

792 quote: str | None = None 

793 start_quote = -1 

794 exit_quote = -1 

795 ignore_matches = False 

796 

797 # Secondary (second found) quote tracking. 

798 alt_quote = None 

799 start_alt_quote = -1 

800 exit_alt_quote = -1 

801 

802 # Track last character 

803 last = '' 

804 

805 for pos in range(index, len(data)): 

806 c = data[pos] 

807 if c == '(': 

808 # Count nested ( 

809 # Don't increment the bracket count if we are sure we're in a title. 

810 if not ignore_matches: 

811 bracket_count += 1 

812 elif backtrack_count > 0: 

813 backtrack_count -= 1 

814 elif c == ')': 

815 # Match nested ) to ( 

816 # Don't decrement if we are sure we are in a title that is unclosed. 

817 if ((exit_quote != -1 and quote == last) or (exit_alt_quote != -1 and alt_quote == last)): 

818 bracket_count = 0 

819 elif not ignore_matches: 

820 bracket_count -= 1 

821 elif backtrack_count > 0: 

822 backtrack_count -= 1 

823 # We've found our backup end location if the title doesn't resolve. 

824 if backtrack_count == 0: 

825 last_bracket = index + 1 

826 

827 elif c in ("'", '"'): 

828 # Quote has started 

829 if not quote: 

830 # We'll assume we are now in a title. 

831 # Brackets are quoted, so no need to match them (except for the final one). 

832 ignore_matches = True 

833 backtrack_count = bracket_count 

834 bracket_count = 1 

835 start_quote = index + 1 

836 quote = c 

837 # Secondary quote (in case the first doesn't resolve): [text](link'"title") 

838 elif c != quote and not alt_quote: 

839 start_alt_quote = index + 1 

840 alt_quote = c 

841 # Update primary quote match 

842 elif c == quote: 

843 exit_quote = index + 1 

844 # Update secondary quote match 

845 elif alt_quote and c == alt_quote: 

846 exit_alt_quote = index + 1 

847 

848 index += 1 

849 

850 # Link is closed, so let's break out of the loop 

851 if bracket_count == 0: 

852 # Get the title if we closed a title string right before link closed 

853 if exit_quote >= 0 and quote == last: 

854 href = data[start_index:start_quote - 1] 

855 title = ''.join(data[start_quote:exit_quote - 1]) 

856 elif exit_alt_quote >= 0 and alt_quote == last: 

857 href = data[start_index:start_alt_quote - 1] 

858 title = ''.join(data[start_alt_quote:exit_alt_quote - 1]) 

859 else: 

860 href = data[start_index:index - 1] 

861 break 

862 

863 if c != ' ': 

864 last = c 

865 

866 # We have a scenario: `[test](link"notitle)` 

867 # When we enter a string, we stop tracking bracket resolution in the main counter, 

868 # but we do keep a backup counter up until we discover where we might resolve all brackets 

869 # if the title string fails to resolve. 

870 if bracket_count != 0 and backtrack_count == 0: 

871 href = data[start_index:last_bracket - 1] 

872 index = last_bracket 

873 bracket_count = 0 

874 

875 handled = bracket_count == 0 

876 

877 if title is not None: 

878 title = self.RE_TITLE_CLEAN.sub(' ', dequote(self.unescape(title.strip()))) 

879 

880 href = self.unescape(href).strip() 

881 

882 return href, title, index, handled 

883 

884 def getText(self, data: str, index: int) -> tuple[str, int, bool]: 

885 """Parse the content between `[]` of the start of an image or link 

886 resolving nested square brackets. 

887 

888 """ 

889 bracket_count = 1 

890 text = [] 

891 for pos in range(index, len(data)): 

892 c = data[pos] 

893 if c == ']': 

894 bracket_count -= 1 

895 elif c == '[': 

896 bracket_count += 1 

897 index += 1 

898 if bracket_count == 0: 

899 break 

900 text.append(c) 

901 return ''.join(text), index, bracket_count == 0 

902 

903 

904class ImageInlineProcessor(LinkInlineProcessor): 

905 """ Return a `img` element from the given match. """ 

906 

907 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]: 

908 """ Return an `img` [`Element`][xml.etree.ElementTree.Element] or `(None, None, None)`. """ 

909 text, index, handled = self.getText(data, m.end(0)) 

910 if not handled: 

911 return None, None, None 

912 

913 src, title, index, handled = self.getLink(data, index) 

914 if not handled: 

915 return None, None, None 

916 

917 el = etree.Element("img") 

918 

919 el.set("src", src) 

920 

921 if title is not None: 

922 el.set("title", title) 

923 

924 el.set('alt', self.unescape(text)) 

925 return el, m.start(0), index 

926 

927 

928class ReferenceInlineProcessor(LinkInlineProcessor): 

929 """ Match to a stored reference and return link element. """ 

930 NEWLINE_CLEANUP_RE = re.compile(r'\s+', re.MULTILINE) 

931 

932 RE_LINK = re.compile(r'\s?\[([^\]]*)\]', re.DOTALL | re.UNICODE) 

933 

934 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]: 

935 """ 

936 Return [`Element`][xml.etree.ElementTree.Element] returned by `makeTag` method or `(None, None, None)`. 

937 

938 """ 

939 text, index, handled = self.getText(data, m.end(0)) 

940 if not handled: 

941 return None, None, None 

942 

943 id, end, handled = self.evalId(data, index, text) 

944 if not handled: 

945 return None, None, None 

946 

947 # Clean up line breaks in id 

948 id = self.NEWLINE_CLEANUP_RE.sub(' ', id) 

949 if id not in self.md.references: # ignore undefined refs 

950 return None, m.start(0), end 

951 

952 href, title = self.md.references[id] 

953 

954 return self.makeTag(href, title, text), m.start(0), end 

955 

956 def evalId(self, data: str, index: int, text: str) -> tuple[str | None, int, bool]: 

957 """ 

958 Evaluate the id portion of `[ref][id]`. 

959 

960 If `[ref][]` use `[ref]`. 

961 """ 

962 m = self.RE_LINK.match(data, pos=index) 

963 if not m: 

964 return None, index, False 

965 else: 

966 id = m.group(1).lower() 

967 end = m.end(0) 

968 if not id: 

969 id = text.lower() 

970 return id, end, True 

971 

972 def makeTag(self, href: str, title: str, text: str) -> etree.Element: 

973 """ Return an `a` [`Element`][xml.etree.ElementTree.Element]. """ 

974 el = etree.Element('a') 

975 

976 el.set('href', href) 

977 if title: 

978 el.set('title', title) 

979 

980 el.text = text 

981 return el 

982 

983 

984class ShortReferenceInlineProcessor(ReferenceInlineProcessor): 

985 """Short form of reference: `[google]`. """ 

986 def evalId(self, data: str, index: int, text: str) -> tuple[str, int, bool]: 

987 """Evaluate the id of `[ref]`. """ 

988 

989 return text.lower(), index, True 

990 

991 

992class ImageReferenceInlineProcessor(ReferenceInlineProcessor): 

993 """ Match to a stored reference and return `img` element. """ 

994 def makeTag(self, href: str, title: str, text: str) -> etree.Element: 

995 """ Return an `img` [`Element`][xml.etree.ElementTree.Element]. """ 

996 el = etree.Element("img") 

997 el.set("src", href) 

998 if title: 

999 el.set("title", title) 

1000 el.set("alt", self.unescape(text)) 

1001 return el 

1002 

1003 

1004class ShortImageReferenceInlineProcessor(ImageReferenceInlineProcessor): 

1005 """ Short form of image reference: `![ref]`. """ 

1006 def evalId(self, data: str, index: int, text: str) -> tuple[str, int, bool]: 

1007 """Evaluate the id of `[ref]`. """ 

1008 

1009 return text.lower(), index, True 

1010 

1011 

1012class AutolinkInlineProcessor(InlineProcessor): 

1013 """ Return a link Element given an auto-link (`<http://example/com>`). """ 

1014 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: 

1015 """ Return an `a` [`Element`][xml.etree.ElementTree.Element] of `group(1)`. """ 

1016 el = etree.Element("a") 

1017 el.set('href', self.unescape(m.group(1))) 

1018 el.text = util.AtomicString(m.group(1)) 

1019 return el, m.start(0), m.end(0) 

1020 

1021 

1022class AutomailInlineProcessor(InlineProcessor): 

1023 """ 

1024 Return a `mailto` link Element given an auto-mail link (`<foo@example.com>`). 

1025 """ 

1026 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: 

1027 """ Return an [`Element`][xml.etree.ElementTree.Element] containing a `mailto` link of `group(1)`. """ 

1028 el = etree.Element('a') 

1029 email = self.unescape(m.group(1)) 

1030 if email.startswith("mailto:"): 

1031 email = email[len("mailto:"):] 

1032 

1033 def codepoint2name(code: int) -> str: 

1034 """Return entity definition by code, or the code if not defined.""" 

1035 entity = entities.codepoint2name.get(code) 

1036 if entity: 

1037 return "{}{};".format(util.AMP_SUBSTITUTE, entity) 

1038 else: 

1039 return "%s#%d;" % (util.AMP_SUBSTITUTE, code) 

1040 

1041 letters = [codepoint2name(ord(letter)) for letter in email] 

1042 el.text = util.AtomicString(''.join(letters)) 

1043 

1044 mailto = "mailto:" + email 

1045 mailto = "".join([util.AMP_SUBSTITUTE + '#%d;' % 

1046 ord(letter) for letter in mailto]) 

1047 el.set('href', mailto) 

1048 return el, m.start(0), m.end(0)