1# Python Markdown
2
3# A Python implementation of John Gruber's Markdown.
4
5# Documentation: https://python-markdown.github.io/
6# GitHub: https://github.com/Python-Markdown/markdown/
7# PyPI: https://pypi.org/project/Markdown/
8
9# Started by Manfred Stienstra (http://www.dwerg.net/).
10# Maintained for a few years by Yuri Takhteyev (http://www.freewisdom.org).
11# Currently maintained by Waylan Limberg (https://github.com/waylan),
12# Dmitry Shachnev (https://github.com/mitya57) and Isaac Muse (https://github.com/facelessuser).
13
14# Copyright 2007-2023 The Python Markdown Project (v. 1.7 and later)
15# Copyright 2004, 2005, 2006 Yuri Takhteyev (v. 0.2-1.6b)
16# Copyright 2004 Manfred Stienstra (the original version)
17
18# License: BSD (see LICENSE.md for details).
19
20"""
21In version 3.0, a new, more flexible inline processor was added, [`markdown.inlinepatterns.InlineProcessor`][]. The
22original inline patterns, which inherit from [`markdown.inlinepatterns.Pattern`][] or one of its children are still
23supported, though users are encouraged to migrate.
24
25The new `InlineProcessor` provides two major enhancements to `Patterns`:
26
271. Inline Processors no longer need to match the entire block, so regular expressions no longer need to start with
28 `r'^(.*?)'` and end with `r'(.*?)%'`. This runs faster. The returned [`Match`][re.Match] object will only contain
29 what is explicitly matched in the pattern, and extension pattern groups now start with `m.group(1)`.
30
312. The `handleMatch` method now takes an additional input called `data`, which is the entire block under analysis,
32 not just what is matched with the specified pattern. The method now returns the element *and* the indexes relative
33 to `data` that the return element is replacing (usually `m.start(0)` and `m.end(0)`). If the boundaries are
34 returned as `None`, it is assumed that the match did not take place, and nothing will be altered in `data`.
35
36 This allows handling of more complex constructs than regular expressions can handle, e.g., matching nested
37 brackets, and explicit control of the span "consumed" by the processor.
38
39"""
40
41from __future__ import annotations
42
43from . import util
44from typing import TYPE_CHECKING, Any, Collection, NamedTuple
45import re
46import xml.etree.ElementTree as etree
47from html import entities
48
49if TYPE_CHECKING: # pragma: no cover
50 from markdown import Markdown
51
52
53def build_inlinepatterns(md: Markdown, **kwargs: Any) -> util.Registry[InlineProcessor]:
54 """
55 Build the default set of inline patterns for Markdown.
56
57 The order in which processors and/or patterns are applied is very important - e.g. if we first replace
58 `http://.../` links with `<a>` tags and _then_ try to replace inline HTML, we would end up with a mess. So, we
59 apply the expressions in the following order:
60
61 * backticks and escaped characters have to be handled before everything else so that we can preempt any markdown
62 patterns by escaping them;
63
64 * then we handle the various types of links (auto-links must be handled before inline HTML);
65
66 * then we handle inline HTML. At this point we will simply replace all inline HTML strings with a placeholder
67 and add the actual HTML to a stash;
68
69 * finally we apply strong, emphasis, etc.
70
71 """
72 inlinePatterns = util.Registry()
73 inlinePatterns.register(BacktickInlineProcessor(BACKTICK_RE), 'backtick', 190)
74 inlinePatterns.register(EscapeInlineProcessor(ESCAPE_RE, md), 'escape', 180)
75 inlinePatterns.register(ReferenceInlineProcessor(REFERENCE_RE, md), 'reference', 170)
76 inlinePatterns.register(LinkInlineProcessor(LINK_RE, md), 'link', 160)
77 inlinePatterns.register(ImageInlineProcessor(IMAGE_LINK_RE, md), 'image_link', 150)
78 inlinePatterns.register(
79 ImageReferenceInlineProcessor(IMAGE_REFERENCE_RE, md), 'image_reference', 140
80 )
81 inlinePatterns.register(
82 ShortReferenceInlineProcessor(REFERENCE_RE, md), 'short_reference', 130
83 )
84 inlinePatterns.register(
85 ShortImageReferenceInlineProcessor(IMAGE_REFERENCE_RE, md), 'short_image_ref', 125
86 )
87 inlinePatterns.register(AutolinkInlineProcessor(AUTOLINK_RE, md), 'autolink', 120)
88 inlinePatterns.register(AutomailInlineProcessor(AUTOMAIL_RE, md), 'automail', 110)
89 inlinePatterns.register(SubstituteTagInlineProcessor(LINE_BREAK_RE, 'br'), 'linebreak', 100)
90 inlinePatterns.register(HtmlInlineProcessor(HTML_RE, md), 'html', 90)
91 inlinePatterns.register(HtmlInlineProcessor(ENTITY_RE, md), 'entity', 80)
92 inlinePatterns.register(SimpleTextInlineProcessor(NOT_STRONG_RE), 'not_strong', 70)
93 inlinePatterns.register(AsteriskProcessor(r'\*'), 'em_strong', 60)
94 inlinePatterns.register(UnderscoreProcessor(r'_'), 'em_strong2', 50)
95 return inlinePatterns
96
97
98# The actual regular expressions for patterns
99# -----------------------------------------------------------------------------
100
101NOIMG = r'(?<!\!)'
102""" Match not an image. Partial regular expression which matches if not preceded by `!`. """
103
104BACKTICK_RE = r'(?:(?<!\\)((?:\\{2})+)(?=`+)|(?<!\\)`)'
105""" Match backtick quoted string (`` `e=f()` `` or ``` ``e=f("`")`` ```). """
106
107ESCAPE_RE = r'\\(.)'
108""" Match a backslash escaped character (`\\<` or `\\*`). """
109
110EMPHASIS_RE = r'(\*)([^\*]+)\1'
111""" Match emphasis with an asterisk (`*emphasis*`). """
112
113STRONG_RE = r'(\*{2})(.+?)\1'
114""" Match strong with an asterisk (`**strong**`). """
115
116SMART_STRONG_RE = r'(?<!\w)(_{2})(?!_)(.+?)(?<!_)\1(?!\w)'
117""" Match strong with underscore while ignoring middle word underscores (`__smart__strong__`). """
118
119SMART_EMPHASIS_RE = r'(?<!\w)(_)(?!_)(.+?)(?<!_)\1(?!\w)'
120""" Match emphasis with underscore while ignoring middle word underscores (`_smart_emphasis_`). """
121
122SMART_STRONG_EM_RE = r'(?<!\w)(\_)\1(?!\1)(.+?)(?<!\w)\1(?!\1)(.+?)\1{3}(?!\w)'
123""" Match strong emphasis with underscores (`__strong _em__`). """
124
125EM_STRONG_RE = r'(\*)\1{2}(.+?)\1(.*?)\1{2}'
126""" Match emphasis strong with asterisk (`***strongem***` or `***em*strong**`). """
127
128EM_STRONG2_RE = r'(_)\1{2}(.+?)\1(.*?)\1{2}'
129""" Match emphasis strong with underscores (`___emstrong___` or `___em_strong__`). """
130
131STRONG_EM_RE = r'(\*)\1{2}(.+?)\1{2}(.*?)\1'
132""" Match strong emphasis with asterisk (`***strong**em*`). """
133
134STRONG_EM2_RE = r'(_)\1{2}(.+?)\1{2}(.*?)\1'
135""" Match strong emphasis with underscores (`___strong__em_`). """
136
137STRONG_EM3_RE = r'(\*)\1(?!\1)([^*]+?)\1(?!\1)(.+?)\1{3}'
138""" Match strong emphasis with asterisk (`**strong*em***`). """
139
140LINK_RE = NOIMG + r'\['
141""" Match start of in-line link (`[text](url)` or `[text](<url>)` or `[text](url "title")`). """
142
143IMAGE_LINK_RE = r'\!\['
144""" Match start of in-line image link (`` or ``). """
145
146REFERENCE_RE = LINK_RE
147""" Match start of reference link (`[Label][3]`). """
148
149IMAGE_REFERENCE_RE = IMAGE_LINK_RE
150""" Match start of image reference (`![alt text][2]`). """
151
152NOT_STRONG_RE = r'((^|(?<=\s))(\*{1,3}|_{1,3})(?=\s|$))'
153""" Match a stand-alone `*` or `_`. """
154
155AUTOLINK_RE = r'<((?:[Ff]|[Hh][Tt])[Tt][Pp][Ss]?://[^<>]*)>'
156""" Match an automatic link (`<http://www.example.com>`). """
157
158AUTOMAIL_RE = r'<([^<> !]+@[^@<> ]+)>'
159""" Match an automatic email link (`<me@example.com>`). """
160
161HTML_RE = (
162 r'(<(\/?[a-zA-Z][^<>@ ]*( [^<>]*)?|' # Tag
163 r'!--(?:(?!<!--|-->).)*--|' # Comment
164 r'[?](?:(?!<[?]|[?]>).)*[?]|' # Processing instruction
165 r'!\[CDATA\[(?:(?!<!\[CDATA\[|\]\]>).)*\]\]' # `CDATA`
166 ')>)'
167)
168""" Match an HTML tag (`<...>`). """
169
170ENTITY_RE = r'(&(?:\#[0-9]+|\#x[0-9a-fA-F]+|[a-zA-Z0-9]+);)'
171""" Match an HTML entity (`&` (decimal) or `&` (hex) or `&` (named)). """
172
173LINE_BREAK_RE = r' \n'
174""" Match two spaces at end of line. """
175
176
177def dequote(string: str) -> str:
178 """Remove quotes from around a string."""
179 if ((string.startswith('"') and string.endswith('"')) or
180 (string.startswith("'") and string.endswith("'"))):
181 return string[1:-1]
182 else:
183 return string
184
185
186class EmStrongItem(NamedTuple):
187 """Emphasis/strong pattern item."""
188 pattern: re.Pattern[str]
189 builder: str
190 tags: str
191
192
193# The pattern classes
194# -----------------------------------------------------------------------------
195
196
197class Pattern: # pragma: no cover
198 """
199 Base class that inline patterns subclass.
200
201 Inline patterns are handled by means of `Pattern` subclasses, one per regular expression.
202 Each pattern object uses a single regular expression and must support the following methods:
203 [`getCompiledRegExp`][markdown.inlinepatterns.Pattern.getCompiledRegExp] and
204 [`handleMatch`][markdown.inlinepatterns.Pattern.handleMatch].
205
206 All the regular expressions used by `Pattern` subclasses must capture the whole block. For this
207 reason, they all start with `^(.*)` and end with `(.*)!`. When passing a regular expression on
208 class initialization, the `^(.*)` and `(.*)!` are added automatically and the regular expression
209 is pre-compiled.
210
211 It is strongly suggested that the newer style [`markdown.inlinepatterns.InlineProcessor`][] that
212 use a more efficient and flexible search approach be used instead. However, the older style
213 `Pattern` remains for backward compatibility with many existing third-party extensions.
214
215 """
216
217 ANCESTOR_EXCLUDES: Collection[str] = tuple()
218 """
219 A collection of elements which are undesirable ancestors. The processor will be skipped if it
220 would cause the content to be a descendant of one of the listed tag names.
221 """
222
223 compiled_re: re.Pattern[str]
224 md: Markdown | None
225
226 def __init__(self, pattern: str, md: Markdown | None = None):
227 """
228 Create an instant of an inline pattern.
229
230 Arguments:
231 pattern: A regular expression that matches a pattern.
232 md: An optional pointer to the instance of `markdown.Markdown` and is available as
233 `self.md` on the class instance.
234
235
236 """
237 self.pattern = pattern
238 self.compiled_re = re.compile(r"^(.*?)%s(.*)$" % pattern,
239 re.DOTALL | re.UNICODE)
240
241 self.md = md
242
243 def getCompiledRegExp(self) -> re.Pattern:
244 """ Return a compiled regular expression. """
245 return self.compiled_re
246
247 def handleMatch(self, m: re.Match[str]) -> etree.Element | str:
248 """Return a ElementTree element from the given match.
249
250 Subclasses should override this method.
251
252 Arguments:
253 m: A match object containing a match of the pattern.
254
255 Returns: An ElementTree Element object.
256
257 """
258 pass # pragma: no cover
259
260 def type(self) -> str:
261 """ Return class name, to define pattern type """
262 return self.__class__.__name__
263
264 def unescape(self, text: str) -> str:
265 """ Return unescaped text given text with an inline placeholder. """
266 try:
267 stash = self.md.treeprocessors['inline'].stashed_nodes
268 except KeyError: # pragma: no cover
269 return text
270
271 def get_stash(m):
272 id = m.group(1)
273 if id in stash:
274 value = stash.get(id)
275 if isinstance(value, str):
276 return value
277 else:
278 # An `etree` Element - return text content only
279 return ''.join(value.itertext())
280 return util.INLINE_PLACEHOLDER_RE.sub(get_stash, text)
281
282
283class InlineProcessor(Pattern):
284 """
285 Base class that inline processors subclass.
286
287 This is the newer style inline processor that uses a more
288 efficient and flexible search approach.
289
290 """
291
292 def __init__(self, pattern: str, md: Markdown | None = None):
293 """
294 Create an instant of an inline processor.
295
296 Arguments:
297 pattern: A regular expression that matches a pattern.
298 md: An optional pointer to the instance of `markdown.Markdown` and is available as
299 `self.md` on the class instance.
300
301 """
302 self.pattern = pattern
303 self.compiled_re = re.compile(pattern, re.DOTALL | re.UNICODE)
304
305 # API for Markdown to pass `safe_mode` into instance
306 self.safe_mode = False
307 self.md = md
308
309 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | str | None, int | None, int | None]:
310 """Return a ElementTree element from the given match and the
311 start and end index of the matched text.
312
313 If `start` and/or `end` are returned as `None`, it will be
314 assumed that the processor did not find a valid region of text.
315
316 Subclasses should override this method.
317
318 Arguments:
319 m: A re match object containing a match of the pattern.
320 data: The buffer currently under analysis.
321
322 Returns:
323 el: The ElementTree element, text or None.
324 start: The start of the region that has been matched or None.
325 end: The end of the region that has been matched or None.
326
327 """
328 pass # pragma: no cover
329
330
331class SimpleTextPattern(Pattern): # pragma: no cover
332 """ Return a simple text of `group(2)` of a Pattern. """
333 def handleMatch(self, m: re.Match[str]) -> str:
334 """ Return string content of `group(2)` of a matching pattern. """
335 return m.group(2)
336
337
338class SimpleTextInlineProcessor(InlineProcessor):
339 """ Return a simple text of `group(1)` of a Pattern. """
340 def handleMatch(self, m: re.Match[str], data: str) -> tuple[str, int, int]:
341 """ Return string content of `group(1)` of a matching pattern. """
342 return m.group(1), m.start(0), m.end(0)
343
344
345class EscapeInlineProcessor(InlineProcessor):
346 """ Return an escaped character. """
347
348 def handleMatch(self, m: re.Match[str], data: str) -> tuple[str | None, int, int]:
349 """
350 If the character matched by `group(1)` of a pattern is in [`ESCAPED_CHARS`][markdown.Markdown.ESCAPED_CHARS]
351 then return the integer representing the character's Unicode code point (as returned by [`ord`][]) wrapped
352 in [`util.STX`][markdown.util.STX] and [`util.ETX`][markdown.util.ETX].
353
354 If the matched character is not in [`ESCAPED_CHARS`][markdown.Markdown.ESCAPED_CHARS], then return `None`.
355 """
356
357 char = m.group(1)
358 if char in self.md.ESCAPED_CHARS:
359 return '{}{}{}'.format(util.STX, ord(char), util.ETX), m.start(0), m.end(0)
360 else:
361 return None, m.start(0), m.end(0)
362
363
364class SimpleTagPattern(Pattern): # pragma: no cover
365 """
366 Return element of type `tag` with a text attribute of `group(3)`
367 of a Pattern.
368
369 """
370 def __init__(self, pattern: str, tag: str):
371 """
372 Create an instant of an simple tag pattern.
373
374 Arguments:
375 pattern: A regular expression that matches a pattern.
376 tag: Tag of element.
377
378 """
379 Pattern.__init__(self, pattern)
380 self.tag = tag
381 """ The tag of the rendered element. """
382
383 def handleMatch(self, m: re.Match[str]) -> etree.Element:
384 """
385 Return [`Element`][xml.etree.ElementTree.Element] of type `tag` with the string in `group(3)` of a
386 matching pattern as the Element's text.
387 """
388 el = etree.Element(self.tag)
389 el.text = m.group(3)
390 return el
391
392
393class SimpleTagInlineProcessor(InlineProcessor):
394 """
395 Return element of type `tag` with a text attribute of `group(2)`
396 of a Pattern.
397
398 """
399 def __init__(self, pattern: str, tag: str):
400 """
401 Create an instant of an simple tag processor.
402
403 Arguments:
404 pattern: A regular expression that matches a pattern.
405 tag: Tag of element.
406
407 """
408 InlineProcessor.__init__(self, pattern)
409 self.tag = tag
410 """ The tag of the rendered element. """
411
412 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: # pragma: no cover
413 """
414 Return [`Element`][xml.etree.ElementTree.Element] of type `tag` with the string in `group(2)` of a
415 matching pattern as the Element's text.
416 """
417 el = etree.Element(self.tag)
418 el.text = m.group(2)
419 return el, m.start(0), m.end(0)
420
421
422class SubstituteTagPattern(SimpleTagPattern): # pragma: no cover
423 """ Return an element of type `tag` with no children. """
424 def handleMatch(self, m: re.Match[str]) -> etree.Element:
425 """ Return empty [`Element`][xml.etree.ElementTree.Element] of type `tag`. """
426 return etree.Element(self.tag)
427
428
429class SubstituteTagInlineProcessor(SimpleTagInlineProcessor):
430 """ Return an element of type `tag` with no children. """
431 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]:
432 """ Return empty [`Element`][xml.etree.ElementTree.Element] of type `tag`. """
433 return etree.Element(self.tag), m.start(0), m.end(0)
434
435
436class BacktickInlineProcessor(InlineProcessor):
437 """ Return a `<code>` element containing the escaped matching text. """
438
439 def __init__(self, pattern: str):
440 InlineProcessor.__init__(self, pattern)
441 self.ESCAPED_BSLASH = '{}{}{}'.format(util.STX, ord('\\'), util.ETX)
442 self.tag = 'code'
443 """ The tag of the rendered element. """
444
445 def find_code_spans(self, start: int, text: str) -> tuple[int, int] | None:
446 """Find code spans."""
447
448 last = len(text)
449
450 # Get the maximum starting ticks
451 max_ticks = 0
452 while start < last and text[start] == '`':
453 max_ticks += 1
454 start += 1
455
456 if not max_ticks: # pragma: no cover
457 # This is not ever expected to happen.
458 return None
459
460 longest_span = 0
461 end = 0
462
463 # Find an ending span of backticks that matches our opening
464 i = start
465 while i < last:
466 span_length = 0
467 while i < last and text[i] == '`':
468 span_length += 1
469 i += 1
470 if not span_length:
471 i += 1
472 continue
473
474 # Did we find the end?
475 if max_ticks == span_length:
476 return start, i - span_length
477
478 # Track the longest span of backticks we find as a fallback.
479 if span_length > longest_span:
480 longest_span = span_length
481 end = i
482
483 # Since we didn't find an exact matching start and end,
484 # adjust start to match the largest end we could calculate.
485 if longest_span:
486 return start - (max_ticks - longest_span), end - longest_span
487
488 # We could not find a suitable pairing
489 return None
490
491 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | str | None, int | None, int | None]:
492 """
493 If the match contains `group(3)` of a pattern, then return a `code`
494 [`Element`][xml.etree.ElementTree.Element] which contains HTML escaped text (with
495 [`code_escape`][markdown.util.code_escape]) as an [`AtomicString`][markdown.util.AtomicString].
496
497 If the match contains `group(1)` then return the text of `group(1)` as backslash escaped.
498
499 """
500 if m.group(1):
501 return m.group(1).replace('\\\\', self.ESCAPED_BSLASH), m.start(0), m.end(0)
502
503 begin = m.start(0)
504 result = self.find_code_spans(begin, data)
505 if result is not None:
506 start, end = result
507 el = etree.Element(self.tag)
508 el.text = util.AtomicString(util.code_escape(data[start:end].strip()))
509 return el, begin, result[1] + (start - begin)
510 return None, None, None
511
512
513class DoubleTagPattern(SimpleTagPattern): # pragma: no cover
514 """Return a ElementTree element nested in tag2 nested in tag1.
515
516 Useful for strong emphasis etc.
517
518 """
519 def handleMatch(self, m: re.Match[str]) -> etree.Element:
520 """
521 Return [`Element`][xml.etree.ElementTree.Element] in following format:
522 `<tag1><tag2>group(3)</tag2>group(4)</tag2>` where `group(4)` is optional.
523
524 """
525 tag1, tag2 = self.tag.split(",")
526 el1 = etree.Element(tag1)
527 el2 = etree.SubElement(el1, tag2)
528 el2.text = m.group(3)
529 if len(m.groups()) == 5:
530 el2.tail = m.group(4)
531 return el1
532
533
534class DoubleTagInlineProcessor(SimpleTagInlineProcessor):
535 """Return a ElementTree element nested in tag2 nested in tag1.
536
537 Useful for strong emphasis etc.
538
539 """
540 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]: # pragma: no cover
541 """
542 Return [`Element`][xml.etree.ElementTree.Element] in following format:
543 `<tag1><tag2>group(2)</tag2>group(3)</tag2>` where `group(3)` is optional.
544
545 """
546 tag1, tag2 = self.tag.split(",")
547 el1 = etree.Element(tag1)
548 el2 = etree.SubElement(el1, tag2)
549 el2.text = m.group(2)
550 if len(m.groups()) == 3:
551 el2.tail = m.group(3)
552 return el1, m.start(0), m.end(0)
553
554
555class HtmlInlineProcessor(InlineProcessor):
556 """ Store raw inline html and return a placeholder. """
557 def handleMatch(self, m: re.Match[str], data: str) -> tuple[str, int, int]:
558 """ Store the text of `group(1)` of a pattern and return a placeholder string. """
559 rawhtml = self.backslash_unescape(self.unescape(m.group(1)))
560 place_holder = self.md.htmlStash.store(rawhtml)
561 return place_holder, m.start(0), m.end(0)
562
563 def unescape(self, text: str) -> str:
564 """ Return unescaped text given text with an inline placeholder. """
565 try:
566 stash = self.md.treeprocessors['inline'].stashed_nodes
567 except KeyError: # pragma: no cover
568 return text
569
570 def get_stash(m: re.Match[str]) -> str:
571 id = m.group(1)
572 value = stash.get(id)
573 if value is not None:
574 try:
575 # Ensure we don't have a placeholder inside a placeholder
576 return self.unescape(self.md.serializer(value))
577 except Exception:
578 return r'\%s' % value
579
580 return util.INLINE_PLACEHOLDER_RE.sub(get_stash, text)
581
582 def backslash_unescape(self, text: str) -> str:
583 """ Return text with backslash escapes undone (backslashes are restored). """
584 try:
585 RE = self.md.treeprocessors['unescape'].RE
586 except KeyError: # pragma: no cover
587 return text
588
589 def _unescape(m: re.Match[str]) -> str:
590 return chr(int(m.group(1)))
591
592 return RE.sub(_unescape, text)
593
594
595class AsteriskProcessor(InlineProcessor):
596 """Emphasis processor for handling strong and em matches inside asterisks."""
597
598 PATTERNS = [
599 EmStrongItem(re.compile(EM_STRONG_RE, re.DOTALL | re.UNICODE), 'double', 'strong,em'),
600 EmStrongItem(re.compile(STRONG_EM_RE, re.DOTALL | re.UNICODE), 'double', 'em,strong'),
601 EmStrongItem(re.compile(STRONG_EM3_RE, re.DOTALL | re.UNICODE), 'double2', 'strong,em'),
602 EmStrongItem(re.compile(STRONG_RE, re.DOTALL | re.UNICODE), 'single', 'strong'),
603 EmStrongItem(re.compile(EMPHASIS_RE, re.DOTALL | re.UNICODE), 'single', 'em')
604 ]
605 """ The various strong and emphasis patterns handled by this processor. """
606
607 def build_single(self, m: re.Match[str], tag: str, idx: int) -> etree.Element:
608 """Return single tag."""
609 el1 = etree.Element(tag)
610 text = m.group(2)
611 self.parse_sub_patterns(text, el1, None, idx)
612 return el1
613
614 def build_double(self, m: re.Match[str], tags: str, idx: int) -> etree.Element:
615 """Return double tag."""
616
617 tag1, tag2 = tags.split(",")
618 el1 = etree.Element(tag1)
619 el2 = etree.Element(tag2)
620 text = m.group(2)
621 self.parse_sub_patterns(text, el2, None, idx)
622 el1.append(el2)
623 if len(m.groups()) == 3:
624 text = m.group(3)
625 self.parse_sub_patterns(text, el1, el2, idx)
626 return el1
627
628 def build_double2(self, m: re.Match[str], tags: str, idx: int) -> etree.Element:
629 """Return double tags (variant 2): `<strong>text <em>text</em></strong>`."""
630
631 tag1, tag2 = tags.split(",")
632 el1 = etree.Element(tag1)
633 el2 = etree.Element(tag2)
634 text = m.group(2)
635 self.parse_sub_patterns(text, el1, None, idx)
636 text = m.group(3)
637 el1.append(el2)
638 self.parse_sub_patterns(text, el2, None, idx)
639 return el1
640
641 def parse_sub_patterns(
642 self, data: str, parent: etree.Element, last: etree.Element | None, idx: int
643 ) -> None:
644 """
645 Parses sub patterns.
646
647 `data`: text to evaluate.
648
649 `parent`: Parent to attach text and sub elements to.
650
651 `last`: Last appended child to parent. Can also be None if parent has no children.
652
653 `idx`: Current pattern index that was used to evaluate the parent.
654 """
655
656 offset = 0
657 pos = 0
658
659 length = len(data)
660 while pos < length:
661 # Find the start of potential emphasis or strong tokens
662 if self.compiled_re.match(data, pos):
663 matched = False
664 # See if the we can match an emphasis/strong pattern
665 for index, item in enumerate(self.PATTERNS):
666 # Only evaluate patterns that are after what was used on the parent
667 if index <= idx:
668 continue
669 m = item.pattern.match(data, pos)
670 if m:
671 # Append child nodes to parent
672 # Text nodes should be appended to the last
673 # child if present, and if not, it should
674 # be added as the parent's text node.
675 text = data[offset:m.start(0)]
676 if text:
677 if last is not None:
678 last.tail = text
679 else:
680 parent.text = text
681 el = self.build_element(m, item.builder, item.tags, index)
682 parent.append(el)
683 last = el
684 # Move our position past the matched hunk
685 offset = pos = m.end(0)
686 matched = True
687 if not matched:
688 # We matched nothing, move on to the next character
689 pos += 1
690 else:
691 # Increment position as no potential emphasis start was found.
692 pos += 1
693
694 # Append any leftover text as a text node.
695 text = data[offset:]
696 if text:
697 if last is not None:
698 last.tail = text
699 else:
700 parent.text = text
701
702 def build_element(self, m: re.Match[str], builder: str, tags: str, index: int) -> etree.Element:
703 """Element builder."""
704
705 if builder == 'double2':
706 return self.build_double2(m, tags, index)
707 elif builder == 'double':
708 return self.build_double(m, tags, index)
709 else:
710 return self.build_single(m, tags, index)
711
712 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]:
713 """Parse patterns."""
714
715 el = None
716 start = None
717 end = None
718
719 for index, item in enumerate(self.PATTERNS):
720 m1 = item.pattern.match(data, m.start(0))
721 if m1:
722 start = m1.start(0)
723 end = m1.end(0)
724 el = self.build_element(m1, item.builder, item.tags, index)
725 break
726 return el, start, end
727
728
729class UnderscoreProcessor(AsteriskProcessor):
730 """Emphasis processor for handling strong and em matches inside underscores."""
731
732 PATTERNS = [
733 EmStrongItem(re.compile(EM_STRONG2_RE, re.DOTALL | re.UNICODE), 'double', 'strong,em'),
734 EmStrongItem(re.compile(STRONG_EM2_RE, re.DOTALL | re.UNICODE), 'double', 'em,strong'),
735 EmStrongItem(re.compile(SMART_STRONG_EM_RE, re.DOTALL | re.UNICODE), 'double2', 'strong,em'),
736 EmStrongItem(re.compile(SMART_STRONG_RE, re.DOTALL | re.UNICODE), 'single', 'strong'),
737 EmStrongItem(re.compile(SMART_EMPHASIS_RE, re.DOTALL | re.UNICODE), 'single', 'em')
738 ]
739 """ The various strong and emphasis patterns handled by this processor. """
740
741
742class LinkInlineProcessor(InlineProcessor):
743 """ Return a link element from the given match. """
744 RE_LINK = re.compile(r'''\(\s*(?:(<[^<>]*>)\s*(?:('[^']*'|"[^"]*")\s*)?\))?''', re.DOTALL | re.UNICODE)
745 RE_TITLE_CLEAN = re.compile(r'\s')
746
747 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]:
748 """ Return an `a` [`Element`][xml.etree.ElementTree.Element] or `(None, None, None)`. """
749 text, index, handled = self.getText(data, m.end(0))
750
751 if not handled:
752 return None, None, None
753
754 href, title, index, handled = self.getLink(data, index)
755 if not handled:
756 return None, None, None
757
758 el = etree.Element("a")
759 el.text = text
760
761 el.set("href", href)
762
763 if title is not None:
764 el.set("title", title)
765
766 return el, m.start(0), index
767
768 def getLink(self, data: str, index: int) -> tuple[str, str | None, int, bool]:
769 """Parse data between `()` of `[Text]()` allowing recursive `()`. """
770
771 href = ''
772 title: str | None = None
773 handled = False
774
775 m = self.RE_LINK.match(data, pos=index)
776 if m and m.group(1):
777 # Matches [Text](<link> "title")
778 href = m.group(1)[1:-1].strip()
779 if m.group(2):
780 title = m.group(2)[1:-1]
781 index = m.end(0)
782 handled = True
783 elif m:
784 # Track bracket nesting and index in string
785 bracket_count = 1
786 backtrack_count = 1
787 start_index = m.end()
788 index = start_index
789 last_bracket = -1
790
791 # Primary (first found) quote tracking.
792 quote: str | None = None
793 start_quote = -1
794 exit_quote = -1
795 ignore_matches = False
796
797 # Secondary (second found) quote tracking.
798 alt_quote = None
799 start_alt_quote = -1
800 exit_alt_quote = -1
801
802 # Track last character
803 last = ''
804
805 for pos in range(index, len(data)):
806 c = data[pos]
807 if c == '(':
808 # Count nested (
809 # Don't increment the bracket count if we are sure we're in a title.
810 if not ignore_matches:
811 bracket_count += 1
812 elif backtrack_count > 0:
813 backtrack_count -= 1
814 elif c == ')':
815 # Match nested ) to (
816 # Don't decrement if we are sure we are in a title that is unclosed.
817 if ((exit_quote != -1 and quote == last) or (exit_alt_quote != -1 and alt_quote == last)):
818 bracket_count = 0
819 elif not ignore_matches:
820 bracket_count -= 1
821 elif backtrack_count > 0:
822 backtrack_count -= 1
823 # We've found our backup end location if the title doesn't resolve.
824 if backtrack_count == 0:
825 last_bracket = index + 1
826
827 elif c in ("'", '"'):
828 # Quote has started
829 if not quote:
830 # We'll assume we are now in a title.
831 # Brackets are quoted, so no need to match them (except for the final one).
832 ignore_matches = True
833 backtrack_count = bracket_count
834 bracket_count = 1
835 start_quote = index + 1
836 quote = c
837 # Secondary quote (in case the first doesn't resolve): [text](link'"title")
838 elif c != quote and not alt_quote:
839 start_alt_quote = index + 1
840 alt_quote = c
841 # Update primary quote match
842 elif c == quote:
843 exit_quote = index + 1
844 # Update secondary quote match
845 elif alt_quote and c == alt_quote:
846 exit_alt_quote = index + 1
847
848 index += 1
849
850 # Link is closed, so let's break out of the loop
851 if bracket_count == 0:
852 # Get the title if we closed a title string right before link closed
853 if exit_quote >= 0 and quote == last:
854 href = data[start_index:start_quote - 1]
855 title = ''.join(data[start_quote:exit_quote - 1])
856 elif exit_alt_quote >= 0 and alt_quote == last:
857 href = data[start_index:start_alt_quote - 1]
858 title = ''.join(data[start_alt_quote:exit_alt_quote - 1])
859 else:
860 href = data[start_index:index - 1]
861 break
862
863 if c != ' ':
864 last = c
865
866 # We have a scenario: `[test](link"notitle)`
867 # When we enter a string, we stop tracking bracket resolution in the main counter,
868 # but we do keep a backup counter up until we discover where we might resolve all brackets
869 # if the title string fails to resolve.
870 if bracket_count != 0 and backtrack_count == 0:
871 href = data[start_index:last_bracket - 1]
872 index = last_bracket
873 bracket_count = 0
874
875 handled = bracket_count == 0
876
877 if title is not None:
878 title = self.RE_TITLE_CLEAN.sub(' ', dequote(self.unescape(title.strip())))
879
880 href = self.unescape(href).strip()
881
882 return href, title, index, handled
883
884 def getText(self, data: str, index: int) -> tuple[str, int, bool]:
885 """Parse the content between `[]` of the start of an image or link
886 resolving nested square brackets.
887
888 """
889 bracket_count = 1
890 text = []
891 for pos in range(index, len(data)):
892 c = data[pos]
893 if c == ']':
894 bracket_count -= 1
895 elif c == '[':
896 bracket_count += 1
897 index += 1
898 if bracket_count == 0:
899 break
900 text.append(c)
901 return ''.join(text), index, bracket_count == 0
902
903
904class ImageInlineProcessor(LinkInlineProcessor):
905 """ Return a `img` element from the given match. """
906
907 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]:
908 """ Return an `img` [`Element`][xml.etree.ElementTree.Element] or `(None, None, None)`. """
909 text, index, handled = self.getText(data, m.end(0))
910 if not handled:
911 return None, None, None
912
913 src, title, index, handled = self.getLink(data, index)
914 if not handled:
915 return None, None, None
916
917 el = etree.Element("img")
918
919 el.set("src", src)
920
921 if title is not None:
922 el.set("title", title)
923
924 el.set('alt', self.unescape(text))
925 return el, m.start(0), index
926
927
928class ReferenceInlineProcessor(LinkInlineProcessor):
929 """ Match to a stored reference and return link element. """
930 NEWLINE_CLEANUP_RE = re.compile(r'\s+', re.MULTILINE)
931
932 RE_LINK = re.compile(r'\s?\[([^\]]*)\]', re.DOTALL | re.UNICODE)
933
934 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element | None, int | None, int | None]:
935 """
936 Return [`Element`][xml.etree.ElementTree.Element] returned by `makeTag` method or `(None, None, None)`.
937
938 """
939 text, index, handled = self.getText(data, m.end(0))
940 if not handled:
941 return None, None, None
942
943 id, end, handled = self.evalId(data, index, text)
944 if not handled:
945 return None, None, None
946
947 # Clean up line breaks in id
948 id = self.NEWLINE_CLEANUP_RE.sub(' ', id)
949 if id not in self.md.references: # ignore undefined refs
950 return None, m.start(0), end
951
952 href, title = self.md.references[id]
953
954 return self.makeTag(href, title, text), m.start(0), end
955
956 def evalId(self, data: str, index: int, text: str) -> tuple[str | None, int, bool]:
957 """
958 Evaluate the id portion of `[ref][id]`.
959
960 If `[ref][]` use `[ref]`.
961 """
962 m = self.RE_LINK.match(data, pos=index)
963 if not m:
964 return None, index, False
965 else:
966 id = m.group(1).lower()
967 end = m.end(0)
968 if not id:
969 id = text.lower()
970 return id, end, True
971
972 def makeTag(self, href: str, title: str, text: str) -> etree.Element:
973 """ Return an `a` [`Element`][xml.etree.ElementTree.Element]. """
974 el = etree.Element('a')
975
976 el.set('href', href)
977 if title:
978 el.set('title', title)
979
980 el.text = text
981 return el
982
983
984class ShortReferenceInlineProcessor(ReferenceInlineProcessor):
985 """Short form of reference: `[google]`. """
986 def evalId(self, data: str, index: int, text: str) -> tuple[str, int, bool]:
987 """Evaluate the id of `[ref]`. """
988
989 return text.lower(), index, True
990
991
992class ImageReferenceInlineProcessor(ReferenceInlineProcessor):
993 """ Match to a stored reference and return `img` element. """
994 def makeTag(self, href: str, title: str, text: str) -> etree.Element:
995 """ Return an `img` [`Element`][xml.etree.ElementTree.Element]. """
996 el = etree.Element("img")
997 el.set("src", href)
998 if title:
999 el.set("title", title)
1000 el.set("alt", self.unescape(text))
1001 return el
1002
1003
1004class ShortImageReferenceInlineProcessor(ImageReferenceInlineProcessor):
1005 """ Short form of image reference: `![ref]`. """
1006 def evalId(self, data: str, index: int, text: str) -> tuple[str, int, bool]:
1007 """Evaluate the id of `[ref]`. """
1008
1009 return text.lower(), index, True
1010
1011
1012class AutolinkInlineProcessor(InlineProcessor):
1013 """ Return a link Element given an auto-link (`<http://example/com>`). """
1014 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]:
1015 """ Return an `a` [`Element`][xml.etree.ElementTree.Element] of `group(1)`. """
1016 el = etree.Element("a")
1017 el.set('href', self.unescape(m.group(1)))
1018 el.text = util.AtomicString(m.group(1))
1019 return el, m.start(0), m.end(0)
1020
1021
1022class AutomailInlineProcessor(InlineProcessor):
1023 """
1024 Return a `mailto` link Element given an auto-mail link (`<foo@example.com>`).
1025 """
1026 def handleMatch(self, m: re.Match[str], data: str) -> tuple[etree.Element, int, int]:
1027 """ Return an [`Element`][xml.etree.ElementTree.Element] containing a `mailto` link of `group(1)`. """
1028 el = etree.Element('a')
1029 email = self.unescape(m.group(1))
1030 if email.startswith("mailto:"):
1031 email = email[len("mailto:"):]
1032
1033 def codepoint2name(code: int) -> str:
1034 """Return entity definition by code, or the code if not defined."""
1035 entity = entities.codepoint2name.get(code)
1036 if entity:
1037 return "{}{};".format(util.AMP_SUBSTITUTE, entity)
1038 else:
1039 return "%s#%d;" % (util.AMP_SUBSTITUTE, code)
1040
1041 letters = [codepoint2name(ord(letter)) for letter in email]
1042 el.text = util.AtomicString(''.join(letters))
1043
1044 mailto = "mailto:" + email
1045 mailto = "".join([util.AMP_SUBSTITUTE + '#%d;' %
1046 ord(letter) for letter in mailto])
1047 el.set('href', mailto)
1048 return el, m.start(0), m.end(0)