1from __future__ import annotations
2
3from dataclasses import dataclass
4from typing import TYPE_CHECKING, Any, Literal, NamedTuple
5
6from ..common.utils import isMdAsciiPunct, isPunctChar, isWhiteSpace
7from ..ruler import StateBase
8from ..token import Token
9from ..utils import EnvType
10
11if TYPE_CHECKING:
12 from markdown_it import MarkdownIt
13
14
15@dataclass(slots=True)
16class Delimiter:
17 # Char code of the starting marker (number).
18 marker: int
19
20 # Total length of these series of delimiters.
21 length: int
22
23 # A position of the token this delimiter corresponds to.
24 token: int
25
26 # If this delimiter is matched as a valid opener, `end` will be
27 # equal to its position, otherwise it's `-1`.
28 end: int
29
30 # Boolean flags that determine if this delimiter could open or close
31 # an emphasis.
32 open: bool
33 close: bool
34
35 level: bool | None = None
36
37
38class Scanned(NamedTuple):
39 can_open: bool
40 can_close: bool
41 length: int
42
43
44class StateInline(StateBase):
45 def __init__(
46 self, src: str, md: MarkdownIt, env: EnvType, outTokens: list[Token]
47 ) -> None:
48 self.src = src
49 self.env = env
50 self.md = md
51 self.tokens = outTokens
52 self.tokens_meta: list[dict[str, Any] | None] = [None] * len(outTokens)
53
54 self.pos = 0
55 self.posMax = len(self.src)
56 self.level = 0
57 # `pending` holds literal text not yet flushed to a token. It is
58 # exposed as a plain `str` (see the property below), but is accumulated
59 # through a list buffer so that appending one character at a time -- the
60 # inline tokenizer's fallback path -- stays amortised O(1). Appending
61 # to a `str` *attribute* cannot use CPython's in-place concatenation
62 # optimisation (the attribute holds a second reference), so each `+=`
63 # copies the whole string, making long runs of non-markup characters
64 # quadratic.
65 self._pending = ""
66 self._pending_buffer: list[str] = []
67 self.pendingLevel = 0
68
69 # Stores { start: end } pairs. Useful for backtrack
70 # optimization of pairs parse (emphasis, strikes).
71 self.cache: dict[int, int] = {}
72
73 # List of emphasis-like delimiters for current tag
74 self.delimiters: list[Delimiter] = []
75
76 # Stack of delimiter lists for upper level tags
77 self._prev_delimiters: list[list[Delimiter]] = []
78
79 # backticklength => last seen position
80 self.backticks: dict[int, int] = {}
81 self.backticksScanned = False
82
83 # Counter used to disable inline linkify-it execution
84 # inside <a> and markdown links
85 self.linkLevel = 0
86
87 # Lazy cache of `terminator -> last index in src`, see
88 # `html_terminator_last`.
89 self._html_terminators: dict[str, int] | None = None
90
91 def html_terminator_last(self, term: str) -> int:
92 """Index of the last occurrence of `term` in `self.src`, or -1.
93
94 The result is cached per terminator, for the life of the state.
95 It is used by the `html_inline` rule to reject, in constant time, a
96 position at which the terminator required to close an HTML construct
97 cannot possibly occur -- without running the tag regex, whose lazy
98 sub-patterns would otherwise rescan to the end of the input on every
99 such (failing) attempt.
100 """
101 if self._html_terminators is None:
102 self._html_terminators = {}
103 elif (last := self._html_terminators.get(term)) is not None:
104 return last
105 last = self._html_terminators[term] = self.src.rfind(term)
106 return last
107
108 def __repr__(self) -> str:
109 return (
110 f"{self.__class__.__name__}"
111 f"(pos=[{self.pos} of {self.posMax}], token={len(self.tokens)})"
112 )
113
114 @property
115 def pending(self) -> str:
116 """Literal text accumulated so far, but not yet flushed to a token."""
117 buffer = self._pending_buffer
118 if buffer:
119 # Move the string into a local and drop the instance's reference
120 # before concatenating. With a single reference left, CPython
121 # resizes the string in place (amortised O(new chars)) rather
122 # than copying it, so a rule that reads `pending` on every
123 # character (e.g. an attribute-syntax plugin) stays linear.
124 text = self._pending
125 self._pending = ""
126 text += "".join(buffer)
127 buffer.clear()
128 self._pending = text
129 return self._pending
130
131 @pending.setter
132 def pending(self, value: str) -> None:
133 self._pending = value
134 # Assign rather than `.clear()` so the setter also works on an
135 # instance whose `__init__` has not run yet (subclasses that set
136 # `pending` before calling `super().__init__()`), and so a copied
137 # state never shares a buffer with its original.
138 self._pending_buffer = []
139
140 def __copy__(self) -> StateInline:
141 """Shallow copy that does not share the pending text buffer."""
142 text = self.pending # materialise (and clear) our own buffer first
143 new = self.__class__.__new__(self.__class__)
144 new.__dict__.update(self.__dict__)
145 new._pending = text
146 new._pending_buffer = []
147 return new
148
149 def append_pending(self, text: str) -> None:
150 """Append literal text to `pending`, in amortised O(1) time.
151
152 Prefer this to ``state.pending += text`` on hot paths: it buffers the
153 fragment rather than rebuilding the whole ``pending`` string per call.
154 """
155 self._pending_buffer.append(text)
156
157 def pushPending(self) -> Token:
158 token = Token("text", "", 0)
159 token.content = self.pending
160 token.level = self.pendingLevel
161 self.tokens.append(token)
162 self.pending = ""
163 return token
164
165 def push(self, ttype: str, tag: str, nesting: Literal[-1, 0, 1]) -> Token:
166 """Push new token to "stream".
167 If pending text exists - flush it as text token
168 """
169 if self.pending:
170 self.pushPending()
171
172 token = Token(ttype, tag, nesting)
173 token_meta = None
174
175 if nesting < 0:
176 # closing tag
177 self.level -= 1
178 self.delimiters = self._prev_delimiters.pop()
179
180 token.level = self.level
181
182 if nesting > 0:
183 # opening tag
184 self.level += 1
185 self._prev_delimiters.append(self.delimiters)
186 self.delimiters = []
187 token_meta = {"delimiters": self.delimiters}
188
189 self.pendingLevel = self.level
190 self.tokens.append(token)
191 self.tokens_meta.append(token_meta)
192 return token
193
194 def scanDelims(self, start: int, canSplitWord: bool) -> Scanned:
195 """
196 Scan a sequence of emphasis-like markers, and determine whether
197 it can start an emphasis sequence or end an emphasis sequence.
198
199 - start - position to scan from (it should point at a valid marker);
200 - canSplitWord - determine if these markers can be found inside a word
201
202 """
203 pos = start
204 maximum = self.posMax
205 marker = self.src[start]
206
207 # treat beginning of the line as a whitespace
208 lastChar = self.src[start - 1] if start > 0 else " "
209
210 while pos < maximum and self.src[pos] == marker:
211 pos += 1
212
213 count = pos - start
214
215 # treat end of the line as a whitespace
216 nextChar = self.src[pos] if pos < maximum else " "
217
218 isLastPunctChar = isMdAsciiPunct(ord(lastChar)) or isPunctChar(lastChar)
219 isNextPunctChar = isMdAsciiPunct(ord(nextChar)) or isPunctChar(nextChar)
220
221 isLastWhiteSpace = isWhiteSpace(ord(lastChar))
222 isNextWhiteSpace = isWhiteSpace(ord(nextChar))
223
224 left_flanking = not (
225 isNextWhiteSpace
226 or (isNextPunctChar and not (isLastWhiteSpace or isLastPunctChar))
227 )
228 right_flanking = not (
229 isLastWhiteSpace
230 or (isLastPunctChar and not (isNextWhiteSpace or isNextPunctChar))
231 )
232
233 can_open = left_flanking and (
234 canSplitWord or (not right_flanking) or isLastPunctChar
235 )
236 can_close = right_flanking and (
237 canSplitWord or (not left_flanking) or isNextPunctChar
238 )
239
240 return Scanned(can_open, can_close, count)