Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/markdown_it/rules_inline/state_inline.py: 93%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

117 statements  

1from __future__ import annotations 

2 

3from dataclasses import dataclass 

4from typing import TYPE_CHECKING, Any, Literal, NamedTuple 

5 

6from ..common.utils import isMdAsciiPunct, isPunctChar, isWhiteSpace 

7from ..ruler import StateBase 

8from ..token import Token 

9from ..utils import EnvType 

10 

11if TYPE_CHECKING: 

12 from markdown_it import MarkdownIt 

13 

14 

15@dataclass(slots=True) 

16class Delimiter: 

17 # Char code of the starting marker (number). 

18 marker: int 

19 

20 # Total length of these series of delimiters. 

21 length: int 

22 

23 # A position of the token this delimiter corresponds to. 

24 token: int 

25 

26 # If this delimiter is matched as a valid opener, `end` will be 

27 # equal to its position, otherwise it's `-1`. 

28 end: int 

29 

30 # Boolean flags that determine if this delimiter could open or close 

31 # an emphasis. 

32 open: bool 

33 close: bool 

34 

35 level: bool | None = None 

36 

37 

38class Scanned(NamedTuple): 

39 can_open: bool 

40 can_close: bool 

41 length: int 

42 

43 

44class StateInline(StateBase): 

45 def __init__( 

46 self, src: str, md: MarkdownIt, env: EnvType, outTokens: list[Token] 

47 ) -> None: 

48 self.src = src 

49 self.env = env 

50 self.md = md 

51 self.tokens = outTokens 

52 self.tokens_meta: list[dict[str, Any] | None] = [None] * len(outTokens) 

53 

54 self.pos = 0 

55 self.posMax = len(self.src) 

56 self.level = 0 

57 # `pending` holds literal text not yet flushed to a token. It is 

58 # exposed as a plain `str` (see the property below), but is accumulated 

59 # through a list buffer so that appending one character at a time -- the 

60 # inline tokenizer's fallback path -- stays amortised O(1). Appending 

61 # to a `str` *attribute* cannot use CPython's in-place concatenation 

62 # optimisation (the attribute holds a second reference), so each `+=` 

63 # copies the whole string, making long runs of non-markup characters 

64 # quadratic. 

65 self._pending = "" 

66 self._pending_buffer: list[str] = [] 

67 self.pendingLevel = 0 

68 

69 # Stores { start: end } pairs. Useful for backtrack 

70 # optimization of pairs parse (emphasis, strikes). 

71 self.cache: dict[int, int] = {} 

72 

73 # List of emphasis-like delimiters for current tag 

74 self.delimiters: list[Delimiter] = [] 

75 

76 # Stack of delimiter lists for upper level tags 

77 self._prev_delimiters: list[list[Delimiter]] = [] 

78 

79 # backticklength => last seen position 

80 self.backticks: dict[int, int] = {} 

81 self.backticksScanned = False 

82 

83 # Counter used to disable inline linkify-it execution 

84 # inside <a> and markdown links 

85 self.linkLevel = 0 

86 

87 # Lazy cache of `terminator -> last index in src`, see 

88 # `html_terminator_last`. 

89 self._html_terminators: dict[str, int] | None = None 

90 

91 def html_terminator_last(self, term: str) -> int: 

92 """Index of the last occurrence of `term` in `self.src`, or -1. 

93 

94 The result is cached per terminator, for the life of the state. 

95 It is used by the `html_inline` rule to reject, in constant time, a 

96 position at which the terminator required to close an HTML construct 

97 cannot possibly occur -- without running the tag regex, whose lazy 

98 sub-patterns would otherwise rescan to the end of the input on every 

99 such (failing) attempt. 

100 """ 

101 if self._html_terminators is None: 

102 self._html_terminators = {} 

103 elif (last := self._html_terminators.get(term)) is not None: 

104 return last 

105 last = self._html_terminators[term] = self.src.rfind(term) 

106 return last 

107 

108 def __repr__(self) -> str: 

109 return ( 

110 f"{self.__class__.__name__}" 

111 f"(pos=[{self.pos} of {self.posMax}], token={len(self.tokens)})" 

112 ) 

113 

114 @property 

115 def pending(self) -> str: 

116 """Literal text accumulated so far, but not yet flushed to a token.""" 

117 buffer = self._pending_buffer 

118 if buffer: 

119 # Move the string into a local and drop the instance's reference 

120 # before concatenating. With a single reference left, CPython 

121 # resizes the string in place (amortised O(new chars)) rather 

122 # than copying it, so a rule that reads `pending` on every 

123 # character (e.g. an attribute-syntax plugin) stays linear. 

124 text = self._pending 

125 self._pending = "" 

126 text += "".join(buffer) 

127 buffer.clear() 

128 self._pending = text 

129 return self._pending 

130 

131 @pending.setter 

132 def pending(self, value: str) -> None: 

133 self._pending = value 

134 # Assign rather than `.clear()` so the setter also works on an 

135 # instance whose `__init__` has not run yet (subclasses that set 

136 # `pending` before calling `super().__init__()`), and so a copied 

137 # state never shares a buffer with its original. 

138 self._pending_buffer = [] 

139 

140 def __copy__(self) -> StateInline: 

141 """Shallow copy that does not share the pending text buffer.""" 

142 text = self.pending # materialise (and clear) our own buffer first 

143 new = self.__class__.__new__(self.__class__) 

144 new.__dict__.update(self.__dict__) 

145 new._pending = text 

146 new._pending_buffer = [] 

147 return new 

148 

149 def append_pending(self, text: str) -> None: 

150 """Append literal text to `pending`, in amortised O(1) time. 

151 

152 Prefer this to ``state.pending += text`` on hot paths: it buffers the 

153 fragment rather than rebuilding the whole ``pending`` string per call. 

154 """ 

155 self._pending_buffer.append(text) 

156 

157 def pushPending(self) -> Token: 

158 token = Token("text", "", 0) 

159 token.content = self.pending 

160 token.level = self.pendingLevel 

161 self.tokens.append(token) 

162 self.pending = "" 

163 return token 

164 

165 def push(self, ttype: str, tag: str, nesting: Literal[-1, 0, 1]) -> Token: 

166 """Push new token to "stream". 

167 If pending text exists - flush it as text token 

168 """ 

169 if self.pending: 

170 self.pushPending() 

171 

172 token = Token(ttype, tag, nesting) 

173 token_meta = None 

174 

175 if nesting < 0: 

176 # closing tag 

177 self.level -= 1 

178 self.delimiters = self._prev_delimiters.pop() 

179 

180 token.level = self.level 

181 

182 if nesting > 0: 

183 # opening tag 

184 self.level += 1 

185 self._prev_delimiters.append(self.delimiters) 

186 self.delimiters = [] 

187 token_meta = {"delimiters": self.delimiters} 

188 

189 self.pendingLevel = self.level 

190 self.tokens.append(token) 

191 self.tokens_meta.append(token_meta) 

192 return token 

193 

194 def scanDelims(self, start: int, canSplitWord: bool) -> Scanned: 

195 """ 

196 Scan a sequence of emphasis-like markers, and determine whether 

197 it can start an emphasis sequence or end an emphasis sequence. 

198 

199 - start - position to scan from (it should point at a valid marker); 

200 - canSplitWord - determine if these markers can be found inside a word 

201 

202 """ 

203 pos = start 

204 maximum = self.posMax 

205 marker = self.src[start] 

206 

207 # treat beginning of the line as a whitespace 

208 lastChar = self.src[start - 1] if start > 0 else " " 

209 

210 while pos < maximum and self.src[pos] == marker: 

211 pos += 1 

212 

213 count = pos - start 

214 

215 # treat end of the line as a whitespace 

216 nextChar = self.src[pos] if pos < maximum else " " 

217 

218 isLastPunctChar = isMdAsciiPunct(ord(lastChar)) or isPunctChar(lastChar) 

219 isNextPunctChar = isMdAsciiPunct(ord(nextChar)) or isPunctChar(nextChar) 

220 

221 isLastWhiteSpace = isWhiteSpace(ord(lastChar)) 

222 isNextWhiteSpace = isWhiteSpace(ord(nextChar)) 

223 

224 left_flanking = not ( 

225 isNextWhiteSpace 

226 or (isNextPunctChar and not (isLastWhiteSpace or isLastPunctChar)) 

227 ) 

228 right_flanking = not ( 

229 isLastWhiteSpace 

230 or (isLastPunctChar and not (isNextWhiteSpace or isNextPunctChar)) 

231 ) 

232 

233 can_open = left_flanking and ( 

234 canSplitWord or (not right_flanking) or isLastPunctChar 

235 ) 

236 can_close = right_flanking and ( 

237 canSplitWord or (not left_flanking) or isNextPunctChar 

238 ) 

239 

240 return Scanned(can_open, can_close, count)