1"""Utilities for parsing source text"""
2
3from __future__ import annotations
4
5import re
6from re import Match
7from typing import TypeVar
8import unicodedata
9
10from .entities import entities
11
12
13def charCodeAt(src: str, pos: int) -> int | None:
14 """
15 Returns the Unicode value of the character at the specified location.
16
17 @param - index The zero-based index of the desired character.
18 If there is no character at the specified index, NaN is returned.
19
20 This was added for compatibility with python
21 """
22 try:
23 return ord(src[pos])
24 except IndexError:
25 return None
26
27
28def charStrAt(src: str, pos: int) -> str | None:
29 """
30 Returns the Unicode value of the character at the specified location.
31
32 @param - index The zero-based index of the desired character.
33 If there is no character at the specified index, NaN is returned.
34
35 This was added for compatibility with python
36 """
37 try:
38 return src[pos]
39 except IndexError:
40 return None
41
42
43_ItemTV = TypeVar("_ItemTV")
44
45
46def arrayReplaceAt(
47 src: list[_ItemTV], pos: int, newElements: list[_ItemTV]
48) -> list[_ItemTV]:
49 """
50 Remove element from array and put another array at those position.
51 Useful for some operations with tokens
52 """
53 return src[:pos] + newElements + src[pos + 1 :]
54
55
56def isValidEntityCode(c: int) -> bool:
57 # broken sequence
58 if c >= 0xD800 and c <= 0xDFFF:
59 return False
60 # never used
61 if c >= 0xFDD0 and c <= 0xFDEF:
62 return False
63 if ((c & 0xFFFF) == 0xFFFF) or ((c & 0xFFFF) == 0xFFFE):
64 return False
65 # control codes
66 if c >= 0x00 and c <= 0x08:
67 return False
68 if c == 0x0B:
69 return False
70 if c >= 0x0E and c <= 0x1F:
71 return False
72 if c >= 0x7F and c <= 0x9F:
73 return False
74 # out of range
75 return not (c > 0x10FFFF)
76
77
78def fromCodePoint(c: int) -> str:
79 """Convert ordinal to unicode.
80
81 Note, in the original Javascript two string characters were required,
82 for codepoints larger than `0xFFFF`.
83 But Python 3 can represent any unicode codepoint in one character.
84 """
85 return chr(c)
86
87
88# UNESCAPE_MD_RE = re.compile(r'\\([!"#$%&\'()*+,\-.\/:;<=>?@[\\\]^_`{|}~])')
89# ENTITY_RE_g = re.compile(r'&([a-z#][a-z0-9]{1,31})', re.IGNORECASE)
90UNESCAPE_ALL_RE = re.compile(
91 r'\\([!"#$%&\'()*+,\-.\/:;<=>?@[\\\]^_`{|}~])' + "|" + r"&([a-z#][a-z0-9]{1,31});",
92 re.IGNORECASE,
93)
94DIGITAL_ENTITY_BASE10_RE = re.compile(r"#([0-9]{1,8})")
95DIGITAL_ENTITY_BASE16_RE = re.compile(r"#x([a-f0-9]{1,8})", re.IGNORECASE)
96
97
98def replaceEntityPattern(match: str, name: str) -> str:
99 """Convert HTML entity patterns,
100 see https://spec.commonmark.org/0.30/#entity-references
101 """
102 if name in entities:
103 return entities[name]
104
105 code: int | None = None
106 if pat := DIGITAL_ENTITY_BASE10_RE.fullmatch(name):
107 code = int(pat.group(1), 10)
108 elif pat := DIGITAL_ENTITY_BASE16_RE.fullmatch(name):
109 code = int(pat.group(1), 16)
110
111 if code is not None and isValidEntityCode(code):
112 return fromCodePoint(code)
113
114 return match
115
116
117def unescapeAll(string: str) -> str:
118 def replacer_func(match: Match[str]) -> str:
119 escaped = match.group(1)
120 if escaped:
121 return escaped
122 entity = match.group(2)
123 return replaceEntityPattern(match.group(), entity)
124
125 if "\\" not in string and "&" not in string:
126 return string
127 return UNESCAPE_ALL_RE.sub(replacer_func, string)
128
129
130ESCAPABLE = r"""\\!"#$%&'()*+,./:;<=>?@\[\]^`{}|_~-"""
131ESCAPE_CHAR = re.compile(r"\\([" + ESCAPABLE + r"])")
132
133
134def stripEscape(string: str) -> str:
135 """Strip escape \\ characters"""
136 return ESCAPE_CHAR.sub(r"\1", string)
137
138
139def escapeHtml(raw: str) -> str:
140 """Replace special characters "&", "<", ">" and '"' to HTML-safe sequences."""
141 # like html.escape, but without escaping single quotes
142 raw = raw.replace("&", "&") # Must be done first!
143 raw = raw.replace("<", "<")
144 raw = raw.replace(">", ">")
145 raw = raw.replace('"', """)
146 return raw
147
148
149# //////////////////////////////////////////////////////////////////////////////
150
151REGEXP_ESCAPE_RE = re.compile(r"[.?*+^$[\]\\(){}|-]")
152
153
154def escapeRE(string: str) -> str:
155 string = REGEXP_ESCAPE_RE.sub("\\$&", string)
156 return string
157
158
159# //////////////////////////////////////////////////////////////////////////////
160
161
162def isSpace(code: int | None) -> bool:
163 """Check if character code is a whitespace."""
164 return code in (0x09, 0x20)
165
166
167def isStrSpace(ch: str | None) -> bool:
168 """Check if character is a whitespace."""
169 return ch in ("\t", " ")
170
171
172MD_WHITESPACE = {
173 0x09, # \t
174 0x0A, # \n
175 0x0B, # \v
176 0x0C, # \f
177 0x0D, # \r
178 0x20, # space
179 0xA0,
180 0x1680,
181 0x202F,
182 0x205F,
183 0x3000,
184}
185
186
187def isWhiteSpace(code: int) -> bool:
188 r"""Zs (unicode class) || [\t\f\v\r\n]"""
189 if code >= 0x2000 and code <= 0x200A:
190 return True
191 return code in MD_WHITESPACE
192
193
194#: The characters ``String.prototype.trim`` removes in JavaScript, which is
195#: what upstream markdown-it strips, minus U+FEFF.
196#:
197#: ``str.strip()`` without an argument uses :py:meth:`str.isspace`, which also
198#: removes U+001C, U+001D, U+001E, U+001F and U+0085. Those are not whitespace
199#: in CommonMark and are not removed by ``trim``, so relying on it drops them
200#: from the output and makes two different reference labels compare equal.
201#:
202#: U+FEFF is deliberately excluded: ``trim`` does remove it, and that has the
203#: very label folding effect this constant exists to avoid.
204MD_TRIM_CHARS = "".join(
205 chr(code)
206 for code in sorted(MD_WHITESPACE | set(range(0x2000, 0x200B)) | {0x2028, 0x2029})
207)
208
209
210#: A run of one or more characters from :data:`MD_TRIM_CHARS`.
211MD_TRIM_RE = re.compile("[" + re.escape(MD_TRIM_CHARS) + "]+")
212
213
214def mdTrim(string: str) -> str:
215 """Strip leading and trailing whitespace, using the CommonMark set."""
216 return string.strip(MD_TRIM_CHARS)
217
218
219# //////////////////////////////////////////////////////////////////////////////
220
221
222def isPunctChar(ch: str) -> bool:
223 """Check if character is a punctuation character."""
224 return unicodedata.category(ch).startswith(("P", "S"))
225
226
227MD_ASCII_PUNCT = {
228 0x21, # /* ! */
229 0x22, # /* " */
230 0x23, # /* # */
231 0x24, # /* $ */
232 0x25, # /* % */
233 0x26, # /* & */
234 0x27, # /* ' */
235 0x28, # /* ( */
236 0x29, # /* ) */
237 0x2A, # /* * */
238 0x2B, # /* + */
239 0x2C, # /* , */
240 0x2D, # /* - */
241 0x2E, # /* . */
242 0x2F, # /* / */
243 0x3A, # /* : */
244 0x3B, # /* ; */
245 0x3C, # /* < */
246 0x3D, # /* = */
247 0x3E, # /* > */
248 0x3F, # /* ? */
249 0x40, # /* @ */
250 0x5B, # /* [ */
251 0x5C, # /* \ */
252 0x5D, # /* ] */
253 0x5E, # /* ^ */
254 0x5F, # /* _ */
255 0x60, # /* ` */
256 0x7B, # /* { */
257 0x7C, # /* | */
258 0x7D, # /* } */
259 0x7E, # /* ~ */
260}
261
262
263def isMdAsciiPunct(ch: int) -> bool:
264 """Markdown ASCII punctuation characters.
265
266 ::
267
268 !, ", #, $, %, &, ', (, ), *, +, ,, -, ., /, :, ;, <, =, >, ?, @, [, \\, ], ^, _, `, {, |, }, or ~
269
270 See http://spec.commonmark.org/0.15/#ascii-punctuation-character
271
272 Don't confuse with unicode punctuation !!! It lacks some chars in ascii range.
273
274 """
275 return ch in MD_ASCII_PUNCT
276
277
278def normalizeReference(string: str) -> str:
279 """Helper to unify [reference labels]."""
280 # Trim and collapse whitespace
281 #
282 string = MD_TRIM_RE.sub(" ", mdTrim(string))
283
284 # In node v10 'ẞ'.toLowerCase() === 'Ṿ', which is presumed to be a bug
285 # fixed in v12 (couldn't find any details).
286 #
287 # So treat this one as a special case
288 # (remove this when node v10 is no longer supported).
289 #
290 # if ('ẞ'.toLowerCase() === 'Ṿ') {
291 # str = str.replace(/ẞ/g, 'ß')
292 # }
293
294 # .toLowerCase().toUpperCase() should get rid of all differences
295 # between letter variants.
296 #
297 # Simple .toLowerCase() doesn't normalize 125 code points correctly,
298 # and .toUpperCase doesn't normalize 6 of them (list of exceptions:
299 # İ, ϴ, ẞ, Ω, K, Å - those are already uppercased, but have differently
300 # uppercased versions).
301 #
302 # Here's an example showing how it happens. Lets take greek letter omega:
303 # uppercase U+0398 (Θ), U+03f4 (ϴ) and lowercase U+03b8 (θ), U+03d1 (ϑ)
304 #
305 # Unicode entries:
306 # 0398;GREEK CAPITAL LETTER THETA;Lu;0;L;;;;;N;;;;03B8
307 # 03B8;GREEK SMALL LETTER THETA;Ll;0;L;;;;;N;;;0398;;0398
308 # 03D1;GREEK THETA SYMBOL;Ll;0;L;<compat> 03B8;;;;N;GREEK SMALL LETTER SCRIPT THETA;;0398;;0398
309 # 03F4;GREEK CAPITAL THETA SYMBOL;Lu;0;L;<compat> 0398;;;;N;;;;03B8
310 #
311 # Case-insensitive comparison should treat all of them as equivalent.
312 #
313 # But .toLowerCase() doesn't change ϑ (it's already lowercase),
314 # and .toUpperCase() doesn't change ϴ (already uppercase).
315 #
316 # Applying first lower then upper case normalizes any character:
317 # '\u0398\u03f4\u03b8\u03d1'.toLowerCase().toUpperCase() === '\u0398\u0398\u0398\u0398'
318 #
319 # Note: this is equivalent to unicode case folding; unicode normalization
320 # is a different step that is not required here.
321 #
322 # Final result should be uppercased, because it's later stored in an object
323 # (this avoid a conflict with Object.prototype members,
324 # most notably, `__proto__`)
325 #
326 return string.lower().upper()
327
328
329LINK_OPEN_RE = re.compile(r"^<a[>\s]", flags=re.IGNORECASE)
330LINK_CLOSE_RE = re.compile(r"^</a\s*>", flags=re.IGNORECASE)
331
332
333def isLinkOpen(string: str) -> bool:
334 return bool(LINK_OPEN_RE.search(string))
335
336
337def isLinkClose(string: str) -> bool:
338 return bool(LINK_CLOSE_RE.search(string))