1"""Implementation of JSONDecoder
2"""
3from __future__ import absolute_import
4import re
5import sys
6from .compat import PY3, unichr
7from .scanner import make_scanner, JSONDecodeError
8
9
10def _import_c_scanstring():
11 try:
12 from ._speedups import scanstring
13 return scanstring
14 except ImportError:
15 return None
16c_scanstring = _import_c_scanstring()
17
18# NOTE (3.1.0): JSONDecodeError may still be imported from this module for
19# compatibility, but it was never in the __all__
20__all__ = ['JSONDecoder']
21
22FLAGS = re.VERBOSE | re.MULTILINE | re.DOTALL
23
24def _floatconstants():
25 return float('nan'), float('inf'), float('-inf')
26
27NaN, PosInf, NegInf = _floatconstants()
28
29_CONSTANTS = {
30 '-Infinity': NegInf,
31 'Infinity': PosInf,
32 'NaN': NaN,
33}
34
35STRINGCHUNK = re.compile(r'(.*?)(["\\\x00-\x1f])', FLAGS)
36BACKSLASH = {
37 '"': u'"', '\\': u'\\', '/': u'/',
38 'b': u'\b', 'f': u'\f', 'n': u'\n', 'r': u'\r', 't': u'\t',
39}
40
41DEFAULT_ENCODING = "utf-8"
42
43if hasattr(sys, 'get_int_max_str_digits'):
44 bounded_int = int
45else:
46 def bounded_int(s, INT_MAX_STR_DIGITS=4300):
47 """Backport of the integer string length conversion limitation
48
49 https://docs.python.org/3/library/stdtypes.html#int-max-str-digits
50 """
51 if len(s) > INT_MAX_STR_DIGITS:
52 raise ValueError("Exceeds the limit (%s) for integer string conversion: value has %s digits" % (INT_MAX_STR_DIGITS, len(s)))
53 return int(s)
54
55
56def scan_four_digit_hex(s, end, _m=re.compile(r'^[0-9a-fA-F]{4}$').match):
57 """Scan a four digit hex number from s[end:end + 4]
58 """
59 msg = "Invalid \\uXXXX escape sequence"
60 esc = s[end:end + 4]
61 if not _m(esc):
62 raise JSONDecodeError(msg, s, end - 2)
63 try:
64 return int(esc, 16), end + 4
65 except ValueError:
66 raise JSONDecodeError(msg, s, end - 2)
67
68def py_scanstring(s, end, encoding=None, strict=True,
69 _b=BACKSLASH, _m=STRINGCHUNK.match, _join=u''.join,
70 _PY3=PY3, _maxunicode=sys.maxunicode,
71 _scan_four_digit_hex=scan_four_digit_hex):
72 """Scan the string s for a JSON string. End is the index of the
73 character in s after the quote that started the JSON string.
74 Unescapes all valid JSON string escape sequences and raises ValueError
75 on attempt to decode an invalid string. If strict is False then literal
76 control characters are allowed in the string.
77
78 Returns a tuple of the decoded string and the index of the character in s
79 after the end quote."""
80 if encoding is None:
81 encoding = DEFAULT_ENCODING
82 chunks = []
83 _append = chunks.append
84 begin = end - 1
85 while 1:
86 chunk = _m(s, end)
87 if chunk is None:
88 raise JSONDecodeError(
89 "Unterminated string starting at", s, begin)
90 prev_end = end
91 end = chunk.end()
92 content, terminator = chunk.groups()
93 # Content is contains zero or more unescaped string characters
94 if content:
95 if not _PY3 and not isinstance(content, unicode):
96 content = unicode(content, encoding)
97 _append(content)
98 # Terminator is the end of string, a literal control character,
99 # or a backslash denoting that an escape sequence follows
100 if terminator == '"':
101 break
102 elif terminator != '\\':
103 if strict:
104 msg = "Invalid control character %r at"
105 raise JSONDecodeError(msg, s, end - 1)
106 else:
107 _append(terminator)
108 continue
109 try:
110 esc = s[end]
111 except IndexError:
112 raise JSONDecodeError(
113 "Unterminated string starting at", s, begin)
114 # If not a unicode escape sequence, must be in the lookup table
115 if esc != 'u':
116 try:
117 char = _b[esc]
118 except KeyError:
119 msg = "Invalid \\X escape sequence %r"
120 raise JSONDecodeError(msg, s, end)
121 end += 1
122 else:
123 # Unicode escape sequence
124 uni, end = _scan_four_digit_hex(s, end + 1)
125 # Check for surrogate pair on UCS-4 systems
126 # Note that this will join high/low surrogate pairs
127 # but will also pass unpaired surrogates through
128 if (_maxunicode > 65535 and
129 uni & 0xfc00 == 0xd800 and
130 s[end:end + 2] == '\\u'):
131 uni2, end2 = _scan_four_digit_hex(s, end + 2)
132 if uni2 & 0xfc00 == 0xdc00:
133 uni = 0x10000 + (((uni - 0xd800) << 10) |
134 (uni2 - 0xdc00))
135 end = end2
136 char = unichr(uni)
137 # Append the unescaped character
138 _append(char)
139 return _join(chunks), end
140
141
142# Use speedup if available
143scanstring = c_scanstring or py_scanstring
144
145WHITESPACE = re.compile(r'[ \t\n\r]*', FLAGS)
146WHITESPACE_STR = ' \t\n\r'
147
148def JSONObject(state, encoding, strict, scan_once, object_hook,
149 object_pairs_hook, memo=None,
150 _w=WHITESPACE.match, _ws=WHITESPACE_STR):
151 (s, end) = state
152 # Backwards compatibility
153 if memo is None:
154 memo = {}
155 memo_get = memo.setdefault
156 pairs = []
157 # Use a slice to prevent IndexError from being raised, the following
158 # check will raise a more specific ValueError if the string is empty
159 nextchar = s[end:end + 1]
160 # Normally we expect nextchar == '"'
161 if nextchar != '"':
162 if nextchar in _ws:
163 end = _w(s, end).end()
164 nextchar = s[end:end + 1]
165 # Trivial empty object
166 if nextchar == '}':
167 if object_pairs_hook is not None:
168 result = object_pairs_hook(pairs)
169 return result, end + 1
170 pairs = {}
171 if object_hook is not None:
172 pairs = object_hook(pairs)
173 return pairs, end + 1
174 elif nextchar != '"':
175 raise JSONDecodeError(
176 "Expecting property name enclosed in double quotes or '}'",
177 s, end)
178 end += 1
179 while True:
180 key, end = scanstring(s, end, encoding, strict)
181 key = memo_get(key, key)
182
183 # To skip some function call overhead we optimize the fast paths where
184 # the JSON key separator is ": " or just ":".
185 if s[end:end + 1] != ':':
186 end = _w(s, end).end()
187 if s[end:end + 1] != ':':
188 raise JSONDecodeError("Expecting ':' delimiter", s, end)
189
190 end += 1
191
192 try:
193 if s[end] in _ws:
194 end += 1
195 if s[end] in _ws:
196 end = _w(s, end + 1).end()
197 except IndexError:
198 pass
199
200 value, end = scan_once(s, end)
201 pairs.append((key, value))
202
203 try:
204 nextchar = s[end]
205 if nextchar in _ws:
206 end = _w(s, end + 1).end()
207 nextchar = s[end]
208 except IndexError:
209 nextchar = ''
210 end += 1
211
212 if nextchar == '}':
213 break
214 elif nextchar != ',':
215 raise JSONDecodeError("Expecting ',' delimiter or '}'", s, end - 1)
216 comma_idx = end - 1
217
218 try:
219 nextchar = s[end]
220 if nextchar in _ws:
221 end += 1
222 nextchar = s[end]
223 if nextchar in _ws:
224 end = _w(s, end + 1).end()
225 nextchar = s[end]
226 except IndexError:
227 nextchar = ''
228
229 end += 1
230 if nextchar != '"':
231 if nextchar == '}':
232 raise JSONDecodeError(
233 "Illegal trailing comma before end of object",
234 s, comma_idx)
235 raise JSONDecodeError(
236 "Expecting property name enclosed in double quotes",
237 s, end - 1)
238
239 if object_pairs_hook is not None:
240 result = object_pairs_hook(pairs)
241 return result, end
242 pairs = dict(pairs)
243 if object_hook is not None:
244 pairs = object_hook(pairs)
245 return pairs, end
246
247def JSONArray(state, scan_once, array_hook=None,
248 _w=WHITESPACE.match, _ws=WHITESPACE_STR):
249 (s, end) = state
250 values = []
251 nextchar = s[end:end + 1]
252 if nextchar in _ws:
253 end = _w(s, end + 1).end()
254 nextchar = s[end:end + 1]
255 # Look-ahead for trivial empty array
256 if nextchar == ']':
257 if array_hook is not None:
258 values = array_hook(values)
259 return values, end + 1
260 elif nextchar == '':
261 raise JSONDecodeError("Expecting value or ']'", s, end)
262 _append = values.append
263 while True:
264 value, end = scan_once(s, end)
265 _append(value)
266 nextchar = s[end:end + 1]
267 if nextchar in _ws:
268 end = _w(s, end + 1).end()
269 nextchar = s[end:end + 1]
270 end += 1
271 if nextchar == ']':
272 break
273 elif nextchar != ',':
274 raise JSONDecodeError("Expecting ',' delimiter or ']'", s, end - 1)
275 comma_idx = end - 1
276
277 try:
278 if s[end] in _ws:
279 end += 1
280 if s[end] in _ws:
281 end = _w(s, end + 1).end()
282 except IndexError:
283 pass
284
285 if s[end:end + 1] == ']':
286 raise JSONDecodeError(
287 "Illegal trailing comma before end of array",
288 s, comma_idx)
289
290 if array_hook is not None:
291 values = array_hook(values)
292 return values, end
293
294class JSONDecoder(object):
295 """Simple JSON <http://json.org> decoder
296
297 Performs the following translations in decoding by default:
298
299 +---------------+-------------------+
300 | JSON | Python |
301 +===============+===================+
302 | object | dict |
303 +---------------+-------------------+
304 | array | list |
305 +---------------+-------------------+
306 | string | str, unicode |
307 +---------------+-------------------+
308 | number (int) | int, long |
309 +---------------+-------------------+
310 | number (real) | float |
311 +---------------+-------------------+
312 | true | True |
313 +---------------+-------------------+
314 | false | False |
315 +---------------+-------------------+
316 | null | None |
317 +---------------+-------------------+
318
319 When allow_nan=True, it also understands
320 ``NaN``, ``Infinity``, and ``-Infinity`` as
321 their corresponding ``float`` values, which is outside the JSON spec.
322
323 """
324
325 def __init__(self, encoding=None, object_hook=None, parse_float=None,
326 parse_int=None, parse_constant=None, strict=True,
327 object_pairs_hook=None, allow_nan=False,
328 array_hook=None):
329 """
330 *encoding* determines the encoding used to interpret any
331 :class:`str` objects decoded by this instance (``'utf-8'`` by
332 default). It has no effect when decoding :class:`unicode` objects.
333
334 Note that currently only encodings that are a superset of ASCII work,
335 strings of other encodings should be passed in as :class:`unicode`.
336
337 *object_hook*, if specified, will be called with the result of every
338 JSON object decoded and its return value will be used in place of the
339 given :class:`dict`. This can be used to provide custom
340 deserializations (e.g. to support JSON-RPC class hinting).
341
342 *object_pairs_hook* is an optional function that will be called with
343 the result of any object literal decode with an ordered list of pairs.
344 The return value of *object_pairs_hook* will be used instead of the
345 :class:`dict`. This feature can be used to implement custom decoders
346 that rely on the order that the key and value pairs are decoded (for
347 example, :func:`collections.OrderedDict` will remember the order of
348 insertion). If *object_hook* is also defined, the *object_pairs_hook*
349 takes priority.
350
351 *parse_float*, if specified, will be called with the string of every
352 JSON float to be decoded. By default, this is equivalent to
353 ``float(num_str)``. This can be used to use another datatype or parser
354 for JSON floats (e.g. :class:`decimal.Decimal`).
355
356 *parse_int*, if specified, will be called with the string of every
357 JSON int to be decoded. By default, this is equivalent to
358 ``int(num_str)``. This can be used to use another datatype or parser
359 for JSON integers (e.g. :class:`float`).
360
361 *allow_nan*, if True (default false), will allow the parser to
362 accept the non-standard floats ``NaN``, ``Infinity``, and ``-Infinity``.
363
364 *parse_constant*, if specified, will be
365 called with one of the following strings: ``'-Infinity'``,
366 ``'Infinity'``, ``'NaN'``. It is not recommended to use this feature,
367 as it is rare to parse non-compliant JSON containing these values.
368
369 *strict* controls the parser's behavior when it encounters an
370 invalid control character in a string. The default setting of
371 ``True`` means that unescaped control characters are parse errors, if
372 ``False`` then control characters will be allowed in strings.
373
374 """
375 if encoding is None:
376 encoding = DEFAULT_ENCODING
377 self.encoding = encoding
378 self.object_hook = object_hook
379 self.object_pairs_hook = object_pairs_hook
380 self.parse_float = parse_float or float
381 self.parse_int = parse_int or bounded_int
382 self.parse_constant = parse_constant or (allow_nan and _CONSTANTS.__getitem__ or None)
383 self.strict = strict
384 self.array_hook = array_hook
385 self.parse_object = JSONObject
386 self.parse_array = JSONArray
387 self.parse_string = scanstring
388 self.memo = {}
389 self.scan_once = make_scanner(self)
390
391 def decode(self, s, _w=WHITESPACE.match, _PY3=PY3):
392 """Return the Python representation of ``s`` (a ``str`` or ``unicode``
393 instance containing a JSON document)
394
395 """
396 if _PY3 and isinstance(s, bytes):
397 s = str(s, self.encoding)
398 obj, end = self.raw_decode(s)
399 end = _w(s, end).end()
400 if end != len(s):
401 raise JSONDecodeError("Extra data", s, end, len(s))
402 return obj
403
404 def raw_decode(self, s, idx=0, _w=WHITESPACE.match, _PY3=PY3):
405 """Decode a JSON document from ``s`` (a ``str`` or ``unicode``
406 beginning with a JSON document) and return a 2-tuple of the Python
407 representation and the index in ``s`` where the document ended.
408 Optionally, ``idx`` can be used to specify an offset in ``s`` where
409 the JSON document begins.
410
411 This can be used to decode a JSON document from a string that may
412 have extraneous data at the end.
413
414 """
415 if idx < 0:
416 # Ensure that raw_decode bails on negative indexes, the regex
417 # would otherwise mask this behavior. #98
418 raise JSONDecodeError('Expecting value', s, idx)
419 if _PY3 and not isinstance(s, str):
420 raise TypeError("Input string must be text, not bytes")
421 # strip UTF-8 bom
422 if len(s) > idx:
423 ord0 = ord(s[idx])
424 if ord0 == 0xfeff:
425 idx += 1
426 elif ord0 == 0xef and s[idx:idx + 3] == '\xef\xbb\xbf':
427 idx += 3
428 return self.scan_once(s, idx=_w(s, idx).end())