/src/cpython3/Parser/tokenizer/decoder.c
Line | Count | Source |
1 | | #include "Python.h" |
2 | | #include "pycore_codecs.h" |
3 | | #include "pycore_global_strings.h" |
4 | | #include "pycore_runtime.h" |
5 | | #include "errcode.h" |
6 | | |
7 | | #include "reader_internal.h" |
8 | | #include "helpers.h" |
9 | | #include "../lexer/state.h" |
10 | | |
11 | | char * |
12 | | _PyTok_CopyBytes(const char *data, Py_ssize_t len) |
13 | 20.7k | { |
14 | 20.7k | if (len < 0 || len == PY_SSIZE_T_MAX) { |
15 | 0 | PyErr_NoMemory(); |
16 | 0 | return NULL; |
17 | 0 | } |
18 | 20.7k | char *copy = PyMem_Malloc((size_t)len + 1); |
19 | 20.7k | if (copy == NULL) { |
20 | 0 | PyErr_NoMemory(); |
21 | 0 | return NULL; |
22 | 0 | } |
23 | 20.7k | memcpy(copy, data, (size_t)len); |
24 | 20.7k | copy[len] = '\0'; |
25 | 20.7k | return copy; |
26 | 20.7k | } |
27 | | |
28 | | static void |
29 | | chunk_release_data(_PyTok_Chunk *chunk) |
30 | 496k | { |
31 | 496k | switch (chunk->ownership) { |
32 | 493k | case _PYTOK_CHUNK_BORROWED: |
33 | 493k | break; |
34 | 0 | case _PYTOK_CHUNK_PYMEM: |
35 | 0 | PyMem_Free(chunk->data); |
36 | 0 | break; |
37 | 3.17k | case _PYTOK_CHUNK_PYOBJECT: |
38 | 3.17k | Py_DECREF(chunk->owner); |
39 | 3.17k | break; |
40 | 496k | } |
41 | 496k | } |
42 | | |
43 | | void |
44 | | _PyTok_ChunkClear(_PyTok_Chunk *chunk) |
45 | 493k | { |
46 | 493k | chunk_release_data(chunk); |
47 | 493k | *chunk = (_PyTok_Chunk){0}; |
48 | 493k | } |
49 | | |
50 | | static int |
51 | | chunk_set_unicode(struct tok_state *tok, _PyTok_Chunk *chunk, |
52 | | PyObject *unicode, int strip_bom) |
53 | 3.20k | { |
54 | 3.20k | Py_ssize_t utf8_len; |
55 | 3.20k | const char *utf8 = PyUnicode_AsUTF8AndSize(unicode, &utf8_len); |
56 | 3.20k | if (utf8 == NULL) { |
57 | 34 | Py_DECREF(unicode); |
58 | 34 | tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) |
59 | 34 | ? E_NOMEM : E_DECODE; |
60 | 34 | return -1; |
61 | 34 | } |
62 | 3.17k | if (strip_bom && PyUnicode_GET_LENGTH(unicode) > 0 && |
63 | 0 | PyUnicode_ReadChar(unicode, 0) == 0xFEFF) { |
64 | 0 | utf8 += 3; |
65 | 0 | utf8_len -= 3; |
66 | 0 | } |
67 | 3.17k | chunk_release_data(chunk); |
68 | 3.17k | chunk->owner = unicode; |
69 | 3.17k | chunk->data = (char *)utf8; |
70 | 3.17k | chunk->len = utf8_len; |
71 | 3.17k | chunk->ownership = _PYTOK_CHUNK_PYOBJECT; |
72 | 3.17k | return 0; |
73 | 3.20k | } |
74 | | |
75 | | char * |
76 | | _PyTok_NormalizeNewlines(const char *data, Py_ssize_t len, int preserve_crlf, |
77 | | int add_final_newline, Py_ssize_t *out_len, |
78 | | int *implicit_newline) |
79 | 153k | { |
80 | 153k | if (len > PY_SSIZE_T_MAX - 2) { |
81 | 0 | PyErr_NoMemory(); |
82 | 0 | return NULL; |
83 | 0 | } |
84 | 153k | char *result = PyMem_Malloc((size_t)len + 2); |
85 | 153k | if (result == NULL) { |
86 | 0 | PyErr_NoMemory(); |
87 | 0 | return NULL; |
88 | 0 | } |
89 | 153k | Py_ssize_t write = 0; |
90 | 19.1M | for (Py_ssize_t read = 0; read < len; read++) { |
91 | 19.0M | char c = data[read]; |
92 | 19.0M | if (!preserve_crlf && c == '\r') { |
93 | 139k | if (read + 1 < len && data[read + 1] == '\n') { |
94 | 924 | read++; |
95 | 924 | } |
96 | 139k | c = '\n'; |
97 | 139k | } |
98 | 19.0M | result[write++] = c; |
99 | 19.0M | } |
100 | 153k | int implicit = add_final_newline && write > 0 && result[write - 1] != '\n'; |
101 | 153k | if (implicit) { |
102 | 13.8k | result[write++] = '\n'; |
103 | 13.8k | } |
104 | 153k | result[write] = '\0'; |
105 | 153k | *out_len = write; |
106 | 153k | *implicit_newline = implicit; |
107 | 153k | return result; |
108 | 153k | } |
109 | | |
110 | | int |
111 | | _PyTok_SetEncoding(struct tok_state *tok, const char *encoding) |
112 | 15.8k | { |
113 | 15.8k | char *copy = _PyTok_CopyBytes(encoding, strlen(encoding)); |
114 | 15.8k | if (copy == NULL) { |
115 | 0 | tok->done = E_NOMEM; |
116 | 0 | return -1; |
117 | 0 | } |
118 | 15.8k | PyMem_Free(tok->encoding); |
119 | 15.8k | tok->encoding = copy; |
120 | 15.8k | return 0; |
121 | 15.8k | } |
122 | | |
123 | | static int |
124 | | find_cookie(const char *line, Py_ssize_t len, char **encoding, int *scan_next) |
125 | 14.2k | { |
126 | 14.2k | Py_ssize_t i = 0; |
127 | 14.2k | *encoding = NULL; |
128 | 14.2k | *scan_next = 1; |
129 | 16.1k | for (; i < len; i++) { |
130 | 16.0k | if (line[i] == '#') { |
131 | 5.07k | break; |
132 | 5.07k | } |
133 | 11.0k | if (line[i] == '\n' || line[i] == '\r') { |
134 | 843 | return 0; |
135 | 843 | } |
136 | 10.1k | if (line[i] != ' ' && line[i] != '\t' && line[i] != '\f') { |
137 | 8.28k | *scan_next = 0; |
138 | 8.28k | return 0; |
139 | 8.28k | } |
140 | 10.1k | } |
141 | 218k | for (; i + 6 < len; i++) { |
142 | 217k | if (memcmp(line + i, "coding", 6) != 0) { |
143 | 212k | continue; |
144 | 212k | } |
145 | 5.88k | const char *cursor = line + i + 6; |
146 | 5.88k | if (*cursor != ':' && *cursor != '=') { |
147 | 428 | continue; |
148 | 428 | } |
149 | 10.5k | do { |
150 | 10.5k | cursor++; |
151 | 10.5k | } while (cursor < line + len && |
152 | 10.5k | (*cursor == ' ' || *cursor == '\t')); |
153 | 5.45k | const char *start = cursor; |
154 | 50.3k | while (cursor < line + len && |
155 | 50.1k | (Py_ISALNUM(*cursor) || *cursor == '-' || |
156 | 44.8k | *cursor == '_' || *cursor == '.')) { |
157 | 44.8k | cursor++; |
158 | 44.8k | } |
159 | 5.45k | if (cursor == start) { |
160 | 613 | continue; |
161 | 613 | } |
162 | 4.84k | char *found = _PyTok_CopyBytes(start, cursor - start); |
163 | 4.84k | if (found == NULL) { |
164 | 0 | return -1; |
165 | 0 | } |
166 | 4.84k | char normalized[13]; |
167 | 4.84k | int n; |
168 | 30.0k | for (n = 0; n < 12 && found[n] != '\0'; n++) { |
169 | 25.2k | normalized[n] = found[n] == '_' ? '-' : Py_TOLOWER(found[n]); |
170 | 25.2k | } |
171 | 4.84k | normalized[n] = '\0'; |
172 | 4.84k | const char *canonical = found; |
173 | 4.84k | if (strcmp(normalized, "utf-8") == 0 || |
174 | 4.83k | strncmp(normalized, "utf-8-", 6) == 0) { |
175 | 3 | canonical = "utf-8"; |
176 | 3 | } |
177 | 4.83k | else if (strcmp(normalized, "latin-1") == 0 || |
178 | 4.82k | strcmp(normalized, "iso-8859-1") == 0 || |
179 | 4.82k | strcmp(normalized, "iso-latin-1") == 0 || |
180 | 4.82k | strncmp(normalized, "latin-1-", 8) == 0 || |
181 | 4.81k | strncmp(normalized, "iso-8859-1-", 11) == 0 || |
182 | 4.81k | strncmp(normalized, "iso-latin-1-", 12) == 0) { |
183 | 21 | canonical = "iso-8859-1"; |
184 | 21 | } |
185 | 4.84k | if (canonical != found) { |
186 | 24 | PyMem_Free(found); |
187 | 24 | found = _PyTok_CopyBytes(canonical, strlen(canonical)); |
188 | 24 | if (found == NULL) { |
189 | 0 | return -1; |
190 | 0 | } |
191 | 24 | } |
192 | 4.84k | *encoding = found; |
193 | 4.84k | *scan_next = 0; |
194 | 4.84k | return 0; |
195 | 4.84k | } |
196 | 269 | return 0; |
197 | 5.10k | } |
198 | | |
199 | | _PyTok_EncodingResult |
200 | | _PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first, |
201 | | const _PyTok_Chunk *second, int final, |
202 | | Py_ssize_t *bom_len) |
203 | 13.5k | { |
204 | 13.5k | int bom = first->len >= 3 && |
205 | 12.7k | (unsigned char)first->data[0] == 0xEF && |
206 | 12 | (unsigned char)first->data[1] == 0xBB && |
207 | 5 | (unsigned char)first->data[2] == 0xBF; |
208 | 13.5k | *bom_len = bom ? 3 : 0; |
209 | | |
210 | 13.5k | char *cookie = NULL; |
211 | 13.5k | int scan_next = 0; |
212 | 13.5k | int cookie_line = 1; |
213 | 13.5k | const char *first_data = first->data + (bom ? 3 : 0); |
214 | 13.5k | Py_ssize_t first_len = first->len - (bom ? 3 : 0); |
215 | 13.5k | if (find_cookie(first_data, first_len, &cookie, &scan_next) < 0) { |
216 | 0 | return _PYTOK_ENCODING_ERROR; |
217 | 0 | } |
218 | 13.5k | if (cookie == NULL && scan_next && second != NULL) { |
219 | 695 | if (find_cookie(second->data, second->len, &cookie, &scan_next) < 0) { |
220 | 0 | return _PYTOK_ENCODING_ERROR; |
221 | 0 | } |
222 | 695 | cookie_line = 2; |
223 | 695 | } |
224 | 12.8k | else if (cookie == NULL && scan_next && !final) { |
225 | 0 | return _PYTOK_ENCODING_NEED_SECOND_LINE; |
226 | 0 | } |
227 | | |
228 | 13.5k | if (bom) { |
229 | 4 | if (_PyTok_SetEncoding(tok, "utf-8") < 0) { |
230 | 0 | PyMem_Free(cookie); |
231 | 0 | return _PYTOK_ENCODING_ERROR; |
232 | 0 | } |
233 | 4 | } |
234 | 13.5k | if (cookie == NULL) { |
235 | 8.70k | return _PYTOK_ENCODING_DONE; |
236 | 8.70k | } |
237 | 4.84k | if (bom && strcmp(cookie, "utf-8") != 0) { |
238 | 2 | const _PyTok_Chunk *line = cookie_line == 2 ? second : first; |
239 | 2 | const char *line_data = line->data + (cookie_line == 1 ? 3 : 0); |
240 | 2 | Py_ssize_t line_len = line->len - (cookie_line == 1 ? 3 : 0); |
241 | 2 | const char *saved_line_start = tok->line_start; |
242 | 2 | char *saved_cur = tok->cur; |
243 | 2 | int saved_lineno = tok->lineno; |
244 | 2 | tok->line_start = line_data; |
245 | 2 | tok->cur = (char *)line_data; |
246 | 2 | tok->lineno = cookie_line; |
247 | 2 | int end_col = (int)Py_MIN(line_len, INT_MAX); |
248 | 2 | if (end_col > 0 && (line_data[end_col - 1] == '\n' || |
249 | 2 | line_data[end_col - 1] == '\r')) { |
250 | 0 | end_col--; |
251 | 0 | } |
252 | 2 | _PyTokenizer_syntaxerror_known_range( |
253 | 2 | tok, 0, end_col, "encoding problem: %s with BOM", cookie); |
254 | 2 | tok->line_start = saved_line_start; |
255 | 2 | tok->cur = saved_cur; |
256 | 2 | tok->lineno = saved_lineno; |
257 | 2 | PyMem_Free(cookie); |
258 | 2 | return _PYTOK_ENCODING_ERROR; |
259 | 2 | } |
260 | 4.83k | if (!bom && _PyTok_SetEncoding(tok, cookie) < 0) { |
261 | 0 | PyMem_Free(cookie); |
262 | 0 | return _PYTOK_ENCODING_ERROR; |
263 | 0 | } |
264 | 4.83k | PyMem_Free(cookie); |
265 | 4.83k | return _PYTOK_ENCODING_DONE; |
266 | 4.83k | } |
267 | | |
268 | | int |
269 | | _PyTok_DecodeOnce(struct tok_state *tok, _PyTok_Chunk *chunk, |
270 | | const char *encoding, const char *errors) |
271 | 4.83k | { |
272 | 4.83k | PyObject *unicode = PyUnicode_Decode( |
273 | 4.83k | chunk->data, chunk->len, encoding, errors); |
274 | 4.83k | if (unicode == NULL) { |
275 | 1.63k | tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) |
276 | 1.63k | ? E_NOMEM : E_DECODE; |
277 | 1.63k | return -1; |
278 | 1.63k | } |
279 | 3.20k | return chunk_set_unicode(tok, chunk, unicode, 0); |
280 | 4.83k | } |
281 | | |
282 | | static Py_ssize_t |
283 | | raw_line_length(const char *data, Py_ssize_t len) |
284 | 449k | { |
285 | 65.0M | for (Py_ssize_t i = 0; i < len; i++) { |
286 | 65.0M | if (data[i] == '\n') { |
287 | 275k | return i + 1; |
288 | 275k | } |
289 | 64.7M | if (data[i] == '\r') { |
290 | 142k | return i + 1 < len && data[i + 1] == '\n' ? i + 2 : i + 1; |
291 | 142k | } |
292 | 64.7M | } |
293 | 31.6k | return len; |
294 | 449k | } |
295 | | |
296 | | static int |
297 | | store_prepared_source(struct tok_state *tok, const char *data, Py_ssize_t len, |
298 | | int preserve_crlf, int add_final_newline) |
299 | 22.9k | { |
300 | 22.9k | Py_ssize_t pos = 0; |
301 | 453k | while (pos < len) { |
302 | 430k | Py_ssize_t raw_line_len; |
303 | 430k | if (preserve_crlf) { |
304 | 0 | const char *newline = memchr(data + pos, '\n', len - pos); |
305 | 0 | raw_line_len = newline == NULL |
306 | 0 | ? len - pos : newline - data - pos + 1; |
307 | 0 | } |
308 | 430k | else { |
309 | 430k | raw_line_len = raw_line_length(data + pos, len - pos); |
310 | 430k | } |
311 | 430k | int terminated = preserve_crlf |
312 | 430k | ? data[pos + raw_line_len - 1] == '\n' |
313 | 430k | : data[pos + raw_line_len - 1] == '\n' || |
314 | 160k | data[pos + raw_line_len - 1] == '\r'; |
315 | 430k | int add_newline = add_final_newline && |
316 | 406k | pos + raw_line_len == len && !terminated; |
317 | 430k | int normalize = add_newline || |
318 | 416k | (!preserve_crlf && |
319 | 416k | memchr(data + pos, '\r', raw_line_len) != NULL); |
320 | | |
321 | 430k | const char *line = data + pos; |
322 | 430k | Py_ssize_t line_len = raw_line_len; |
323 | 430k | char *normalized = NULL; |
324 | 430k | int implicit = 0; |
325 | 430k | if (normalize) { |
326 | 153k | normalized = _PyTok_NormalizeNewlines( |
327 | 153k | line, line_len, preserve_crlf, add_newline, |
328 | 153k | &line_len, &implicit); |
329 | 153k | if (normalized == NULL) { |
330 | 0 | tok->done = E_NOMEM; |
331 | 0 | return -1; |
332 | 0 | } |
333 | 153k | line = normalized; |
334 | 153k | } |
335 | 430k | _PyTok_Off appended = _PyTok_SourceAppendLine( |
336 | 430k | &tok->source, line, line_len, implicit); |
337 | 430k | PyMem_Free(normalized); |
338 | 430k | if (appended < 0) { |
339 | 0 | tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) |
340 | 0 | ? E_NOMEM : E_ERROR; |
341 | 0 | return -1; |
342 | 0 | } |
343 | 430k | pos += raw_line_len; |
344 | 430k | } |
345 | 22.9k | return 0; |
346 | 22.9k | } |
347 | | |
348 | | int |
349 | | _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, |
350 | | int exec_input, int preserve_crlf) |
351 | 24.5k | { |
352 | 24.5k | Py_ssize_t raw_len = strlen(input); |
353 | 24.5k | char *raw = (char *)input; |
354 | | |
355 | 24.5k | if (utf8_only) { |
356 | 11.0k | if (_PyTok_SetEncoding(tok, "utf-8") < 0) { |
357 | 0 | return -1; |
358 | 0 | } |
359 | 11.0k | } |
360 | 13.5k | else { |
361 | 13.5k | Py_ssize_t first_original_len = raw_line_length(raw, raw_len); |
362 | 13.5k | _PyTok_Chunk first = { |
363 | 13.5k | .data = raw, |
364 | 13.5k | .len = first_original_len, |
365 | 13.5k | .ownership = _PYTOK_CHUNK_BORROWED, |
366 | 13.5k | }; |
367 | 13.5k | _PyTok_Chunk second = {0}; |
368 | 13.5k | int have_second = first_original_len < raw_len; |
369 | 13.5k | if (have_second) { |
370 | 5.32k | second.data = raw + first_original_len; |
371 | 5.32k | second.len = raw_line_length(second.data, |
372 | 5.32k | raw_len - first_original_len); |
373 | 5.32k | } |
374 | 13.5k | Py_ssize_t bom_len; |
375 | 13.5k | _PyTok_EncodingResult detection = _PyTok_DetectEncoding( |
376 | 13.5k | tok, &first, have_second ? &second : NULL, 1, &bom_len); |
377 | 13.5k | if (detection == _PYTOK_ENCODING_ERROR) { |
378 | 2 | return -1; |
379 | 2 | } |
380 | 13.5k | raw += bom_len; |
381 | 13.5k | raw_len -= bom_len; |
382 | 13.5k | } |
383 | | |
384 | 24.5k | _PyTok_Chunk decoded = { |
385 | 24.5k | .data = raw, |
386 | 24.5k | .len = raw_len, |
387 | 24.5k | .ownership = _PYTOK_CHUNK_BORROWED, |
388 | 24.5k | }; |
389 | 24.5k | if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) { |
390 | 4.83k | if (_PyTok_DecodeOnce( |
391 | 4.83k | tok, &decoded, tok->encoding, NULL) < 0) { |
392 | 1.66k | return -1; |
393 | 1.66k | } |
394 | 4.83k | } |
395 | | |
396 | 22.9k | int stored = store_prepared_source( |
397 | 22.9k | tok, decoded.data, decoded.len, preserve_crlf, exec_input); |
398 | 22.9k | _PyTok_ChunkClear(&decoded); |
399 | 22.9k | if (stored < 0) { |
400 | 0 | return -1; |
401 | 0 | } |
402 | 22.9k | tok->str = tok->source.bytes != NULL ? tok->source.bytes : (char *)""; |
403 | 22.9k | if (!utf8_only && |
404 | 11.8k | (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && |
405 | 8.70k | !_PyTokenizer_ensure_utf8(tok->str, tok, 1)) { |
406 | 238 | return -1; |
407 | 238 | } |
408 | 22.6k | return 0; |
409 | 22.9k | } |
410 | | |
411 | | int |
412 | | _PyTok_StartDecoder(struct tok_state *tok, const char *errors) |
413 | 0 | { |
414 | 0 | _PyTok_Reader *reader = tok->reader; |
415 | 0 | if (tok->encoding == NULL || reader->decoder != NULL) { |
416 | 0 | return 0; |
417 | 0 | } |
418 | 0 | if (reader->kind == _PYTOK_READER_FILE && |
419 | 0 | strcmp(tok->encoding, "utf-8") == 0) { |
420 | 0 | return 0; |
421 | 0 | } |
422 | | |
423 | 0 | PyObject *codec = _PyCodec_LookupTextEncoding(tok->encoding, NULL); |
424 | 0 | if (codec != NULL) { |
425 | 0 | PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder"); |
426 | 0 | Py_DECREF(codec); |
427 | 0 | if (factory != NULL) { |
428 | 0 | reader->decoder = PyObject_CallFunction(factory, "s", errors); |
429 | 0 | Py_DECREF(factory); |
430 | 0 | } |
431 | 0 | } |
432 | 0 | if (reader->decoder == NULL) { |
433 | 0 | tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) |
434 | 0 | ? E_NOMEM : E_DECODE; |
435 | 0 | if (reader->kind == _PYTOK_READER_FILE) { |
436 | 0 | _PyTokenizer_raise_init_error( |
437 | 0 | tok->filename != NULL ? tok->filename : Py_None); |
438 | 0 | } |
439 | 0 | return -1; |
440 | 0 | } |
441 | 0 | return 0; |
442 | 0 | } |
443 | | |
444 | | int |
445 | | _PyTok_DecodeChunk(struct tok_state *tok, _PyTok_Chunk *chunk, int final) |
446 | 0 | { |
447 | 0 | _PyTok_Reader *reader = tok->reader; |
448 | 0 | if (reader->decoder == NULL) { |
449 | 0 | return 0; |
450 | 0 | } |
451 | 0 | int strip_bom = reader->kind == _PYTOK_READER_READLINE && |
452 | 0 | chunk->len >= 2 && |
453 | 0 | (((unsigned char)chunk->data[0] == 0xFF && |
454 | 0 | (unsigned char)chunk->data[1] == 0xFE) || |
455 | 0 | ((unsigned char)chunk->data[0] == 0xFE && |
456 | 0 | (unsigned char)chunk->data[1] == 0xFF)); |
457 | 0 | PyObject *input; |
458 | 0 | if (chunk->ownership == _PYTOK_CHUNK_PYOBJECT && |
459 | 0 | PyBytes_Check(chunk->owner) && |
460 | 0 | chunk->data == PyBytes_AS_STRING(chunk->owner)) { |
461 | 0 | input = Py_NewRef(chunk->owner); |
462 | 0 | } |
463 | 0 | else { |
464 | 0 | input = PyBytes_FromStringAndSize(chunk->data, chunk->len); |
465 | 0 | } |
466 | 0 | if (input == NULL) { |
467 | 0 | tok->done = E_NOMEM; |
468 | 0 | return -1; |
469 | 0 | } |
470 | 0 | PyObject *unicode = PyObject_CallMethodObjArgs( |
471 | 0 | reader->decoder, &_Py_ID(decode), input, |
472 | 0 | final ? Py_True : Py_False, NULL); |
473 | 0 | Py_DECREF(input); |
474 | 0 | if (unicode == NULL) { |
475 | 0 | tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) |
476 | 0 | ? E_NOMEM : E_DECODE; |
477 | 0 | if (reader->kind == _PYTOK_READER_FILE) { |
478 | 0 | _PyTokenizer_raise_init_error( |
479 | 0 | tok->filename != NULL ? tok->filename : Py_None); |
480 | 0 | } |
481 | 0 | return -1; |
482 | 0 | } |
483 | 0 | if (!PyUnicode_Check(unicode)) { |
484 | 0 | PyErr_Format(PyExc_TypeError, |
485 | 0 | "decoder should return a string result, not '%.200s'", |
486 | 0 | Py_TYPE(unicode)->tp_name); |
487 | 0 | Py_DECREF(unicode); |
488 | 0 | tok->done = E_DECODE; |
489 | 0 | return -1; |
490 | 0 | } |
491 | 0 | return chunk_set_unicode(tok, chunk, unicode, strip_bom); |
492 | 0 | } |
493 | | |
494 | | int |
495 | | _PyTok_DecoderHasBufferedInput(struct tok_state *tok) |
496 | 0 | { |
497 | 0 | if (tok->reader->decoder == NULL) { |
498 | 0 | return 0; |
499 | 0 | } |
500 | 0 | PyObject *state = PyObject_CallMethodNoArgs( |
501 | 0 | tok->reader->decoder, &_Py_ID(getstate)); |
502 | 0 | if (state == NULL) { |
503 | 0 | tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) |
504 | 0 | ? E_NOMEM : E_DECODE; |
505 | 0 | return -1; |
506 | 0 | } |
507 | 0 | if (!PyTuple_Check(state) || PyTuple_GET_SIZE(state) != 2 || |
508 | 0 | !PyBytes_Check(PyTuple_GET_ITEM(state, 0)) || |
509 | 0 | !PyLong_Check(PyTuple_GET_ITEM(state, 1))) { |
510 | 0 | Py_DECREF(state); |
511 | 0 | PyErr_SetString(PyExc_TypeError, |
512 | 0 | "incremental decoder getstate() must return (bytes, int)"); |
513 | 0 | tok->done = E_DECODE; |
514 | 0 | return -1; |
515 | 0 | } |
516 | 0 | int pending = PyBytes_GET_SIZE(PyTuple_GET_ITEM(state, 0)) != 0; |
517 | 0 | Py_DECREF(state); |
518 | 0 | return pending; |
519 | 0 | } |