Coverage Report

Created: 2026-09-06 06:49

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/cpython3/Parser/tokenizer/decoder.c
Line
Count
Source
1
#include "Python.h"
2
#include "pycore_codecs.h"
3
#include "pycore_global_strings.h"
4
#include "pycore_runtime.h"
5
#include "errcode.h"
6
7
#include "reader_internal.h"
8
#include "helpers.h"
9
#include "../lexer/state.h"
10
11
char *
12
_PyTok_CopyBytes(const char *data, Py_ssize_t len)
13
20.7k
{
14
20.7k
    if (len < 0 || len == PY_SSIZE_T_MAX) {
15
0
        PyErr_NoMemory();
16
0
        return NULL;
17
0
    }
18
20.7k
    char *copy = PyMem_Malloc((size_t)len + 1);
19
20.7k
    if (copy == NULL) {
20
0
        PyErr_NoMemory();
21
0
        return NULL;
22
0
    }
23
20.7k
    memcpy(copy, data, (size_t)len);
24
20.7k
    copy[len] = '\0';
25
20.7k
    return copy;
26
20.7k
}
27
28
static void
29
chunk_release_data(_PyTok_Chunk *chunk)
30
496k
{
31
496k
    switch (chunk->ownership) {
32
493k
        case _PYTOK_CHUNK_BORROWED:
33
493k
            break;
34
0
        case _PYTOK_CHUNK_PYMEM:
35
0
            PyMem_Free(chunk->data);
36
0
            break;
37
3.17k
        case _PYTOK_CHUNK_PYOBJECT:
38
3.17k
            Py_DECREF(chunk->owner);
39
3.17k
            break;
40
496k
    }
41
496k
}
42
43
void
44
_PyTok_ChunkClear(_PyTok_Chunk *chunk)
45
493k
{
46
493k
    chunk_release_data(chunk);
47
493k
    *chunk = (_PyTok_Chunk){0};
48
493k
}
49
50
static int
51
chunk_set_unicode(struct tok_state *tok, _PyTok_Chunk *chunk,
52
                  PyObject *unicode, int strip_bom)
53
3.20k
{
54
3.20k
    Py_ssize_t utf8_len;
55
3.20k
    const char *utf8 = PyUnicode_AsUTF8AndSize(unicode, &utf8_len);
56
3.20k
    if (utf8 == NULL) {
57
34
        Py_DECREF(unicode);
58
34
        tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)
59
34
            ? E_NOMEM : E_DECODE;
60
34
        return -1;
61
34
    }
62
3.17k
    if (strip_bom && PyUnicode_GET_LENGTH(unicode) > 0 &&
63
0
            PyUnicode_ReadChar(unicode, 0) == 0xFEFF) {
64
0
        utf8 += 3;
65
0
        utf8_len -= 3;
66
0
    }
67
3.17k
    chunk_release_data(chunk);
68
3.17k
    chunk->owner = unicode;
69
3.17k
    chunk->data = (char *)utf8;
70
3.17k
    chunk->len = utf8_len;
71
3.17k
    chunk->ownership = _PYTOK_CHUNK_PYOBJECT;
72
3.17k
    return 0;
73
3.20k
}
74
75
char *
76
_PyTok_NormalizeNewlines(const char *data, Py_ssize_t len, int preserve_crlf,
77
                         int add_final_newline, Py_ssize_t *out_len,
78
                         int *implicit_newline)
79
153k
{
80
153k
    if (len > PY_SSIZE_T_MAX - 2) {
81
0
        PyErr_NoMemory();
82
0
        return NULL;
83
0
    }
84
153k
    char *result = PyMem_Malloc((size_t)len + 2);
85
153k
    if (result == NULL) {
86
0
        PyErr_NoMemory();
87
0
        return NULL;
88
0
    }
89
153k
    Py_ssize_t write = 0;
90
19.1M
    for (Py_ssize_t read = 0; read < len; read++) {
91
19.0M
        char c = data[read];
92
19.0M
        if (!preserve_crlf && c == '\r') {
93
139k
            if (read + 1 < len && data[read + 1] == '\n') {
94
924
                read++;
95
924
            }
96
139k
            c = '\n';
97
139k
        }
98
19.0M
        result[write++] = c;
99
19.0M
    }
100
153k
    int implicit = add_final_newline && write > 0 && result[write - 1] != '\n';
101
153k
    if (implicit) {
102
13.8k
        result[write++] = '\n';
103
13.8k
    }
104
153k
    result[write] = '\0';
105
153k
    *out_len = write;
106
153k
    *implicit_newline = implicit;
107
153k
    return result;
108
153k
}
109
110
int
111
_PyTok_SetEncoding(struct tok_state *tok, const char *encoding)
112
15.8k
{
113
15.8k
    char *copy = _PyTok_CopyBytes(encoding, strlen(encoding));
114
15.8k
    if (copy == NULL) {
115
0
        tok->done = E_NOMEM;
116
0
        return -1;
117
0
    }
118
15.8k
    PyMem_Free(tok->encoding);
119
15.8k
    tok->encoding = copy;
120
15.8k
    return 0;
121
15.8k
}
122
123
static int
124
find_cookie(const char *line, Py_ssize_t len, char **encoding, int *scan_next)
125
14.2k
{
126
14.2k
    Py_ssize_t i = 0;
127
14.2k
    *encoding = NULL;
128
14.2k
    *scan_next = 1;
129
16.1k
    for (; i < len; i++) {
130
16.0k
        if (line[i] == '#') {
131
5.07k
            break;
132
5.07k
        }
133
11.0k
        if (line[i] == '\n' || line[i] == '\r') {
134
843
            return 0;
135
843
        }
136
10.1k
        if (line[i] != ' ' && line[i] != '\t' && line[i] != '\f') {
137
8.28k
            *scan_next = 0;
138
8.28k
            return 0;
139
8.28k
        }
140
10.1k
    }
141
218k
    for (; i + 6 < len; i++) {
142
217k
        if (memcmp(line + i, "coding", 6) != 0) {
143
212k
            continue;
144
212k
        }
145
5.88k
        const char *cursor = line + i + 6;
146
5.88k
        if (*cursor != ':' && *cursor != '=') {
147
428
            continue;
148
428
        }
149
10.5k
        do {
150
10.5k
            cursor++;
151
10.5k
        } while (cursor < line + len &&
152
10.5k
                 (*cursor == ' ' || *cursor == '\t'));
153
5.45k
        const char *start = cursor;
154
50.3k
        while (cursor < line + len &&
155
50.1k
                (Py_ISALNUM(*cursor) || *cursor == '-' ||
156
44.8k
                 *cursor == '_' || *cursor == '.')) {
157
44.8k
            cursor++;
158
44.8k
        }
159
5.45k
        if (cursor == start) {
160
613
            continue;
161
613
        }
162
4.84k
        char *found = _PyTok_CopyBytes(start, cursor - start);
163
4.84k
        if (found == NULL) {
164
0
            return -1;
165
0
        }
166
4.84k
        char normalized[13];
167
4.84k
        int n;
168
30.0k
        for (n = 0; n < 12 && found[n] != '\0'; n++) {
169
25.2k
            normalized[n] = found[n] == '_' ? '-' : Py_TOLOWER(found[n]);
170
25.2k
        }
171
4.84k
        normalized[n] = '\0';
172
4.84k
        const char *canonical = found;
173
4.84k
        if (strcmp(normalized, "utf-8") == 0 ||
174
4.83k
                strncmp(normalized, "utf-8-", 6) == 0) {
175
3
            canonical = "utf-8";
176
3
        }
177
4.83k
        else if (strcmp(normalized, "latin-1") == 0 ||
178
4.82k
                 strcmp(normalized, "iso-8859-1") == 0 ||
179
4.82k
                 strcmp(normalized, "iso-latin-1") == 0 ||
180
4.82k
                 strncmp(normalized, "latin-1-", 8) == 0 ||
181
4.81k
                 strncmp(normalized, "iso-8859-1-", 11) == 0 ||
182
4.81k
                 strncmp(normalized, "iso-latin-1-", 12) == 0) {
183
21
            canonical = "iso-8859-1";
184
21
        }
185
4.84k
        if (canonical != found) {
186
24
            PyMem_Free(found);
187
24
            found = _PyTok_CopyBytes(canonical, strlen(canonical));
188
24
            if (found == NULL) {
189
0
                return -1;
190
0
            }
191
24
        }
192
4.84k
        *encoding = found;
193
4.84k
        *scan_next = 0;
194
4.84k
        return 0;
195
4.84k
    }
196
269
    return 0;
197
5.10k
}
198
199
_PyTok_EncodingResult
200
_PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first,
201
                      const _PyTok_Chunk *second, int final,
202
                      Py_ssize_t *bom_len)
203
13.5k
{
204
13.5k
    int bom = first->len >= 3 &&
205
12.7k
        (unsigned char)first->data[0] == 0xEF &&
206
12
        (unsigned char)first->data[1] == 0xBB &&
207
5
        (unsigned char)first->data[2] == 0xBF;
208
13.5k
    *bom_len = bom ? 3 : 0;
209
210
13.5k
    char *cookie = NULL;
211
13.5k
    int scan_next = 0;
212
13.5k
    int cookie_line = 1;
213
13.5k
    const char *first_data = first->data + (bom ? 3 : 0);
214
13.5k
    Py_ssize_t first_len = first->len - (bom ? 3 : 0);
215
13.5k
    if (find_cookie(first_data, first_len, &cookie, &scan_next) < 0) {
216
0
        return _PYTOK_ENCODING_ERROR;
217
0
    }
218
13.5k
    if (cookie == NULL && scan_next && second != NULL) {
219
695
        if (find_cookie(second->data, second->len, &cookie, &scan_next) < 0) {
220
0
            return _PYTOK_ENCODING_ERROR;
221
0
        }
222
695
        cookie_line = 2;
223
695
    }
224
12.8k
    else if (cookie == NULL && scan_next && !final) {
225
0
        return _PYTOK_ENCODING_NEED_SECOND_LINE;
226
0
    }
227
228
13.5k
    if (bom) {
229
4
        if (_PyTok_SetEncoding(tok, "utf-8") < 0) {
230
0
            PyMem_Free(cookie);
231
0
            return _PYTOK_ENCODING_ERROR;
232
0
        }
233
4
    }
234
13.5k
    if (cookie == NULL) {
235
8.70k
        return _PYTOK_ENCODING_DONE;
236
8.70k
    }
237
4.84k
    if (bom && strcmp(cookie, "utf-8") != 0) {
238
2
        const _PyTok_Chunk *line = cookie_line == 2 ? second : first;
239
2
        const char *line_data = line->data + (cookie_line == 1 ? 3 : 0);
240
2
        Py_ssize_t line_len = line->len - (cookie_line == 1 ? 3 : 0);
241
2
        const char *saved_line_start = tok->line_start;
242
2
        char *saved_cur = tok->cur;
243
2
        int saved_lineno = tok->lineno;
244
2
        tok->line_start = line_data;
245
2
        tok->cur = (char *)line_data;
246
2
        tok->lineno = cookie_line;
247
2
        int end_col = (int)Py_MIN(line_len, INT_MAX);
248
2
        if (end_col > 0 && (line_data[end_col - 1] == '\n' ||
249
2
                            line_data[end_col - 1] == '\r')) {
250
0
            end_col--;
251
0
        }
252
2
        _PyTokenizer_syntaxerror_known_range(
253
2
            tok, 0, end_col, "encoding problem: %s with BOM", cookie);
254
2
        tok->line_start = saved_line_start;
255
2
        tok->cur = saved_cur;
256
2
        tok->lineno = saved_lineno;
257
2
        PyMem_Free(cookie);
258
2
        return _PYTOK_ENCODING_ERROR;
259
2
    }
260
4.83k
    if (!bom && _PyTok_SetEncoding(tok, cookie) < 0) {
261
0
        PyMem_Free(cookie);
262
0
        return _PYTOK_ENCODING_ERROR;
263
0
    }
264
4.83k
    PyMem_Free(cookie);
265
4.83k
    return _PYTOK_ENCODING_DONE;
266
4.83k
}
267
268
int
269
_PyTok_DecodeOnce(struct tok_state *tok, _PyTok_Chunk *chunk,
270
                  const char *encoding, const char *errors)
271
4.83k
{
272
4.83k
    PyObject *unicode = PyUnicode_Decode(
273
4.83k
        chunk->data, chunk->len, encoding, errors);
274
4.83k
    if (unicode == NULL) {
275
1.63k
        tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)
276
1.63k
            ? E_NOMEM : E_DECODE;
277
1.63k
        return -1;
278
1.63k
    }
279
3.20k
    return chunk_set_unicode(tok, chunk, unicode, 0);
280
4.83k
}
281
282
static Py_ssize_t
283
raw_line_length(const char *data, Py_ssize_t len)
284
449k
{
285
65.0M
    for (Py_ssize_t i = 0; i < len; i++) {
286
65.0M
        if (data[i] == '\n') {
287
275k
            return i + 1;
288
275k
        }
289
64.7M
        if (data[i] == '\r') {
290
142k
            return i + 1 < len && data[i + 1] == '\n' ? i + 2 : i + 1;
291
142k
        }
292
64.7M
    }
293
31.6k
    return len;
294
449k
}
295
296
static int
297
store_prepared_source(struct tok_state *tok, const char *data, Py_ssize_t len,
298
                      int preserve_crlf, int add_final_newline)
299
22.9k
{
300
22.9k
    Py_ssize_t pos = 0;
301
453k
    while (pos < len) {
302
430k
        Py_ssize_t raw_line_len;
303
430k
        if (preserve_crlf) {
304
0
            const char *newline = memchr(data + pos, '\n', len - pos);
305
0
            raw_line_len = newline == NULL
306
0
                ? len - pos : newline - data - pos + 1;
307
0
        }
308
430k
        else {
309
430k
            raw_line_len = raw_line_length(data + pos, len - pos);
310
430k
        }
311
430k
        int terminated = preserve_crlf
312
430k
            ? data[pos + raw_line_len - 1] == '\n'
313
430k
            : data[pos + raw_line_len - 1] == '\n' ||
314
160k
              data[pos + raw_line_len - 1] == '\r';
315
430k
        int add_newline = add_final_newline &&
316
406k
            pos + raw_line_len == len && !terminated;
317
430k
        int normalize = add_newline ||
318
416k
            (!preserve_crlf &&
319
416k
             memchr(data + pos, '\r', raw_line_len) != NULL);
320
321
430k
        const char *line = data + pos;
322
430k
        Py_ssize_t line_len = raw_line_len;
323
430k
        char *normalized = NULL;
324
430k
        int implicit = 0;
325
430k
        if (normalize) {
326
153k
            normalized = _PyTok_NormalizeNewlines(
327
153k
                line, line_len, preserve_crlf, add_newline,
328
153k
                &line_len, &implicit);
329
153k
            if (normalized == NULL) {
330
0
                tok->done = E_NOMEM;
331
0
                return -1;
332
0
            }
333
153k
            line = normalized;
334
153k
        }
335
430k
        _PyTok_Off appended = _PyTok_SourceAppendLine(
336
430k
            &tok->source, line, line_len, implicit);
337
430k
        PyMem_Free(normalized);
338
430k
        if (appended < 0) {
339
0
            tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)
340
0
                ? E_NOMEM : E_ERROR;
341
0
            return -1;
342
0
        }
343
430k
        pos += raw_line_len;
344
430k
    }
345
22.9k
    return 0;
346
22.9k
}
347
348
int
349
_PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only,
350
                     int exec_input, int preserve_crlf)
351
24.5k
{
352
24.5k
    Py_ssize_t raw_len = strlen(input);
353
24.5k
    char *raw = (char *)input;
354
355
24.5k
    if (utf8_only) {
356
11.0k
        if (_PyTok_SetEncoding(tok, "utf-8") < 0) {
357
0
            return -1;
358
0
        }
359
11.0k
    }
360
13.5k
    else {
361
13.5k
        Py_ssize_t first_original_len = raw_line_length(raw, raw_len);
362
13.5k
        _PyTok_Chunk first = {
363
13.5k
            .data = raw,
364
13.5k
            .len = first_original_len,
365
13.5k
            .ownership = _PYTOK_CHUNK_BORROWED,
366
13.5k
        };
367
13.5k
        _PyTok_Chunk second = {0};
368
13.5k
        int have_second = first_original_len < raw_len;
369
13.5k
        if (have_second) {
370
5.32k
            second.data = raw + first_original_len;
371
5.32k
            second.len = raw_line_length(second.data,
372
5.32k
                                         raw_len - first_original_len);
373
5.32k
        }
374
13.5k
        Py_ssize_t bom_len;
375
13.5k
        _PyTok_EncodingResult detection = _PyTok_DetectEncoding(
376
13.5k
            tok, &first, have_second ? &second : NULL, 1, &bom_len);
377
13.5k
        if (detection == _PYTOK_ENCODING_ERROR) {
378
2
            return -1;
379
2
        }
380
13.5k
        raw += bom_len;
381
13.5k
        raw_len -= bom_len;
382
13.5k
    }
383
384
24.5k
    _PyTok_Chunk decoded = {
385
24.5k
        .data = raw,
386
24.5k
        .len = raw_len,
387
24.5k
        .ownership = _PYTOK_CHUNK_BORROWED,
388
24.5k
    };
389
24.5k
    if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) {
390
4.83k
        if (_PyTok_DecodeOnce(
391
4.83k
                tok, &decoded, tok->encoding, NULL) < 0) {
392
1.66k
            return -1;
393
1.66k
        }
394
4.83k
    }
395
396
22.9k
    int stored = store_prepared_source(
397
22.9k
        tok, decoded.data, decoded.len, preserve_crlf, exec_input);
398
22.9k
    _PyTok_ChunkClear(&decoded);
399
22.9k
    if (stored < 0) {
400
0
        return -1;
401
0
    }
402
22.9k
    tok->str = tok->source.bytes != NULL ? tok->source.bytes : (char *)"";
403
22.9k
    if (!utf8_only &&
404
11.8k
            (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) &&
405
8.70k
            !_PyTokenizer_ensure_utf8(tok->str, tok, 1)) {
406
238
        return -1;
407
238
    }
408
22.6k
    return 0;
409
22.9k
}
410
411
int
412
_PyTok_StartDecoder(struct tok_state *tok, const char *errors)
413
0
{
414
0
    _PyTok_Reader *reader = tok->reader;
415
0
    if (tok->encoding == NULL || reader->decoder != NULL) {
416
0
        return 0;
417
0
    }
418
0
    if (reader->kind == _PYTOK_READER_FILE &&
419
0
            strcmp(tok->encoding, "utf-8") == 0) {
420
0
        return 0;
421
0
    }
422
423
0
    PyObject *codec = _PyCodec_LookupTextEncoding(tok->encoding, NULL);
424
0
    if (codec != NULL) {
425
0
        PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder");
426
0
        Py_DECREF(codec);
427
0
        if (factory != NULL) {
428
0
            reader->decoder = PyObject_CallFunction(factory, "s", errors);
429
0
            Py_DECREF(factory);
430
0
        }
431
0
    }
432
0
    if (reader->decoder == NULL) {
433
0
        tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)
434
0
            ? E_NOMEM : E_DECODE;
435
0
        if (reader->kind == _PYTOK_READER_FILE) {
436
0
            _PyTokenizer_raise_init_error(
437
0
                tok->filename != NULL ? tok->filename : Py_None);
438
0
        }
439
0
        return -1;
440
0
    }
441
0
    return 0;
442
0
}
443
444
int
445
_PyTok_DecodeChunk(struct tok_state *tok, _PyTok_Chunk *chunk, int final)
446
0
{
447
0
    _PyTok_Reader *reader = tok->reader;
448
0
    if (reader->decoder == NULL) {
449
0
        return 0;
450
0
    }
451
0
    int strip_bom = reader->kind == _PYTOK_READER_READLINE &&
452
0
        chunk->len >= 2 &&
453
0
        (((unsigned char)chunk->data[0] == 0xFF &&
454
0
          (unsigned char)chunk->data[1] == 0xFE) ||
455
0
         ((unsigned char)chunk->data[0] == 0xFE &&
456
0
          (unsigned char)chunk->data[1] == 0xFF));
457
0
    PyObject *input;
458
0
    if (chunk->ownership == _PYTOK_CHUNK_PYOBJECT &&
459
0
            PyBytes_Check(chunk->owner) &&
460
0
            chunk->data == PyBytes_AS_STRING(chunk->owner)) {
461
0
        input = Py_NewRef(chunk->owner);
462
0
    }
463
0
    else {
464
0
        input = PyBytes_FromStringAndSize(chunk->data, chunk->len);
465
0
    }
466
0
    if (input == NULL) {
467
0
        tok->done = E_NOMEM;
468
0
        return -1;
469
0
    }
470
0
    PyObject *unicode = PyObject_CallMethodObjArgs(
471
0
        reader->decoder, &_Py_ID(decode), input,
472
0
        final ? Py_True : Py_False, NULL);
473
0
    Py_DECREF(input);
474
0
    if (unicode == NULL) {
475
0
        tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)
476
0
            ? E_NOMEM : E_DECODE;
477
0
        if (reader->kind == _PYTOK_READER_FILE) {
478
0
            _PyTokenizer_raise_init_error(
479
0
                tok->filename != NULL ? tok->filename : Py_None);
480
0
        }
481
0
        return -1;
482
0
    }
483
0
    if (!PyUnicode_Check(unicode)) {
484
0
        PyErr_Format(PyExc_TypeError,
485
0
                     "decoder should return a string result, not '%.200s'",
486
0
                     Py_TYPE(unicode)->tp_name);
487
0
        Py_DECREF(unicode);
488
0
        tok->done = E_DECODE;
489
0
        return -1;
490
0
    }
491
0
    return chunk_set_unicode(tok, chunk, unicode, strip_bom);
492
0
}
493
494
int
495
_PyTok_DecoderHasBufferedInput(struct tok_state *tok)
496
0
{
497
0
    if (tok->reader->decoder == NULL) {
498
0
        return 0;
499
0
    }
500
0
    PyObject *state = PyObject_CallMethodNoArgs(
501
0
        tok->reader->decoder, &_Py_ID(getstate));
502
0
    if (state == NULL) {
503
0
        tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)
504
0
            ? E_NOMEM : E_DECODE;
505
0
        return -1;
506
0
    }
507
0
    if (!PyTuple_Check(state) || PyTuple_GET_SIZE(state) != 2 ||
508
0
            !PyBytes_Check(PyTuple_GET_ITEM(state, 0)) ||
509
0
            !PyLong_Check(PyTuple_GET_ITEM(state, 1))) {
510
0
        Py_DECREF(state);
511
0
        PyErr_SetString(PyExc_TypeError,
512
0
                        "incremental decoder getstate() must return (bytes, int)");
513
0
        tok->done = E_DECODE;
514
0
        return -1;
515
0
    }
516
0
    int pending = PyBytes_GET_SIZE(PyTuple_GET_ITEM(state, 0)) != 0;
517
0
    Py_DECREF(state);
518
0
    return pending;
519
0
}