/src/cpython3/Parser/pegen.c
Line | Count | Source |
1 | | #include <Python.h> |
2 | | #include "pycore_ast.h" // _PyAST_Validate(), |
3 | | #include "pycore_pystate.h" // _PyThreadState_GET() |
4 | | #include "pycore_parser.h" // _PYPEGEN_NSTATISTICS |
5 | | #include "pycore_pyerrors.h" // PyExc_IncompleteInputError |
6 | | #include "pycore_runtime.h" // _PyRuntime |
7 | | #include "pycore_unicodeobject.h" // _PyUnicode_InternImmortal |
8 | | #include <errcode.h> |
9 | | |
10 | | #include "lexer/lexer.h" |
11 | | #include "tokenizer/tokenizer.h" |
12 | | #include "tokenizer/helpers.h" |
13 | | #include "pegen.h" |
14 | | |
15 | 7.76M | #define IDENTIFIER_CACHE_SIZE 2048 // Must be a power of two. |
16 | 3.87M | #define IDENTIFIER_CACHE_MAX_PROBES 8 |
17 | | |
18 | | struct _identifier_cache_entry { |
19 | | const char *key; // Borrowed from arena-owned token bytes. |
20 | | Py_ssize_t len; |
21 | | Py_hash_t hash; |
22 | | PyObject *value; // Borrowed from an arena-owned identifier. |
23 | | }; |
24 | | |
25 | | // Internal parser functions |
26 | | |
27 | | asdl_stmt_seq* |
28 | | _PyPegen_interactive_exit(Parser *p) |
29 | 25 | { |
30 | 25 | if (p->errcode) { |
31 | 0 | *(p->errcode) = E_EOF; |
32 | 0 | } |
33 | 25 | return NULL; |
34 | 25 | } |
35 | | |
36 | | Py_ssize_t |
37 | | _PyPegen_byte_offset_to_character_offset_line(PyObject *line, Py_ssize_t col_offset, Py_ssize_t end_col_offset) |
38 | 0 | { |
39 | 0 | const unsigned char *data = (const unsigned char*)PyUnicode_AsUTF8(line); |
40 | |
|
41 | 0 | Py_ssize_t len = 0; |
42 | 0 | while (col_offset < end_col_offset) { |
43 | 0 | Py_UCS4 ch = data[col_offset]; |
44 | 0 | if (ch < 0x80) { |
45 | 0 | col_offset += 1; |
46 | 0 | } else if ((ch & 0xe0) == 0xc0) { |
47 | 0 | col_offset += 2; |
48 | 0 | } else if ((ch & 0xf0) == 0xe0) { |
49 | 0 | col_offset += 3; |
50 | 0 | } else if ((ch & 0xf8) == 0xf0) { |
51 | 0 | col_offset += 4; |
52 | 0 | } else { |
53 | 0 | PyErr_SetString(PyExc_ValueError, "Invalid UTF-8 sequence"); |
54 | 0 | return -1; |
55 | 0 | } |
56 | 0 | len++; |
57 | 0 | } |
58 | 0 | return len; |
59 | 0 | } |
60 | | |
61 | | Py_ssize_t |
62 | | _PyPegen_byte_offset_to_character_offset_raw(const char* str, Py_ssize_t col_offset) |
63 | 14.7k | { |
64 | 14.7k | Py_ssize_t len = (Py_ssize_t)strlen(str); |
65 | 14.7k | if (col_offset > len + 1) { |
66 | 21 | col_offset = len + 1; |
67 | 21 | } |
68 | 14.7k | assert(col_offset >= 0); |
69 | 14.7k | PyObject *text = PyUnicode_DecodeUTF8(str, col_offset, "replace"); |
70 | 14.7k | if (!text) { |
71 | 0 | return -1; |
72 | 0 | } |
73 | 14.7k | Py_ssize_t size = PyUnicode_GET_LENGTH(text); |
74 | 14.7k | Py_DECREF(text); |
75 | 14.7k | return size; |
76 | 14.7k | } |
77 | | |
78 | | Py_ssize_t |
79 | | _PyPegen_byte_offset_to_character_offset(PyObject *line, Py_ssize_t col_offset) |
80 | 14.7k | { |
81 | 14.7k | const char *str = PyUnicode_AsUTF8(line); |
82 | 14.7k | if (!str) { |
83 | 0 | return -1; |
84 | 0 | } |
85 | 14.7k | return _PyPegen_byte_offset_to_character_offset_raw(str, col_offset); |
86 | 14.7k | } |
87 | | |
88 | | // Here, mark is the start of the node, while p->mark is the end. |
89 | | // If node==NULL, they should be the same. |
90 | | int |
91 | | _PyPegen_insert_memo(Parser *p, int mark, int type, void *node) |
92 | 23.1M | { |
93 | | // Insert in front |
94 | 23.1M | Memo *m = _PyArena_Malloc(p->arena, sizeof(Memo)); |
95 | 23.1M | if (m == NULL) { |
96 | 0 | return -1; |
97 | 0 | } |
98 | 23.1M | m->type = type; |
99 | 23.1M | m->node = node; |
100 | 23.1M | m->mark = p->mark; |
101 | 23.1M | m->next = p->tokens[mark]->memo; |
102 | 23.1M | p->tokens[mark]->memo = m; |
103 | 23.1M | p->tokens[mark]->memo_mask |= 1ULL << (type & 63); |
104 | 23.1M | return 0; |
105 | 23.1M | } |
106 | | |
107 | | // Like _PyPegen_insert_memo(), but updates an existing node if found. |
108 | | int |
109 | | _PyPegen_update_memo(Parser *p, int mark, int type, void *node) |
110 | 20.5M | { |
111 | 90.7M | for (Memo *m = p->tokens[mark]->memo; m != NULL; m = m->next) { |
112 | 79.2M | if (m->type == type) { |
113 | | // Update existing node. |
114 | 9.05M | m->node = node; |
115 | 9.05M | m->mark = p->mark; |
116 | 9.05M | return 0; |
117 | 9.05M | } |
118 | 79.2M | } |
119 | | // Insert new node. |
120 | 11.4M | return _PyPegen_insert_memo(p, mark, type, node); |
121 | 20.5M | } |
122 | | |
123 | | static int |
124 | | init_normalization(Parser *p) |
125 | 14.5k | { |
126 | 14.5k | if (p->normalize) { |
127 | 11.4k | return 1; |
128 | 11.4k | } |
129 | 3.01k | p->normalize = PyImport_ImportModuleAttrString("unicodedata", "normalize"); |
130 | 3.01k | if (!p->normalize) |
131 | 0 | { |
132 | 0 | return 0; |
133 | 0 | } |
134 | 3.01k | return 1; |
135 | 3.01k | } |
136 | | |
137 | | static int |
138 | 22.6k | growable_comment_array_init(growable_comment_array *arr, size_t initial_size) { |
139 | 22.6k | assert(initial_size > 0); |
140 | 22.6k | arr->items = PyMem_Malloc(initial_size * sizeof(*arr->items)); |
141 | 22.6k | arr->size = initial_size; |
142 | 22.6k | arr->num_items = 0; |
143 | | |
144 | 22.6k | return arr->items != NULL; |
145 | 22.6k | } |
146 | | |
147 | | static int |
148 | 1.06k | growable_comment_array_add(growable_comment_array *arr, int lineno, char *comment) { |
149 | 1.06k | if (arr->num_items >= arr->size) { |
150 | 50 | size_t new_size = arr->size * 2; |
151 | 50 | void *new_items_array = PyMem_Realloc(arr->items, new_size * sizeof(*arr->items)); |
152 | 50 | if (!new_items_array) { |
153 | 0 | return 0; |
154 | 0 | } |
155 | 50 | arr->items = new_items_array; |
156 | 50 | arr->size = new_size; |
157 | 50 | } |
158 | | |
159 | 1.06k | arr->items[arr->num_items].lineno = lineno; |
160 | 1.06k | arr->items[arr->num_items].comment = comment; // Take ownership |
161 | 1.06k | arr->num_items++; |
162 | 1.06k | return 1; |
163 | 1.06k | } |
164 | | |
165 | | static void |
166 | 22.6k | growable_comment_array_deallocate(growable_comment_array *arr) { |
167 | 23.7k | for (unsigned i = 0; i < arr->num_items; i++) { |
168 | 1.06k | PyMem_Free(arr->items[i].comment); |
169 | 1.06k | } |
170 | 22.6k | PyMem_Free(arr->items); |
171 | 22.6k | } |
172 | | |
173 | | static int |
174 | | _get_keyword_or_name_type(Parser *p, struct token *new_token) |
175 | 1.07M | { |
176 | 1.07M | Py_ssize_t name_len = new_token->end_col_offset - new_token->col_offset; |
177 | 1.07M | assert(name_len > 0); |
178 | | |
179 | 1.07M | if (name_len >= p->n_keyword_lists || |
180 | 971k | p->keywords[name_len] == NULL || |
181 | 971k | p->keywords[name_len]->type == -1) { |
182 | 421k | return NAME; |
183 | 421k | } |
184 | 3.74M | for (KeywordToken *k = p->keywords[name_len]; k != NULL && k->type != -1; k++) { |
185 | 3.28M | if (strncmp(k->str, new_token->start, (size_t)name_len) == 0) { |
186 | 194k | return k->type; |
187 | 194k | } |
188 | 3.28M | } |
189 | 457k | return NAME; |
190 | 651k | } |
191 | | |
192 | | static int |
193 | 5.04M | initialize_token(Parser *p, Token *parser_token, struct token *new_token, int token_type) { |
194 | 5.04M | assert(parser_token != NULL); |
195 | | |
196 | 5.04M | parser_token->type = (token_type == NAME) ? _get_keyword_or_name_type(p, new_token) : token_type; |
197 | 5.04M | parser_token->bytes = PyBytes_FromStringAndSize(new_token->start, new_token->end - new_token->start); |
198 | 5.04M | if (parser_token->bytes == NULL) { |
199 | 0 | return -1; |
200 | 0 | } |
201 | 5.04M | if (_PyArena_AddPyObject(p->arena, parser_token->bytes) < 0) { |
202 | 0 | Py_DECREF(parser_token->bytes); |
203 | 0 | return -1; |
204 | 0 | } |
205 | | |
206 | 5.04M | parser_token->metadata = NULL; |
207 | 5.04M | if (new_token->metadata != NULL) { |
208 | 12.0k | if (_PyArena_AddPyObject(p->arena, new_token->metadata) < 0) { |
209 | 0 | Py_DECREF(new_token->metadata); |
210 | 0 | return -1; |
211 | 0 | } |
212 | 12.0k | parser_token->metadata = new_token->metadata; |
213 | 12.0k | new_token->metadata = NULL; |
214 | 12.0k | } |
215 | | |
216 | 5.04M | parser_token->level = new_token->level; |
217 | 5.04M | parser_token->lineno = new_token->lineno; |
218 | 5.04M | parser_token->col_offset = p->tok->lineno == p->starting_lineno ? p->starting_col_offset + new_token->col_offset |
219 | 5.04M | : new_token->col_offset; |
220 | 5.04M | parser_token->end_lineno = new_token->end_lineno; |
221 | 5.04M | parser_token->end_col_offset = p->tok->lineno == p->starting_lineno ? p->starting_col_offset + new_token->end_col_offset |
222 | 5.04M | : new_token->end_col_offset; |
223 | | |
224 | 5.04M | p->fill += 1; |
225 | | |
226 | 5.04M | if (token_type == ERRORTOKEN && p->tok->done == E_DECODE) { |
227 | 673 | return _Pypegen_raise_decode_error(p); |
228 | 673 | } |
229 | | |
230 | 5.04M | return (token_type == ERRORTOKEN ? _Pypegen_tokenizer_error(p) : 0); |
231 | 5.04M | } |
232 | | |
233 | | static int |
234 | 120k | _resize_tokens_array(Parser *p) { |
235 | 120k | int newsize = p->size * 2; |
236 | 120k | Token **new_tokens = PyMem_Realloc(p->tokens, (size_t)newsize * sizeof(Token *)); |
237 | 120k | if (new_tokens == NULL) { |
238 | 0 | PyErr_NoMemory(); |
239 | 0 | return -1; |
240 | 0 | } |
241 | 120k | p->tokens = new_tokens; |
242 | | |
243 | 7.36M | for (int i = p->size; i < newsize; i++) { |
244 | 7.24M | p->tokens[i] = PyMem_Calloc(1, sizeof(Token)); |
245 | 7.24M | if (p->tokens[i] == NULL) { |
246 | 0 | p->size = i; // Needed, in order to cleanup correctly after parser fails |
247 | 0 | PyErr_NoMemory(); |
248 | 0 | return -1; |
249 | 0 | } |
250 | 7.24M | } |
251 | 120k | p->size = newsize; |
252 | 120k | return 0; |
253 | 120k | } |
254 | | |
255 | | int |
256 | | _PyPegen_fill_token(Parser *p) |
257 | 5.04M | { |
258 | 5.04M | struct token new_token; |
259 | 5.04M | _PyToken_Init(&new_token); |
260 | 5.04M | int type = _PyTokenizer_Get(p->tok, &new_token); |
261 | | |
262 | | // Record and skip '# type: ignore' comments |
263 | 5.04M | while (type == TYPE_IGNORE) { |
264 | 1.06k | Py_ssize_t len = new_token.end_col_offset - new_token.col_offset; |
265 | 1.06k | char *tag = PyMem_Malloc((size_t)len + 1); |
266 | 1.06k | if (tag == NULL) { |
267 | 0 | PyErr_NoMemory(); |
268 | 0 | goto error; |
269 | 0 | } |
270 | 1.06k | strncpy(tag, new_token.start, (size_t)len); |
271 | 1.06k | tag[len] = '\0'; |
272 | | // Ownership of tag passes to the growable array |
273 | 1.06k | if (!growable_comment_array_add(&p->type_ignore_comments, p->tok->lineno, tag)) { |
274 | 0 | PyErr_NoMemory(); |
275 | 0 | goto error; |
276 | 0 | } |
277 | 1.06k | type = _PyTokenizer_Get(p->tok, &new_token); |
278 | 1.06k | } |
279 | | |
280 | | // If we have reached the end and we are in single input mode we need to insert a newline and reset the parsing |
281 | 5.04M | if (p->start_rule == Py_single_input && type == ENDMARKER && p->parsing_started) { |
282 | 2.58k | type = NEWLINE; /* Add an extra newline */ |
283 | 2.58k | p->parsing_started = 0; |
284 | | |
285 | 2.58k | if (p->tok->indent && !(p->flags & PyPARSE_DONT_IMPLY_DEDENT)) { |
286 | 95 | p->tok->pendin = -p->tok->indent; |
287 | 95 | p->tok->indent = 0; |
288 | 95 | } |
289 | 2.58k | } |
290 | 5.04M | else { |
291 | 5.04M | p->parsing_started = 1; |
292 | 5.04M | } |
293 | | |
294 | | // Check if we are at the limit of the token array capacity and resize if needed |
295 | 5.04M | if ((p->fill == p->size) && (_resize_tokens_array(p) != 0)) { |
296 | 0 | goto error; |
297 | 0 | } |
298 | | |
299 | 5.04M | Token *t = p->tokens[p->fill]; |
300 | 5.04M | return initialize_token(p, t, &new_token, type); |
301 | 0 | error: |
302 | 0 | _PyToken_Free(&new_token); |
303 | 0 | return -1; |
304 | 5.04M | } |
305 | | |
306 | | #if defined(Py_DEBUG) |
307 | | // Instrumentation to count the effectiveness of memoization. |
308 | | // The array counts the number of tokens skipped by memoization, |
309 | | // indexed by type. |
310 | | |
311 | | #define NSTATISTICS _PYPEGEN_NSTATISTICS |
312 | | #define memo_statistics _PyRuntime.parser.memo_statistics |
313 | | |
314 | | void |
315 | | _PyPegen_clear_memo_statistics(void) |
316 | | { |
317 | | PyMutex_Lock(&_PyRuntime.parser.mutex); |
318 | | for (int i = 0; i < NSTATISTICS; i++) { |
319 | | memo_statistics[i] = 0; |
320 | | } |
321 | | PyMutex_Unlock(&_PyRuntime.parser.mutex); |
322 | | } |
323 | | |
324 | | PyObject * |
325 | | _PyPegen_get_memo_statistics(void) |
326 | | { |
327 | | PyObject *ret = PyList_New(NSTATISTICS); |
328 | | if (ret == NULL) { |
329 | | return NULL; |
330 | | } |
331 | | |
332 | | PyMutex_Lock(&_PyRuntime.parser.mutex); |
333 | | for (int i = 0; i < NSTATISTICS; i++) { |
334 | | PyObject *value = PyLong_FromLong(memo_statistics[i]); |
335 | | if (value == NULL) { |
336 | | PyMutex_Unlock(&_PyRuntime.parser.mutex); |
337 | | Py_DECREF(ret); |
338 | | return NULL; |
339 | | } |
340 | | // PyList_SetItem borrows a reference to value. |
341 | | if (PyList_SetItem(ret, i, value) < 0) { |
342 | | PyMutex_Unlock(&_PyRuntime.parser.mutex); |
343 | | Py_DECREF(ret); |
344 | | return NULL; |
345 | | } |
346 | | } |
347 | | PyMutex_Unlock(&_PyRuntime.parser.mutex); |
348 | | return ret; |
349 | | } |
350 | | #endif |
351 | | |
352 | | int // bool |
353 | | _PyPegen_is_memoized(Parser *p, int type, void *pres) |
354 | 97.4M | { |
355 | 97.4M | if (p->mark == p->fill) { |
356 | 1.89M | if (_PyPegen_fill_token(p) < 0) { |
357 | 689 | p->error_indicator = 1; |
358 | 689 | return -1; |
359 | 689 | } |
360 | 1.89M | } |
361 | | |
362 | 97.4M | Token *t = p->tokens[p->mark]; |
363 | | |
364 | 97.4M | if (!(t->memo_mask & (1ULL << (type & 63)))) { |
365 | 20.3M | return 0; |
366 | 20.3M | } |
367 | | |
368 | 189M | for (Memo *m = t->memo; m != NULL; m = m->next) { |
369 | 186M | if (m->type == type) { |
370 | | #if defined(Py_DEBUG) |
371 | | if (0 <= type && type < NSTATISTICS) { |
372 | | long count = m->mark - p->mark; |
373 | | // A memoized negative result counts for one. |
374 | | if (count <= 0) { |
375 | | count = 1; |
376 | | } |
377 | | PyMutex_Lock(&_PyRuntime.parser.mutex); |
378 | | memo_statistics[type] += count; |
379 | | PyMutex_Unlock(&_PyRuntime.parser.mutex); |
380 | | } |
381 | | #endif |
382 | 74.0M | p->mark = m->mark; |
383 | 74.0M | *(void **)(pres) = m->node; |
384 | 74.0M | return 1; |
385 | 74.0M | } |
386 | 186M | } |
387 | 3.10M | return 0; |
388 | 77.1M | } |
389 | | |
390 | | #define LOOKAHEAD1(NAME, RES_TYPE) \ |
391 | | int \ |
392 | | NAME (int positive, RES_TYPE (func)(Parser *), Parser *p) \ |
393 | 5.20M | { \ |
394 | 5.20M | int mark = p->mark; \ |
395 | 5.20M | void *res = func(p); \ |
396 | 5.20M | p->mark = mark; \ |
397 | 5.20M | return (res != NULL) == positive; \ |
398 | 5.20M | } |
399 | | |
400 | 5.20M | LOOKAHEAD1(_PyPegen_lookahead, void *) |
401 | 1.02k | LOOKAHEAD1(_PyPegen_lookahead_for_expr, expr_ty) |
402 | 0 | LOOKAHEAD1(_PyPegen_lookahead_for_stmt, stmt_ty) |
403 | | #undef LOOKAHEAD1 |
404 | | |
405 | | #define LOOKAHEAD2(NAME, RES_TYPE, T) \ |
406 | | int \ |
407 | | NAME (int positive, RES_TYPE (func)(Parser *, T), Parser *p, T arg) \ |
408 | 4.75M | { \ |
409 | 4.75M | int mark = p->mark; \ |
410 | 4.75M | void *res = func(p, arg); \ |
411 | 4.75M | p->mark = mark; \ |
412 | 4.75M | return (res != NULL) == positive; \ |
413 | 4.75M | } |
414 | | |
415 | 4.45M | LOOKAHEAD2(_PyPegen_lookahead_with_int, Token *, int) |
416 | 303k | LOOKAHEAD2(_PyPegen_lookahead_with_string, expr_ty, const char *) |
417 | | #undef LOOKAHEAD2 |
418 | | |
419 | | Token * |
420 | | _PyPegen_expect_token(Parser *p, int type) |
421 | 112M | { |
422 | 112M | if (p->mark == p->fill) { |
423 | 2.53M | if (_PyPegen_fill_token(p) < 0) { |
424 | 1.93k | p->error_indicator = 1; |
425 | 1.93k | return NULL; |
426 | 1.93k | } |
427 | 2.53M | } |
428 | 112M | Token *t = p->tokens[p->mark]; |
429 | 112M | if (t->type != type) { |
430 | 99.5M | return NULL; |
431 | 99.5M | } |
432 | 12.7M | p->mark += 1; |
433 | 12.7M | return t; |
434 | 112M | } |
435 | | |
436 | | void* |
437 | 0 | _PyPegen_expect_forced_result(Parser *p, void* result, const char* expected) { |
438 | |
|
439 | 0 | if (p->error_indicator == 1) { |
440 | 0 | return NULL; |
441 | 0 | } |
442 | 0 | if (result == NULL) { |
443 | 0 | RAISE_SYNTAX_ERROR("expected (%s)", expected); |
444 | 0 | return NULL; |
445 | 0 | } |
446 | 0 | return result; |
447 | 0 | } |
448 | | |
449 | | Token * |
450 | 30.1k | _PyPegen_expect_forced_token(Parser *p, int type, const char* expected) { |
451 | | |
452 | 30.1k | if (p->error_indicator == 1) { |
453 | 0 | return NULL; |
454 | 0 | } |
455 | | |
456 | 30.1k | if (p->mark == p->fill) { |
457 | 11.0k | if (_PyPegen_fill_token(p) < 0) { |
458 | 5 | p->error_indicator = 1; |
459 | 5 | return NULL; |
460 | 5 | } |
461 | 11.0k | } |
462 | 30.1k | Token *t = p->tokens[p->mark]; |
463 | 30.1k | if (t->type != type) { |
464 | 77 | RAISE_SYNTAX_ERROR_KNOWN_LOCATION(t, "expected '%s'", expected); |
465 | 77 | return NULL; |
466 | 77 | } |
467 | 30.1k | p->mark += 1; |
468 | 30.1k | return t; |
469 | 30.1k | } |
470 | | |
471 | | expr_ty |
472 | | _PyPegen_expect_soft_keyword(Parser *p, const char *keyword) |
473 | 833k | { |
474 | 833k | if (p->mark == p->fill) { |
475 | 4.96k | if (_PyPegen_fill_token(p) < 0) { |
476 | 2 | p->error_indicator = 1; |
477 | 2 | return NULL; |
478 | 2 | } |
479 | 4.96k | } |
480 | 833k | Token *t = p->tokens[p->mark]; |
481 | 833k | if (t->type != NAME) { |
482 | 345k | return NULL; |
483 | 345k | } |
484 | 487k | const char *s = PyBytes_AsString(t->bytes); |
485 | 487k | if (!s) { |
486 | 0 | p->error_indicator = 1; |
487 | 0 | return NULL; |
488 | 0 | } |
489 | 487k | if (strcmp(s, keyword) != 0) { |
490 | 471k | return NULL; |
491 | 471k | } |
492 | 15.9k | return _PyPegen_name_token(p); |
493 | 487k | } |
494 | | |
495 | | Token * |
496 | | _PyPegen_get_last_nonnwhitespace_token(Parser *p) |
497 | 3.13M | { |
498 | 3.13M | assert(p->mark >= 0); |
499 | 3.13M | Token *token = NULL; |
500 | 3.22M | for (int m = p->mark - 1; m >= 0; m--) { |
501 | 3.22M | token = p->tokens[m]; |
502 | 3.22M | if (token->type != ENDMARKER && (token->type < NEWLINE || token->type > DEDENT)) { |
503 | 3.13M | break; |
504 | 3.13M | } |
505 | 3.22M | } |
506 | 3.13M | return token; |
507 | 3.13M | } |
508 | | |
509 | | PyObject * |
510 | | _PyPegen_new_identifier(Parser *p, const char *n) |
511 | 123k | { |
512 | 123k | PyObject *id = PyUnicode_DecodeUTF8(n, (Py_ssize_t)strlen(n), NULL); |
513 | 123k | if (!id) { |
514 | 0 | goto error; |
515 | 0 | } |
516 | | /* Check whether there are non-ASCII characters in the |
517 | | identifier; if so, normalize to NFKC. */ |
518 | 123k | if (!PyUnicode_IS_ASCII(id)) |
519 | 14.5k | { |
520 | 14.5k | if (!init_normalization(p)) |
521 | 0 | { |
522 | 0 | Py_DECREF(id); |
523 | 0 | goto error; |
524 | 0 | } |
525 | 14.5k | PyObject *form = PyUnicode_InternFromString("NFKC"); |
526 | 14.5k | if (form == NULL) |
527 | 0 | { |
528 | 0 | Py_DECREF(id); |
529 | 0 | goto error; |
530 | 0 | } |
531 | 14.5k | PyObject *args[2] = {form, id}; |
532 | 14.5k | PyObject *id2 = PyObject_Vectorcall(p->normalize, args, 2, NULL); |
533 | 14.5k | Py_DECREF(id); |
534 | 14.5k | Py_DECREF(form); |
535 | 14.5k | if (!id2) { |
536 | 0 | goto error; |
537 | 0 | } |
538 | | |
539 | 14.5k | if (!PyUnicode_Check(id2)) |
540 | 0 | { |
541 | 0 | PyErr_Format(PyExc_TypeError, |
542 | 0 | "unicodedata.normalize() must return a string, not " |
543 | 0 | "%.200s", |
544 | 0 | _PyType_Name(Py_TYPE(id2))); |
545 | 0 | Py_DECREF(id2); |
546 | 0 | goto error; |
547 | 0 | } |
548 | 14.5k | id = id2; |
549 | 14.5k | } |
550 | 123k | static const char * const forbidden[] = { |
551 | 123k | "None", |
552 | 123k | "True", |
553 | 123k | "False", |
554 | 123k | NULL |
555 | 123k | }; |
556 | 495k | for (int i = 0; forbidden[i] != NULL; i++) { |
557 | 371k | if (_PyUnicode_EqualToASCIIString(id, forbidden[i])) { |
558 | 0 | PyErr_Format(PyExc_ValueError, |
559 | 0 | "identifier field can't represent '%s' constant", |
560 | 0 | forbidden[i]); |
561 | 0 | Py_DECREF(id); |
562 | 0 | goto error; |
563 | 0 | } |
564 | 371k | } |
565 | 123k | PyInterpreterState *interp = _PyInterpreterState_GET(); |
566 | 123k | _PyUnicode_InternImmortal(interp, &id); |
567 | 123k | if (_PyArena_AddPyObject(p->arena, id) < 0) |
568 | 0 | { |
569 | 0 | Py_DECREF(id); |
570 | 0 | goto error; |
571 | 0 | } |
572 | 123k | return id; |
573 | | |
574 | 0 | error: |
575 | 0 | p->error_indicator = 1; |
576 | 0 | return NULL; |
577 | 123k | } |
578 | | |
579 | | static expr_ty |
580 | | _PyPegen_name_from_token(Parser *p, Token* t) |
581 | 9.98M | { |
582 | 9.98M | if (t == NULL) { |
583 | 6.12M | return NULL; |
584 | 6.12M | } |
585 | 3.86M | const char *s = PyBytes_AsString(t->bytes); |
586 | 3.86M | if (!s) { |
587 | 0 | p->error_indicator = 1; |
588 | 0 | return NULL; |
589 | 0 | } |
590 | | // Identifiers repeat constantly; a small span-keyed cache skips the |
591 | | // UTF-8 decode + intern for repeated occurrences. Keys point into |
592 | | // arena-owned token bytes and values are arena-owned interned strings, |
593 | | // so borrowed references are valid for the lifetime of the parse |
594 | | // (including the second error pass, which reuses parser and arena). |
595 | 3.86M | Py_ssize_t len = PyBytes_GET_SIZE(t->bytes); |
596 | 3.86M | Py_hash_t hash = PyObject_Hash(t->bytes); |
597 | 3.86M | if (hash == -1) { |
598 | 0 | p->error_indicator = 1; |
599 | 0 | return NULL; |
600 | 0 | } |
601 | 3.86M | IdentifierCacheEntry *free_slot = NULL; |
602 | 3.86M | size_t idx = (size_t)hash & (IDENTIFIER_CACHE_SIZE - 1); |
603 | 3.87M | for (int probe = 0; probe < IDENTIFIER_CACHE_MAX_PROBES; probe++) { |
604 | 3.87M | IdentifierCacheEntry *entry = &p->identifier_cache[ |
605 | 3.87M | (idx + probe) & (IDENTIFIER_CACHE_SIZE - 1)]; |
606 | 3.87M | if (entry->key == NULL) { |
607 | 118k | free_slot = entry; |
608 | 118k | break; |
609 | 118k | } |
610 | 3.75M | if (entry->hash == hash && entry->len == len && |
611 | 3.74M | memcmp(entry->key, s, len) == 0) |
612 | 3.74M | { |
613 | 3.74M | return _PyAST_Name(entry->value, Load, t->lineno, t->col_offset, |
614 | 3.74M | t->end_lineno, t->end_col_offset, p->arena); |
615 | 3.74M | } |
616 | 3.75M | } |
617 | 118k | PyObject *id = _PyPegen_new_identifier(p, s); |
618 | 118k | if (id == NULL) { |
619 | 0 | p->error_indicator = 1; |
620 | 0 | return NULL; |
621 | 0 | } |
622 | 118k | if (free_slot != NULL) { |
623 | 118k | free_slot->key = s; |
624 | 118k | free_slot->len = len; |
625 | 118k | free_slot->hash = hash; |
626 | 118k | free_slot->value = id; |
627 | 118k | } |
628 | 118k | return _PyAST_Name(id, Load, t->lineno, t->col_offset, t->end_lineno, |
629 | 118k | t->end_col_offset, p->arena); |
630 | 118k | } |
631 | | |
632 | | expr_ty |
633 | | _PyPegen_name_token(Parser *p) |
634 | 9.98M | { |
635 | 9.98M | Token *t = _PyPegen_expect_token(p, NAME); |
636 | 9.98M | return _PyPegen_name_from_token(p, t); |
637 | 9.98M | } |
638 | | |
639 | | void * |
640 | | _PyPegen_string_token(Parser *p) |
641 | 3.72M | { |
642 | 3.72M | return _PyPegen_expect_token(p, STRING); |
643 | 3.72M | } |
644 | | |
645 | 291k | expr_ty _PyPegen_soft_keyword_token(Parser *p) { |
646 | 291k | Token *t = _PyPegen_expect_token(p, NAME); |
647 | 291k | if (t == NULL) { |
648 | 183k | return NULL; |
649 | 183k | } |
650 | 108k | char *the_token; |
651 | 108k | Py_ssize_t size; |
652 | 108k | PyBytes_AsStringAndSize(t->bytes, &the_token, &size); |
653 | 636k | for (char **keyword = p->soft_keywords; *keyword != NULL; keyword++) { |
654 | 531k | if (strlen(*keyword) == (size_t)size && |
655 | 91.3k | strncmp(*keyword, the_token, (size_t)size) == 0) { |
656 | 3.37k | return _PyPegen_name_from_token(p, t); |
657 | 3.37k | } |
658 | 531k | } |
659 | 104k | return NULL; |
660 | 108k | } |
661 | | |
662 | | static PyObject * |
663 | | parsenumber_raw(const char *s) |
664 | 2.26M | { |
665 | 2.26M | const char *end; |
666 | 2.26M | long x; |
667 | 2.26M | double dx; |
668 | 2.26M | Py_complex compl; |
669 | 2.26M | int imflag; |
670 | | |
671 | 2.26M | assert(s != NULL); |
672 | 2.26M | errno = 0; |
673 | 2.26M | end = s + strlen(s) - 1; |
674 | 2.26M | imflag = *end == 'j' || *end == 'J'; |
675 | 2.26M | if (s[0] == '0') { |
676 | 152k | x = (long)PyOS_strtoul(s, (char **)&end, 0); |
677 | 152k | if (x < 0 && errno == 0) { |
678 | 549 | return PyLong_FromString(s, (char **)0, 0); |
679 | 549 | } |
680 | 152k | } |
681 | 2.10M | else { |
682 | 2.10M | x = PyOS_strtol(s, (char **)&end, 0); |
683 | 2.10M | } |
684 | 2.25M | if (*end == '\0') { |
685 | 1.80M | if (errno != 0) { |
686 | 77.1k | return PyLong_FromString(s, (char **)0, 0); |
687 | 77.1k | } |
688 | 1.73M | return PyLong_FromLong(x); |
689 | 1.80M | } |
690 | | /* XXX Huge floats may silently fail */ |
691 | 451k | if (imflag) { |
692 | 197k | compl.real = 0.; |
693 | 197k | compl.imag = PyOS_string_to_double(s, (char **)&end, NULL); |
694 | 197k | if (compl.imag == -1.0 && PyErr_Occurred()) { |
695 | 0 | return NULL; |
696 | 0 | } |
697 | 197k | return PyComplex_FromCComplex(compl); |
698 | 197k | } |
699 | 253k | dx = PyOS_string_to_double(s, NULL, NULL); |
700 | 253k | if (dx == -1.0 && PyErr_Occurred()) { |
701 | 0 | return NULL; |
702 | 0 | } |
703 | 253k | return PyFloat_FromDouble(dx); |
704 | 253k | } |
705 | | |
706 | | static PyObject * |
707 | | parsenumber(const char *s) |
708 | 2.26M | { |
709 | 2.26M | char *dup; |
710 | 2.26M | char *end; |
711 | 2.26M | PyObject *res = NULL; |
712 | | |
713 | 2.26M | assert(s != NULL); |
714 | | |
715 | 2.26M | if (strchr(s, '_') == NULL) { |
716 | 2.22M | return parsenumber_raw(s); |
717 | 2.22M | } |
718 | | /* Create a duplicate without underscores. */ |
719 | 31.2k | dup = PyMem_Malloc(strlen(s) + 1); |
720 | 31.2k | if (dup == NULL) { |
721 | 0 | return PyErr_NoMemory(); |
722 | 0 | } |
723 | 31.2k | end = dup; |
724 | 1.83M | for (; *s; s++) { |
725 | 1.80M | if (*s != '_') { |
726 | 1.76M | *end++ = *s; |
727 | 1.76M | } |
728 | 1.80M | } |
729 | 31.2k | *end = '\0'; |
730 | 31.2k | res = parsenumber_raw(dup); |
731 | 31.2k | PyMem_Free(dup); |
732 | 31.2k | return res; |
733 | 31.2k | } |
734 | | |
735 | | expr_ty |
736 | | _PyPegen_number_token(Parser *p) |
737 | 3.08M | { |
738 | 3.08M | Token *t = _PyPegen_expect_token(p, NUMBER); |
739 | 3.08M | if (t == NULL) { |
740 | 822k | return NULL; |
741 | 822k | } |
742 | | |
743 | 2.26M | const char *num_raw = PyBytes_AsString(t->bytes); |
744 | 2.26M | if (num_raw == NULL) { |
745 | 0 | p->error_indicator = 1; |
746 | 0 | return NULL; |
747 | 0 | } |
748 | | |
749 | 2.26M | if (p->feature_version < 6 && strchr(num_raw, '_') != NULL) { |
750 | 0 | p->error_indicator = 1; |
751 | 0 | return RAISE_SYNTAX_ERROR("Underscores in numeric literals are only supported " |
752 | 0 | "in Python 3.6 and greater"); |
753 | 0 | } |
754 | | |
755 | 2.26M | PyObject *c = parsenumber(num_raw); |
756 | | |
757 | 2.26M | if (c == NULL) { |
758 | 2 | p->error_indicator = 1; |
759 | 2 | PyThreadState *tstate = _PyThreadState_GET(); |
760 | | // The only way a ValueError should happen in _this_ code is via |
761 | | // PyLong_FromString hitting a length limit. |
762 | 2 | if (tstate->current_exception != NULL && |
763 | 2 | Py_TYPE(tstate->current_exception) == (PyTypeObject *)PyExc_ValueError |
764 | 2 | ) { |
765 | 2 | PyObject *exc = PyErr_GetRaisedException(); |
766 | | /* Intentionally omitting columns to avoid a wall of 1000s of '^'s |
767 | | * on the error message. Nobody is going to overlook their huge |
768 | | * numeric literal once given the line. */ |
769 | 2 | RAISE_ERROR_KNOWN_LOCATION( |
770 | 2 | p, PyExc_SyntaxError, |
771 | 2 | t->lineno, -1 /* col_offset */, |
772 | 2 | t->end_lineno, -1 /* end_col_offset */, |
773 | 2 | "%S - Consider hexadecimal for huge integer literals " |
774 | 2 | "to avoid decimal conversion limits.", |
775 | 2 | exc); |
776 | 2 | Py_DECREF(exc); |
777 | 2 | } |
778 | 2 | return NULL; |
779 | 2 | } |
780 | | |
781 | 2.26M | if (_PyArena_AddPyObject(p->arena, c) < 0) { |
782 | 0 | Py_DECREF(c); |
783 | 0 | p->error_indicator = 1; |
784 | 0 | return NULL; |
785 | 0 | } |
786 | | |
787 | 2.26M | return _PyAST_Constant(c, NULL, t->lineno, t->col_offset, t->end_lineno, |
788 | 2.26M | t->end_col_offset, p->arena); |
789 | 2.26M | } |
790 | | |
791 | | /* Check that the source for a single input statement really is a single |
792 | | statement by looking at what is left in the buffer after parsing. |
793 | | Trailing whitespace and comments are OK. */ |
794 | | static int // bool |
795 | | bad_single_statement(Parser *p) |
796 | 2.25k | { |
797 | 2.25k | char *cur = p->tok->cur; |
798 | 2.25k | char c = *cur; |
799 | | |
800 | 2.46k | for (;;) { |
801 | 3.11k | while (c == ' ' || c == '\t' || c == '\n' || c == '\014') { |
802 | 646 | c = *++cur; |
803 | 646 | } |
804 | | |
805 | 2.46k | if (!c) { |
806 | 2.23k | return 0; |
807 | 2.23k | } |
808 | | |
809 | 231 | if (c != '#') { |
810 | 18 | return 1; |
811 | 18 | } |
812 | | |
813 | | /* Suck up comment. */ |
814 | 6.93k | while (c && c != '\n') { |
815 | 6.72k | c = *++cur; |
816 | 6.72k | } |
817 | 213 | } |
818 | 2.25k | } |
819 | | |
820 | | static int |
821 | | compute_parser_flags(PyCompilerFlags *flags) |
822 | 22.6k | { |
823 | 22.6k | int parser_flags = 0; |
824 | 22.6k | if (!flags) { |
825 | 0 | return 0; |
826 | 0 | } |
827 | 22.6k | if (flags->cf_flags & PyCF_DONT_IMPLY_DEDENT) { |
828 | 10.3k | parser_flags |= PyPARSE_DONT_IMPLY_DEDENT; |
829 | 10.3k | } |
830 | 22.6k | if (flags->cf_flags & PyCF_IGNORE_COOKIE) { |
831 | 11.0k | parser_flags |= PyPARSE_IGNORE_COOKIE; |
832 | 11.0k | } |
833 | 22.6k | if (flags->cf_flags & CO_FUTURE_BARRY_AS_BDFL) { |
834 | 0 | parser_flags |= PyPARSE_BARRY_AS_BDFL; |
835 | 0 | } |
836 | 22.6k | if (flags->cf_flags & PyCF_TYPE_COMMENTS) { |
837 | 10.2k | parser_flags |= PyPARSE_TYPE_COMMENTS; |
838 | 10.2k | } |
839 | 22.6k | if (flags->cf_flags & PyCF_ALLOW_INCOMPLETE_INPUT) { |
840 | 16.9k | parser_flags |= PyPARSE_ALLOW_INCOMPLETE_INPUT; |
841 | 16.9k | } |
842 | 22.6k | return parser_flags; |
843 | 22.6k | } |
844 | | |
845 | | // Parser API |
846 | | |
847 | | Parser * |
848 | | _PyPegen_Parser_New(struct tok_state *tok, int start_rule, int flags, |
849 | | int feature_version, int *errcode, const char* source, PyArena *arena) |
850 | 22.6k | { |
851 | 22.6k | Parser *p = PyMem_Malloc(sizeof(Parser)); |
852 | 22.6k | if (p == NULL) { |
853 | 0 | return (Parser *) PyErr_NoMemory(); |
854 | 0 | } |
855 | 22.6k | assert(tok != NULL); |
856 | 22.6k | tok->type_comments = (flags & PyPARSE_TYPE_COMMENTS) > 0; |
857 | 22.6k | p->tok = tok; |
858 | 22.6k | p->keywords = NULL; |
859 | 22.6k | p->n_keyword_lists = -1; |
860 | 22.6k | p->soft_keywords = NULL; |
861 | 22.6k | p->tokens = PyMem_Malloc(sizeof(Token *)); |
862 | 22.6k | if (!p->tokens) { |
863 | 0 | PyMem_Free(p); |
864 | 0 | return (Parser *) PyErr_NoMemory(); |
865 | 0 | } |
866 | 22.6k | p->tokens[0] = PyMem_Calloc(1, sizeof(Token)); |
867 | 22.6k | if (!p->tokens[0]) { |
868 | 0 | PyMem_Free(p->tokens); |
869 | 0 | PyMem_Free(p); |
870 | 0 | return (Parser *) PyErr_NoMemory(); |
871 | 0 | } |
872 | 22.6k | if (!growable_comment_array_init(&p->type_ignore_comments, 10)) { |
873 | 0 | PyMem_Free(p->tokens[0]); |
874 | 0 | PyMem_Free(p->tokens); |
875 | 0 | PyMem_Free(p); |
876 | 0 | return (Parser *) PyErr_NoMemory(); |
877 | 0 | } |
878 | | |
879 | 22.6k | p->mark = 0; |
880 | 22.6k | p->fill = 0; |
881 | 22.6k | p->size = 1; |
882 | | |
883 | 22.6k | p->errcode = errcode; |
884 | 22.6k | p->arena = arena; |
885 | 22.6k | p->start_rule = start_rule; |
886 | 22.6k | p->parsing_started = 0; |
887 | 22.6k | p->normalize = NULL; |
888 | 22.6k | p->error_indicator = 0; |
889 | | |
890 | 22.6k | p->starting_lineno = 0; |
891 | 22.6k | p->starting_col_offset = 0; |
892 | 22.6k | p->flags = flags; |
893 | 22.6k | p->feature_version = feature_version; |
894 | 22.6k | p->known_err_token = NULL; |
895 | 22.6k | p->identifier_cache = PyMem_Calloc( |
896 | 22.6k | IDENTIFIER_CACHE_SIZE, sizeof(*p->identifier_cache)); |
897 | 22.6k | if (p->identifier_cache == NULL) { |
898 | 0 | growable_comment_array_deallocate(&p->type_ignore_comments); |
899 | 0 | PyMem_Free(p->tokens[0]); |
900 | 0 | PyMem_Free(p->tokens); |
901 | 0 | PyMem_Free(p); |
902 | 0 | return (Parser *) PyErr_NoMemory(); |
903 | 0 | } |
904 | 22.6k | p->level = 0; |
905 | 22.6k | p->call_invalid_rules = 0; |
906 | 22.6k | p->last_stmt_location.lineno = 0; |
907 | 22.6k | p->last_stmt_location.col_offset = 0; |
908 | 22.6k | p->last_stmt_location.end_lineno = 0; |
909 | 22.6k | p->last_stmt_location.end_col_offset = 0; |
910 | | #ifdef Py_DEBUG |
911 | | p->debug = _Py_GetConfig()->parser_debug; |
912 | | #endif |
913 | 22.6k | return p; |
914 | 22.6k | } |
915 | | |
916 | | void |
917 | | _PyPegen_Parser_Free(Parser *p) |
918 | 22.6k | { |
919 | 22.6k | PyMem_Free(p->identifier_cache); |
920 | 22.6k | Py_XDECREF(p->normalize); |
921 | 7.28M | for (int i = 0; i < p->size; i++) { |
922 | 7.26M | PyMem_Free(p->tokens[i]); |
923 | 7.26M | } |
924 | 22.6k | PyMem_Free(p->tokens); |
925 | 22.6k | growable_comment_array_deallocate(&p->type_ignore_comments); |
926 | 22.6k | PyMem_Free(p); |
927 | 22.6k | } |
928 | | |
929 | | static void |
930 | | reset_parser_state_for_error_pass(Parser *p) |
931 | 9.07k | { |
932 | 9.07k | p->last_stmt_location.lineno = 0; |
933 | 9.07k | p->last_stmt_location.col_offset = 0; |
934 | 9.07k | p->last_stmt_location.end_lineno = 0; |
935 | 9.07k | p->last_stmt_location.end_col_offset = 0; |
936 | 739k | for (int i = 0; i < p->fill; i++) { |
937 | 730k | p->tokens[i]->memo = NULL; |
938 | 730k | } |
939 | 9.07k | p->mark = 0; |
940 | 9.07k | p->call_invalid_rules = 1; |
941 | | // Don't try to get extra tokens in interactive mode when trying to |
942 | | // raise specialized errors in the second pass. |
943 | 9.07k | p->tok->interactive_underflow = IUNDERFLOW_STOP; |
944 | 9.07k | } |
945 | | |
946 | | static inline int |
947 | 7.48k | _is_end_of_source(Parser *p) { |
948 | 7.48k | int err = p->tok->done; |
949 | 7.48k | return err == E_EOF || err == E_EOFS || err == E_EOLS; |
950 | 7.48k | } |
951 | | |
952 | | static void |
953 | 8.76k | _PyPegen_set_syntax_error_metadata(Parser *p) { |
954 | 8.76k | PyObject *exc = PyErr_GetRaisedException(); |
955 | 8.76k | if (!exc || !PyObject_TypeCheck(exc, (PyTypeObject *)PyExc_SyntaxError)) { |
956 | 0 | PyErr_SetRaisedException(exc); |
957 | 0 | return; |
958 | 0 | } |
959 | 8.76k | const char *source = NULL; |
960 | 8.76k | if (p->tok->str != NULL) { |
961 | 8.76k | source = p->tok->str; |
962 | 8.76k | } |
963 | 8.76k | if (!source && p->tok->fp_interactive && p->tok->interactive_src_start) { |
964 | 0 | source = p->tok->interactive_src_start; |
965 | 0 | } |
966 | 8.76k | PyObject* the_source = NULL; |
967 | 8.76k | if (source) { |
968 | 8.76k | if (p->tok->encoding == NULL) { |
969 | 3.82k | the_source = PyUnicode_FromString(source); |
970 | 4.94k | } else { |
971 | 4.94k | the_source = PyUnicode_Decode(source, strlen(source), p->tok->encoding, NULL); |
972 | 4.94k | } |
973 | 8.76k | } |
974 | 8.76k | if (!the_source) { |
975 | 1.53k | PyErr_Clear(); |
976 | 1.53k | the_source = Py_None; |
977 | 1.53k | Py_INCREF(the_source); |
978 | 1.53k | } |
979 | 8.76k | PyObject* metadata = Py_BuildValue( |
980 | 8.76k | "(iiN)", |
981 | 8.76k | p->last_stmt_location.lineno, |
982 | 8.76k | p->last_stmt_location.col_offset, |
983 | 8.76k | the_source // N gives ownership to metadata |
984 | 8.76k | ); |
985 | 8.76k | if (!metadata) { |
986 | 0 | PyErr_Clear(); |
987 | 0 | return; |
988 | 0 | } |
989 | 8.76k | PySyntaxErrorObject *syntax_error = (PySyntaxErrorObject *)exc; |
990 | | |
991 | 8.76k | Py_XDECREF(syntax_error->metadata); |
992 | 8.76k | syntax_error->metadata = metadata; |
993 | 8.76k | PyErr_SetRaisedException(exc); |
994 | 8.76k | } |
995 | | |
996 | | void * |
997 | | _PyPegen_run_parser(Parser *p) |
998 | 22.6k | { |
999 | 22.6k | void *res = _PyPegen_parse(p); |
1000 | 22.6k | assert(p->level == 0); |
1001 | 22.6k | if (res != NULL && PyErr_Occurred()) { |
1002 | | // Discard a result returned with an exception still pending |
1003 | | // (e.g. a MemoryError from a recovered-from allocation failure). |
1004 | 0 | return NULL; |
1005 | 0 | } |
1006 | 22.6k | if (res == NULL) { |
1007 | 9.90k | if ((p->flags & PyPARSE_ALLOW_INCOMPLETE_INPUT) && _is_end_of_source(p)) { |
1008 | 802 | PyErr_Clear(); |
1009 | 802 | return _PyPegen_raise_error(p, PyExc_IncompleteInputError, 0, "incomplete input"); |
1010 | 802 | } |
1011 | 9.10k | if (PyErr_Occurred() && !PyErr_ExceptionMatches(PyExc_SyntaxError)) { |
1012 | 27 | return NULL; |
1013 | 27 | } |
1014 | | // Make a second parser pass. In this pass we activate heavier and slower checks |
1015 | | // to produce better error messages and more complete diagnostics. Extra "invalid_*" |
1016 | | // rules will be active during parsing. |
1017 | 9.07k | Token *last_token = p->tokens[p->fill - 1]; |
1018 | 9.07k | reset_parser_state_for_error_pass(p); |
1019 | 9.07k | _PyPegen_parse(p); |
1020 | | |
1021 | | // Set SyntaxErrors accordingly depending on the parser/tokenizer status at the failure |
1022 | | // point. |
1023 | 9.07k | _Pypegen_set_syntax_error(p, last_token); |
1024 | | |
1025 | | // Set the metadata in the exception from p->last_stmt_location |
1026 | 9.07k | if (PyErr_ExceptionMatches(PyExc_SyntaxError)) { |
1027 | 8.76k | _PyPegen_set_syntax_error_metadata(p); |
1028 | 8.76k | } |
1029 | 9.07k | return NULL; |
1030 | 9.10k | } |
1031 | | |
1032 | 12.7k | if (p->start_rule == Py_single_input && bad_single_statement(p)) { |
1033 | 18 | p->tok->done = E_BADSINGLE; // This is not necessary for now, but might be in the future |
1034 | 18 | return RAISE_SYNTAX_ERROR("multiple statements found while compiling a single statement"); |
1035 | 18 | } |
1036 | | |
1037 | | // test_peg_generator defines _Py_TEST_PEGEN to not call PyAST_Validate() |
1038 | | #if defined(Py_DEBUG) && !defined(_Py_TEST_PEGEN) |
1039 | | if (p->start_rule == Py_single_input || |
1040 | | p->start_rule == Py_file_input || |
1041 | | p->start_rule == Py_eval_input) |
1042 | | { |
1043 | | if (!_PyAST_Validate(res)) { |
1044 | | return NULL; |
1045 | | } |
1046 | | } |
1047 | | #endif |
1048 | 12.7k | return res; |
1049 | 12.7k | } |
1050 | | |
1051 | | mod_ty |
1052 | | _PyPegen_run_parser_from_file_pointer(FILE *fp, int start_rule, PyObject *filename_ob, |
1053 | | const char *enc, const char *ps1, const char *ps2, |
1054 | | PyCompilerFlags *flags, int *errcode, |
1055 | | PyObject **interactive_src, PyArena *arena) |
1056 | 0 | { |
1057 | 0 | struct tok_state *tok = _PyTokenizer_FromFile(fp, enc, ps1, ps2); |
1058 | 0 | if (tok == NULL) { |
1059 | 0 | if (PyErr_Occurred()) { |
1060 | 0 | _PyTokenizer_raise_init_error(filename_ob); |
1061 | 0 | } |
1062 | 0 | else { |
1063 | | // The only silent tokenizer init failure is a failed allocation. |
1064 | 0 | PyErr_NoMemory(); |
1065 | 0 | } |
1066 | 0 | return NULL; |
1067 | 0 | } |
1068 | 0 | if (!tok->fp || ps1 != NULL || ps2 != NULL || |
1069 | 0 | PyUnicode_CompareWithASCIIString(filename_ob, "<stdin>") == 0) { |
1070 | 0 | tok->fp_interactive = 1; |
1071 | 0 | } |
1072 | | // This transfers the ownership to the tokenizer |
1073 | 0 | tok->filename = Py_NewRef(filename_ob); |
1074 | | |
1075 | | // From here on we need to clean up even if there's an error |
1076 | 0 | mod_ty result = NULL; |
1077 | |
|
1078 | 0 | tok->module = PyUnicode_FromString("__main__"); |
1079 | 0 | if (tok->module == NULL) { |
1080 | 0 | goto error; |
1081 | 0 | } |
1082 | | |
1083 | 0 | int parser_flags = compute_parser_flags(flags); |
1084 | 0 | Parser *p = _PyPegen_Parser_New(tok, start_rule, parser_flags, PY_MINOR_VERSION, |
1085 | 0 | errcode, NULL, arena); |
1086 | 0 | if (p == NULL) { |
1087 | 0 | goto error; |
1088 | 0 | } |
1089 | | |
1090 | 0 | result = _PyPegen_run_parser(p); |
1091 | 0 | _PyPegen_Parser_Free(p); |
1092 | |
|
1093 | 0 | if (tok->fp_interactive && tok->interactive_src_start && result && interactive_src != NULL) { |
1094 | 0 | *interactive_src = PyUnicode_FromString(tok->interactive_src_start); |
1095 | 0 | if (!*interactive_src || _PyArena_AddPyObject(arena, *interactive_src) < 0) { |
1096 | 0 | Py_XDECREF(*interactive_src); |
1097 | 0 | result = NULL; |
1098 | 0 | goto error; |
1099 | 0 | } |
1100 | 0 | } |
1101 | | |
1102 | 0 | error: |
1103 | 0 | _PyTokenizer_Free(tok); |
1104 | 0 | return result; |
1105 | 0 | } |
1106 | | |
1107 | | mod_ty |
1108 | | _PyPegen_run_parser_from_string(const char *str, int start_rule, PyObject *filename_ob, |
1109 | | PyCompilerFlags *flags, PyArena *arena, PyObject *module) |
1110 | 24.5k | { |
1111 | 24.5k | int exec_input = start_rule == Py_file_input; |
1112 | | |
1113 | 24.5k | struct tok_state *tok; |
1114 | 24.5k | if (flags != NULL && flags->cf_flags & PyCF_IGNORE_COOKIE) { |
1115 | 11.0k | tok = _PyTokenizer_FromUTF8(str, exec_input, 0); |
1116 | 13.5k | } else { |
1117 | 13.5k | tok = _PyTokenizer_FromString(str, exec_input, 0); |
1118 | 13.5k | } |
1119 | 24.5k | if (tok == NULL) { |
1120 | 1.90k | if (PyErr_Occurred()) { |
1121 | 1.90k | _PyTokenizer_raise_init_error(filename_ob); |
1122 | 1.90k | } |
1123 | 0 | else { |
1124 | | // The only silent tokenizer init failure is a failed allocation. |
1125 | 0 | PyErr_NoMemory(); |
1126 | 0 | } |
1127 | 1.90k | return NULL; |
1128 | 1.90k | } |
1129 | | // This transfers the ownership to the tokenizer |
1130 | 22.6k | tok->filename = Py_NewRef(filename_ob); |
1131 | 22.6k | tok->module = Py_XNewRef(module); |
1132 | | |
1133 | | // We need to clear up from here on |
1134 | 22.6k | mod_ty result = NULL; |
1135 | | |
1136 | 22.6k | int parser_flags = compute_parser_flags(flags); |
1137 | 22.6k | int feature_version = flags && (flags->cf_flags & PyCF_ONLY_AST) ? |
1138 | 14.7k | flags->cf_feature_version : PY_MINOR_VERSION; |
1139 | 22.6k | Parser *p = _PyPegen_Parser_New(tok, start_rule, parser_flags, feature_version, |
1140 | 22.6k | NULL, str, arena); |
1141 | 22.6k | if (p == NULL) { |
1142 | 0 | goto error; |
1143 | 0 | } |
1144 | | |
1145 | 22.6k | result = _PyPegen_run_parser(p); |
1146 | 22.6k | _PyPegen_Parser_Free(p); |
1147 | | |
1148 | 22.6k | error: |
1149 | 22.6k | _PyTokenizer_Free(tok); |
1150 | 22.6k | return result; |
1151 | 22.6k | } |