/src/jsonnet/core/lexer.cpp
Line | Count | Source |
1 | | /* |
2 | | Copyright 2015 Google Inc. All rights reserved. |
3 | | |
4 | | Licensed under the Apache License, Version 2.0 (the "License"); |
5 | | you may not use this file except in compliance with the License. |
6 | | You may obtain a copy of the License at |
7 | | |
8 | | http://www.apache.org/licenses/LICENSE-2.0 |
9 | | |
10 | | Unless required by applicable law or agreed to in writing, software |
11 | | distributed under the License is distributed on an "AS IS" BASIS, |
12 | | WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
13 | | See the License for the specific language governing permissions and |
14 | | limitations under the License. |
15 | | */ |
16 | | |
17 | | #include <cassert> |
18 | | |
19 | | #include <map> |
20 | | #include <sstream> |
21 | | #include <string> |
22 | | |
23 | | #include "lexer.h" |
24 | | #include "static_error.h" |
25 | | #include "unicode.h" |
26 | | |
27 | | namespace jsonnet::internal { |
28 | | |
29 | | static const std::vector<std::string> EMPTY; |
30 | | |
31 | | /** Is the char whitespace (excluding \n). */ |
32 | | static bool is_horz_ws(char c) |
33 | 485M | { |
34 | 485M | return c == ' ' || c == '\t' || c == '\r'; |
35 | 485M | } |
36 | | |
37 | | /** Is the char whitespace. */ |
38 | | static bool is_ws(char c) |
39 | 436M | { |
40 | 436M | return c == '\n' || is_horz_ws(c); |
41 | 436M | } |
42 | | |
43 | | /** Strip whitespace from both ends of a string, but only up to margin on the left hand side. */ |
44 | | static std::string strip_ws(const std::string &s, unsigned margin) |
45 | 10.2M | { |
46 | 10.2M | if (s.size() == 0) |
47 | 8.74M | return s; // Avoid underflow below. |
48 | 1.54M | size_t i = 0; |
49 | 4.66M | while (i < s.length() && is_horz_ws(s[i]) && i < margin) |
50 | 3.12M | i++; |
51 | 1.54M | size_t j = s.size(); |
52 | 4.45M | while (j > i && is_horz_ws(s[j - 1])) { |
53 | 2.91M | j--; |
54 | 2.91M | } |
55 | 1.54M | return std::string(&s[i], &s[j]); |
56 | 10.2M | } |
57 | | |
58 | | /** Split a string by \n and also strip left (up to margin) & right whitespace from each line. */ |
59 | | static std::vector<std::string> line_split(const std::string &s, unsigned margin) |
60 | 154k | { |
61 | 154k | std::vector<std::string> ret; |
62 | 154k | std::stringstream ss; |
63 | 71.6M | for (size_t i = 0; i < s.length(); ++i) { |
64 | 71.4M | if (s[i] == '\n') { |
65 | 10.1M | ret.emplace_back(strip_ws(ss.str(), margin)); |
66 | 10.1M | ss.str(""); |
67 | 61.3M | } else { |
68 | 61.3M | ss << s[i]; |
69 | 61.3M | } |
70 | 71.4M | } |
71 | 154k | ret.emplace_back(strip_ws(ss.str(), margin)); |
72 | 154k | return ret; |
73 | 154k | } |
74 | | |
75 | | /** Consume whitespace. |
76 | | * |
77 | | * Return number of \n and number of spaces after last \n. Convert \t to spaces. |
78 | | */ |
79 | | static void lex_ws(const char *&c, unsigned &new_lines, unsigned &indent, const char *&line_start, |
80 | | unsigned long &line_number) |
81 | 188M | { |
82 | 188M | indent = 0; |
83 | 188M | new_lines = 0; |
84 | 436M | for (; *c != '\0' && is_ws(*c); c++) { |
85 | 248M | switch (*c) { |
86 | 564k | case '\r': |
87 | | // Ignore. |
88 | 564k | break; |
89 | | |
90 | 28.5M | case '\n': |
91 | 28.5M | indent = 0; |
92 | 28.5M | new_lines++; |
93 | 28.5M | line_number++; |
94 | 28.5M | line_start = c + 1; |
95 | 28.5M | break; |
96 | | |
97 | 219M | case ' ': indent += 1; break; |
98 | | |
99 | | // This only works for \t at the beginning of lines, but we strip it everywhere else |
100 | | // anyway. The only case where this will cause a problem is spaces followed by \t |
101 | | // at the beginning of a line. However that is rare, ill-advised, and if re-indentation |
102 | | // is enabled it will be fixed later. |
103 | 50.7k | case '\t': indent += 8; break; |
104 | 248M | } |
105 | 248M | } |
106 | 188M | } |
107 | | |
108 | | /** |
109 | | # Consume all text until the end of the line, return number of newlines after that and indent |
110 | | */ |
111 | | static void lex_until_newline(const char *&c, std::string &text, unsigned &blanks, unsigned &indent, |
112 | | const char *&line_start, unsigned long &line_number) |
113 | 4.74M | { |
114 | 4.74M | const char *original_c = c; |
115 | 4.74M | const char *last_non_space = c; |
116 | 73.1M | for (; *c != '\0' && *c != '\n'; c++) { |
117 | 68.3M | if (!is_horz_ws(*c)) |
118 | 58.7M | last_non_space = c; |
119 | 68.3M | } |
120 | 4.74M | text = std::string(original_c, last_non_space - original_c + 1); |
121 | | // Consume subsequent whitespace including the '\n'. |
122 | 4.74M | unsigned new_lines; |
123 | 4.74M | lex_ws(c, new_lines, indent, line_start, line_number); |
124 | 4.74M | blanks = new_lines == 0 ? 0 : new_lines - 1; |
125 | 4.74M | } |
126 | | |
127 | | static bool is_upper(char c) |
128 | 525M | { |
129 | 525M | return c >= 'A' && c <= 'Z'; |
130 | 525M | } |
131 | | |
132 | | static bool is_lower(char c) |
133 | 515M | { |
134 | 515M | return c >= 'a' && c <= 'z'; |
135 | 515M | } |
136 | | |
137 | | static bool is_number(char c) |
138 | 97.5M | { |
139 | 97.5M | return c >= '0' && c <= '9'; |
140 | 97.5M | } |
141 | | |
142 | | static bool is_identifier_first(char c) |
143 | 525M | { |
144 | 525M | return is_upper(c) || is_lower(c) || c == '_'; |
145 | 525M | } |
146 | | |
147 | | static bool is_identifier(char c) |
148 | 423M | { |
149 | 423M | return is_identifier_first(c) || is_number(c); |
150 | 423M | } |
151 | | |
152 | | static bool is_symbol(char c) |
153 | 91.6M | { |
154 | 91.6M | switch (c) { |
155 | 2.09M | case '!': |
156 | 2.31M | case '$': |
157 | 11.8M | case ':': |
158 | 12.5M | case '~': |
159 | 25.5M | case '+': |
160 | 27.1M | case '-': |
161 | 28.9M | case '&': |
162 | 29.9M | case '|': |
163 | 29.9M | case '^': |
164 | 47.3M | case '=': |
165 | 48.9M | case '<': |
166 | 51.1M | case '>': |
167 | 65.6M | case '*': |
168 | 67.5M | case '/': |
169 | 68.3M | case '%': return true; |
170 | 91.6M | } |
171 | 23.2M | return false; |
172 | 91.6M | } |
173 | | |
174 | 9.81M | bool allowed_at_end_of_operator(char c) { |
175 | 9.81M | switch (c) { |
176 | 2.82M | case '+': |
177 | 2.98M | case '-': |
178 | 3.65M | case '~': |
179 | 4.20M | case '!': |
180 | 4.31M | case '$': return false; |
181 | 9.81M | } |
182 | 5.50M | return true; |
183 | 9.81M | } |
184 | | |
185 | | static const std::map<std::string, Token::Kind> keywords = { |
186 | | {"assert", Token::ASSERT}, |
187 | | {"else", Token::ELSE}, |
188 | | {"error", Token::ERROR}, |
189 | | {"false", Token::FALSE}, |
190 | | {"for", Token::FOR}, |
191 | | {"function", Token::FUNCTION}, |
192 | | {"if", Token::IF}, |
193 | | {"import", Token::IMPORT}, |
194 | | {"importstr", Token::IMPORTSTR}, |
195 | | {"importbin", Token::IMPORTBIN}, |
196 | | {"in", Token::IN}, |
197 | | {"local", Token::LOCAL}, |
198 | | {"null", Token::NULL_LIT}, |
199 | | {"self", Token::SELF}, |
200 | | {"super", Token::SUPER}, |
201 | | {"tailstrict", Token::TAILSTRICT}, |
202 | | {"then", Token::THEN}, |
203 | | {"true", Token::TRUE}, |
204 | | }; |
205 | | |
206 | | Token::Kind lex_get_keyword_kind(const std::string &identifier) |
207 | 77.1M | { |
208 | 77.1M | auto it = keywords.find(identifier); |
209 | 77.1M | if (it == keywords.end()) |
210 | 56.4M | return Token::IDENTIFIER; |
211 | 20.7M | return it->second; |
212 | 77.1M | } |
213 | | |
214 | | std::string lex_number(const char *&c, const std::string &filename, const Location &begin) |
215 | 8.19M | { |
216 | | // This function should be understood with reference to the linked image: |
217 | | // https://www.json.org/img/number.png |
218 | | |
219 | | // Note, we deviate from the json.org documentation as follows: |
220 | | // * There is no reason to lex negative numbers as atomic tokens, it is better to parse them |
221 | | // as a unary operator combined with a numeric literal. This avoids x-1 being tokenized as |
222 | | // <identifier> <number> instead of the intended <identifier> <binop> <number>. |
223 | | // * We support digit separators using the _ character for readability in |
224 | | // large numeric literals. |
225 | | |
226 | 8.19M | enum State { |
227 | 8.19M | BEGIN, |
228 | 8.19M | AFTER_ZERO, |
229 | 8.19M | AFTER_ONE_TO_NINE, |
230 | 8.19M | AFTER_INT_UNDERSCORE, |
231 | 8.19M | AFTER_DOT, |
232 | 8.19M | AFTER_DIGIT, |
233 | 8.19M | AFTER_FRAC_UNDERSCORE, |
234 | 8.19M | AFTER_E, |
235 | 8.19M | AFTER_EXP_SIGN, |
236 | 8.19M | AFTER_EXP_DIGIT, |
237 | 8.19M | AFTER_EXP_UNDERSCORE |
238 | 8.19M | } state; |
239 | | |
240 | 8.19M | std::string r; |
241 | | |
242 | 8.19M | state = BEGIN; |
243 | 18.0M | while (true) { |
244 | 18.0M | switch (state) { |
245 | 8.19M | case BEGIN: |
246 | 8.19M | switch (*c) { |
247 | 1.98M | case '0': state = AFTER_ZERO; break; |
248 | | |
249 | 2.33M | case '1': |
250 | 2.80M | case '2': |
251 | 3.23M | case '3': |
252 | 3.57M | case '4': |
253 | 3.64M | case '5': |
254 | 4.19M | case '6': |
255 | 4.35M | case '7': |
256 | 5.38M | case '8': |
257 | 6.21M | case '9': state = AFTER_ONE_TO_NINE; break; |
258 | | |
259 | 0 | default: throw StaticError(filename, begin, "couldn't lex number"); |
260 | 8.19M | } |
261 | 8.19M | break; |
262 | | |
263 | 8.19M | case AFTER_ZERO: |
264 | 1.98M | switch (*c) { |
265 | 27.8k | case '.': state = AFTER_DOT; break; |
266 | | |
267 | 640 | case 'e': |
268 | 1.39k | case 'E': state = AFTER_E; break; |
269 | | |
270 | 3 | case '_': { |
271 | 3 | std::stringstream ss; |
272 | 3 | ss << "couldn't lex number, _ not allowed after leading 0"; |
273 | 3 | throw StaticError(filename, begin, ss.str()); |
274 | 640 | } |
275 | | |
276 | 1.95M | default: goto end; |
277 | 1.98M | } |
278 | 29.2k | break; |
279 | | |
280 | 7.47M | case AFTER_ONE_TO_NINE: |
281 | 7.47M | switch (*c) { |
282 | 17.2k | case '.': state = AFTER_DOT; break; |
283 | | |
284 | 1.56k | case 'e': |
285 | 3.01k | case 'E': state = AFTER_E; break; |
286 | | |
287 | 621k | case '0': |
288 | 648k | case '1': |
289 | 796k | case '2': |
290 | 851k | case '3': |
291 | 914k | case '4': |
292 | 1.04M | case '5': |
293 | 1.13M | case '6': |
294 | 1.17M | case '7': |
295 | 1.22M | case '8': |
296 | 1.26M | case '9': state = AFTER_ONE_TO_NINE; break; |
297 | | |
298 | 378 | case '_': state = AFTER_INT_UNDERSCORE; goto skip_char; |
299 | | |
300 | 6.19M | default: goto end; |
301 | 7.47M | } |
302 | 1.28M | break; |
303 | | |
304 | 1.28M | case AFTER_INT_UNDERSCORE: |
305 | 378 | switch (*c) { |
306 | | // The only valid transition from _ is to a digit. |
307 | 215 | case '0': |
308 | 240 | case '1': |
309 | 248 | case '2': |
310 | 248 | case '3': |
311 | 257 | case '4': |
312 | 257 | case '5': |
313 | 257 | case '6': |
314 | 257 | case '7': |
315 | 359 | case '8': |
316 | 359 | case '9': state = AFTER_ONE_TO_NINE; break; |
317 | | |
318 | 19 | default: { |
319 | 19 | std::stringstream ss; |
320 | 19 | ss << "couldn't lex number, junk after _: " << *c; |
321 | 19 | throw StaticError(filename, begin, ss.str()); |
322 | 359 | } |
323 | 378 | } |
324 | 359 | break; |
325 | | |
326 | 45.1k | case AFTER_DOT: |
327 | 45.1k | switch (*c) { |
328 | 1.58k | case '0': |
329 | 15.0k | case '1': |
330 | 15.8k | case '2': |
331 | 16.6k | case '3': |
332 | 17.4k | case '4': |
333 | 43.2k | case '5': |
334 | 43.7k | case '6': |
335 | 43.8k | case '7': |
336 | 44.8k | case '8': |
337 | 45.1k | case '9': state = AFTER_DIGIT; break; |
338 | | |
339 | 15 | default: { |
340 | 15 | std::stringstream ss; |
341 | 15 | ss << "couldn't lex number, junk after decimal point: " << *c; |
342 | 15 | throw StaticError(filename, begin, ss.str()); |
343 | 44.8k | } |
344 | 45.1k | } |
345 | 45.1k | break; |
346 | | |
347 | 315k | case AFTER_DIGIT: |
348 | 315k | switch (*c) { |
349 | 690 | case 'e': |
350 | 1.39k | case 'E': state = AFTER_E; break; |
351 | | |
352 | 29.8k | case '0': |
353 | 70.3k | case '1': |
354 | 84.8k | case '2': |
355 | 112k | case '3': |
356 | 126k | case '4': |
357 | 167k | case '5': |
358 | 195k | case '6': |
359 | 209k | case '7': |
360 | 230k | case '8': |
361 | 270k | case '9': state = AFTER_DIGIT; break; |
362 | | |
363 | 435 | case '_': state = AFTER_FRAC_UNDERSCORE; goto skip_char; |
364 | | |
365 | 43.7k | default: goto end; |
366 | 315k | } |
367 | 271k | break; |
368 | | |
369 | 271k | case AFTER_FRAC_UNDERSCORE: |
370 | 435 | switch (*c) { |
371 | | // The only valid transition from _ is to a digit. |
372 | 159 | case '0': |
373 | 198 | case '1': |
374 | 199 | case '2': |
375 | 199 | case '3': |
376 | 207 | case '4': |
377 | 210 | case '5': |
378 | 210 | case '6': |
379 | 210 | case '7': |
380 | 430 | case '8': |
381 | 430 | case '9': state = AFTER_DIGIT; break; |
382 | | |
383 | 5 | default: { |
384 | 5 | std::stringstream ss; |
385 | 5 | ss << "couldn't lex number, junk after _: " << *c; |
386 | 5 | throw StaticError(filename, begin, ss.str()); |
387 | 430 | } |
388 | 435 | } |
389 | 430 | break; |
390 | | |
391 | 5.80k | case AFTER_E: |
392 | 5.80k | switch (*c) { |
393 | 860 | case '+': |
394 | 2.24k | case '-': state = AFTER_EXP_SIGN; break; |
395 | | |
396 | 1.16k | case '0': |
397 | 1.94k | case '1': |
398 | 2.22k | case '2': |
399 | 2.55k | case '3': |
400 | 2.69k | case '4': |
401 | 2.73k | case '5': |
402 | 2.93k | case '6': |
403 | 3.02k | case '7': |
404 | 3.21k | case '8': |
405 | 3.51k | case '9': state = AFTER_EXP_DIGIT; break; |
406 | | |
407 | 43 | default: { |
408 | 43 | std::stringstream ss; |
409 | 43 | ss << "couldn't lex number, junk after 'E': " << *c; |
410 | 43 | throw StaticError(filename, begin, ss.str()); |
411 | 3.21k | } |
412 | 5.80k | } |
413 | 5.75k | break; |
414 | | |
415 | 5.75k | case AFTER_EXP_SIGN: |
416 | 2.24k | switch (*c) { |
417 | 780 | case '0': |
418 | 1.12k | case '1': |
419 | 1.43k | case '2': |
420 | 1.95k | case '3': |
421 | 2.21k | case '4': |
422 | 2.21k | case '5': |
423 | 2.22k | case '6': |
424 | 2.22k | case '7': |
425 | 2.22k | case '8': |
426 | 2.23k | case '9': state = AFTER_EXP_DIGIT; break; |
427 | | |
428 | 9 | default: { |
429 | 9 | std::stringstream ss; |
430 | 9 | ss << "couldn't lex number, junk after exponent sign: " << *c; |
431 | 9 | throw StaticError(filename, begin, ss.str()); |
432 | 2.22k | } |
433 | 2.24k | } |
434 | 2.23k | break; |
435 | | |
436 | 34.3k | case AFTER_EXP_DIGIT: |
437 | 34.3k | switch (*c) { |
438 | 9.56k | case '0': |
439 | 12.1k | case '1': |
440 | 14.3k | case '2': |
441 | 16.5k | case '3': |
442 | 19.0k | case '4': |
443 | 20.3k | case '5': |
444 | 21.8k | case '6': |
445 | 24.6k | case '7': |
446 | 26.5k | case '8': |
447 | 28.3k | case '9': state = AFTER_EXP_DIGIT; break; |
448 | | |
449 | 302 | case '_': state = AFTER_EXP_UNDERSCORE; goto skip_char; |
450 | | |
451 | 5.74k | default: goto end; |
452 | 34.3k | } |
453 | 28.3k | break; |
454 | | |
455 | 28.3k | case AFTER_EXP_UNDERSCORE: |
456 | 302 | switch (*c) { |
457 | | // The only valid transition from _ is to a digit. |
458 | 13 | case '0': |
459 | 172 | case '1': |
460 | 175 | case '2': |
461 | 176 | case '3': |
462 | 273 | case '4': |
463 | 273 | case '5': |
464 | 281 | case '6': |
465 | 283 | case '7': |
466 | 289 | case '8': |
467 | 296 | case '9': state = AFTER_EXP_DIGIT; break; |
468 | | |
469 | 6 | default: { |
470 | 6 | std::stringstream ss; |
471 | 6 | ss << "couldn't lex number, junk after _: " << *c; |
472 | 6 | throw StaticError(filename, begin, ss.str()); |
473 | 289 | } |
474 | 302 | } |
475 | 296 | break; |
476 | 18.0M | } |
477 | 9.86M | r += *c; |
478 | | |
479 | 9.86M | skip_char: |
480 | 9.86M | c++; |
481 | 9.86M | } |
482 | 8.19M | end: |
483 | 8.19M | return r; |
484 | 8.19M | } |
485 | | |
486 | | // Check that b has at least the same whitespace prefix as a and returns the amount of this |
487 | | // whitespace, otherwise returns 0. If a has no whitespace prefix than return 0. |
488 | | static int whitespace_check(const char *a, const char *b) |
489 | 23.9k | { |
490 | 23.9k | int i = 0; |
491 | 252k | while (a[i] == ' ' || a[i] == '\t') { |
492 | 239k | if (b[i] != a[i]) |
493 | 10.4k | return 0; |
494 | 228k | i++; |
495 | 228k | } |
496 | 13.5k | return i; |
497 | 23.9k | } |
498 | | |
499 | 154 | static void describe_whitespace(std::stringstream& msg, const std::string& ws) { |
500 | 154 | int spaces = 0; |
501 | 154 | int tabs = 0; |
502 | 22.7k | for (char c : ws) { |
503 | 22.7k | if (c == ' ') |
504 | 801 | spaces++; |
505 | 21.9k | else if (c == '\t') |
506 | 21.9k | tabs++; |
507 | 22.7k | } |
508 | 154 | if (spaces > 0 && tabs > 0) { |
509 | 30 | msg << spaces << (spaces == 1 ? " space" : " spaces") << " and " << tabs |
510 | 30 | << (tabs == 1 ? " tab" : " tabs"); |
511 | 124 | } else if (spaces > 0) { |
512 | 55 | msg << spaces << (spaces == 1 ? " space" : " spaces"); |
513 | 69 | } else if (tabs > 0) { |
514 | 69 | msg << tabs << (tabs == 1 ? " tab" : " tabs"); |
515 | 69 | } else { |
516 | 0 | msg << "no indentation"; |
517 | 0 | } |
518 | 154 | } |
519 | | |
520 | | Tokens jsonnet_lex(const std::string &filename, const char *input) |
521 | 25.6k | { |
522 | 25.6k | unsigned long line_number = 1; |
523 | 25.6k | const char *line_start = input; |
524 | | |
525 | 25.6k | Tokens r; |
526 | | |
527 | 25.6k | const char *c = input; |
528 | | |
529 | 25.6k | Fodder fodder; |
530 | 25.6k | bool fresh_line = true; // Are we tokenizing from the beginning of a new line? |
531 | | |
532 | 183M | while (*c != '\0') { |
533 | | // Used to ensure we have actually advanced the pointer by the end of the iteration. |
534 | 183M | const char *original_c = c; |
535 | | |
536 | 183M | Token::Kind kind; |
537 | 183M | std::string data; |
538 | 183M | std::string string_block_indent; |
539 | 183M | std::string string_block_term_indent; |
540 | | |
541 | 183M | unsigned new_lines, indent; |
542 | 183M | lex_ws(c, new_lines, indent, line_start, line_number); |
543 | | |
544 | | // If it's the end of the file, discard final whitespace. |
545 | 183M | if (*c == '\0') |
546 | 13.1k | break; |
547 | | |
548 | 183M | if (new_lines > 0) { |
549 | | // Otherwise store whitespace in fodder. |
550 | 19.3M | unsigned blanks = new_lines - 1; |
551 | 19.3M | fodder.emplace_back(FodderElement::LINE_END, blanks, indent, EMPTY); |
552 | 19.3M | fresh_line = true; |
553 | 19.3M | } |
554 | | |
555 | 183M | Location begin(line_number, c - line_start + 1); |
556 | | |
557 | 183M | switch (*c) { |
558 | | // The following operators should never be combined with subsequent symbols. |
559 | 773k | case '{': |
560 | 773k | kind = Token::BRACE_L; |
561 | 773k | c++; |
562 | 773k | break; |
563 | | |
564 | 758k | case '}': |
565 | 758k | kind = Token::BRACE_R; |
566 | 758k | c++; |
567 | 758k | break; |
568 | | |
569 | 4.15M | case '[': |
570 | 4.15M | kind = Token::BRACKET_L; |
571 | 4.15M | c++; |
572 | 4.15M | break; |
573 | | |
574 | 4.13M | case ']': |
575 | 4.13M | kind = Token::BRACKET_R; |
576 | 4.13M | c++; |
577 | 4.13M | break; |
578 | | |
579 | 16.0M | case ',': |
580 | 16.0M | kind = Token::COMMA; |
581 | 16.0M | c++; |
582 | 16.0M | break; |
583 | | |
584 | 9.25M | case '.': |
585 | 9.25M | kind = Token::DOT; |
586 | 9.25M | c++; |
587 | 9.25M | break; |
588 | | |
589 | 13.9M | case '(': |
590 | 13.9M | kind = Token::PAREN_L; |
591 | 13.9M | c++; |
592 | 13.9M | break; |
593 | | |
594 | 13.9M | case ')': |
595 | 13.9M | kind = Token::PAREN_R; |
596 | 13.9M | c++; |
597 | 13.9M | break; |
598 | | |
599 | 3.64M | case ';': |
600 | 3.64M | kind = Token::SEMICOLON; |
601 | 3.64M | c++; |
602 | 3.64M | break; |
603 | | |
604 | | // Numeric literals. |
605 | 1.98M | case '0': |
606 | 4.31M | case '1': |
607 | 4.79M | case '2': |
608 | 5.21M | case '3': |
609 | 5.55M | case '4': |
610 | 5.62M | case '5': |
611 | 6.17M | case '6': |
612 | 6.33M | case '7': |
613 | 7.36M | case '8': |
614 | 8.19M | case '9': |
615 | 8.19M | kind = Token::NUMBER; |
616 | 8.19M | data = lex_number(c, filename, begin); |
617 | 8.19M | break; |
618 | | |
619 | | // UString literals. |
620 | 283k | case '"': { |
621 | 283k | c++; |
622 | 48.9M | for (;; ++c) { |
623 | 48.9M | if (*c == '\0') { |
624 | 42 | throw StaticError(filename, begin, "unterminated string"); |
625 | 42 | } |
626 | 48.9M | if (*c == '"') { |
627 | 283k | break; |
628 | 283k | } |
629 | 48.6M | if (*c == '\\' && *(c + 1) != '\0') { |
630 | 126k | data += *c; |
631 | 126k | ++c; |
632 | 126k | } |
633 | 48.6M | if (*c == '\n') { |
634 | | // Maintain line/column counters. |
635 | 2.51M | line_number++; |
636 | 2.51M | line_start = c + 1; |
637 | 2.51M | } |
638 | 48.6M | data += *c; |
639 | 48.6M | } |
640 | 283k | c++; // Advance beyond the ". |
641 | 283k | kind = Token::STRING_DOUBLE; |
642 | 283k | } break; |
643 | | |
644 | | // UString literals. |
645 | 6.05M | case '\'': { |
646 | 6.05M | c++; |
647 | 85.3M | for (;; ++c) { |
648 | 85.3M | if (*c == '\0') { |
649 | 45 | throw StaticError(filename, begin, "unterminated string"); |
650 | 45 | } |
651 | 85.3M | if (*c == '\'') { |
652 | 6.05M | break; |
653 | 6.05M | } |
654 | 79.2M | if (*c == '\\' && *(c + 1) != '\0') { |
655 | 716k | data += *c; |
656 | 716k | ++c; |
657 | 716k | } |
658 | 79.2M | if (*c == '\n') { |
659 | | // Maintain line/column counters. |
660 | 2.21M | line_number++; |
661 | 2.21M | line_start = c + 1; |
662 | 2.21M | } |
663 | 79.2M | data += *c; |
664 | 79.2M | } |
665 | 6.05M | c++; // Advance beyond the '. |
666 | 6.05M | kind = Token::STRING_SINGLE; |
667 | 6.05M | } break; |
668 | | |
669 | | // Verbatim string literals. |
670 | | // ' and " quoting is interpreted here, unlike non-verbatim strings |
671 | | // where it is done later by jsonnet_string_unescape. This is OK |
672 | | // in this case because no information is lost by resoving the |
673 | | // repeated quote into a single quote, so we can go back to the |
674 | | // original form in the formatter. |
675 | 6.17k | case '@': { |
676 | 6.17k | c++; |
677 | 6.17k | if (*c != '"' && *c != '\'') { |
678 | 24 | std::stringstream ss; |
679 | 24 | ss << "couldn't lex verbatim string, junk after '@': " << *c; |
680 | 24 | throw StaticError(filename, begin, ss.str()); |
681 | 24 | } |
682 | 6.15k | const char quot = *c; |
683 | 6.15k | c++; // Advance beyond the opening quote. |
684 | 55.0k | for (;; ++c) { |
685 | 55.0k | if (*c == '\0') { |
686 | 48 | throw StaticError(filename, begin, "unterminated verbatim string"); |
687 | 48 | } |
688 | 55.0k | if (*c == quot) { |
689 | 9.39k | if (*(c + 1) == quot) { |
690 | 3.29k | c++; |
691 | 6.10k | } else { |
692 | 6.10k | break; |
693 | 6.10k | } |
694 | 9.39k | } |
695 | 48.9k | data += *c; |
696 | 48.9k | } |
697 | 6.10k | c++; // Advance beyond the closing quote. |
698 | 6.10k | if (quot == '"') { |
699 | 2.97k | kind = Token::VERBATIM_STRING_DOUBLE; |
700 | 3.13k | } else { |
701 | 3.13k | kind = Token::VERBATIM_STRING_SINGLE; |
702 | 3.13k | } |
703 | 6.10k | } break; |
704 | | |
705 | | // Keywords |
706 | 101M | default: |
707 | 101M | if (is_identifier_first(*c)) { |
708 | 77.1M | std::string id; |
709 | 423M | for (; is_identifier(*c); ++c) |
710 | 345M | id += *c; |
711 | 77.1M | kind = lex_get_keyword_kind(id); |
712 | 77.1M | data = id; |
713 | | |
714 | 77.1M | } else if (is_symbol(*c) || *c == '#') { |
715 | | // Single line C++ and Python style comments. |
716 | 24.7M | if (*c == '#' || (*c == '/' && *(c + 1) == '/')) { |
717 | 4.74M | std::vector<std::string> comment(1); |
718 | 4.74M | unsigned blanks; |
719 | 4.74M | unsigned indent; |
720 | 4.74M | lex_until_newline(c, comment[0], blanks, indent, line_start, line_number); |
721 | 4.74M | auto kind = fresh_line ? FodderElement::PARAGRAPH : FodderElement::LINE_END; |
722 | 4.74M | fodder.emplace_back(kind, blanks, indent, comment); |
723 | 4.74M | fresh_line = true; |
724 | 4.74M | continue; // We've not got a token, just fodder, so keep scanning. |
725 | 4.74M | } |
726 | | |
727 | | // Multi-line C style comment. |
728 | 20.0M | if (*c == '/' && *(c + 1) == '*') { |
729 | 267k | unsigned margin = c - line_start; |
730 | | |
731 | 267k | const char *initial_c = c; |
732 | 267k | c += 2; // Avoid matching /*/: skip the /* before starting the search for |
733 | | // */. |
734 | | |
735 | 77.7M | while (!(*c == '*' && *(c + 1) == '/')) { |
736 | 77.5M | if (*c == '\0') { |
737 | 127 | auto msg = "multi-line comment has no terminating */."; |
738 | 127 | throw StaticError(filename, begin, msg); |
739 | 127 | } |
740 | 77.5M | if (*c == '\n') { |
741 | | // Just keep track of the line / column counters. |
742 | 10.1M | line_number++; |
743 | 10.1M | line_start = c + 1; |
744 | 10.1M | } |
745 | 77.5M | ++c; |
746 | 77.5M | } |
747 | 267k | c += 2; // Move the pointer to the char after the closing '/'. |
748 | | |
749 | 267k | std::string comment(initial_c, |
750 | 267k | c - initial_c); // Includes the "/*" and "*/". |
751 | | |
752 | | // Lex whitespace after comment |
753 | 267k | unsigned new_lines_after, indent_after; |
754 | 267k | lex_ws(c, new_lines_after, indent_after, line_start, line_number); |
755 | 267k | std::vector<std::string> lines; |
756 | 267k | if (comment.find('\n') >= comment.length()) { |
757 | | // Comment looks like /* foo */ |
758 | 112k | lines.push_back(comment); |
759 | 112k | fodder.emplace_back(FodderElement::INTERSTITIAL, 0, 0, lines); |
760 | 112k | if (new_lines_after > 0) { |
761 | 109k | fodder.emplace_back(FodderElement::LINE_END, |
762 | 109k | new_lines_after - 1, |
763 | 109k | indent_after, |
764 | 109k | EMPTY); |
765 | 109k | fresh_line = true; |
766 | 109k | } |
767 | 154k | } else { |
768 | 154k | lines = line_split(comment, margin); |
769 | 154k | assert(lines[0][0] == '/'); |
770 | | // Little hack to support PARAGRAPHs with * down the LHS: |
771 | | // Add a space to lines that start with a '*' |
772 | 154k | bool all_star = true; |
773 | 10.2M | for (auto &l : lines) { |
774 | 10.2M | if (l[0] != '*') |
775 | 10.1M | all_star = false; |
776 | 10.2M | } |
777 | 154k | if (all_star) { |
778 | 0 | for (auto &l : lines) { |
779 | 0 | if (l[0] == '*') |
780 | 0 | l = " " + l; |
781 | 0 | } |
782 | 0 | } |
783 | 154k | if (new_lines_after == 0) { |
784 | | // Ensure a line end after the paragraph. |
785 | 11.3k | new_lines_after = 1; |
786 | 11.3k | indent_after = 0; |
787 | 11.3k | } |
788 | 154k | fodder_push_back(fodder, |
789 | 154k | FodderElement(FodderElement::PARAGRAPH, |
790 | 154k | new_lines_after - 1, |
791 | 154k | indent_after, |
792 | 154k | lines)); |
793 | 154k | fresh_line = true; |
794 | 154k | } |
795 | 267k | continue; // We've not got a token, just fodder, so keep scanning. |
796 | 267k | } |
797 | | |
798 | | // Text block |
799 | 19.7M | if (*c == '|' && *(c + 1) == '|' && *(c + 2) == '|') { |
800 | 10.7k | c += 3; // Skip the "|||". |
801 | | |
802 | 10.7k | bool chomp_trailing_nl = false; |
803 | 10.7k | if (*c == '-') { |
804 | 540 | chomp_trailing_nl = true; |
805 | 540 | c++; |
806 | 540 | } |
807 | | |
808 | 13.7k | while (is_horz_ws(*c)) ++c; // Chomp whitespace at end of line. |
809 | 10.7k | if (*c != '\n') { |
810 | 79 | auto msg = "text block syntax requires new line after |||."; |
811 | 79 | throw StaticError(filename, begin, msg); |
812 | 79 | } |
813 | 10.6k | std::stringstream block; |
814 | 10.6k | c++; // Skip the "\n" |
815 | 10.6k | line_number++; |
816 | | // Skip any blank lines at the beginning of the block. |
817 | 12.8k | while (*c == '\n') { |
818 | 2.20k | line_number++; |
819 | 2.20k | ++c; |
820 | 2.20k | block << '\n'; |
821 | 2.20k | } |
822 | 10.6k | line_start = c; |
823 | 10.6k | const char *first_line = c; |
824 | 10.6k | int ws_chars = whitespace_check(first_line, c); |
825 | 10.6k | string_block_indent = std::string(first_line, ws_chars); |
826 | 10.6k | if (ws_chars == 0) { |
827 | 42 | auto msg = "text block's first line must start with whitespace."; |
828 | 42 | throw StaticError(filename, begin, msg); |
829 | 42 | } |
830 | 13.4k | while (true) { |
831 | 13.4k | assert(ws_chars > 0); |
832 | | // Read up to the \n |
833 | 160k | for (c = &c[ws_chars]; *c != '\n'; ++c) { |
834 | 147k | if (*c == '\0') |
835 | 109 | throw StaticError(filename, begin, "unexpected EOF"); |
836 | 147k | block << *c; |
837 | 147k | } |
838 | | // Add the \n |
839 | 13.3k | block << '\n'; |
840 | 13.3k | ++c; |
841 | 13.3k | line_number++; |
842 | 13.3k | line_start = c; |
843 | | // Skip any blank lines |
844 | 15.5k | while (*c == '\n') { |
845 | 2.16k | line_number++; |
846 | 2.16k | ++c; |
847 | 2.16k | block << '\n'; |
848 | 2.16k | } |
849 | | // Examine next line |
850 | 13.3k | ws_chars = whitespace_check(first_line, c); |
851 | 13.3k | if (ws_chars == 0) { |
852 | | // End of text block (or indentation error). |
853 | | // Count actual whitespace on this line. |
854 | 10.4k | int actual_ws = 0; |
855 | 75.1k | while (c[actual_ws] == ' ' || |
856 | 64.6k | c[actual_ws] == '\t') { |
857 | 64.6k | actual_ws++; |
858 | 64.6k | } |
859 | | |
860 | | // Check if this is the terminator ||| |
861 | 10.4k | bool is_terminator = ( |
862 | 10.4k | c[actual_ws] == '|' && |
863 | 10.3k | c[actual_ws + 1] == '|' && |
864 | 10.3k | c[actual_ws + 2] == '|'); |
865 | | |
866 | 10.4k | if (!is_terminator) { |
867 | | // Not a terminator - check if it's an |
868 | | // indentation issue. |
869 | 158 | if (actual_ws > 0) { |
870 | | // Has whitespace but doesn't match expected |
871 | | // indentation. |
872 | 77 | std::stringstream msg; |
873 | 77 | msg << "text block indentation mismatch: " |
874 | 77 | "expected at least "; |
875 | 77 | describe_whitespace(msg, string_block_indent); |
876 | 77 | msg << ", found "; |
877 | 77 | describe_whitespace(msg, std::string(c, actual_ws)); |
878 | 77 | throw StaticError(filename, begin, msg.str()); |
879 | 81 | } else { |
880 | | // No whitespace and no ||| - missing |
881 | | // terminator. |
882 | 81 | auto msg = |
883 | 81 | "text block not terminated with |||"; |
884 | 81 | throw StaticError(filename, begin, msg); |
885 | 81 | } |
886 | 158 | } |
887 | | |
888 | | // Valid termination - skip over any whitespace. |
889 | 62.3k | while (*c == ' ' || *c == '\t') { |
890 | 52.0k | string_block_term_indent += *c; |
891 | 52.0k | ++c; |
892 | 52.0k | } |
893 | | // Skip the ||| |
894 | 10.3k | c += 3; // Leave after the last | |
895 | 10.3k | data = block.str(); |
896 | 10.3k | kind = Token::STRING_BLOCK; |
897 | 10.3k | if (chomp_trailing_nl) { |
898 | 518 | assert(data.back() == '\n'); |
899 | 518 | data.pop_back(); |
900 | 518 | } |
901 | 10.3k | break; // Out of the while loop. |
902 | 10.3k | } |
903 | 13.3k | } |
904 | | |
905 | 10.3k | break; // Out of the switch. |
906 | 10.5k | } |
907 | | |
908 | 19.7M | const char *operator_begin = c; |
909 | 66.8M | for (; is_symbol(*c); ++c) { |
910 | | // Not allowed // in operators |
911 | 47.0M | if (*c == '/' && *(c + 1) == '/') |
912 | 628 | break; |
913 | | // Not allowed /* in operators |
914 | 47.0M | if (*c == '/' && *(c + 1) == '*') |
915 | 664 | break; |
916 | | // Not allowed ||| in operators |
917 | 47.0M | if (*c == '|' && *(c + 1) == '|' && *(c + 2) == '|') |
918 | 3.03k | break; |
919 | 47.0M | } |
920 | | // Not allowed to end with a + - ~ ! unless a single char. |
921 | | // So, wind it back if we need to (but not too far). |
922 | 24.0M | while (c > operator_begin + 1 && !allowed_at_end_of_operator(*(c - 1))) { |
923 | 4.31M | c--; |
924 | 4.31M | } |
925 | 19.7M | data += std::string(operator_begin, c); |
926 | 19.7M | if (data == "$") { |
927 | 55.2k | kind = Token::DOLLAR; |
928 | 55.2k | data = ""; |
929 | 19.6M | } else { |
930 | 19.6M | kind = Token::OPERATOR; |
931 | 19.6M | } |
932 | 19.7M | } else { |
933 | 162 | std::stringstream ss; |
934 | 162 | ss << "Could not lex the character "; |
935 | 162 | auto uc = (unsigned char)(*c); |
936 | 162 | if (*c < 32) |
937 | 147 | ss << "code " << unsigned(uc); |
938 | 15 | else |
939 | 15 | ss << "'" << *c << "'"; |
940 | 162 | throw StaticError(filename, begin, ss.str()); |
941 | 162 | } |
942 | 183M | } |
943 | | |
944 | | // Ensure that a bug in the above code does not cause an infinite memory consuming loop due |
945 | | // to pushing empty tokens. |
946 | 178M | if (c == original_c) { |
947 | 0 | throw StaticError(filename, begin, "internal lexing error: pointer did not advance"); |
948 | 0 | } |
949 | | |
950 | 178M | Location end(line_number, (c + 1) - line_start); |
951 | 178M | r.emplace_back(kind, |
952 | 178M | fodder, |
953 | 178M | data, |
954 | 178M | string_block_indent, |
955 | 178M | string_block_term_indent, |
956 | 178M | LocationRange(filename, begin, end)); |
957 | 178M | fodder.clear(); |
958 | 178M | fresh_line = false; |
959 | 178M | } |
960 | | |
961 | 24.7k | Location begin(line_number, c - line_start + 1); |
962 | 24.7k | Location end(line_number, (c + 1) - line_start + 1); |
963 | 24.7k | r.emplace_back(Token::END_OF_FILE, fodder, "", "", "", LocationRange(filename, begin, end)); |
964 | 24.7k | return r; |
965 | 25.6k | } |
966 | | |
967 | | std::string jsonnet_unlex(const Tokens &tokens) |
968 | 0 | { |
969 | 0 | std::stringstream ss; |
970 | 0 | for (const auto &t : tokens) { |
971 | 0 | for (const auto &f : t.fodder) { |
972 | 0 | switch (f.kind) { |
973 | 0 | case FodderElement::LINE_END: { |
974 | 0 | if (f.comment.size() > 0) { |
975 | 0 | ss << "LineEnd(" << f.blanks << ", " << f.indent << ", " << f.comment[0] |
976 | 0 | << ")\n"; |
977 | 0 | } else { |
978 | 0 | ss << "LineEnd(" << f.blanks << ", " << f.indent << ")\n"; |
979 | 0 | } |
980 | 0 | } break; |
981 | | |
982 | 0 | case FodderElement::INTERSTITIAL: { |
983 | 0 | ss << "Interstitial(" << f.comment[0] << ")\n"; |
984 | 0 | } break; |
985 | | |
986 | 0 | case FodderElement::PARAGRAPH: { |
987 | 0 | ss << "Paragraph(\n"; |
988 | 0 | for (const auto &line : f.comment) { |
989 | 0 | ss << " " << line << '\n'; |
990 | 0 | } |
991 | 0 | ss << ")" << f.blanks << "\n"; |
992 | 0 | } break; |
993 | 0 | } |
994 | 0 | } |
995 | 0 | if (t.kind == Token::END_OF_FILE) { |
996 | 0 | ss << "EOF\n"; |
997 | 0 | break; |
998 | 0 | } |
999 | 0 | if (t.kind == Token::STRING_DOUBLE) { |
1000 | 0 | ss << "\"" << t.data << "\"\n"; |
1001 | 0 | } else if (t.kind == Token::STRING_SINGLE) { |
1002 | 0 | ss << "'" << t.data << "'\n"; |
1003 | 0 | } else if (t.kind == Token::STRING_BLOCK) { |
1004 | 0 | ss << "|||\n"; |
1005 | 0 | ss << t.stringBlockIndent; |
1006 | 0 | for (const char *cp = t.data.c_str(); *cp != '\0'; ++cp) { |
1007 | 0 | ss << *cp; |
1008 | 0 | if (*cp == '\n' && *(cp + 1) != '\n' && *(cp + 1) != '\0') { |
1009 | 0 | ss << t.stringBlockIndent; |
1010 | 0 | } |
1011 | 0 | } |
1012 | 0 | ss << t.stringBlockTermIndent << "|||\n"; |
1013 | 0 | } else { |
1014 | 0 | ss << t.data << "\n"; |
1015 | 0 | } |
1016 | 0 | } |
1017 | 0 | return ss.str(); |
1018 | 0 | } |
1019 | | |
1020 | | } // namespace jsonnet::internal |