Coverage Report

Created: 2026-09-13 06:14

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/jsonnet/core/lexer.cpp
Line
Count
Source
1
/*
2
Copyright 2015 Google Inc. All rights reserved.
3
4
Licensed under the Apache License, Version 2.0 (the "License");
5
you may not use this file except in compliance with the License.
6
You may obtain a copy of the License at
7
8
    http://www.apache.org/licenses/LICENSE-2.0
9
10
Unless required by applicable law or agreed to in writing, software
11
distributed under the License is distributed on an "AS IS" BASIS,
12
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
See the License for the specific language governing permissions and
14
limitations under the License.
15
*/
16
17
#include <cassert>
18
19
#include <map>
20
#include <sstream>
21
#include <string>
22
23
#include "lexer.h"
24
#include "static_error.h"
25
#include "unicode.h"
26
27
namespace jsonnet::internal {
28
29
static const std::vector<std::string> EMPTY;
30
31
/** Is the char whitespace (excluding \n). */
32
static bool is_horz_ws(char c)
33
485M
{
34
485M
    return c == ' ' || c == '\t' || c == '\r';
35
485M
}
36
37
/** Is the char whitespace. */
38
static bool is_ws(char c)
39
436M
{
40
436M
    return c == '\n' || is_horz_ws(c);
41
436M
}
42
43
/** Strip whitespace from both ends of a string, but only up to margin on the left hand side. */
44
static std::string strip_ws(const std::string &s, unsigned margin)
45
10.2M
{
46
10.2M
    if (s.size() == 0)
47
8.74M
        return s;  // Avoid underflow below.
48
1.54M
    size_t i = 0;
49
4.66M
    while (i < s.length() && is_horz_ws(s[i]) && i < margin)
50
3.12M
        i++;
51
1.54M
    size_t j = s.size();
52
4.45M
    while (j > i && is_horz_ws(s[j - 1])) {
53
2.91M
        j--;
54
2.91M
    }
55
1.54M
    return std::string(&s[i], &s[j]);
56
10.2M
}
57
58
/** Split a string by \n and also strip left (up to margin) & right whitespace from each line. */
59
static std::vector<std::string> line_split(const std::string &s, unsigned margin)
60
154k
{
61
154k
    std::vector<std::string> ret;
62
154k
    std::stringstream ss;
63
71.6M
    for (size_t i = 0; i < s.length(); ++i) {
64
71.4M
        if (s[i] == '\n') {
65
10.1M
            ret.emplace_back(strip_ws(ss.str(), margin));
66
10.1M
            ss.str("");
67
61.3M
        } else {
68
61.3M
            ss << s[i];
69
61.3M
        }
70
71.4M
    }
71
154k
    ret.emplace_back(strip_ws(ss.str(), margin));
72
154k
    return ret;
73
154k
}
74
75
/** Consume whitespace.
76
 *
77
 * Return number of \n and number of spaces after last \n.  Convert \t to spaces.
78
 */
79
static void lex_ws(const char *&c, unsigned &new_lines, unsigned &indent, const char *&line_start,
80
                   unsigned long &line_number)
81
188M
{
82
188M
    indent = 0;
83
188M
    new_lines = 0;
84
436M
    for (; *c != '\0' && is_ws(*c); c++) {
85
248M
        switch (*c) {
86
564k
            case '\r':
87
                // Ignore.
88
564k
                break;
89
90
28.5M
            case '\n':
91
28.5M
                indent = 0;
92
28.5M
                new_lines++;
93
28.5M
                line_number++;
94
28.5M
                line_start = c + 1;
95
28.5M
                break;
96
97
219M
            case ' ': indent += 1; break;
98
99
            // This only works for \t at the beginning of lines, but we strip it everywhere else
100
            // anyway.  The only case where this will cause a problem is spaces followed by \t
101
            // at the beginning of a line.  However that is rare, ill-advised, and if re-indentation
102
            // is enabled it will be fixed later.
103
50.7k
            case '\t': indent += 8; break;
104
248M
        }
105
248M
    }
106
188M
}
107
108
/**
109
# Consume all text until the end of the line, return number of newlines after that and indent
110
*/
111
static void lex_until_newline(const char *&c, std::string &text, unsigned &blanks, unsigned &indent,
112
                              const char *&line_start, unsigned long &line_number)
113
4.74M
{
114
4.74M
    const char *original_c = c;
115
4.74M
    const char *last_non_space = c;
116
73.1M
    for (; *c != '\0' && *c != '\n'; c++) {
117
68.3M
        if (!is_horz_ws(*c))
118
58.7M
            last_non_space = c;
119
68.3M
    }
120
4.74M
    text = std::string(original_c, last_non_space - original_c + 1);
121
    // Consume subsequent whitespace including the '\n'.
122
4.74M
    unsigned new_lines;
123
4.74M
    lex_ws(c, new_lines, indent, line_start, line_number);
124
4.74M
    blanks = new_lines == 0 ? 0 : new_lines - 1;
125
4.74M
}
126
127
static bool is_upper(char c)
128
525M
{
129
525M
    return c >= 'A' && c <= 'Z';
130
525M
}
131
132
static bool is_lower(char c)
133
515M
{
134
515M
    return c >= 'a' && c <= 'z';
135
515M
}
136
137
static bool is_number(char c)
138
97.5M
{
139
97.5M
    return c >= '0' && c <= '9';
140
97.5M
}
141
142
static bool is_identifier_first(char c)
143
525M
{
144
525M
    return is_upper(c) || is_lower(c) || c == '_';
145
525M
}
146
147
static bool is_identifier(char c)
148
423M
{
149
423M
    return is_identifier_first(c) || is_number(c);
150
423M
}
151
152
static bool is_symbol(char c)
153
91.6M
{
154
91.6M
    switch (c) {
155
2.09M
        case '!':
156
2.31M
        case '$':
157
11.8M
        case ':':
158
12.5M
        case '~':
159
25.5M
        case '+':
160
27.1M
        case '-':
161
28.9M
        case '&':
162
29.9M
        case '|':
163
29.9M
        case '^':
164
47.3M
        case '=':
165
48.9M
        case '<':
166
51.1M
        case '>':
167
65.6M
        case '*':
168
67.5M
        case '/':
169
68.3M
        case '%': return true;
170
91.6M
    }
171
23.2M
    return false;
172
91.6M
}
173
174
9.81M
bool allowed_at_end_of_operator(char c) {
175
9.81M
    switch (c) {
176
2.82M
        case '+':
177
2.98M
        case '-':
178
3.65M
        case '~':
179
4.20M
        case '!':
180
4.31M
        case '$': return false;
181
9.81M
    }
182
5.50M
    return true;
183
9.81M
}
184
185
static const std::map<std::string, Token::Kind> keywords = {
186
    {"assert", Token::ASSERT},
187
    {"else", Token::ELSE},
188
    {"error", Token::ERROR},
189
    {"false", Token::FALSE},
190
    {"for", Token::FOR},
191
    {"function", Token::FUNCTION},
192
    {"if", Token::IF},
193
    {"import", Token::IMPORT},
194
    {"importstr", Token::IMPORTSTR},
195
    {"importbin", Token::IMPORTBIN},
196
    {"in", Token::IN},
197
    {"local", Token::LOCAL},
198
    {"null", Token::NULL_LIT},
199
    {"self", Token::SELF},
200
    {"super", Token::SUPER},
201
    {"tailstrict", Token::TAILSTRICT},
202
    {"then", Token::THEN},
203
    {"true", Token::TRUE},
204
};
205
206
Token::Kind lex_get_keyword_kind(const std::string &identifier)
207
77.1M
{
208
77.1M
    auto it = keywords.find(identifier);
209
77.1M
    if (it == keywords.end())
210
56.4M
        return Token::IDENTIFIER;
211
20.7M
    return it->second;
212
77.1M
}
213
214
std::string lex_number(const char *&c, const std::string &filename, const Location &begin)
215
8.19M
{
216
    // This function should be understood with reference to the linked image:
217
    // https://www.json.org/img/number.png
218
219
    // Note, we deviate from the json.org documentation as follows:
220
    // * There is no reason to lex negative numbers as atomic tokens, it is better to parse them
221
    //   as a unary operator combined with a numeric literal.  This avoids x-1 being tokenized as
222
    //   <identifier> <number> instead of the intended <identifier> <binop> <number>.
223
    // * We support digit separators using the _ character for readability in
224
    //   large numeric literals.
225
226
8.19M
    enum State {
227
8.19M
        BEGIN,
228
8.19M
        AFTER_ZERO,
229
8.19M
        AFTER_ONE_TO_NINE,
230
8.19M
        AFTER_INT_UNDERSCORE,
231
8.19M
        AFTER_DOT,
232
8.19M
        AFTER_DIGIT,
233
8.19M
        AFTER_FRAC_UNDERSCORE,
234
8.19M
        AFTER_E,
235
8.19M
        AFTER_EXP_SIGN,
236
8.19M
        AFTER_EXP_DIGIT,
237
8.19M
        AFTER_EXP_UNDERSCORE
238
8.19M
    } state;
239
240
8.19M
    std::string r;
241
242
8.19M
    state = BEGIN;
243
18.0M
    while (true) {
244
18.0M
        switch (state) {
245
8.19M
            case BEGIN:
246
8.19M
                switch (*c) {
247
1.98M
                    case '0': state = AFTER_ZERO; break;
248
249
2.33M
                    case '1':
250
2.80M
                    case '2':
251
3.23M
                    case '3':
252
3.57M
                    case '4':
253
3.64M
                    case '5':
254
4.19M
                    case '6':
255
4.35M
                    case '7':
256
5.38M
                    case '8':
257
6.21M
                    case '9': state = AFTER_ONE_TO_NINE; break;
258
259
0
                    default: throw StaticError(filename, begin, "couldn't lex number");
260
8.19M
                }
261
8.19M
                break;
262
263
8.19M
            case AFTER_ZERO:
264
1.98M
                switch (*c) {
265
27.8k
                    case '.': state = AFTER_DOT; break;
266
267
640
                    case 'e':
268
1.39k
                    case 'E': state = AFTER_E; break;
269
270
3
                    case '_': {
271
3
                        std::stringstream ss;
272
3
                        ss << "couldn't lex number, _ not allowed after leading 0";
273
3
                        throw StaticError(filename, begin, ss.str());
274
640
                    }
275
276
1.95M
                    default: goto end;
277
1.98M
                }
278
29.2k
                break;
279
280
7.47M
            case AFTER_ONE_TO_NINE:
281
7.47M
                switch (*c) {
282
17.2k
                    case '.': state = AFTER_DOT; break;
283
284
1.56k
                    case 'e':
285
3.01k
                    case 'E': state = AFTER_E; break;
286
287
621k
                    case '0':
288
648k
                    case '1':
289
796k
                    case '2':
290
851k
                    case '3':
291
914k
                    case '4':
292
1.04M
                    case '5':
293
1.13M
                    case '6':
294
1.17M
                    case '7':
295
1.22M
                    case '8':
296
1.26M
                    case '9': state = AFTER_ONE_TO_NINE; break;
297
298
378
                    case '_': state = AFTER_INT_UNDERSCORE; goto skip_char;
299
300
6.19M
                    default: goto end;
301
7.47M
                }
302
1.28M
                break;
303
304
1.28M
            case AFTER_INT_UNDERSCORE:
305
378
                switch (*c) {
306
                    // The only valid transition from _ is to a digit.
307
215
                    case '0':
308
240
                    case '1':
309
248
                    case '2':
310
248
                    case '3':
311
257
                    case '4':
312
257
                    case '5':
313
257
                    case '6':
314
257
                    case '7':
315
359
                    case '8':
316
359
                    case '9': state = AFTER_ONE_TO_NINE; break;
317
318
19
                    default: {
319
19
                        std::stringstream ss;
320
19
                        ss << "couldn't lex number, junk after _: " << *c;
321
19
                        throw StaticError(filename, begin, ss.str());
322
359
                    }
323
378
                }
324
359
                break;
325
326
45.1k
            case AFTER_DOT:
327
45.1k
                switch (*c) {
328
1.58k
                    case '0':
329
15.0k
                    case '1':
330
15.8k
                    case '2':
331
16.6k
                    case '3':
332
17.4k
                    case '4':
333
43.2k
                    case '5':
334
43.7k
                    case '6':
335
43.8k
                    case '7':
336
44.8k
                    case '8':
337
45.1k
                    case '9': state = AFTER_DIGIT; break;
338
339
15
                    default: {
340
15
                        std::stringstream ss;
341
15
                        ss << "couldn't lex number, junk after decimal point: " << *c;
342
15
                        throw StaticError(filename, begin, ss.str());
343
44.8k
                    }
344
45.1k
                }
345
45.1k
                break;
346
347
315k
            case AFTER_DIGIT:
348
315k
                switch (*c) {
349
690
                    case 'e':
350
1.39k
                    case 'E': state = AFTER_E; break;
351
352
29.8k
                    case '0':
353
70.3k
                    case '1':
354
84.8k
                    case '2':
355
112k
                    case '3':
356
126k
                    case '4':
357
167k
                    case '5':
358
195k
                    case '6':
359
209k
                    case '7':
360
230k
                    case '8':
361
270k
                    case '9': state = AFTER_DIGIT; break;
362
363
435
                    case '_': state = AFTER_FRAC_UNDERSCORE; goto skip_char;
364
365
43.7k
                    default: goto end;
366
315k
                }
367
271k
                break;
368
369
271k
            case AFTER_FRAC_UNDERSCORE:
370
435
                switch (*c) {
371
                    // The only valid transition from _ is to a digit.
372
159
                    case '0':
373
198
                    case '1':
374
199
                    case '2':
375
199
                    case '3':
376
207
                    case '4':
377
210
                    case '5':
378
210
                    case '6':
379
210
                    case '7':
380
430
                    case '8':
381
430
                    case '9': state = AFTER_DIGIT; break;
382
383
5
                    default: {
384
5
                        std::stringstream ss;
385
5
                        ss << "couldn't lex number, junk after _: " << *c;
386
5
                        throw StaticError(filename, begin, ss.str());
387
430
                    }
388
435
                }
389
430
                break;
390
391
5.80k
            case AFTER_E:
392
5.80k
                switch (*c) {
393
860
                    case '+':
394
2.24k
                    case '-': state = AFTER_EXP_SIGN; break;
395
396
1.16k
                    case '0':
397
1.94k
                    case '1':
398
2.22k
                    case '2':
399
2.55k
                    case '3':
400
2.69k
                    case '4':
401
2.73k
                    case '5':
402
2.93k
                    case '6':
403
3.02k
                    case '7':
404
3.21k
                    case '8':
405
3.51k
                    case '9': state = AFTER_EXP_DIGIT; break;
406
407
43
                    default: {
408
43
                        std::stringstream ss;
409
43
                        ss << "couldn't lex number, junk after 'E': " << *c;
410
43
                        throw StaticError(filename, begin, ss.str());
411
3.21k
                    }
412
5.80k
                }
413
5.75k
                break;
414
415
5.75k
            case AFTER_EXP_SIGN:
416
2.24k
                switch (*c) {
417
780
                    case '0':
418
1.12k
                    case '1':
419
1.43k
                    case '2':
420
1.95k
                    case '3':
421
2.21k
                    case '4':
422
2.21k
                    case '5':
423
2.22k
                    case '6':
424
2.22k
                    case '7':
425
2.22k
                    case '8':
426
2.23k
                    case '9': state = AFTER_EXP_DIGIT; break;
427
428
9
                    default: {
429
9
                        std::stringstream ss;
430
9
                        ss << "couldn't lex number, junk after exponent sign: " << *c;
431
9
                        throw StaticError(filename, begin, ss.str());
432
2.22k
                    }
433
2.24k
                }
434
2.23k
                break;
435
436
34.3k
            case AFTER_EXP_DIGIT:
437
34.3k
                switch (*c) {
438
9.56k
                    case '0':
439
12.1k
                    case '1':
440
14.3k
                    case '2':
441
16.5k
                    case '3':
442
19.0k
                    case '4':
443
20.3k
                    case '5':
444
21.8k
                    case '6':
445
24.6k
                    case '7':
446
26.5k
                    case '8':
447
28.3k
                    case '9': state = AFTER_EXP_DIGIT; break;
448
449
302
                    case '_': state = AFTER_EXP_UNDERSCORE; goto skip_char;
450
451
5.74k
                    default: goto end;
452
34.3k
                }
453
28.3k
                break;
454
455
28.3k
            case AFTER_EXP_UNDERSCORE:
456
302
                switch (*c) {
457
                    // The only valid transition from _ is to a digit.
458
13
                    case '0':
459
172
                    case '1':
460
175
                    case '2':
461
176
                    case '3':
462
273
                    case '4':
463
273
                    case '5':
464
281
                    case '6':
465
283
                    case '7':
466
289
                    case '8':
467
296
                    case '9': state = AFTER_EXP_DIGIT; break;
468
469
6
                    default: {
470
6
                        std::stringstream ss;
471
6
                        ss << "couldn't lex number, junk after _: " << *c;
472
6
                        throw StaticError(filename, begin, ss.str());
473
289
                    }
474
302
                }
475
296
                break;
476
18.0M
        }
477
9.86M
        r += *c;
478
479
9.86M
skip_char:
480
9.86M
        c++;
481
9.86M
    }
482
8.19M
end:
483
8.19M
    return r;
484
8.19M
}
485
486
// Check that b has at least the same whitespace prefix as a and returns the amount of this
487
// whitespace, otherwise returns 0.  If a has no whitespace prefix than return 0.
488
static int whitespace_check(const char *a, const char *b)
489
23.9k
{
490
23.9k
    int i = 0;
491
252k
    while (a[i] == ' ' || a[i] == '\t') {
492
239k
        if (b[i] != a[i])
493
10.4k
            return 0;
494
228k
        i++;
495
228k
    }
496
13.5k
    return i;
497
23.9k
}
498
499
154
static void describe_whitespace(std::stringstream& msg, const std::string& ws) {
500
154
    int spaces = 0;
501
154
    int tabs = 0;
502
22.7k
    for (char c : ws) {
503
22.7k
        if (c == ' ')
504
801
            spaces++;
505
21.9k
        else if (c == '\t')
506
21.9k
            tabs++;
507
22.7k
    }
508
154
    if (spaces > 0 && tabs > 0) {
509
30
        msg << spaces << (spaces == 1 ? " space" : " spaces") << " and " << tabs
510
30
            << (tabs == 1 ? " tab" : " tabs");
511
124
    } else if (spaces > 0) {
512
55
        msg << spaces << (spaces == 1 ? " space" : " spaces");
513
69
    } else if (tabs > 0) {
514
69
        msg << tabs << (tabs == 1 ? " tab" : " tabs");
515
69
    } else {
516
0
        msg << "no indentation";
517
0
    }
518
154
}
519
520
Tokens jsonnet_lex(const std::string &filename, const char *input)
521
25.6k
{
522
25.6k
    unsigned long line_number = 1;
523
25.6k
    const char *line_start = input;
524
525
25.6k
    Tokens r;
526
527
25.6k
    const char *c = input;
528
529
25.6k
    Fodder fodder;
530
25.6k
    bool fresh_line = true;  // Are we tokenizing from the beginning of a new line?
531
532
183M
    while (*c != '\0') {
533
        // Used to ensure we have actually advanced the pointer by the end of the iteration.
534
183M
        const char *original_c = c;
535
536
183M
        Token::Kind kind;
537
183M
        std::string data;
538
183M
        std::string string_block_indent;
539
183M
        std::string string_block_term_indent;
540
541
183M
        unsigned new_lines, indent;
542
183M
        lex_ws(c, new_lines, indent, line_start, line_number);
543
544
        // If it's the end of the file, discard final whitespace.
545
183M
        if (*c == '\0')
546
13.1k
            break;
547
548
183M
        if (new_lines > 0) {
549
            // Otherwise store whitespace in fodder.
550
19.3M
            unsigned blanks = new_lines - 1;
551
19.3M
            fodder.emplace_back(FodderElement::LINE_END, blanks, indent, EMPTY);
552
19.3M
            fresh_line = true;
553
19.3M
        }
554
555
183M
        Location begin(line_number, c - line_start + 1);
556
557
183M
        switch (*c) {
558
            // The following operators should never be combined with subsequent symbols.
559
773k
            case '{':
560
773k
                kind = Token::BRACE_L;
561
773k
                c++;
562
773k
                break;
563
564
758k
            case '}':
565
758k
                kind = Token::BRACE_R;
566
758k
                c++;
567
758k
                break;
568
569
4.15M
            case '[':
570
4.15M
                kind = Token::BRACKET_L;
571
4.15M
                c++;
572
4.15M
                break;
573
574
4.13M
            case ']':
575
4.13M
                kind = Token::BRACKET_R;
576
4.13M
                c++;
577
4.13M
                break;
578
579
16.0M
            case ',':
580
16.0M
                kind = Token::COMMA;
581
16.0M
                c++;
582
16.0M
                break;
583
584
9.25M
            case '.':
585
9.25M
                kind = Token::DOT;
586
9.25M
                c++;
587
9.25M
                break;
588
589
13.9M
            case '(':
590
13.9M
                kind = Token::PAREN_L;
591
13.9M
                c++;
592
13.9M
                break;
593
594
13.9M
            case ')':
595
13.9M
                kind = Token::PAREN_R;
596
13.9M
                c++;
597
13.9M
                break;
598
599
3.64M
            case ';':
600
3.64M
                kind = Token::SEMICOLON;
601
3.64M
                c++;
602
3.64M
                break;
603
604
            // Numeric literals.
605
1.98M
            case '0':
606
4.31M
            case '1':
607
4.79M
            case '2':
608
5.21M
            case '3':
609
5.55M
            case '4':
610
5.62M
            case '5':
611
6.17M
            case '6':
612
6.33M
            case '7':
613
7.36M
            case '8':
614
8.19M
            case '9':
615
8.19M
                kind = Token::NUMBER;
616
8.19M
                data = lex_number(c, filename, begin);
617
8.19M
                break;
618
619
            // UString literals.
620
283k
            case '"': {
621
283k
                c++;
622
48.9M
                for (;; ++c) {
623
48.9M
                    if (*c == '\0') {
624
42
                        throw StaticError(filename, begin, "unterminated string");
625
42
                    }
626
48.9M
                    if (*c == '"') {
627
283k
                        break;
628
283k
                    }
629
48.6M
                    if (*c == '\\' && *(c + 1) != '\0') {
630
126k
                        data += *c;
631
126k
                        ++c;
632
126k
                    }
633
48.6M
                    if (*c == '\n') {
634
                        // Maintain line/column counters.
635
2.51M
                        line_number++;
636
2.51M
                        line_start = c + 1;
637
2.51M
                    }
638
48.6M
                    data += *c;
639
48.6M
                }
640
283k
                c++;  // Advance beyond the ".
641
283k
                kind = Token::STRING_DOUBLE;
642
283k
            } break;
643
644
            // UString literals.
645
6.05M
            case '\'': {
646
6.05M
                c++;
647
85.3M
                for (;; ++c) {
648
85.3M
                    if (*c == '\0') {
649
45
                        throw StaticError(filename, begin, "unterminated string");
650
45
                    }
651
85.3M
                    if (*c == '\'') {
652
6.05M
                        break;
653
6.05M
                    }
654
79.2M
                    if (*c == '\\' && *(c + 1) != '\0') {
655
716k
                        data += *c;
656
716k
                        ++c;
657
716k
                    }
658
79.2M
                    if (*c == '\n') {
659
                        // Maintain line/column counters.
660
2.21M
                        line_number++;
661
2.21M
                        line_start = c + 1;
662
2.21M
                    }
663
79.2M
                    data += *c;
664
79.2M
                }
665
6.05M
                c++;  // Advance beyond the '.
666
6.05M
                kind = Token::STRING_SINGLE;
667
6.05M
            } break;
668
669
            // Verbatim string literals.
670
            // ' and " quoting is interpreted here, unlike non-verbatim strings
671
            // where it is done later by jsonnet_string_unescape.  This is OK
672
            // in this case because no information is lost by resoving the
673
            // repeated quote into a single quote, so we can go back to the
674
            // original form in the formatter.
675
6.17k
            case '@': {
676
6.17k
                c++;
677
6.17k
                if (*c != '"' && *c != '\'') {
678
24
                    std::stringstream ss;
679
24
                    ss << "couldn't lex verbatim string, junk after '@': " << *c;
680
24
                    throw StaticError(filename, begin, ss.str());
681
24
                }
682
6.15k
                const char quot = *c;
683
6.15k
                c++;  // Advance beyond the opening quote.
684
55.0k
                for (;; ++c) {
685
55.0k
                    if (*c == '\0') {
686
48
                        throw StaticError(filename, begin, "unterminated verbatim string");
687
48
                    }
688
55.0k
                    if (*c == quot) {
689
9.39k
                        if (*(c + 1) == quot) {
690
3.29k
                            c++;
691
6.10k
                        } else {
692
6.10k
                            break;
693
6.10k
                        }
694
9.39k
                    }
695
48.9k
                    data += *c;
696
48.9k
                }
697
6.10k
                c++;  // Advance beyond the closing quote.
698
6.10k
                if (quot == '"') {
699
2.97k
                    kind = Token::VERBATIM_STRING_DOUBLE;
700
3.13k
                } else {
701
3.13k
                    kind = Token::VERBATIM_STRING_SINGLE;
702
3.13k
                }
703
6.10k
            } break;
704
705
            // Keywords
706
101M
            default:
707
101M
                if (is_identifier_first(*c)) {
708
77.1M
                    std::string id;
709
423M
                    for (; is_identifier(*c); ++c)
710
345M
                        id += *c;
711
77.1M
                    kind = lex_get_keyword_kind(id);
712
77.1M
                    data = id;
713
714
77.1M
                } else if (is_symbol(*c) || *c == '#') {
715
                    // Single line C++ and Python style comments.
716
24.7M
                    if (*c == '#' || (*c == '/' && *(c + 1) == '/')) {
717
4.74M
                        std::vector<std::string> comment(1);
718
4.74M
                        unsigned blanks;
719
4.74M
                        unsigned indent;
720
4.74M
                        lex_until_newline(c, comment[0], blanks, indent, line_start, line_number);
721
4.74M
                        auto kind = fresh_line ? FodderElement::PARAGRAPH : FodderElement::LINE_END;
722
4.74M
                        fodder.emplace_back(kind, blanks, indent, comment);
723
4.74M
                        fresh_line = true;
724
4.74M
                        continue;  // We've not got a token, just fodder, so keep scanning.
725
4.74M
                    }
726
727
                    // Multi-line C style comment.
728
20.0M
                    if (*c == '/' && *(c + 1) == '*') {
729
267k
                        unsigned margin = c - line_start;
730
731
267k
                        const char *initial_c = c;
732
267k
                        c += 2;  // Avoid matching /*/: skip the /* before starting the search for
733
                                 // */.
734
735
77.7M
                        while (!(*c == '*' && *(c + 1) == '/')) {
736
77.5M
                            if (*c == '\0') {
737
127
                                auto msg = "multi-line comment has no terminating */.";
738
127
                                throw StaticError(filename, begin, msg);
739
127
                            }
740
77.5M
                            if (*c == '\n') {
741
                                // Just keep track of the line / column counters.
742
10.1M
                                line_number++;
743
10.1M
                                line_start = c + 1;
744
10.1M
                            }
745
77.5M
                            ++c;
746
77.5M
                        }
747
267k
                        c += 2;  // Move the pointer to the char after the closing '/'.
748
749
267k
                        std::string comment(initial_c,
750
267k
                                            c - initial_c);  // Includes the "/*" and "*/".
751
752
                        // Lex whitespace after comment
753
267k
                        unsigned new_lines_after, indent_after;
754
267k
                        lex_ws(c, new_lines_after, indent_after, line_start, line_number);
755
267k
                        std::vector<std::string> lines;
756
267k
                        if (comment.find('\n') >= comment.length()) {
757
                            // Comment looks like /* foo */
758
112k
                            lines.push_back(comment);
759
112k
                            fodder.emplace_back(FodderElement::INTERSTITIAL, 0, 0, lines);
760
112k
                            if (new_lines_after > 0) {
761
109k
                                fodder.emplace_back(FodderElement::LINE_END,
762
109k
                                                    new_lines_after - 1,
763
109k
                                                    indent_after,
764
109k
                                                    EMPTY);
765
109k
                                fresh_line = true;
766
109k
                            }
767
154k
                        } else {
768
154k
                            lines = line_split(comment, margin);
769
154k
                            assert(lines[0][0] == '/');
770
                            // Little hack to support PARAGRAPHs with * down the LHS:
771
                            // Add a space to lines that start with a '*'
772
154k
                            bool all_star = true;
773
10.2M
                            for (auto &l : lines) {
774
10.2M
                                if (l[0] != '*')
775
10.1M
                                    all_star = false;
776
10.2M
                            }
777
154k
                            if (all_star) {
778
0
                                for (auto &l : lines) {
779
0
                                    if (l[0] == '*')
780
0
                                        l = " " + l;
781
0
                                }
782
0
                            }
783
154k
                            if (new_lines_after == 0) {
784
                                // Ensure a line end after the paragraph.
785
11.3k
                                new_lines_after = 1;
786
11.3k
                                indent_after = 0;
787
11.3k
                            }
788
154k
                            fodder_push_back(fodder,
789
154k
                                             FodderElement(FodderElement::PARAGRAPH,
790
154k
                                                           new_lines_after - 1,
791
154k
                                                           indent_after,
792
154k
                                                           lines));
793
154k
                            fresh_line = true;
794
154k
                        }
795
267k
                        continue;  // We've not got a token, just fodder, so keep scanning.
796
267k
                    }
797
798
                    // Text block
799
19.7M
                    if (*c == '|' && *(c + 1) == '|' && *(c + 2) == '|') {
800
10.7k
                        c += 3;  // Skip the "|||".
801
802
10.7k
                        bool chomp_trailing_nl = false;
803
10.7k
                        if (*c == '-') {
804
540
                            chomp_trailing_nl = true;
805
540
                            c++;
806
540
                        }
807
808
13.7k
                        while (is_horz_ws(*c)) ++c;  // Chomp whitespace at end of line.
809
10.7k
                        if (*c != '\n') {
810
79
                            auto msg = "text block syntax requires new line after |||.";
811
79
                            throw StaticError(filename, begin, msg);
812
79
                        }
813
10.6k
                        std::stringstream block;
814
10.6k
                        c++;  // Skip the "\n"
815
10.6k
                        line_number++;
816
                        // Skip any blank lines at the beginning of the block.
817
12.8k
                        while (*c == '\n') {
818
2.20k
                            line_number++;
819
2.20k
                            ++c;
820
2.20k
                            block << '\n';
821
2.20k
                        }
822
10.6k
                        line_start = c;
823
10.6k
                        const char *first_line = c;
824
10.6k
                        int ws_chars = whitespace_check(first_line, c);
825
10.6k
                        string_block_indent = std::string(first_line, ws_chars);
826
10.6k
                        if (ws_chars == 0) {
827
42
                            auto msg = "text block's first line must start with whitespace.";
828
42
                            throw StaticError(filename, begin, msg);
829
42
                        }
830
13.4k
                        while (true) {
831
13.4k
                            assert(ws_chars > 0);
832
                            // Read up to the \n
833
160k
                            for (c = &c[ws_chars]; *c != '\n'; ++c) {
834
147k
                                if (*c == '\0')
835
109
                                    throw StaticError(filename, begin, "unexpected EOF");
836
147k
                                block << *c;
837
147k
                            }
838
                            // Add the \n
839
13.3k
                            block << '\n';
840
13.3k
                            ++c;
841
13.3k
                            line_number++;
842
13.3k
                            line_start = c;
843
                            // Skip any blank lines
844
15.5k
                            while (*c == '\n') {
845
2.16k
                                line_number++;
846
2.16k
                                ++c;
847
2.16k
                                block << '\n';
848
2.16k
                            }
849
                            // Examine next line
850
13.3k
                            ws_chars = whitespace_check(first_line, c);
851
13.3k
                            if (ws_chars == 0) {
852
                                // End of text block (or indentation error).
853
                                // Count actual whitespace on this line.
854
10.4k
                                int actual_ws = 0;
855
75.1k
                                while (c[actual_ws] == ' ' ||
856
64.6k
                                       c[actual_ws] == '\t') {
857
64.6k
                                    actual_ws++;
858
64.6k
                                }
859
860
                                // Check if this is the terminator |||
861
10.4k
                                bool is_terminator = (
862
10.4k
                                    c[actual_ws] == '|' &&
863
10.3k
                                    c[actual_ws + 1] == '|' &&
864
10.3k
                                    c[actual_ws + 2] == '|');
865
866
10.4k
                                if (!is_terminator) {
867
                                    // Not a terminator - check if it's an
868
                                    // indentation issue.
869
158
                                    if (actual_ws > 0) {
870
                                        // Has whitespace but doesn't match expected
871
                                        // indentation.
872
77
                                        std::stringstream msg;
873
77
                                        msg << "text block indentation mismatch: "
874
77
                                                "expected at least ";
875
77
                                        describe_whitespace(msg, string_block_indent);
876
77
                                        msg << ", found ";
877
77
                                        describe_whitespace(msg, std::string(c, actual_ws));
878
77
                                        throw StaticError(filename, begin, msg.str());
879
81
                                    } else {
880
                                        // No whitespace and no ||| - missing
881
                                        // terminator.
882
81
                                        auto msg =
883
81
                                            "text block not terminated with |||";
884
81
                                        throw StaticError(filename, begin, msg);
885
81
                                    }
886
158
                                }
887
888
                                // Valid termination - skip over any whitespace.
889
62.3k
                                while (*c == ' ' || *c == '\t') {
890
52.0k
                                    string_block_term_indent += *c;
891
52.0k
                                    ++c;
892
52.0k
                                }
893
                                // Skip the |||
894
10.3k
                                c += 3;  // Leave after the last |
895
10.3k
                                data = block.str();
896
10.3k
                                kind = Token::STRING_BLOCK;
897
10.3k
                                if (chomp_trailing_nl) {
898
518
                                    assert(data.back() == '\n');
899
518
                                    data.pop_back();
900
518
                                }
901
10.3k
                                break;  // Out of the while loop.
902
10.3k
                            }
903
13.3k
                        }
904
905
10.3k
                        break;  // Out of the switch.
906
10.5k
                    }
907
908
19.7M
                    const char *operator_begin = c;
909
66.8M
                    for (; is_symbol(*c); ++c) {
910
                        // Not allowed // in operators
911
47.0M
                        if (*c == '/' && *(c + 1) == '/')
912
628
                            break;
913
                        // Not allowed /* in operators
914
47.0M
                        if (*c == '/' && *(c + 1) == '*')
915
664
                            break;
916
                        // Not allowed ||| in operators
917
47.0M
                        if (*c == '|' && *(c + 1) == '|' && *(c + 2) == '|')
918
3.03k
                            break;
919
47.0M
                    }
920
                    // Not allowed to end with a + - ~ ! unless a single char.
921
                    // So, wind it back if we need to (but not too far).
922
24.0M
                    while (c > operator_begin + 1 && !allowed_at_end_of_operator(*(c - 1))) {
923
4.31M
                        c--;
924
4.31M
                    }
925
19.7M
                    data += std::string(operator_begin, c);
926
19.7M
                    if (data == "$") {
927
55.2k
                        kind = Token::DOLLAR;
928
55.2k
                        data = "";
929
19.6M
                    } else {
930
19.6M
                        kind = Token::OPERATOR;
931
19.6M
                    }
932
19.7M
                } else {
933
162
                    std::stringstream ss;
934
162
                    ss << "Could not lex the character ";
935
162
                    auto uc = (unsigned char)(*c);
936
162
                    if (*c < 32)
937
147
                        ss << "code " << unsigned(uc);
938
15
                    else
939
15
                        ss << "'" << *c << "'";
940
162
                    throw StaticError(filename, begin, ss.str());
941
162
                }
942
183M
        }
943
944
        // Ensure that a bug in the above code does not cause an infinite memory consuming loop due
945
        // to pushing empty tokens.
946
178M
        if (c == original_c) {
947
0
            throw StaticError(filename, begin, "internal lexing error:  pointer did not advance");
948
0
        }
949
950
178M
        Location end(line_number, (c + 1) - line_start);
951
178M
        r.emplace_back(kind,
952
178M
                       fodder,
953
178M
                       data,
954
178M
                       string_block_indent,
955
178M
                       string_block_term_indent,
956
178M
                       LocationRange(filename, begin, end));
957
178M
        fodder.clear();
958
178M
        fresh_line = false;
959
178M
    }
960
961
24.7k
    Location begin(line_number, c - line_start + 1);
962
24.7k
    Location end(line_number, (c + 1) - line_start + 1);
963
24.7k
    r.emplace_back(Token::END_OF_FILE, fodder, "", "", "", LocationRange(filename, begin, end));
964
24.7k
    return r;
965
25.6k
}
966
967
std::string jsonnet_unlex(const Tokens &tokens)
968
0
{
969
0
    std::stringstream ss;
970
0
    for (const auto &t : tokens) {
971
0
        for (const auto &f : t.fodder) {
972
0
            switch (f.kind) {
973
0
                case FodderElement::LINE_END: {
974
0
                    if (f.comment.size() > 0) {
975
0
                        ss << "LineEnd(" << f.blanks << ", " << f.indent << ", " << f.comment[0]
976
0
                           << ")\n";
977
0
                    } else {
978
0
                        ss << "LineEnd(" << f.blanks << ", " << f.indent << ")\n";
979
0
                    }
980
0
                } break;
981
982
0
                case FodderElement::INTERSTITIAL: {
983
0
                    ss << "Interstitial(" << f.comment[0] << ")\n";
984
0
                } break;
985
986
0
                case FodderElement::PARAGRAPH: {
987
0
                    ss << "Paragraph(\n";
988
0
                    for (const auto &line : f.comment) {
989
0
                        ss << "    " << line << '\n';
990
0
                    }
991
0
                    ss << ")" << f.blanks << "\n";
992
0
                } break;
993
0
            }
994
0
        }
995
0
        if (t.kind == Token::END_OF_FILE) {
996
0
            ss << "EOF\n";
997
0
            break;
998
0
        }
999
0
        if (t.kind == Token::STRING_DOUBLE) {
1000
0
            ss << "\"" << t.data << "\"\n";
1001
0
        } else if (t.kind == Token::STRING_SINGLE) {
1002
0
            ss << "'" << t.data << "'\n";
1003
0
        } else if (t.kind == Token::STRING_BLOCK) {
1004
0
            ss << "|||\n";
1005
0
            ss << t.stringBlockIndent;
1006
0
            for (const char *cp = t.data.c_str(); *cp != '\0'; ++cp) {
1007
0
                ss << *cp;
1008
0
                if (*cp == '\n' && *(cp + 1) != '\n' && *(cp + 1) != '\0') {
1009
0
                    ss << t.stringBlockIndent;
1010
0
                }
1011
0
            }
1012
0
            ss << t.stringBlockTermIndent << "|||\n";
1013
0
        } else {
1014
0
            ss << t.data << "\n";
1015
0
        }
1016
0
    }
1017
0
    return ss.str();
1018
0
}
1019
1020
}  // namespace jsonnet::internal