/src/build-dir/_deps/simdjson-src/src/fallback.cpp
Line | Count | Source |
1 | | #ifndef SIMDJSON_SRC_FALLBACK_CPP |
2 | | #define SIMDJSON_SRC_FALLBACK_CPP |
3 | | |
4 | | #ifndef SIMDJSON_CONDITIONAL_INCLUDE |
5 | | #include <base.h> |
6 | | #endif // SIMDJSON_CONDITIONAL_INCLUDE |
7 | | |
8 | | #include <simdjson/fallback.h> |
9 | | #include <simdjson/fallback/implementation.h> |
10 | | |
11 | | #include <simdjson/fallback/begin.h> |
12 | | #include <generic/stage1/find_next_document_index.h> |
13 | | #include <generic/stage2/stringparsing.h> |
14 | | #include <generic/stage2/logger.h> |
15 | | #include <generic/stage2/json_iterator.h> |
16 | | #include <generic/stage2/tape_writer.h> |
17 | | #include <generic/stage2/tape_builder.h> |
18 | | |
19 | | // |
20 | | // Stage 1 |
21 | | // |
22 | | |
23 | | namespace simdjson { |
24 | | namespace fallback { |
25 | | |
26 | | simdjson_warn_unused error_code implementation::create_dom_parser_implementation( |
27 | | size_t capacity, |
28 | | size_t max_depth, |
29 | | std::unique_ptr<internal::dom_parser_implementation>& dst |
30 | 0 | ) const noexcept { |
31 | 0 | dst.reset( new (std::nothrow) SIMDJSON_IMPLEMENTATION::dom_parser_implementation() ); |
32 | 0 | if (!dst) { return MEMALLOC; } |
33 | 0 | if (auto err = dst->set_capacity(capacity)) |
34 | 0 | return err; |
35 | 0 | if (auto err = dst->set_max_depth(max_depth)) |
36 | 0 | return err; |
37 | 0 | return SUCCESS; |
38 | 0 | } |
39 | | |
40 | | namespace { |
41 | | namespace stage1 { |
42 | | |
43 | | class structural_scanner { |
44 | | public: |
45 | | |
46 | | simdjson_inline structural_scanner(dom_parser_implementation &_parser, stage1_mode _partial) |
47 | 0 | : buf{_parser.buf}, |
48 | 0 | next_structural_index{_parser.structural_indexes.get()}, |
49 | 0 | parser{_parser}, |
50 | 0 | len{static_cast<uint32_t>(_parser.len)}, |
51 | 0 | partial{_partial} { |
52 | 0 | } |
53 | | |
54 | 0 | simdjson_inline void add_structural() { |
55 | 0 | *next_structural_index = idx; |
56 | 0 | next_structural_index++; |
57 | 0 | } |
58 | | |
59 | 0 | simdjson_inline bool is_continuation(uint8_t c) { |
60 | 0 | return (c & 0xc0) == 0x80; |
61 | 0 | } |
62 | | |
63 | 0 | simdjson_inline void validate_utf8_character() { |
64 | | // Continuation |
65 | 0 | if (simdjson_unlikely((buf[idx] & 0x40) == 0)) { |
66 | | // extra continuation |
67 | 0 | error = UTF8_ERROR; |
68 | 0 | idx++; |
69 | 0 | return; |
70 | 0 | } |
71 | | |
72 | | // 2-byte |
73 | 0 | if ((buf[idx] & 0x20) == 0) { |
74 | | // missing continuation |
75 | 0 | if (simdjson_unlikely(idx+1 > len || !is_continuation(buf[idx+1]))) { |
76 | 0 | if (idx+1 > len && is_streaming(partial)) { idx = len; return; } |
77 | 0 | error = UTF8_ERROR; |
78 | 0 | idx++; |
79 | 0 | return; |
80 | 0 | } |
81 | | // overlong: 1100000_ 10______ |
82 | 0 | if (buf[idx] <= 0xc1) { error = UTF8_ERROR; } |
83 | 0 | idx += 2; |
84 | 0 | return; |
85 | 0 | } |
86 | | |
87 | | // 3-byte |
88 | 0 | if ((buf[idx] & 0x10) == 0) { |
89 | | // missing continuation |
90 | 0 | if (simdjson_unlikely(idx+2 > len || !is_continuation(buf[idx+1]) || !is_continuation(buf[idx+2]))) { |
91 | 0 | if (idx+2 > len && is_streaming(partial)) { idx = len; return; } |
92 | 0 | error = UTF8_ERROR; |
93 | 0 | idx++; |
94 | 0 | return; |
95 | 0 | } |
96 | | // overlong: 11100000 100_____ ________ |
97 | 0 | if (buf[idx] == 0xe0 && buf[idx+1] <= 0x9f) { error = UTF8_ERROR; } |
98 | | // surrogates: U+D800-U+DFFF 11101101 101_____ |
99 | 0 | if (buf[idx] == 0xed && buf[idx+1] >= 0xa0) { error = UTF8_ERROR; } |
100 | 0 | idx += 3; |
101 | 0 | return; |
102 | 0 | } |
103 | | |
104 | | // 4-byte |
105 | | // missing continuation |
106 | 0 | if (simdjson_unlikely(idx+3 > len || !is_continuation(buf[idx+1]) || !is_continuation(buf[idx+2]) || !is_continuation(buf[idx+3]))) { |
107 | 0 | if (idx+2 > len && is_streaming(partial)) { idx = len; return; } |
108 | 0 | error = UTF8_ERROR; |
109 | 0 | idx++; |
110 | 0 | return; |
111 | 0 | } |
112 | | // overlong: 11110000 1000____ ________ ________ |
113 | 0 | if (buf[idx] == 0xf0 && buf[idx+1] <= 0x8f) { error = UTF8_ERROR; } |
114 | | // too large: > U+10FFFF: |
115 | | // 11110100 (1001|101_)____ |
116 | | // 1111(1___|011_|0101) 10______ |
117 | | // also includes 5, 6, 7 and 8 byte characters: |
118 | | // 11111___ |
119 | 0 | if (buf[idx] == 0xf4 && buf[idx+1] >= 0x90) { error = UTF8_ERROR; } |
120 | 0 | if (buf[idx] >= 0xf5) { error = UTF8_ERROR; } |
121 | 0 | idx += 4; |
122 | 0 | } |
123 | | |
124 | | static const uint8_t CHAR_TYPE_SPACE = 1 << 0; |
125 | | static const uint8_t CHAR_TYPE_OPERATOR = 1 << 1; |
126 | | static const uint8_t CHAR_TYPE_ESC_ASCII = 1 << 2; |
127 | | static const uint8_t CHAR_TYPE_NON_ASCII = 1 << 3; |
128 | | |
129 | | const uint8_t char_table[256] = { |
130 | | 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, |
131 | | 0x04, 0x05, 0x05, 0x04, 0x04, 0x05, 0x04, 0x04, |
132 | | 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, |
133 | | 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, |
134 | | 0x01, 0x00, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, |
135 | | 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, |
136 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
137 | | 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, |
138 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
139 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
140 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
141 | | 0x00, 0x00, 0x00, 0x02, 0x04, 0x02, 0x00, 0x00, |
142 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
143 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
144 | | 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, |
145 | | 0x00, 0x00, 0x00, 0x02, 0x00, 0x02, 0x00, 0x00, |
146 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
147 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
148 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
149 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
150 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
151 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
152 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
153 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
154 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
155 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
156 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
157 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
158 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
159 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
160 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, |
161 | | 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08 |
162 | | }; |
163 | | |
164 | 0 | simdjson_inline bool char_is_type(uint8_t c, uint8_t type) { |
165 | 0 | return (char_table[c] & type); |
166 | 0 | } |
167 | | |
168 | 0 | simdjson_inline bool char_is_space(uint8_t c) { |
169 | 0 | return char_is_type(c, CHAR_TYPE_SPACE); |
170 | 0 | } |
171 | | |
172 | 0 | simdjson_inline bool char_is_operator(uint8_t c) { |
173 | 0 | return char_is_type(c, CHAR_TYPE_OPERATOR); |
174 | 0 | } |
175 | | |
176 | 0 | simdjson_inline bool char_is_space_or_operator(uint8_t c) { |
177 | 0 | return char_is_type(c, CHAR_TYPE_SPACE | CHAR_TYPE_OPERATOR); |
178 | 0 | } |
179 | | |
180 | 0 | simdjson_inline bool char_is_ascii_stop(uint8_t c) { |
181 | 0 | return char_is_type(c, CHAR_TYPE_ESC_ASCII | CHAR_TYPE_NON_ASCII); |
182 | 0 | } |
183 | | |
184 | | // Returns true if the string is unclosed. |
185 | 0 | simdjson_inline bool validate_string() { |
186 | 0 | idx++; // skip first quote |
187 | 0 | while (idx < len) { |
188 | 0 | do { |
189 | 0 | if (char_is_ascii_stop(buf[idx])) { break; } |
190 | 0 | idx++; |
191 | 0 | } while (idx < len); |
192 | 0 | if (idx >= len) { return true; } |
193 | 0 | if (buf[idx] == '"') { |
194 | 0 | return false; |
195 | 0 | } |
196 | 0 | if (buf[idx] == '\\') { |
197 | 0 | idx += 2; |
198 | 0 | } else if (simdjson_unlikely(buf[idx] & 0x80)) { |
199 | 0 | validate_utf8_character(); |
200 | 0 | } else { |
201 | 0 | if (buf[idx] < 0x20) { error = UNESCAPED_CHARS; } |
202 | 0 | idx++; |
203 | 0 | } |
204 | 0 | } |
205 | 0 | if (idx >= len) { return true; } |
206 | 0 | return false; |
207 | 0 | } |
208 | | |
209 | | // |
210 | | // Parse the entire input in STEP_SIZE-byte chunks. |
211 | | // |
212 | 0 | simdjson_warn_unused simdjson_inline error_code scan() { |
213 | 0 | bool unclosed_string = false; |
214 | 0 | for (;idx<len;idx++) { |
215 | 0 | do { |
216 | 0 | if (!char_is_space(buf[idx])) { break; } |
217 | 0 | idx++; |
218 | 0 | } while (idx < len); |
219 | 0 | if (idx >= len) { break; } |
220 | | // String |
221 | 0 | if (buf[idx] == '"') { |
222 | 0 | add_structural(); |
223 | 0 | unclosed_string |= validate_string(); |
224 | | // Operator |
225 | 0 | } else if (char_is_operator(buf[idx])) { |
226 | 0 | add_structural(); |
227 | | // Primitive or invalid character (invalid characters will be checked in stage 2) |
228 | 0 | } else { |
229 | | // Anything else, add the structural and go until we find the next one |
230 | 0 | add_structural(); |
231 | 0 | while (idx+1<len && !char_is_space_or_operator(buf[idx+1])) { |
232 | 0 | idx++; |
233 | 0 | }; |
234 | 0 | } |
235 | 0 | } |
236 | | // We pad beyond. |
237 | | // https://github.com/simdjson/simdjson/issues/906 |
238 | | // See json_structural_indexer.h for an explanation. |
239 | 0 | *next_structural_index = len; // assumed later in partial == stage1_mode::streaming_final |
240 | 0 | next_structural_index[1] = len; |
241 | 0 | next_structural_index[2] = 0; |
242 | 0 | parser.n_structural_indexes = uint32_t(next_structural_index - parser.structural_indexes.get()); |
243 | 0 | if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return EMPTY; } |
244 | 0 | parser.next_structural_index = 0; |
245 | 0 | if (partial == stage1_mode::streaming_partial) { |
246 | 0 | if(unclosed_string) { |
247 | 0 | parser.n_structural_indexes--; |
248 | 0 | if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return CAPACITY; } |
249 | 0 | } |
250 | | // We truncate the input to the end of the last complete document (or zero). |
251 | 0 | auto new_structural_indexes = find_next_document_index(parser); |
252 | 0 | if (new_structural_indexes == 0 && parser.n_structural_indexes > 0) { |
253 | 0 | if(parser.structural_indexes[0] == 0) { |
254 | | // If the buffer is partial and we started at index 0 but the document is |
255 | | // incomplete, it's too big to parse. |
256 | 0 | return CAPACITY; |
257 | 0 | } else { |
258 | | // It is possible that the document could be parsed, we just had a lot |
259 | | // of white space. |
260 | 0 | parser.n_structural_indexes = 0; |
261 | 0 | return EMPTY; |
262 | 0 | } |
263 | 0 | } |
264 | 0 | parser.n_structural_indexes = new_structural_indexes; |
265 | 0 | } else if(partial == stage1_mode::streaming_final) { |
266 | 0 | if(unclosed_string) { parser.n_structural_indexes--; } |
267 | | // We truncate the input to the end of the last complete document (or zero). |
268 | | // Because partial == stage1_mode::streaming_final, it means that we may |
269 | | // silently ignore trailing garbage. Though it sounds bad, we do it |
270 | | // deliberately because many people who have streams of JSON documents |
271 | | // will truncate them for processing. E.g., imagine that you are uncompressing |
272 | | // the data from a size file or receiving it in chunks from the network. You |
273 | | // may not know where exactly the last document will be. Meanwhile the |
274 | | // document_stream instances allow people to know the JSON documents they are |
275 | | // parsing (see the iterator.source() method). |
276 | 0 | parser.n_structural_indexes = find_next_document_index(parser); |
277 | | // We store the initial n_structural_indexes so that the client can see |
278 | | // whether we used truncation. If initial_n_structural_indexes == parser.n_structural_indexes, |
279 | | // then this will query parser.structural_indexes[parser.n_structural_indexes] which is len, |
280 | | // otherwise, it will copy some prior index. |
281 | 0 | parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes]; |
282 | | // This next line is critical, do not change it unless you understand what you are |
283 | | // doing. |
284 | 0 | parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len); |
285 | 0 | if (parser.n_structural_indexes == 0) { return EMPTY; } |
286 | 0 | } else if(unclosed_string) { error = UNCLOSED_STRING; } |
287 | 0 | return error; |
288 | 0 | } |
289 | | |
290 | | private: |
291 | | const uint8_t *buf; |
292 | | uint32_t *next_structural_index; |
293 | | dom_parser_implementation &parser; |
294 | | uint32_t len; |
295 | | uint32_t idx{0}; |
296 | | error_code error{SUCCESS}; |
297 | | stage1_mode partial; |
298 | | }; // structural_scanner |
299 | | |
300 | | } // namespace stage1 |
301 | | } // unnamed namespace |
302 | | |
303 | 0 | simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode partial) noexcept { |
304 | 0 | this->buf = _buf; |
305 | 0 | this->len = _len; |
306 | 0 | stage1::structural_scanner scanner(*this, partial); |
307 | 0 | return scanner.scan(); |
308 | 0 | } |
309 | | |
310 | | // big table for the minifier |
311 | | static uint8_t jump_table[256 * 3] = { |
312 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
313 | | 1, 1, 0, 1, 0, 0, 1, 0, 0, 1, 1, 0, 1, 1, 0, 1, 0, 0, 1, 1, 0, 1, 1, 0, 1, |
314 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
315 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 0, 0, |
316 | | 1, 1, 1, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
317 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
318 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
319 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
320 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
321 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
322 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
323 | | 1, 0, 0, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
324 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
325 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
326 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
327 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
328 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
329 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
330 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
331 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
332 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
333 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
334 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
335 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
336 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
337 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
338 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
339 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, |
340 | | 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, |
341 | | 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
342 | | 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1, 1, |
343 | | }; |
344 | | |
345 | 0 | simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept { |
346 | 0 | size_t i = 0, pos = 0; |
347 | 0 | uint8_t quote = 0; |
348 | 0 | uint8_t nonescape = 1; |
349 | |
|
350 | 0 | while (i < len) { |
351 | 0 | unsigned char c = buf[i]; |
352 | 0 | uint8_t *meta = jump_table + 3 * c; |
353 | |
|
354 | 0 | quote = quote ^ (meta[0] & nonescape); |
355 | 0 | dst[pos] = c; |
356 | 0 | pos += meta[2] | quote; |
357 | |
|
358 | 0 | i += 1; |
359 | 0 | nonescape = uint8_t(~nonescape) | (meta[1]); |
360 | 0 | } |
361 | 0 | dst_len = pos; // we intentionally do not work with a reference |
362 | | // for fear of aliasing |
363 | 0 | return quote ? UNCLOSED_STRING : SUCCESS; |
364 | 0 | } |
365 | | |
366 | | // credit: based on code from Google Fuchsia (Apache Licensed) |
367 | 0 | simdjson_warn_unused bool implementation::validate_utf8(const char *buf, size_t len) const noexcept { |
368 | 0 | const uint8_t *data = reinterpret_cast<const uint8_t *>(buf); |
369 | 0 | uint64_t pos = 0; |
370 | 0 | uint32_t code_point = 0; |
371 | 0 | while (pos < len) { |
372 | | // check of the next 8 bytes are ascii. |
373 | 0 | uint64_t next_pos = pos + 16; |
374 | 0 | if (next_pos <= len) { // if it is safe to read 8 more bytes, check that they are ascii |
375 | 0 | uint64_t v1; |
376 | 0 | memcpy(&v1, data + pos, sizeof(uint64_t)); |
377 | 0 | uint64_t v2; |
378 | 0 | memcpy(&v2, data + pos + sizeof(uint64_t), sizeof(uint64_t)); |
379 | 0 | uint64_t v{v1 | v2}; |
380 | 0 | if ((v & 0x8080808080808080) == 0) { |
381 | 0 | pos = next_pos; |
382 | 0 | continue; |
383 | 0 | } |
384 | 0 | } |
385 | 0 | unsigned char byte = data[pos]; |
386 | 0 | if (byte < 0x80) { |
387 | 0 | pos++; |
388 | 0 | continue; |
389 | 0 | } else if ((byte & 0xe0) == 0xc0) { |
390 | 0 | next_pos = pos + 2; |
391 | 0 | if (next_pos > len) { return false; } |
392 | 0 | if ((data[pos + 1] & 0xc0) != 0x80) { return false; } |
393 | | // range check |
394 | 0 | code_point = (byte & 0x1f) << 6 | (data[pos + 1] & 0x3f); |
395 | 0 | if (code_point < 0x80 || 0x7ff < code_point) { return false; } |
396 | 0 | } else if ((byte & 0xf0) == 0xe0) { |
397 | 0 | next_pos = pos + 3; |
398 | 0 | if (next_pos > len) { return false; } |
399 | 0 | if ((data[pos + 1] & 0xc0) != 0x80) { return false; } |
400 | 0 | if ((data[pos + 2] & 0xc0) != 0x80) { return false; } |
401 | | // range check |
402 | 0 | code_point = (byte & 0x0f) << 12 | |
403 | 0 | (data[pos + 1] & 0x3f) << 6 | |
404 | 0 | (data[pos + 2] & 0x3f); |
405 | 0 | if (code_point < 0x800 || 0xffff < code_point || |
406 | 0 | (0xd7ff < code_point && code_point < 0xe000)) { |
407 | 0 | return false; |
408 | 0 | } |
409 | 0 | } else if ((byte & 0xf8) == 0xf0) { // 0b11110000 |
410 | 0 | next_pos = pos + 4; |
411 | 0 | if (next_pos > len) { return false; } |
412 | 0 | if ((data[pos + 1] & 0xc0) != 0x80) { return false; } |
413 | 0 | if ((data[pos + 2] & 0xc0) != 0x80) { return false; } |
414 | 0 | if ((data[pos + 3] & 0xc0) != 0x80) { return false; } |
415 | | // range check |
416 | 0 | code_point = |
417 | 0 | (byte & 0x07) << 18 | (data[pos + 1] & 0x3f) << 12 | |
418 | 0 | (data[pos + 2] & 0x3f) << 6 | (data[pos + 3] & 0x3f); |
419 | 0 | if (code_point <= 0xffff || 0x10ffff < code_point) { return false; } |
420 | 0 | } else { |
421 | | // we may have a continuation |
422 | 0 | return false; |
423 | 0 | } |
424 | 0 | pos = next_pos; |
425 | 0 | } |
426 | 0 | return true; |
427 | 0 | } |
428 | | |
429 | | } // namespace fallback |
430 | | } // namespace simdjson |
431 | | |
432 | | // |
433 | | // Stage 2 |
434 | | // |
435 | | |
436 | | namespace simdjson { |
437 | | namespace fallback { |
438 | | |
439 | 0 | simdjson_warn_unused error_code dom_parser_implementation::stage2(dom::document &_doc) noexcept { |
440 | 0 | return stage2::tape_builder::parse_document<false>(*this, _doc); |
441 | 0 | } |
442 | | |
443 | 0 | simdjson_warn_unused error_code dom_parser_implementation::stage2_next(dom::document &_doc) noexcept { |
444 | 0 | return stage2::tape_builder::parse_document<true>(*this, _doc); |
445 | 0 | } |
446 | | |
447 | | SIMDJSON_NO_SANITIZE_MEMORY |
448 | 0 | simdjson_warn_unused uint8_t *dom_parser_implementation::parse_string(const uint8_t *src, uint8_t *dst, bool replacement_char) const noexcept { |
449 | 0 | return fallback::stringparsing::parse_string(src, dst, replacement_char); |
450 | 0 | } |
451 | | |
452 | 0 | simdjson_warn_unused uint8_t *dom_parser_implementation::parse_wobbly_string(const uint8_t *src, uint8_t *dst) const noexcept { |
453 | 0 | return fallback::stringparsing::parse_wobbly_string(src, dst); |
454 | 0 | } |
455 | | |
456 | 0 | simdjson_warn_unused error_code dom_parser_implementation::parse(const uint8_t *_buf, size_t _len, dom::document &_doc) noexcept { |
457 | 0 | auto error = stage1(_buf, _len, stage1_mode::regular); |
458 | 0 | if (error) { return error; } |
459 | 0 | return stage2(_doc); |
460 | 0 | } |
461 | | |
462 | | } // namespace fallback |
463 | | } // namespace simdjson |
464 | | |
465 | | #include <simdjson/fallback/end.h> |
466 | | |
467 | | #endif // SIMDJSON_SRC_FALLBACK_CPP |