/src/wireshark/wsutil/wsjson.c
Line | Count | Source |
1 | | /* wsjson.c |
2 | | * JSON parsing functions. |
3 | | * |
4 | | * Copyright 2016, Dario Lombardo |
5 | | * |
6 | | * Wireshark - Network traffic analyzer |
7 | | * By Gerald Combs <gerald@wireshark.org> |
8 | | * Copyright 1998 Gerald Combs |
9 | | * |
10 | | * SPDX-License-Identifier: GPL-2.0-or-later |
11 | | */ |
12 | | |
13 | | #include "config.h" |
14 | 0 | #define WS_LOG_DOMAIN LOG_DOMAIN_MAIN |
15 | | |
16 | | #include "wsjson.h" |
17 | | |
18 | | #include <string.h> |
19 | | #include <errno.h> |
20 | | #include <wsutil/jsmn.h> |
21 | | #include <wsutil/str_util.h> |
22 | | #include <wsutil/strtoi.h> |
23 | | #include <wsutil/unicode-utils.h> |
24 | | #include <wsutil/wslog.h> |
25 | | |
26 | | bool |
27 | | json_validate(const uint8_t *buf, const size_t len) |
28 | 0 | { |
29 | 0 | bool ret = true; |
30 | | /* We expect no more than 1024 tokens */ |
31 | 0 | unsigned max_tokens = 1024; |
32 | 0 | jsmntok_t* t; |
33 | 0 | jsmn_parser p; |
34 | 0 | int rcode; |
35 | | |
36 | | /* |
37 | | * Make sure the buffer isn't empty and the first octet isn't a NUL; |
38 | | * otherwise, the parser will immediately stop parsing and not validate |
39 | | * anything after that, so it'll just think it was handed an empty string. |
40 | | * |
41 | | * XXX - should we check for NULs anywhere in the buffer? |
42 | | */ |
43 | 0 | if (len == 0) { |
44 | 0 | ws_debug("JSON string is empty"); |
45 | 0 | return false; |
46 | 0 | } |
47 | 0 | if (buf[0] == '\0') { |
48 | 0 | ws_debug("invalid character inside JSON string"); |
49 | 0 | return false; |
50 | 0 | } |
51 | | |
52 | | /* |
53 | | * XXX - We create the token array and have jsmn_parse fill it in, only |
54 | | * to free it. It might make more sense to pass in NULL for tokens, and |
55 | | * for our sanity check just check that len isn't too big. |
56 | | */ |
57 | 0 | t = g_new0(jsmntok_t, max_tokens); |
58 | |
|
59 | 0 | if (!t) |
60 | 0 | return false; |
61 | | |
62 | 0 | jsmn_init(&p); |
63 | 0 | rcode = jsmn_parse(&p, (const char*)buf, len, t, max_tokens); |
64 | 0 | if (rcode < 0) { |
65 | 0 | switch (rcode) { |
66 | 0 | case JSMN_ERROR_NOMEM: |
67 | 0 | ws_debug("not enough tokens were provided"); |
68 | 0 | break; |
69 | 0 | case JSMN_ERROR_INVAL: |
70 | 0 | ws_debug("invalid character inside JSON string"); |
71 | 0 | break; |
72 | 0 | case JSMN_ERROR_PART: |
73 | 0 | ws_debug("the string is not a full JSON packet, " |
74 | 0 | "more bytes expected"); |
75 | 0 | break; |
76 | 0 | default: |
77 | 0 | ws_debug("unexpected error"); |
78 | 0 | break; |
79 | 0 | } |
80 | 0 | ret = false; |
81 | 0 | } |
82 | | |
83 | 0 | g_free(t); |
84 | 0 | return ret; |
85 | 0 | } |
86 | | |
87 | | int |
88 | | json_parse(const char *buf, jsmntok_t *tokens, unsigned int max_tokens) |
89 | 0 | { |
90 | 0 | jsmn_parser p; |
91 | |
|
92 | 0 | jsmn_init(&p); |
93 | 0 | return jsmn_parse(&p, buf, strlen(buf), tokens, max_tokens); |
94 | 0 | } |
95 | | |
96 | | int |
97 | | json_parse_len(const char *buf, size_t len, jsmntok_t *tokens, unsigned int max_tokens) |
98 | 0 | { |
99 | 0 | jsmn_parser p; |
100 | |
|
101 | 0 | jsmn_init(&p); |
102 | 0 | return jsmn_parse(&p, buf, len, tokens, max_tokens); |
103 | 0 | } |
104 | | |
105 | | jsmntok_t *json_get_next_object(jsmntok_t *cur) |
106 | 0 | { |
107 | 0 | for (size_t tokens_remaining = 1; tokens_remaining > 0; tokens_remaining--) { |
108 | 0 | tokens_remaining += cur->size; |
109 | 0 | cur++; |
110 | 0 | } |
111 | 0 | return cur; |
112 | 0 | } |
113 | | |
114 | | jsmntok_t *json_get_object(const char *buf, jsmntok_t *parent, const char *name) |
115 | 0 | { |
116 | 0 | int i; |
117 | 0 | jsmntok_t *cur = parent+1; |
118 | |
|
119 | 0 | for (i = 0; i < parent->size; i++) { |
120 | 0 | if (cur->type == JSMN_STRING && |
121 | 0 | !strncmp(&buf[cur->start], name, cur->end - cur->start) |
122 | 0 | && strlen(name) == (size_t)(cur->end - cur->start) && |
123 | 0 | cur->size == 1 && (cur+1)->type == JSMN_OBJECT) { |
124 | 0 | return cur+1; |
125 | 0 | } |
126 | 0 | cur = json_get_next_object(cur); |
127 | 0 | } |
128 | 0 | return NULL; |
129 | 0 | } |
130 | | |
131 | | jsmntok_t *json_get_array(const char *buf, jsmntok_t *parent, const char *name) |
132 | 0 | { |
133 | 0 | int i; |
134 | 0 | jsmntok_t *cur = parent+1; |
135 | |
|
136 | 0 | for (i = 0; i < parent->size; i++) { |
137 | 0 | if (cur->type == JSMN_STRING && |
138 | 0 | !strncmp(&buf[cur->start], name, cur->end - cur->start) |
139 | 0 | && strlen(name) == (size_t)(cur->end - cur->start) && |
140 | 0 | cur->size == 1 && (cur+1)->type == JSMN_ARRAY) { |
141 | 0 | return cur+1; |
142 | 0 | } |
143 | 0 | cur = json_get_next_object(cur); |
144 | 0 | } |
145 | 0 | return NULL; |
146 | 0 | } |
147 | | |
148 | | int json_get_array_len(jsmntok_t *array) |
149 | 0 | { |
150 | 0 | if (array->type != JSMN_ARRAY) |
151 | 0 | return -1; |
152 | 0 | return array->size; |
153 | 0 | } |
154 | | |
155 | | jsmntok_t *json_get_array_index(jsmntok_t *array, int idx) |
156 | 0 | { |
157 | 0 | int i; |
158 | 0 | jsmntok_t *cur = array+1; |
159 | | |
160 | |
|
161 | 0 | if (array->type != JSMN_ARRAY || idx < 0 || idx >= array->size) |
162 | 0 | return NULL; |
163 | 0 | for (i = 0; i < idx; i++) |
164 | 0 | cur = json_get_next_object(cur); |
165 | 0 | return cur; |
166 | 0 | } |
167 | | |
168 | | char *json_get_string(char *buf, jsmntok_t *parent, const char *name) |
169 | 0 | { |
170 | 0 | int i; |
171 | 0 | jsmntok_t *cur = parent+1; |
172 | |
|
173 | 0 | for (i = 0; i < parent->size; i++) { |
174 | 0 | if (cur->type == JSMN_STRING && |
175 | 0 | !strncmp(&buf[cur->start], name, cur->end - cur->start) |
176 | 0 | && strlen(name) == (size_t)(cur->end - cur->start) && |
177 | 0 | cur->size == 1 && (cur+1)->type == JSMN_STRING) { |
178 | 0 | buf[(cur+1)->end] = '\0'; |
179 | 0 | if (!json_decode_string_inplace(&buf[(cur+1)->start])) |
180 | 0 | return NULL; |
181 | 0 | return &buf[(cur+1)->start]; |
182 | 0 | } |
183 | 0 | cur = json_get_next_object(cur); |
184 | 0 | } |
185 | 0 | return NULL; |
186 | 0 | } |
187 | | |
188 | | bool json_get_double(char *buf, jsmntok_t *parent, const char *name, double *val) |
189 | 0 | { |
190 | 0 | int i; |
191 | 0 | jsmntok_t *cur = parent+1; |
192 | |
|
193 | 0 | for (i = 0; i < parent->size; i++) { |
194 | 0 | if (cur->type == JSMN_STRING && |
195 | 0 | !strncmp(&buf[cur->start], name, cur->end - cur->start) |
196 | 0 | && strlen(name) == (size_t)(cur->end - cur->start) && |
197 | 0 | cur->size == 1 && (cur+1)->type == JSMN_PRIMITIVE) { |
198 | 0 | buf[(cur+1)->end] = '\0'; |
199 | 0 | errno = 0; // GLib says it resets errno but this doesn't hurt. |
200 | 0 | *val = g_ascii_strtod(&buf[(cur+1)->start], NULL); |
201 | 0 | if (errno != 0) |
202 | 0 | return false; |
203 | 0 | return true; |
204 | 0 | } |
205 | 0 | cur = json_get_next_object(cur); |
206 | 0 | } |
207 | 0 | return false; |
208 | 0 | } |
209 | | |
210 | | bool json_get_int(char *buf, jsmntok_t *parent, const char *name, int64_t *val) |
211 | 0 | { |
212 | 0 | int i; |
213 | 0 | jsmntok_t *cur = parent+1; |
214 | |
|
215 | 0 | for (i = 0; i < parent->size; i++) { |
216 | 0 | if (cur->type == JSMN_STRING && |
217 | 0 | !strncmp(&buf[cur->start], name, cur->end - cur->start) |
218 | 0 | && strlen(name) == (size_t)(cur->end - cur->start) && |
219 | 0 | cur->size == 1 && (cur+1)->type == JSMN_PRIMITIVE) { |
220 | 0 | buf[(cur+1)->end] = '\0'; |
221 | 0 | return ws_strtoi64(&buf[(cur+1)->start], NULL, val); |
222 | 0 | } |
223 | 0 | cur = json_get_next_object(cur); |
224 | 0 | } |
225 | 0 | return false; |
226 | 0 | } |
227 | | |
228 | | bool json_get_boolean(char *buf, jsmntok_t *parent, const char *name, bool *val) |
229 | 0 | { |
230 | 0 | int i; |
231 | 0 | size_t tok_len; |
232 | 0 | jsmntok_t *cur = parent+1; |
233 | |
|
234 | 0 | for (i = 0; i < parent->size; i++) { |
235 | 0 | if (cur->type == JSMN_STRING && |
236 | 0 | !strncmp(&buf[cur->start], name, cur->end - cur->start) |
237 | 0 | && strlen(name) == (size_t)(cur->end - cur->start) && |
238 | 0 | cur->size == 1 && (cur+1)->type == JSMN_PRIMITIVE) { |
239 | | /* JSMN_STRICT guarantees that a primitive starts with the |
240 | | * correct character. |
241 | | */ |
242 | 0 | tok_len = (cur+1)->end - (cur+1)->start; |
243 | 0 | switch (buf[(cur+1)->start]) { |
244 | 0 | case 't': |
245 | 0 | if (tok_len == 4 && strncmp(&buf[(cur+1)->start], "true", tok_len) == 0) { |
246 | 0 | *val = true; |
247 | 0 | return true; |
248 | 0 | } |
249 | 0 | return false; |
250 | 0 | case 'f': |
251 | 0 | if (tok_len == 5 && strncmp(&buf[(cur+1)->start], "false", tok_len) == 0) { |
252 | 0 | *val = false; |
253 | 0 | return true; |
254 | 0 | } |
255 | 0 | return false; |
256 | 0 | default: |
257 | 0 | return false; |
258 | 0 | } |
259 | 0 | } |
260 | 0 | cur = json_get_next_object(cur); |
261 | 0 | } |
262 | 0 | return false; |
263 | 0 | } |
264 | | |
265 | | bool |
266 | | json_decode_string_inplace(char *text) |
267 | 0 | { |
268 | 0 | const char *input = text; |
269 | 0 | char *output = text; |
270 | 0 | while (*input) { |
271 | 0 | char ch = *input++; |
272 | |
|
273 | 0 | if (ch == '\\') { |
274 | 0 | ch = *input++; |
275 | |
|
276 | 0 | switch (ch) { |
277 | 0 | case '\"': |
278 | 0 | case '\\': |
279 | 0 | case '/': |
280 | 0 | *output++ = ch; |
281 | 0 | break; |
282 | | |
283 | 0 | case 'b': |
284 | 0 | *output++ = '\b'; |
285 | 0 | break; |
286 | 0 | case 'f': |
287 | 0 | *output++ = '\f'; |
288 | 0 | break; |
289 | 0 | case 'n': |
290 | 0 | *output++ = '\n'; |
291 | 0 | break; |
292 | 0 | case 'r': |
293 | 0 | *output++ = '\r'; |
294 | 0 | break; |
295 | 0 | case 't': |
296 | 0 | *output++ = '\t'; |
297 | 0 | break; |
298 | | |
299 | 0 | case 'u': |
300 | 0 | { |
301 | 0 | uint32_t unicode_hex = 0; |
302 | 0 | int k; |
303 | 0 | int bin; |
304 | |
|
305 | 0 | for (k = 0; k < 4; k++) { |
306 | 0 | unicode_hex <<= 4; |
307 | |
|
308 | 0 | ch = *input++; |
309 | 0 | bin = ws_xton(ch); |
310 | 0 | if (bin == -1) |
311 | 0 | return false; |
312 | 0 | unicode_hex |= bin; |
313 | 0 | } |
314 | | |
315 | 0 | if ((IS_LEAD_SURROGATE(unicode_hex))) { |
316 | 0 | uint16_t lead_surrogate = unicode_hex; |
317 | 0 | uint16_t trail_surrogate = 0; |
318 | |
|
319 | 0 | if (input[0] != '\\' || input[1] != 'u') |
320 | 0 | return false; |
321 | 0 | input += 2; |
322 | |
|
323 | 0 | for (k = 0; k < 4; k++) { |
324 | 0 | trail_surrogate <<= 4; |
325 | |
|
326 | 0 | ch = *input++; |
327 | 0 | bin = ws_xton(ch); |
328 | 0 | if (bin == -1) |
329 | 0 | return false; |
330 | 0 | trail_surrogate |= bin; |
331 | 0 | } |
332 | | |
333 | 0 | if ((!IS_TRAIL_SURROGATE(trail_surrogate))) |
334 | 0 | return false; |
335 | | |
336 | 0 | unicode_hex = SURROGATE_VALUE(lead_surrogate,trail_surrogate); |
337 | |
|
338 | 0 | } else if ((IS_TRAIL_SURROGATE(unicode_hex))) { |
339 | 0 | return false; |
340 | 0 | } |
341 | | |
342 | 0 | if (!g_unichar_validate(unicode_hex)) |
343 | 0 | return false; |
344 | | |
345 | | /* Don't allow NUL byte injection. */ |
346 | 0 | if (unicode_hex == 0) |
347 | 0 | return false; |
348 | | |
349 | | /* \uXXXX => 6 bytes, and g_unichar_to_utf8() requires to have output buffer at least 6 bytes -> OK. */ |
350 | 0 | k = g_unichar_to_utf8(unicode_hex, output); |
351 | 0 | output += k; |
352 | 0 | break; |
353 | 0 | } |
354 | | |
355 | 0 | default: |
356 | 0 | return false; |
357 | 0 | } |
358 | |
|
359 | 0 | } else { |
360 | 0 | *output = ch; |
361 | 0 | output++; |
362 | 0 | } |
363 | 0 | } |
364 | | |
365 | 0 | *output = '\0'; |
366 | 0 | return true; |
367 | 0 | } |
368 | | |
369 | | bool |
370 | | json_strip_jsonc_comments(char *text) |
371 | 0 | { |
372 | 0 | bool in_string = false; |
373 | 0 | size_t len = strlen(text); |
374 | 0 | size_t pos = 0; |
375 | |
|
376 | 0 | while (pos < len) { |
377 | 0 | char ch = text[pos]; |
378 | |
|
379 | 0 | if (in_string) { |
380 | 0 | if (ch == '\\') { |
381 | | /* Skip the escaped character, whatever it is; it can't |
382 | | * end the string or start a comment. */ |
383 | 0 | pos += (pos + 1 < len) ? 2 : 1; |
384 | 0 | continue; |
385 | 0 | } |
386 | 0 | if (ch == '"') |
387 | 0 | in_string = false; |
388 | 0 | pos++; |
389 | 0 | continue; |
390 | 0 | } |
391 | | |
392 | 0 | if (ch == '"') { |
393 | 0 | in_string = true; |
394 | 0 | pos++; |
395 | 0 | continue; |
396 | 0 | } |
397 | | |
398 | 0 | if (ch == '/' && pos + 1 < len && text[pos + 1] == '/') { |
399 | 0 | while (pos < len && text[pos] != '\n') { |
400 | 0 | text[pos] = ' '; |
401 | 0 | pos++; |
402 | 0 | } |
403 | 0 | continue; |
404 | 0 | } |
405 | | |
406 | 0 | if (ch == '/' && pos + 1 < len && text[pos + 1] == '*') { |
407 | 0 | bool closed = false; |
408 | |
|
409 | 0 | text[pos] = ' '; |
410 | 0 | text[pos + 1] = ' '; |
411 | 0 | pos += 2; |
412 | 0 | while (pos < len) { |
413 | 0 | if (text[pos] == '*' && pos + 1 < len && text[pos + 1] == '/') { |
414 | 0 | text[pos] = ' '; |
415 | 0 | text[pos + 1] = ' '; |
416 | 0 | pos += 2; |
417 | 0 | closed = true; |
418 | 0 | break; |
419 | 0 | } |
420 | 0 | if (text[pos] != '\n') |
421 | 0 | text[pos] = ' '; |
422 | 0 | pos++; |
423 | 0 | } |
424 | 0 | if (!closed) { |
425 | 0 | ws_debug("unterminated block comment in JSONC data"); |
426 | 0 | return false; |
427 | 0 | } |
428 | 0 | continue; |
429 | 0 | } |
430 | 0 | pos++; |
431 | 0 | } |
432 | | |
433 | 0 | return true; |
434 | 0 | } |
435 | | |
436 | | /* |
437 | | * Editor modelines - https://www.wireshark.org/tools/modelines.html |
438 | | * |
439 | | * Local variables: |
440 | | * c-basic-offset: 4 |
441 | | * tab-width: 8 |
442 | | * indent-tabs-mode: nil |
443 | | * End: |
444 | | * |
445 | | * vi: set shiftwidth=4 tabstop=8 expandtab: |
446 | | * :indentSize=4:tabSize=8:noTabs=true: |
447 | | */ |