/src/curl/lib/curlx/strparse.c
Line | Count | Source |
1 | | /*************************************************************************** |
2 | | * _ _ ____ _ |
3 | | * Project ___| | | | _ \| | |
4 | | * / __| | | | |_) | | |
5 | | * | (__| |_| | _ <| |___ |
6 | | * \___|\___/|_| \_\_____| |
7 | | * |
8 | | * Copyright (C) Daniel Stenberg, <daniel@haxx.se>, et al. |
9 | | * |
10 | | * This software is licensed as described in the file COPYING, which |
11 | | * you should have received as part of this distribution. The terms |
12 | | * are also available at https://curl.se/docs/copyright.html. |
13 | | * |
14 | | * You may opt to use, copy, modify, merge, publish, distribute and/or sell |
15 | | * copies of the Software, and permit persons to whom the Software is |
16 | | * furnished to do so, under the terms of the COPYING file. |
17 | | * |
18 | | * This software is distributed on an "AS IS" basis, WITHOUT WARRANTY OF ANY |
19 | | * KIND, either express or implied. |
20 | | * |
21 | | * SPDX-License-Identifier: curl |
22 | | * |
23 | | ***************************************************************************/ |
24 | | #include "curlx/strparse.h" |
25 | | |
26 | | void curlx_str_init(struct Curl_str *out) |
27 | 796k | { |
28 | 796k | out->str = NULL; |
29 | 796k | out->len = 0; |
30 | 796k | } |
31 | | |
32 | | void curlx_str_assign(struct Curl_str *out, const char *str, size_t len) |
33 | 20.7k | { |
34 | 20.7k | out->str = str; |
35 | 20.7k | out->len = len; |
36 | 20.7k | } |
37 | | |
38 | | /* remove bytes from the end of the string, never remove more bytes than what |
39 | | the string holds! */ |
40 | | void curlx_str_trim(struct Curl_str *out, size_t len) |
41 | 0 | { |
42 | 0 | DEBUGASSERT(out); |
43 | 0 | DEBUGASSERT(out->len >= len); |
44 | 0 | out->len -= len; |
45 | 0 | } |
46 | | |
47 | | /* Get a word until the first DELIM or end of string. At least one byte long. |
48 | | return non-zero on error. If 'max' is zero, it will always return error. */ |
49 | | int curlx_str_until(const char **linep, struct Curl_str *out, |
50 | | const size_t max, char delim) |
51 | 507k | { |
52 | 507k | const char *s; |
53 | 507k | size_t len = 0; |
54 | 507k | DEBUGASSERT(linep); |
55 | 507k | DEBUGASSERT(*linep); |
56 | 507k | DEBUGASSERT(out); |
57 | 507k | DEBUGASSERT(delim); |
58 | 507k | s = *linep; |
59 | | |
60 | 507k | curlx_str_init(out); |
61 | 9.73M | while(*s && (*s != delim)) { |
62 | 9.23M | s++; |
63 | 9.23M | if(++len > max) { |
64 | 4.79k | return STRE_BIG; |
65 | 4.79k | } |
66 | 9.23M | } |
67 | 502k | if(!len) |
68 | 21.3k | return STRE_SHORT; |
69 | 481k | out->str = *linep; |
70 | 481k | out->len = len; |
71 | 481k | *linep = s; /* point to the first byte after the word */ |
72 | 481k | return STRE_OK; |
73 | 502k | } |
74 | | |
75 | | /* Get a word until the first space or end of string. At least one byte long. |
76 | | return non-zero on error */ |
77 | | int curlx_str_word(const char **linep, struct Curl_str *out, const size_t max) |
78 | 205k | { |
79 | 205k | return curlx_str_until(linep, out, max, ' '); |
80 | 205k | } |
81 | | |
82 | | /* Get a word until a newline byte or end of string. At least one byte long. |
83 | | return non-zero on error */ |
84 | | int curlx_str_untilnl(const char **linep, struct Curl_str *out, |
85 | | const size_t max) |
86 | 72.5k | { |
87 | 72.5k | const char *s = *linep; |
88 | 72.5k | size_t len = 0; |
89 | 72.5k | DEBUGASSERT(linep && *linep && out && max); |
90 | | |
91 | 72.5k | curlx_str_init(out); |
92 | 8.62M | while(*s && !ISNEWLINE(*s)) { |
93 | 8.55M | s++; |
94 | 8.55M | if(++len > max) |
95 | 0 | return STRE_BIG; |
96 | 8.55M | } |
97 | 72.5k | if(!len) |
98 | 2.13k | return STRE_SHORT; |
99 | 70.4k | out->str = *linep; |
100 | 70.4k | out->len = len; |
101 | 70.4k | *linep = s; /* point to the first byte after the word */ |
102 | 70.4k | return STRE_OK; |
103 | 72.5k | } |
104 | | |
105 | | /* Get a "quoted" word. Escaped quotes are supported. |
106 | | return non-zero on error */ |
107 | | int curlx_str_quotedword(const char **linep, struct Curl_str *out, |
108 | | const size_t max) |
109 | 9.04k | { |
110 | 9.04k | const char *s = *linep; |
111 | 9.04k | size_t len = 0; |
112 | 9.04k | DEBUGASSERT(linep && *linep && out && max); |
113 | | |
114 | 9.04k | curlx_str_init(out); |
115 | 9.04k | if(*s != '\"') |
116 | 50 | return STRE_BEGQUOTE; |
117 | 8.99k | s++; |
118 | 3.79M | while(*s && (*s != '\"')) { |
119 | 3.78M | if(*s == '\\' && s[1]) { |
120 | 1.28k | s++; |
121 | 1.28k | if(++len > max) |
122 | 10 | return STRE_BIG; |
123 | 1.28k | } |
124 | 3.78M | s++; |
125 | 3.78M | if(++len > max) |
126 | 15 | return STRE_BIG; |
127 | 3.78M | } |
128 | 8.97k | if(*s != '\"') |
129 | 2.79k | return STRE_ENDQUOTE; |
130 | 6.18k | out->str = (*linep) + 1; |
131 | 6.18k | out->len = len; |
132 | 6.18k | *linep = s + 1; |
133 | 6.18k | return STRE_OK; |
134 | 8.97k | } |
135 | | |
136 | | /* Advance over a single character. |
137 | | return non-zero on error */ |
138 | | int curlx_str_single(const char **linep, char byte) |
139 | 7.65M | { |
140 | 7.65M | DEBUGASSERT(linep && *linep); |
141 | 7.65M | if(**linep != byte) |
142 | 6.55M | return STRE_BYTE; |
143 | 1.10M | (*linep)++; /* move over it */ |
144 | 1.10M | return STRE_OK; |
145 | 7.65M | } |
146 | | |
147 | | /* Advance over a single space. |
148 | | return non-zero on error */ |
149 | | int curlx_str_singlespace(const char **linep) |
150 | 205k | { |
151 | 205k | return curlx_str_single(linep, ' '); |
152 | 205k | } |
153 | | |
154 | | /* given an ASCII character and max ascii, return TRUE if valid */ |
155 | | #define valid_digit(x, m) \ |
156 | 41.7M | (((x) >= '0') && ((x) <= (m)) && curlx_hexasciitable[(x) - '0']) |
157 | | |
158 | | /* We use 16 for the zero index (and the necessary bitwise AND in the loop) |
159 | | to be able to have a non-zero value there to make valid_digit() able to |
160 | | use the info */ |
161 | | const unsigned char curlx_hexasciitable[] = { |
162 | | 16, 1, 2, 3, 4, 5, 6, 7, 8, 9, /* 0x30: 0 - 9 */ |
163 | | 0, 0, 0, 0, 0, 0, 0, |
164 | | 10, 11, 12, 13, 14, 15, /* 0x41: A - F */ |
165 | | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
166 | | 10, 11, 12, 13, 14, 15 /* 0x61: a - f */ |
167 | | }; |
168 | | |
169 | | /* no support for 0x prefix nor leading spaces */ |
170 | | static int str_num_base(const char **linep, curl_off_t *nump, curl_off_t max, |
171 | | int base) /* 8, 10 or 16, nothing else */ |
172 | 17.0M | { |
173 | 17.0M | curl_off_t num = 0; |
174 | 17.0M | const char *p; |
175 | 17.0M | int m = (base == 10) ? '9' : /* the largest digit possible */ |
176 | 17.0M | (base == 16) ? 'f' : '7'; |
177 | 17.0M | DEBUGASSERT(linep && *linep && nump); |
178 | 17.0M | DEBUGASSERT((base == 8) || (base == 10) || (base == 16)); |
179 | 17.0M | DEBUGASSERT(max >= 0); /* mostly to catch SIZE_MAX, which is too large */ |
180 | 17.0M | *nump = 0; |
181 | 17.0M | p = *linep; |
182 | 17.0M | if(!valid_digit(*p, m)) |
183 | 4.74M | return STRE_NO_NUM; |
184 | 12.2M | if(max < base) { |
185 | | /* special-case low max scenario because check needs to be different */ |
186 | 0 | do { |
187 | 0 | int n = curlx_hexval(*p++); |
188 | 0 | num = (num * base) + n; |
189 | 0 | if(num > max) |
190 | 0 | return STRE_OVERFLOW; |
191 | 0 | } while(valid_digit(*p, m)); |
192 | 0 | } |
193 | 12.2M | else { |
194 | 24.6M | do { |
195 | 24.6M | int n = curlx_hexval(*p++); |
196 | 24.6M | if(num > ((max - n) / base)) |
197 | 7.13k | return STRE_OVERFLOW; |
198 | 24.6M | num = (num * base) + n; |
199 | 24.6M | } while(valid_digit(*p, m)); |
200 | 12.2M | } |
201 | 12.2M | *nump = num; |
202 | 12.2M | *linep = p; |
203 | 12.2M | return STRE_OK; |
204 | 12.2M | } |
205 | | |
206 | | /* Get an unsigned decimal number with no leading space or minus. Leading |
207 | | zeroes are accepted. return non-zero on error */ |
208 | | int curlx_str_number(const char **linep, curl_off_t *nump, curl_off_t max) |
209 | 16.8M | { |
210 | 16.8M | return str_num_base(linep, nump, max, 10); |
211 | 16.8M | } |
212 | | |
213 | | /* Get an unsigned hexadecimal number with no leading space or minus and no |
214 | | "0x" support. Leading zeroes are accepted. return non-zero on error */ |
215 | | int curlx_str_hex(const char **linep, curl_off_t *nump, curl_off_t max) |
216 | 2.90k | { |
217 | 2.90k | return str_num_base(linep, nump, max, 16); |
218 | 2.90k | } |
219 | | |
220 | | /* Get an unsigned octal number with no leading space or minus and no "0" |
221 | | prefix support. Leading zeroes are accepted. return non-zero on error */ |
222 | | int curlx_str_octal(const char **linep, curl_off_t *nump, curl_off_t max) |
223 | 177k | { |
224 | 177k | return str_num_base(linep, nump, max, 8); |
225 | 177k | } |
226 | | |
227 | | /* |
228 | | * Parse a positive number up to 63-bit number written in ASCII. Skip leading |
229 | | * blanks. No support for prefixes. |
230 | | */ |
231 | | int curlx_str_numblanks(const char **str, curl_off_t *num) |
232 | 7.78k | { |
233 | 7.78k | curlx_str_passblanks(str); |
234 | 7.78k | return curlx_str_number(str, num, CURL_OFF_T_MAX); |
235 | 7.78k | } |
236 | | |
237 | | /* CR or LF |
238 | | return non-zero on error */ |
239 | | int curlx_str_newline(const char **linep) |
240 | 5.14k | { |
241 | 5.14k | DEBUGASSERT(linep && *linep); |
242 | 5.14k | if(ISNEWLINE(**linep)) { |
243 | 5.10k | (*linep)++; |
244 | 5.10k | return STRE_OK; /* yessir */ |
245 | 5.10k | } |
246 | 32 | return STRE_NEWLINE; |
247 | 5.14k | } |
248 | | |
249 | | #ifndef WITHOUT_LIBCURL |
250 | | /* case insensitive compare that the parsed string matches the given string. |
251 | | Returns non-zero on match. */ |
252 | | int curlx_str_casecompare(struct Curl_str *str, const char *check) |
253 | 1.77M | { |
254 | 1.77M | size_t clen = check ? strlen(check) : 0; |
255 | 1.77M | return ((str->len == clen) && curl_strnequal(str->str, check, clen)); |
256 | 1.77M | } |
257 | | #endif |
258 | | |
259 | | /* case sensitive string compare. Returns non-zero on match. */ |
260 | | int curlx_str_cmp(struct Curl_str *str, const char *check) |
261 | 76.8k | { |
262 | 76.8k | if(check) { |
263 | 76.8k | size_t clen = strlen(check); |
264 | 76.8k | return ((str->len == clen) && !strncmp(str->str, check, clen)); |
265 | 76.8k | } |
266 | 0 | return !!(str->len); |
267 | 76.8k | } |
268 | | |
269 | | /* Trim off 'num' number of bytes from the beginning (left side) of the |
270 | | string. If 'num' is larger than the string, return error. */ |
271 | | int curlx_str_nudge(struct Curl_str *str, size_t num) |
272 | 98.1k | { |
273 | 98.1k | if(num <= str->len) { |
274 | 98.1k | str->str += num; |
275 | 98.1k | str->len -= num; |
276 | 98.1k | return STRE_OK; |
277 | 98.1k | } |
278 | 0 | return STRE_OVERFLOW; |
279 | 98.1k | } |
280 | | |
281 | | /* Get the following character sequence that consists only of bytes not |
282 | | present in the 'reject' string. Like strcspn(). */ |
283 | | int curlx_str_cspn(const char **linep, struct Curl_str *out, |
284 | | const char *reject) |
285 | 981k | { |
286 | 981k | const char *s = *linep; |
287 | 981k | size_t len; |
288 | 981k | DEBUGASSERT(linep && *linep); |
289 | | |
290 | 981k | len = strcspn(s, reject); |
291 | 981k | if(len) { |
292 | 873k | out->str = s; |
293 | 873k | out->len = len; |
294 | 873k | *linep = &s[len]; |
295 | 873k | return STRE_OK; |
296 | 873k | } |
297 | 107k | curlx_str_init(out); |
298 | 107k | return STRE_SHORT; |
299 | 981k | } |
300 | | |
301 | | /* remove ISBLANK()s from both ends of the string */ |
302 | | void curlx_str_trimblanks(struct Curl_str *out) |
303 | 997k | { |
304 | 1.09M | while(out->len && ISBLANK(*out->str)) |
305 | 94.0k | curlx_str_nudge(out, 1); |
306 | | |
307 | | /* trim trailing spaces and tabs */ |
308 | 1.03M | while(out->len && ISBLANK(out->str[out->len - 1])) |
309 | 34.3k | out->len--; |
310 | 997k | } |
311 | | |
312 | | /* increase the pointer until it has moved over all blanks */ |
313 | | void curlx_str_passblanks(const char **linep) |
314 | 1.29M | { |
315 | 1.44M | while(ISBLANK(**linep)) |
316 | 152k | (*linep)++; /* move over it */ |
317 | 1.29M | } |