/src/simdutf/fuzz/roundtrip.cpp
Line | Count | Source |
1 | | #include <cstring> |
2 | | #include <fuzzer/FuzzedDataProvider.h> |
3 | | #include <memory> |
4 | | #include <cstdlib> |
5 | | #include <string> |
6 | | #include <iostream> |
7 | | |
8 | | #include "simdutf.h" |
9 | | |
10 | | // useful for debugging |
11 | | static void print_input(const std::string& s, |
12 | 0 | const simdutf::implementation* const e) { |
13 | 0 | printf("We are about to abort on the following input: "); |
14 | 0 | for (auto c : s) { |
15 | 0 | printf("%02x ", (unsigned char)c); |
16 | 0 | } |
17 | 0 | printf("\n"); |
18 | 0 | std::cout << "string length : " << s.size() << " bytes" << std::endl; |
19 | 0 | std::cout << "implementation->name() = " << e->name() << std::endl; |
20 | 0 | } |
21 | | |
22 | | /** |
23 | | * We do round trips from UTF-8 to UTF-16, from UTF-8 to UTF-32, from UTF-16 to |
24 | | * UTF-8. |
25 | | * We do round trips from Latin 1 to UTF-8, from Latin 1 to UTF-16, from Latin 1 |
26 | | * to UTF-32. We test all available kernels. We also try to transcode invalid |
27 | | * inputs. |
28 | | */ |
29 | 5.30k | extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { |
30 | 5.30k | FuzzedDataProvider fdp(data, size); |
31 | 5.30k | constexpr int kMaxStringSize = 1024; |
32 | 5.30k | std::string source = fdp.ConsumeRandomLengthString(kMaxStringSize); |
33 | 21.2k | for (auto& e : simdutf::get_available_implementations()) { |
34 | 21.2k | if (!e->supported_by_runtime_system()) { |
35 | 5.30k | continue; |
36 | 5.30k | } |
37 | | /** |
38 | | * Transcoding from UTF-8 to UTF-16LE. |
39 | | */ |
40 | 15.9k | bool validutf8 = e->validate_utf8(source.c_str(), source.size()); |
41 | 15.9k | auto rutf8 = e->validate_utf8_with_errors(source.c_str(), source.size()); |
42 | 15.9k | if (validutf8 != (rutf8.error == simdutf::SUCCESS)) { // they should agree |
43 | 0 | print_input(source, e); |
44 | 0 | abort(); |
45 | 0 | } |
46 | 15.9k | if (validutf8) { |
47 | | // We need a buffer of size where to write the UTF-16LE words. |
48 | 14.6k | size_t expected_utf16words = |
49 | 14.6k | e->utf16_length_from_utf8(source.c_str(), source.size()); |
50 | 14.6k | std::unique_ptr<char16_t[]> utf16_output{ |
51 | 14.6k | new char16_t[expected_utf16words]}; |
52 | | // convert to UTF-16LE |
53 | 14.6k | size_t utf16words = e->convert_utf8_to_utf16le( |
54 | 14.6k | source.c_str(), source.size(), utf16_output.get()); |
55 | | // It wrote utf16words * sizeof(char16_t) bytes. |
56 | 14.6k | bool validutf16 = e->validate_utf16le(utf16_output.get(), utf16words); |
57 | 14.6k | if (!validutf16) { |
58 | 0 | print_input(source, e); |
59 | 0 | abort(); |
60 | 0 | } |
61 | | // convert it back: |
62 | | // We need a buffer of size where to write the UTF-8 words. |
63 | 14.6k | size_t expected_utf8words = |
64 | 14.6k | e->utf8_length_from_utf16le(utf16_output.get(), utf16words); |
65 | 14.6k | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
66 | | // convert to UTF-8 |
67 | 14.6k | size_t utf8words = e->convert_utf16le_to_utf8( |
68 | 14.6k | utf16_output.get(), utf16words, utf8_output.get()); |
69 | 14.6k | std::string final_string(utf8_output.get(), utf8words); |
70 | 14.6k | if (final_string != source) { |
71 | 0 | print_input(source, e); |
72 | 0 | abort(); |
73 | 0 | } |
74 | 14.6k | } else { |
75 | | // invalid input!!! |
76 | | // We need a buffer of size where to write the UTF-16LE words. |
77 | 1.27k | size_t expected_utf16words = |
78 | 1.27k | e->utf16_length_from_utf8(source.c_str(), source.size()); |
79 | 1.27k | std::unique_ptr<char16_t[]> utf16_output{ |
80 | 1.27k | new char16_t[expected_utf16words]}; |
81 | | // convert to UTF-16LE |
82 | 1.27k | size_t utf16words = e->convert_utf8_to_utf16le( |
83 | 1.27k | source.c_str(), source.size(), utf16_output.get()); |
84 | 1.27k | if (utf16words != 0) { |
85 | 0 | print_input(source, e); |
86 | 0 | abort(); |
87 | 0 | } |
88 | 1.27k | } |
89 | | |
90 | | /** |
91 | | * Transcoding from UTF-8 to UTF-16BE. |
92 | | */ |
93 | 15.9k | if (validutf8) { |
94 | | // We need a buffer of size where to write the UTF-16BE words. |
95 | 14.6k | size_t expected_utf16words = |
96 | 14.6k | e->utf16_length_from_utf8(source.c_str(), source.size()); |
97 | 14.6k | std::unique_ptr<char16_t[]> utf16_output{ |
98 | 14.6k | new char16_t[expected_utf16words]}; |
99 | | // convert to UTF-16BE |
100 | 14.6k | size_t utf16words = e->convert_utf8_to_utf16be( |
101 | 14.6k | source.c_str(), source.size(), utf16_output.get()); |
102 | | // It wrote utf16words * sizeof(char16_t) bytes. |
103 | 14.6k | bool validutf16 = e->validate_utf16be(utf16_output.get(), utf16words); |
104 | 14.6k | if (!validutf16) { |
105 | 0 | print_input(source, e); |
106 | 0 | abort(); |
107 | 0 | } |
108 | | // convert it back: |
109 | | // We need a buffer of size where to write the UTF-8 words. |
110 | 14.6k | size_t expected_utf8words = |
111 | 14.6k | e->utf8_length_from_utf16be(utf16_output.get(), utf16words); |
112 | 14.6k | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
113 | | // convert to UTF-8 |
114 | 14.6k | size_t utf8words = e->convert_utf16be_to_utf8( |
115 | 14.6k | utf16_output.get(), utf16words, utf8_output.get()); |
116 | 14.6k | std::string final_string(utf8_output.get(), utf8words); |
117 | 14.6k | if (final_string != source) { |
118 | 0 | print_input(source, e); |
119 | 0 | abort(); |
120 | 0 | } |
121 | 14.6k | } else { |
122 | | // invalid input!!! |
123 | | // We need a buffer of size where to write the UTF-16BE words. |
124 | 1.27k | size_t expected_utf16words = |
125 | 1.27k | e->utf16_length_from_utf8(source.c_str(), source.size()); |
126 | 1.27k | std::unique_ptr<char16_t[]> utf16_output{ |
127 | 1.27k | new char16_t[expected_utf16words]}; |
128 | | // convert to UTF-16BE |
129 | 1.27k | size_t utf16words = e->convert_utf8_to_utf16be( |
130 | 1.27k | source.c_str(), source.size(), utf16_output.get()); |
131 | 1.27k | if (utf16words != 0) { |
132 | 0 | print_input(source, e); |
133 | 0 | abort(); |
134 | 0 | } |
135 | 1.27k | } |
136 | | /** |
137 | | * Transcoding from UTF-8 to UTF-32. |
138 | | */ |
139 | 15.9k | if (validutf8) { |
140 | | // We need a buffer of size where to write the UTF-32 words. |
141 | 14.6k | size_t expected_utf32words = |
142 | 14.6k | e->utf32_length_from_utf8(source.c_str(), source.size()); |
143 | 14.6k | std::unique_ptr<char32_t[]> utf32_output{ |
144 | 14.6k | new char32_t[expected_utf32words]}; |
145 | | // convert to UTF-32 |
146 | 14.6k | size_t utf32words = e->convert_utf8_to_utf32( |
147 | 14.6k | source.c_str(), source.size(), utf32_output.get()); |
148 | | // It wrote utf32words * sizeof(char32_t) bytes. |
149 | 14.6k | bool validutf32 = e->validate_utf32(utf32_output.get(), utf32words); |
150 | 14.6k | if (!validutf32) { |
151 | 0 | return -1; |
152 | 0 | } |
153 | | // convert it back: |
154 | | // We need a buffer of size where to write the UTF-8 words. |
155 | 14.6k | size_t expected_utf8words = |
156 | 14.6k | e->utf8_length_from_utf32(utf32_output.get(), utf32words); |
157 | 14.6k | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
158 | | // convert to UTF-8 |
159 | 14.6k | size_t utf8words = e->convert_utf32_to_utf8( |
160 | 14.6k | utf32_output.get(), utf32words, utf8_output.get()); |
161 | 14.6k | std::string final_string(utf8_output.get(), utf8words); |
162 | 14.6k | if (source != final_string) { |
163 | 0 | print_input(source, e); |
164 | 0 | abort(); |
165 | 0 | } |
166 | 14.6k | } else { |
167 | | // invalid input!!! |
168 | 1.27k | size_t expected_utf32words = |
169 | 1.27k | e->utf32_length_from_utf8(source.c_str(), source.size()); |
170 | 1.27k | std::unique_ptr<char32_t[]> utf32_output{ |
171 | 1.27k | new char32_t[expected_utf32words]}; |
172 | | // convert to UTF-32 |
173 | 1.27k | size_t utf32words = e->convert_utf8_to_utf32( |
174 | 1.27k | source.c_str(), source.size(), utf32_output.get()); |
175 | 1.27k | if (utf32words != 0) { |
176 | 0 | print_input(source, e); |
177 | 0 | abort(); |
178 | 0 | } |
179 | 1.27k | } |
180 | | |
181 | | /** |
182 | | * Transcoding from UTF-8 to Latin 1 |
183 | | */ |
184 | 15.9k | if (validutf8) { |
185 | | // We need a buffer of size where to write the UTF-16LE words. |
186 | 14.6k | size_t expected_latin1words = |
187 | 14.6k | e->latin1_length_from_utf8(source.c_str(), source.size()); |
188 | 14.6k | std::unique_ptr<char[]> latin1_output{new char[expected_latin1words]}; |
189 | | // convert to latin1 |
190 | 14.6k | size_t latin1words = e->convert_utf8_to_latin1( |
191 | 14.6k | source.c_str(), source.size(), latin1_output.get()); |
192 | 14.6k | if (latin1words != 0) { |
193 | | // convert it back: |
194 | | // We need a buffer of size where to write the UTF-8 words. |
195 | 756 | size_t expected_utf8words = |
196 | 756 | e->utf8_length_from_latin1(latin1_output.get(), latin1words); |
197 | 756 | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
198 | | // convert to UTF-8 |
199 | 756 | size_t utf8words = e->convert_latin1_to_utf8( |
200 | 756 | latin1_output.get(), latin1words, utf8_output.get()); |
201 | 756 | std::string final_string(utf8_output.get(), utf8words); |
202 | 756 | if (final_string != source) { |
203 | 0 | print_input(source, e); |
204 | 0 | abort(); |
205 | 0 | } |
206 | 756 | } |
207 | 14.6k | } else { |
208 | | // invalid input!!! |
209 | | // We need a buffer of size where to write the Latin 1 words. |
210 | 1.27k | size_t expected_latin1words = |
211 | 1.27k | e->latin1_length_from_utf8(source.c_str(), source.size()); |
212 | 1.27k | std::unique_ptr<char[]> latin1_output{new char[expected_latin1words]}; |
213 | | // convert to Latin 1 |
214 | 1.27k | size_t latin1words = e->convert_utf8_to_latin1( |
215 | 1.27k | source.c_str(), source.size(), latin1_output.get()); |
216 | 1.27k | if (latin1words != 0) { |
217 | 0 | print_input(source, e); |
218 | 0 | abort(); |
219 | 0 | } |
220 | 1.27k | } |
221 | | /** |
222 | | * Transcoding from UTF-16LE to UTF-8. |
223 | | */ |
224 | 15.9k | { |
225 | | // Get new source data here as this will allow the fuzzer to optimize it's |
226 | | // input for UTF16-LE. |
227 | 15.9k | source = fdp.ConsumeRandomLengthString(kMaxStringSize); |
228 | | // We copy to avoid alignment issues. |
229 | 15.9k | std::unique_ptr<char16_t[]> utf16_source{new char16_t[source.size() / 2]}; |
230 | 15.9k | if (source.data() != nullptr) { |
231 | 15.9k | std::memcpy(utf16_source.get(), source.data(), source.size() / 2 * 2); |
232 | 15.9k | } |
233 | 15.9k | bool validutf16le = |
234 | 15.9k | e->validate_utf16le(utf16_source.get(), source.size() / 2); |
235 | 15.9k | auto rutf16le = e->validate_utf16le_with_errors(utf16_source.get(), |
236 | 15.9k | source.size() / 2); |
237 | 15.9k | if (validutf16le != |
238 | 15.9k | (rutf16le.error == simdutf::SUCCESS)) { // they should agree |
239 | 0 | print_input(source, e); |
240 | 0 | abort(); |
241 | 0 | } |
242 | 15.9k | if (validutf16le) { |
243 | | // We need a buffer of size where to write the UTF-16 words. |
244 | 15.6k | size_t expected_utf8words = |
245 | 15.6k | e->utf8_length_from_utf16le(utf16_source.get(), source.size() / 2); |
246 | 15.6k | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
247 | 15.6k | size_t utf8words = e->convert_utf16le_to_utf8( |
248 | 15.6k | utf16_source.get(), source.size() / 2, utf8_output.get()); |
249 | | // It wrote utf16words * sizeof(char16_t) bytes. |
250 | 15.6k | bool validutf8 = e->validate_utf8(utf8_output.get(), utf8words); |
251 | 15.6k | if (!validutf8) { |
252 | 0 | print_input(source, e); |
253 | 0 | abort(); |
254 | 0 | } |
255 | | // convert it back: |
256 | | // We need a buffer of size where to write the UTF-16 words. |
257 | 15.6k | size_t expected_utf16words = |
258 | 15.6k | e->utf16_length_from_utf8(utf8_output.get(), utf8words); |
259 | 15.6k | std::unique_ptr<char16_t[]> utf16_output{ |
260 | 15.6k | new char16_t[expected_utf16words]}; |
261 | | // convert to UTF-8 |
262 | 15.6k | size_t utf16words = e->convert_utf8_to_utf16le( |
263 | 15.6k | utf8_output.get(), utf8words, utf16_output.get()); |
264 | 87.5k | for (size_t i = 0; i < source.size() / 2; i++) { |
265 | 71.9k | if (utf16_output.get()[i] != (utf16_source.get())[i]) { |
266 | 0 | print_input(source, e); |
267 | 0 | abort(); |
268 | 0 | } |
269 | 71.9k | } |
270 | 15.6k | } else { |
271 | | // invalid input!!! |
272 | | // We need a buffer of size where to write the UTF-16 words. |
273 | 243 | size_t expected_utf8words = |
274 | 243 | e->utf8_length_from_utf16le(utf16_source.get(), source.size() / 2); |
275 | 243 | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
276 | 243 | size_t utf8words = e->convert_utf16le_to_utf8( |
277 | 243 | utf16_source.get(), source.size() / 2, utf8_output.get()); |
278 | 243 | if (utf8words != 0) { |
279 | 0 | print_input(source, e); |
280 | 0 | abort(); |
281 | 0 | } |
282 | 243 | } |
283 | 15.9k | } |
284 | | |
285 | | /** |
286 | | * Transcoding from UTF-16BE to UTF-8. |
287 | | */ |
288 | 15.9k | { |
289 | | // Get new source data here as this will allow the fuzzer to optimize it's |
290 | | // input for UTF16-BE. |
291 | 15.9k | source = fdp.ConsumeRandomLengthString(kMaxStringSize); |
292 | 15.9k | std::unique_ptr<char16_t[]> utf16_source{new char16_t[source.size() / 2]}; |
293 | 15.9k | if (source.data() != nullptr) { |
294 | 15.9k | std::memcpy(utf16_source.get(), source.data(), source.size() / 2 * 2); |
295 | 15.9k | } |
296 | 15.9k | bool validutf16be = |
297 | 15.9k | e->validate_utf16be(utf16_source.get(), source.size() / 2); |
298 | 15.9k | auto rutf16be = e->validate_utf16be_with_errors(utf16_source.get(), |
299 | 15.9k | source.size() / 2); |
300 | 15.9k | if (validutf16be != |
301 | 15.9k | (rutf16be.error == simdutf::SUCCESS)) { // they should agree |
302 | 0 | print_input(source, e); |
303 | 0 | abort(); |
304 | 0 | } |
305 | 15.9k | if (validutf16be) { |
306 | | // We need a buffer of size where to write the UTF-16 words. |
307 | 15.6k | size_t expected_utf8words = |
308 | 15.6k | e->utf8_length_from_utf16be(utf16_source.get(), source.size() / 2); |
309 | 15.6k | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
310 | 15.6k | size_t utf8words = e->convert_utf16be_to_utf8( |
311 | 15.6k | utf16_source.get(), source.size() / 2, utf8_output.get()); |
312 | | // It wrote utf16words * sizeof(char16_t) bytes. |
313 | 15.6k | bool validutf8 = e->validate_utf8(utf8_output.get(), utf8words); |
314 | 15.6k | if (!validutf8) { |
315 | 0 | print_input(source, e); |
316 | 0 | abort(); |
317 | 0 | } |
318 | | // convert it back: |
319 | | // We need a buffer of size where to write the UTF-16 words. |
320 | 15.6k | size_t expected_utf16words = |
321 | 15.6k | e->utf16_length_from_utf8(utf8_output.get(), utf8words); |
322 | 15.6k | std::unique_ptr<char16_t[]> utf16_output{ |
323 | 15.6k | new char16_t[expected_utf16words]}; |
324 | | // convert to UTF-8 |
325 | 15.6k | size_t utf16words = e->convert_utf8_to_utf16be( |
326 | 15.6k | utf8_output.get(), utf8words, utf16_output.get()); |
327 | 90.5k | for (size_t i = 0; i < source.size() / 2; i++) { |
328 | 74.8k | if (utf16_output.get()[i] != (utf16_source.get())[i]) { |
329 | 0 | print_input(source, e); |
330 | 0 | abort(); |
331 | 0 | } |
332 | 74.8k | } |
333 | 15.6k | } else { |
334 | | // invalid input!!! |
335 | | // We need a buffer of size where to write the UTF-16 words. |
336 | 251 | size_t expected_utf8words = |
337 | 251 | e->utf8_length_from_utf16be(utf16_source.get(), source.size() / 2); |
338 | 251 | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
339 | 251 | size_t utf8words = e->convert_utf16be_to_utf8( |
340 | 251 | utf16_source.get(), source.size() / 2, utf8_output.get()); |
341 | 251 | if (utf8words != 0) { |
342 | 0 | print_input(source, e); |
343 | 0 | abort(); |
344 | 0 | } |
345 | 251 | } |
346 | 15.9k | } |
347 | | |
348 | | /** |
349 | | * Transcoding from latin1 to UTF-8. |
350 | | */ |
351 | | // Get new source data here as this will allow the fuzzer to optimize it's |
352 | | // input for latin1. |
353 | 15.9k | source = fdp.ConsumeRandomLengthString(kMaxStringSize); |
354 | 15.9k | bool validlatin1 = true; // has to be |
355 | 15.9k | if (validlatin1) { |
356 | | // We need a buffer of size where to write the UTF-8 words. |
357 | 15.9k | size_t expected_utf8words = |
358 | 15.9k | e->utf8_length_from_latin1(source.c_str(), source.size()); |
359 | 15.9k | std::unique_ptr<char[]> utf8_output{new char[expected_utf8words]}; |
360 | 15.9k | size_t utf8words = e->convert_latin1_to_utf8( |
361 | 15.9k | source.c_str(), source.size(), utf8_output.get()); |
362 | | // It wrote utf8words * sizeof(char) bytes. |
363 | 15.9k | bool validutf8 = e->validate_utf8(utf8_output.get(), utf8words); |
364 | 15.9k | if (!validutf8) { |
365 | 0 | print_input(source, e); |
366 | 0 | abort(); |
367 | 0 | } |
368 | | // convert it back: |
369 | | // We need a buffer of size where to write the latin1 words. |
370 | 15.9k | size_t expected_latin1words = |
371 | 15.9k | e->latin1_length_from_utf8(utf8_output.get(), utf8words); |
372 | 15.9k | std::unique_ptr<char[]> latin1_output{new char[expected_latin1words]}; |
373 | | // convert to latin1 |
374 | 15.9k | size_t latin1words = e->convert_utf8_to_latin1( |
375 | 15.9k | utf8_output.get(), utf8words, latin1_output.get()); |
376 | 388k | for (size_t i = 0; i < source.size(); i++) { |
377 | 372k | if (latin1_output.get()[i] != (source.c_str())[i]) { |
378 | 0 | print_input(source, e); |
379 | 0 | abort(); |
380 | 0 | } |
381 | 372k | } |
382 | 15.9k | } |
383 | 15.9k | if (validlatin1) { |
384 | | // We need a buffer of size where to write the UTF-16 words. |
385 | 15.9k | size_t expected_utf16words = e->utf16_length_from_latin1(source.size()); |
386 | 15.9k | std::unique_ptr<char16_t[]> utf16_output{ |
387 | 15.9k | new char16_t[expected_utf16words]}; |
388 | 15.9k | size_t utf16words = e->convert_latin1_to_utf16le( |
389 | 15.9k | source.c_str(), source.size(), utf16_output.get()); |
390 | | // It wrote utf16words * sizeof(char16_t) bytes. |
391 | 15.9k | bool validutf16 = e->validate_utf16le(utf16_output.get(), utf16words); |
392 | 15.9k | if (!validutf16) { |
393 | 0 | print_input(source, e); |
394 | 0 | abort(); |
395 | 0 | } |
396 | | // convert it back: |
397 | | // We need a buffer of size where to write the latin1 words. |
398 | 15.9k | size_t expected_latin1words = e->latin1_length_from_utf16(utf16words); |
399 | 15.9k | std::unique_ptr<char[]> latin1_output{new char[expected_latin1words]}; |
400 | | // convert to latin1 |
401 | 15.9k | size_t latin1words = e->convert_utf16le_to_latin1( |
402 | 15.9k | utf16_output.get(), utf16words, latin1_output.get()); |
403 | 388k | for (size_t i = 0; i < source.size(); i++) { |
404 | 372k | if (latin1_output.get()[i] != (source.c_str())[i]) { |
405 | 0 | print_input(source, e); |
406 | 0 | abort(); |
407 | 0 | } |
408 | 372k | } |
409 | 15.9k | } |
410 | 15.9k | if (validlatin1) { |
411 | | // We need a buffer of size where to write the UTF-16 words. |
412 | 15.9k | size_t expected_utf16words = e->utf16_length_from_latin1(source.size()); |
413 | 15.9k | std::unique_ptr<char16_t[]> utf16_output{ |
414 | 15.9k | new char16_t[expected_utf16words]}; |
415 | 15.9k | size_t utf16words = e->convert_latin1_to_utf16be( |
416 | 15.9k | source.c_str(), source.size(), utf16_output.get()); |
417 | | // It wrote utf16words * sizeof(char16_t) bytes. |
418 | 15.9k | bool validutf16 = e->validate_utf16be(utf16_output.get(), utf16words); |
419 | 15.9k | if (!validutf16) { |
420 | 0 | print_input(source, e); |
421 | 0 | abort(); |
422 | 0 | } |
423 | | // convert it back: |
424 | | // We need a buffer of size where to write the latin1 words. |
425 | 15.9k | size_t expected_latin1words = e->latin1_length_from_utf16(utf16words); |
426 | 15.9k | std::unique_ptr<char[]> latin1_output{new char[expected_latin1words]}; |
427 | | // convert to latin1 |
428 | 15.9k | size_t latin1words = e->convert_utf16be_to_latin1( |
429 | 15.9k | utf16_output.get(), utf16words, latin1_output.get()); |
430 | 388k | for (size_t i = 0; i < source.size(); i++) { |
431 | 372k | if (latin1_output.get()[i] != (source.c_str())[i]) { |
432 | 0 | print_input(source, e); |
433 | 0 | abort(); |
434 | 0 | } |
435 | 372k | } |
436 | 15.9k | } |
437 | | |
438 | 15.9k | if (validlatin1) { |
439 | | // We need a buffer of size where to write the UTF-16 words. |
440 | 15.9k | size_t expected_utf32words = e->utf32_length_from_latin1(source.size()); |
441 | 15.9k | std::unique_ptr<char32_t[]> utf32_output{ |
442 | 15.9k | new char32_t[expected_utf32words]}; |
443 | 15.9k | size_t utf32words = e->convert_latin1_to_utf32( |
444 | 15.9k | source.c_str(), source.size(), utf32_output.get()); |
445 | | // It wrote utf16words * sizeof(char16_t) bytes. |
446 | 15.9k | bool validutf32 = e->validate_utf32(utf32_output.get(), utf32words); |
447 | 15.9k | if (!validutf32) { |
448 | 0 | print_input(source, e); |
449 | 0 | abort(); |
450 | 0 | } |
451 | | // convert it back: |
452 | | // We need a buffer of size where to write the latin1 words. |
453 | 15.9k | size_t expected_latin1words = e->latin1_length_from_utf32(utf32words); |
454 | 15.9k | std::unique_ptr<char[]> latin1_output{new char[expected_latin1words]}; |
455 | | // convert to latin1 |
456 | 15.9k | size_t latin1words = e->convert_utf32_to_latin1( |
457 | 15.9k | utf32_output.get(), utf32words, latin1_output.get()); |
458 | 388k | for (size_t i = 0; i < source.size(); i++) { |
459 | 372k | if (latin1_output.get()[i] != (source.c_str())[i]) { |
460 | 0 | print_input(source, e); |
461 | 0 | abort(); |
462 | 0 | } |
463 | 372k | } |
464 | 15.9k | } |
465 | | |
466 | | /// Base64 tests. We begin by trying to decode the input, even if we |
467 | | /// expect it to fail. |
468 | 15.9k | { |
469 | 15.9k | size_t max_length_needed = |
470 | 15.9k | e->maximal_binary_length_from_base64(source.data(), source.size()); |
471 | 15.9k | std::vector<char> back(max_length_needed); |
472 | 15.9k | simdutf::result r = |
473 | 15.9k | e->base64_to_binary(source.data(), source.size(), back.data()); |
474 | 15.9k | if (r.error == simdutf::error_code::SUCCESS) { |
475 | | // We expect failure but if we succeed, then we should have a roundtrip. |
476 | 14.5k | back.resize(r.count); |
477 | 14.5k | std::vector<char> back2(e->base64_length_from_binary(back.size())); |
478 | 14.5k | size_t base64size = |
479 | 14.5k | e->binary_to_base64(back.data(), back.size(), back2.data()); |
480 | 14.5k | back2.resize(base64size); |
481 | 14.5k | std::vector<char> back3( |
482 | 14.5k | e->maximal_binary_length_from_base64(back2.data(), back2.size())); |
483 | 14.5k | simdutf::result r2 = |
484 | 14.5k | e->base64_to_binary(back2.data(), back2.size(), back3.data()); |
485 | 14.5k | if (r2.error != simdutf::error_code::SUCCESS) { |
486 | 0 | print_input(source, e); |
487 | 0 | return false; |
488 | 0 | } |
489 | 14.5k | if (r2.count != back.size()) { |
490 | 0 | print_input(source, e); |
491 | 0 | return false; |
492 | 0 | } |
493 | 14.5k | if (back3.size() != back.size()) { |
494 | 0 | print_input(source, e); |
495 | 0 | return false; |
496 | 0 | } |
497 | 14.5k | } |
498 | 15.9k | } |
499 | | |
500 | | // Same as above, but we use the safe decoder version. |
501 | 15.9k | { |
502 | 15.9k | size_t max_length_needed = |
503 | 15.9k | e->maximal_binary_length_from_base64(source.data(), source.size()); |
504 | 15.9k | std::vector<char> back(max_length_needed); |
505 | 15.9k | simdutf::result r = simdutf::base64_to_binary_safe( |
506 | 15.9k | source.data(), source.size(), back.data(), max_length_needed); |
507 | 15.9k | if (r.error == simdutf::error_code::SUCCESS) { |
508 | | // We expect failure but if we succeed, then we should have a roundtrip. |
509 | 14.5k | back.resize(max_length_needed); |
510 | 14.5k | std::vector<char> back2(e->base64_length_from_binary(back.size())); |
511 | 14.5k | size_t base64size = |
512 | 14.5k | e->binary_to_base64(back.data(), back.size(), back2.data()); |
513 | 14.5k | back2.resize(base64size); |
514 | 14.5k | size_t max_length_needed2 = |
515 | 14.5k | e->maximal_binary_length_from_base64(back2.data(), back2.size()); |
516 | 14.5k | std::vector<char> back3(max_length_needed2); |
517 | 14.5k | simdutf::result r2 = simdutf::base64_to_binary_safe( |
518 | 14.5k | back2.data(), back2.size(), back3.data(), max_length_needed2); |
519 | 14.5k | if (r2.error != simdutf::error_code::SUCCESS) { |
520 | 0 | print_input(source, e); |
521 | 0 | return false; |
522 | 0 | } |
523 | 14.5k | if (max_length_needed != back.size()) { |
524 | 0 | print_input(source, e); |
525 | 0 | return false; |
526 | 0 | } |
527 | 14.5k | if (back3.size() != back.size()) { |
528 | 0 | print_input(source, e); |
529 | 0 | return false; |
530 | 0 | } |
531 | 14.5k | } |
532 | 15.9k | } |
533 | | /// Base64 tests. We encode the content as binary in base64 and we decode |
534 | | /// it, it should always succeed. |
535 | 15.9k | { |
536 | 15.9k | source = fdp.ConsumeRandomLengthString(kMaxStringSize); |
537 | 15.9k | std::vector<char> base64buffer( |
538 | 15.9k | e->base64_length_from_binary(source.size())); |
539 | 15.9k | size_t base64size = e->binary_to_base64(source.data(), source.size(), |
540 | 15.9k | base64buffer.data()); |
541 | 15.9k | if (base64size != base64buffer.size()) { |
542 | 0 | print_input(source, e); |
543 | 0 | abort(); |
544 | 0 | } |
545 | 15.9k | std::vector<char> back(e->maximal_binary_length_from_base64( |
546 | 15.9k | base64buffer.data(), base64buffer.size())); |
547 | 15.9k | simdutf::result r = e->base64_to_binary(base64buffer.data(), |
548 | 15.9k | base64buffer.size(), back.data()); |
549 | 15.9k | if (r.error != simdutf::error_code::SUCCESS) { |
550 | 0 | print_input(source, e); |
551 | 0 | abort(); |
552 | 0 | } |
553 | 15.9k | if (r.count != source.size()) { |
554 | 0 | print_input(source, e); |
555 | 0 | abort(); |
556 | 0 | } |
557 | 345k | for (size_t i = 0; i < source.size(); i++) { |
558 | 330k | if (back[i] != (source.c_str())[i]) { |
559 | 0 | print_input(source, e); |
560 | 0 | abort(); |
561 | 0 | } |
562 | 330k | } |
563 | 15.9k | size_t max_length = back.size(); |
564 | 15.9k | r = simdutf::base64_to_binary_safe( |
565 | 15.9k | base64buffer.data(), base64buffer.size(), back.data(), max_length); |
566 | 15.9k | if (r.error != simdutf::error_code::SUCCESS) { |
567 | 0 | printf("base64 round trip failed, error code %d\n", r.error); |
568 | 0 | print_input(source, e); |
569 | 0 | return false; |
570 | 0 | } |
571 | 15.9k | if (max_length != source.size()) { |
572 | 0 | printf("base64 safe round trip failed, not the same size %zu %zu\n", |
573 | 0 | max_length, source.size()); |
574 | 0 | print_input(source, e); |
575 | 0 | return false; |
576 | 0 | } |
577 | 345k | for (size_t i = 0; i < source.size(); i++) { |
578 | 330k | if (back[i] != (source.c_str())[i]) { |
579 | 0 | printf("base64 round trip failed, same size, different content\n"); |
580 | 0 | print_input(source, e); |
581 | 0 | return false; |
582 | 0 | } |
583 | 330k | } |
584 | 15.9k | } |
585 | | |
586 | 15.9k | } // for (auto &e : simdutf::get_available_implementations()) { |
587 | | |
588 | 5.30k | return 0; |
589 | 5.30k | } // extern "C" int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) { |