/src/icu/icu4c/source/common/ustring.cpp
Line | Count | Source |
1 | | // © 2016 and later: Unicode, Inc. and others. |
2 | | // License & terms of use: http://www.unicode.org/copyright.html |
3 | | /* |
4 | | ****************************************************************************** |
5 | | * |
6 | | * Copyright (C) 1998-2016, International Business Machines |
7 | | * Corporation and others. All Rights Reserved. |
8 | | * |
9 | | ****************************************************************************** |
10 | | * |
11 | | * File ustring.cpp |
12 | | * |
13 | | * Modification History: |
14 | | * |
15 | | * Date Name Description |
16 | | * 12/07/98 bertrand Creation. |
17 | | ****************************************************************************** |
18 | | */ |
19 | | |
20 | | #include "unicode/utypes.h" |
21 | | #include "unicode/putil.h" |
22 | | #include "unicode/uchar.h" |
23 | | #include "unicode/ustring.h" |
24 | | #include "unicode/utf16.h" |
25 | | #include "cstring.h" |
26 | | #include "cwchar.h" |
27 | | #include "cmemory.h" |
28 | | #include "ustr_imp.h" |
29 | | |
30 | | /* ANSI string.h - style functions ------------------------------------------ */ |
31 | | |
32 | | /* U+ffff is the highest BMP code point, the highest one that fits into a 16-bit char16_t */ |
33 | 0 | #define U_BMP_MAX 0xffff |
34 | | |
35 | | /* Forward binary string search functions ----------------------------------- */ |
36 | | |
37 | | /* |
38 | | * Test if a substring match inside a string is at code point boundaries. |
39 | | * All pointers refer to the same buffer. |
40 | | * The limit pointer may be nullptr, all others must be real pointers. |
41 | | */ |
42 | | static inline UBool |
43 | 235k | isMatchAtCPBoundary(const char16_t *start, const char16_t *match, const char16_t *matchLimit, const char16_t *limit) { |
44 | 235k | if(U16_IS_TRAIL(*match) && start!=match && U16_IS_LEAD(*(match-1))) { |
45 | | /* the leading edge of the match is in the middle of a surrogate pair */ |
46 | 0 | return false; |
47 | 0 | } |
48 | 235k | if(U16_IS_LEAD(*(matchLimit-1)) && matchLimit!=limit && U16_IS_TRAIL(*matchLimit)) { |
49 | | /* the trailing edge of the match is in the middle of a surrogate pair */ |
50 | 0 | return false; |
51 | 0 | } |
52 | 235k | return true; |
53 | 235k | } |
54 | | |
55 | | U_CAPI char16_t* U_EXPORT2 |
56 | | u_strFindFirst(const char16_t* s U_LIFETIME_BOUND, |
57 | | int32_t length, |
58 | | const char16_t* sub, |
59 | 2.94M | int32_t subLength) { |
60 | 2.94M | const char16_t *start, *p, *q, *subLimit; |
61 | 2.94M | char16_t c, cs, cq; |
62 | | |
63 | 2.94M | if(sub==nullptr || subLength<-1) { |
64 | 0 | return (char16_t *)s; |
65 | 0 | } |
66 | 2.94M | if(s==nullptr || length<-1) { |
67 | 0 | return nullptr; |
68 | 0 | } |
69 | | |
70 | 2.94M | start=s; |
71 | | |
72 | 2.94M | if(length<0 && subLength<0) { |
73 | | /* both strings are NUL-terminated */ |
74 | 0 | if((cs=*sub++)==0) { |
75 | 0 | return (char16_t *)s; |
76 | 0 | } |
77 | 0 | if(*sub==0 && !U16_IS_SURROGATE(cs)) { |
78 | | /* the substring consists of a single, non-surrogate BMP code point */ |
79 | 0 | return u_strchr(s, cs); |
80 | 0 | } |
81 | | |
82 | 0 | while((c=*s++)!=0) { |
83 | 0 | if(c==cs) { |
84 | | /* found first substring char16_t, compare rest */ |
85 | 0 | p=s; |
86 | 0 | q=sub; |
87 | 0 | for(;;) { |
88 | 0 | if((cq=*q)==0) { |
89 | 0 | if(isMatchAtCPBoundary(start, s-1, p, nullptr)) { |
90 | 0 | return (char16_t *)(s-1); /* well-formed match */ |
91 | 0 | } else { |
92 | 0 | break; /* no match because surrogate pair is split */ |
93 | 0 | } |
94 | 0 | } |
95 | 0 | if((c=*p)==0) { |
96 | 0 | return nullptr; /* no match, and none possible after s */ |
97 | 0 | } |
98 | 0 | if(c!=cq) { |
99 | 0 | break; /* no match */ |
100 | 0 | } |
101 | 0 | ++p; |
102 | 0 | ++q; |
103 | 0 | } |
104 | 0 | } |
105 | 0 | } |
106 | | |
107 | | /* not found */ |
108 | 0 | return nullptr; |
109 | 0 | } |
110 | | |
111 | 2.94M | if(subLength<0) { |
112 | 1.62M | subLength=u_strlen(sub); |
113 | 1.62M | } |
114 | 2.94M | if(subLength==0) { |
115 | 0 | return (char16_t *)s; |
116 | 0 | } |
117 | | |
118 | | /* get sub[0] to search for it fast */ |
119 | 2.94M | cs=*sub++; |
120 | 2.94M | --subLength; |
121 | 2.94M | subLimit=sub+subLength; |
122 | | |
123 | 2.94M | if(subLength==0 && !U16_IS_SURROGATE(cs)) { |
124 | | /* the substring consists of a single, non-surrogate BMP code point */ |
125 | 106k | return length<0 ? u_strchr(s, cs) : u_memchr(s, cs, length); |
126 | 106k | } |
127 | | |
128 | 2.83M | if(length<0) { |
129 | | /* s is NUL-terminated */ |
130 | 0 | while((c=*s++)!=0) { |
131 | 0 | if(c==cs) { |
132 | | /* found first substring char16_t, compare rest */ |
133 | 0 | p=s; |
134 | 0 | q=sub; |
135 | 0 | for(;;) { |
136 | 0 | if(q==subLimit) { |
137 | 0 | if(isMatchAtCPBoundary(start, s-1, p, nullptr)) { |
138 | 0 | return (char16_t *)(s-1); /* well-formed match */ |
139 | 0 | } else { |
140 | 0 | break; /* no match because surrogate pair is split */ |
141 | 0 | } |
142 | 0 | } |
143 | 0 | if((c=*p)==0) { |
144 | 0 | return nullptr; /* no match, and none possible after s */ |
145 | 0 | } |
146 | 0 | if(c!=*q) { |
147 | 0 | break; /* no match */ |
148 | 0 | } |
149 | 0 | ++p; |
150 | 0 | ++q; |
151 | 0 | } |
152 | 0 | } |
153 | 0 | } |
154 | 2.83M | } else { |
155 | 2.83M | const char16_t *limit, *preLimit; |
156 | | |
157 | | /* subLength was decremented above */ |
158 | 2.83M | if(length<=subLength) { |
159 | 949k | return nullptr; /* s is shorter than sub */ |
160 | 949k | } |
161 | | |
162 | 1.88M | limit=s+length; |
163 | | |
164 | | /* the substring must start before preLimit */ |
165 | 1.88M | preLimit=limit-subLength; |
166 | | |
167 | 59.9M | while(s!=preLimit) { |
168 | 58.2M | c=*s++; |
169 | 58.2M | if(c==cs) { |
170 | | /* found first substring char16_t, compare rest */ |
171 | 2.77M | p=s; |
172 | 2.77M | q=sub; |
173 | 3.12M | for(;;) { |
174 | 3.12M | if(q==subLimit) { |
175 | 218k | if(isMatchAtCPBoundary(start, s-1, p, limit)) { |
176 | 218k | return (char16_t *)(s-1); /* well-formed match */ |
177 | 218k | } else { |
178 | 0 | break; /* no match because surrogate pair is split */ |
179 | 0 | } |
180 | 218k | } |
181 | 2.90M | if(*p!=*q) { |
182 | 2.55M | break; /* no match */ |
183 | 2.55M | } |
184 | 349k | ++p; |
185 | 349k | ++q; |
186 | 349k | } |
187 | 2.77M | } |
188 | 58.2M | } |
189 | 1.88M | } |
190 | | |
191 | | /* not found */ |
192 | 1.66M | return nullptr; |
193 | 2.83M | } |
194 | | |
195 | | U_CAPI char16_t* U_EXPORT2 |
196 | 0 | u_strstr(const char16_t* s U_LIFETIME_BOUND, const char16_t* substring) { |
197 | 0 | return u_strFindFirst(s, -1, substring, -1); |
198 | 0 | } |
199 | | |
200 | | U_CAPI char16_t* U_EXPORT2 |
201 | 77.3k | u_strchr(const char16_t* s U_LIFETIME_BOUND, char16_t c) { |
202 | 77.3k | if(U16_IS_SURROGATE(c)) { |
203 | | /* make sure to not find half of a surrogate pair */ |
204 | 0 | return u_strFindFirst(s, -1, &c, 1); |
205 | 77.3k | } else { |
206 | 77.3k | char16_t cs; |
207 | | |
208 | | /* trivial search for a BMP code point */ |
209 | 942k | for(;;) { |
210 | 942k | if((cs=*s)==c) { |
211 | 37.2k | return (char16_t *)s; |
212 | 37.2k | } |
213 | 904k | if(cs==0) { |
214 | 40.0k | return nullptr; |
215 | 40.0k | } |
216 | 864k | ++s; |
217 | 864k | } |
218 | 77.3k | } |
219 | 77.3k | } |
220 | | |
221 | | U_CAPI char16_t* U_EXPORT2 |
222 | 0 | u_strchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c) { |
223 | 0 | if((uint32_t)c<=U_BMP_MAX) { |
224 | | /* find BMP code point */ |
225 | 0 | return u_strchr(s, (char16_t)c); |
226 | 0 | } else if((uint32_t)c<=UCHAR_MAX_VALUE) { |
227 | | /* find supplementary code point as surrogate pair */ |
228 | 0 | char16_t cs, lead=U16_LEAD(c), trail=U16_TRAIL(c); |
229 | |
|
230 | 0 | while((cs=*s++)!=0) { |
231 | 0 | if(cs==lead && *s==trail) { |
232 | 0 | return (char16_t *)(s-1); |
233 | 0 | } |
234 | 0 | } |
235 | 0 | return nullptr; |
236 | 0 | } else { |
237 | | /* not a Unicode code point, not findable */ |
238 | 0 | return nullptr; |
239 | 0 | } |
240 | 0 | } |
241 | | |
242 | | U_CAPI char16_t* U_EXPORT2 |
243 | 32.7M | u_memchr(const char16_t* s U_LIFETIME_BOUND, char16_t c, int32_t count) { |
244 | 32.7M | if(count<=0) { |
245 | 626k | return nullptr; /* no string */ |
246 | 32.0M | } else if(U16_IS_SURROGATE(c)) { |
247 | | /* make sure to not find half of a surrogate pair */ |
248 | 0 | return u_strFindFirst(s, count, &c, 1); |
249 | 32.0M | } else { |
250 | | /* trivial search for a BMP code point */ |
251 | 32.0M | const char16_t *limit=s+count; |
252 | 159M | do { |
253 | 159M | if(*s==c) { |
254 | 8.40M | return (char16_t *)s; |
255 | 8.40M | } |
256 | 159M | } while(++s!=limit); |
257 | 23.6M | return nullptr; |
258 | 32.0M | } |
259 | 32.7M | } |
260 | | |
261 | | U_CAPI char16_t* U_EXPORT2 |
262 | 0 | u_memchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c, int32_t count) { |
263 | 0 | if((uint32_t)c<=U_BMP_MAX) { |
264 | | /* find BMP code point */ |
265 | 0 | return u_memchr(s, (char16_t)c, count); |
266 | 0 | } else if(count<2) { |
267 | | /* too short for a surrogate pair */ |
268 | 0 | return nullptr; |
269 | 0 | } else if((uint32_t)c<=UCHAR_MAX_VALUE) { |
270 | | /* find supplementary code point as surrogate pair */ |
271 | 0 | const char16_t *limit=s+count-1; /* -1 so that we do not need a separate check for the trail unit */ |
272 | 0 | char16_t lead=U16_LEAD(c), trail=U16_TRAIL(c); |
273 | |
|
274 | 0 | do { |
275 | 0 | if(*s==lead && *(s+1)==trail) { |
276 | 0 | return (char16_t *)s; |
277 | 0 | } |
278 | 0 | } while(++s!=limit); |
279 | 0 | return nullptr; |
280 | 0 | } else { |
281 | | /* not a Unicode code point, not findable */ |
282 | 0 | return nullptr; |
283 | 0 | } |
284 | 0 | } |
285 | | |
286 | | /* Backward binary string search functions ---------------------------------- */ |
287 | | |
288 | | U_CAPI char16_t* U_EXPORT2 |
289 | | u_strFindLast(const char16_t* s U_LIFETIME_BOUND, |
290 | | int32_t length, |
291 | | const char16_t* sub, |
292 | 16.2k | int32_t subLength) { |
293 | 16.2k | const char16_t *start, *limit, *p, *q, *subLimit; |
294 | 16.2k | char16_t c, cs; |
295 | | |
296 | 16.2k | if(sub==nullptr || subLength<-1) { |
297 | 0 | return (char16_t *)s; |
298 | 0 | } |
299 | 16.2k | if(s==nullptr || length<-1) { |
300 | 0 | return nullptr; |
301 | 0 | } |
302 | | |
303 | | /* |
304 | | * This implementation is more lazy than the one for u_strFindFirst(): |
305 | | * There is no special search code for NUL-terminated strings. |
306 | | * It does not seem to be worth it for searching substrings to |
307 | | * search forward and find all matches like in u_strrchr() and similar. |
308 | | * Therefore, we simply get both string lengths and search backward. |
309 | | * |
310 | | * markus 2002oct23 |
311 | | */ |
312 | | |
313 | 16.2k | if(subLength<0) { |
314 | 0 | subLength=u_strlen(sub); |
315 | 0 | } |
316 | 16.2k | if(subLength==0) { |
317 | 0 | return (char16_t *)s; |
318 | 0 | } |
319 | | |
320 | | /* get sub[subLength-1] to search for it fast */ |
321 | 16.2k | subLimit=sub+subLength; |
322 | 16.2k | cs=*(--subLimit); |
323 | 16.2k | --subLength; |
324 | | |
325 | 16.2k | if(subLength==0 && !U16_IS_SURROGATE(cs)) { |
326 | | /* the substring consists of a single, non-surrogate BMP code point */ |
327 | 0 | return length<0 ? u_strrchr(s, cs) : u_memrchr(s, cs, length); |
328 | 0 | } |
329 | | |
330 | 16.2k | if(length<0) { |
331 | 0 | length=u_strlen(s); |
332 | 0 | } |
333 | | |
334 | | /* subLength was decremented above */ |
335 | 16.2k | if(length<=subLength) { |
336 | 0 | return nullptr; /* s is shorter than sub */ |
337 | 0 | } |
338 | | |
339 | 16.2k | start=s; |
340 | 16.2k | limit=s+length; |
341 | | |
342 | | /* the substring must start no later than s+subLength */ |
343 | 16.2k | s+=subLength; |
344 | | |
345 | 32.5k | while(s!=limit) { |
346 | 32.4k | c=*(--limit); |
347 | 32.4k | if(c==cs) { |
348 | | /* found last substring char16_t, compare rest */ |
349 | 16.2k | p=limit; |
350 | 16.2k | q=subLimit; |
351 | 32.4k | for(;;) { |
352 | 32.4k | if(q==sub) { |
353 | 16.2k | if(isMatchAtCPBoundary(start, p, limit+1, start+length)) { |
354 | 16.2k | return (char16_t *)p; /* well-formed match */ |
355 | 16.2k | } else { |
356 | 0 | break; /* no match because surrogate pair is split */ |
357 | 0 | } |
358 | 16.2k | } |
359 | 16.2k | if(*(--p)!=*(--q)) { |
360 | 42 | break; /* no match */ |
361 | 42 | } |
362 | 16.2k | } |
363 | 16.2k | } |
364 | 32.4k | } |
365 | | |
366 | | /* not found */ |
367 | 42 | return nullptr; |
368 | 16.2k | } |
369 | | |
370 | | U_CAPI char16_t* U_EXPORT2 |
371 | 0 | u_strrstr(const char16_t* s U_LIFETIME_BOUND, const char16_t* substring) { |
372 | 0 | return u_strFindLast(s, -1, substring, -1); |
373 | 0 | } |
374 | | |
375 | | U_CAPI char16_t* U_EXPORT2 |
376 | 0 | u_strrchr(const char16_t* s U_LIFETIME_BOUND, char16_t c) { |
377 | 0 | if(U16_IS_SURROGATE(c)) { |
378 | | /* make sure to not find half of a surrogate pair */ |
379 | 0 | return u_strFindLast(s, -1, &c, 1); |
380 | 0 | } else { |
381 | 0 | const char16_t *result=nullptr; |
382 | 0 | char16_t cs; |
383 | | |
384 | | /* trivial search for a BMP code point */ |
385 | 0 | for(;;) { |
386 | 0 | if((cs=*s)==c) { |
387 | 0 | result=s; |
388 | 0 | } |
389 | 0 | if(cs==0) { |
390 | 0 | return (char16_t *)result; |
391 | 0 | } |
392 | 0 | ++s; |
393 | 0 | } |
394 | 0 | } |
395 | 0 | } |
396 | | |
397 | | U_CAPI char16_t* U_EXPORT2 |
398 | 0 | u_strrchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c) { |
399 | 0 | if((uint32_t)c<=U_BMP_MAX) { |
400 | | /* find BMP code point */ |
401 | 0 | return u_strrchr(s, (char16_t)c); |
402 | 0 | } else if((uint32_t)c<=UCHAR_MAX_VALUE) { |
403 | | /* find supplementary code point as surrogate pair */ |
404 | 0 | const char16_t *result=nullptr; |
405 | 0 | char16_t cs, lead=U16_LEAD(c), trail=U16_TRAIL(c); |
406 | |
|
407 | 0 | while((cs=*s++)!=0) { |
408 | 0 | if(cs==lead && *s==trail) { |
409 | 0 | result=s-1; |
410 | 0 | } |
411 | 0 | } |
412 | 0 | return (char16_t *)result; |
413 | 0 | } else { |
414 | | /* not a Unicode code point, not findable */ |
415 | 0 | return nullptr; |
416 | 0 | } |
417 | 0 | } |
418 | | |
419 | | U_CAPI char16_t* U_EXPORT2 |
420 | 127k | u_memrchr(const char16_t* s U_LIFETIME_BOUND, char16_t c, int32_t count) { |
421 | 127k | if(count<=0) { |
422 | 0 | return nullptr; /* no string */ |
423 | 127k | } else if(U16_IS_SURROGATE(c)) { |
424 | | /* make sure to not find half of a surrogate pair */ |
425 | 0 | return u_strFindLast(s, count, &c, 1); |
426 | 127k | } else { |
427 | | /* trivial search for a BMP code point */ |
428 | 127k | const char16_t *limit=s+count; |
429 | 922k | do { |
430 | 922k | if(*(--limit)==c) { |
431 | 110k | return (char16_t *)limit; |
432 | 110k | } |
433 | 922k | } while(s!=limit); |
434 | 16.9k | return nullptr; |
435 | 127k | } |
436 | 127k | } |
437 | | |
438 | | U_CAPI char16_t* U_EXPORT2 |
439 | 0 | u_memrchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c, int32_t count) { |
440 | 0 | if((uint32_t)c<=U_BMP_MAX) { |
441 | | /* find BMP code point */ |
442 | 0 | return u_memrchr(s, (char16_t)c, count); |
443 | 0 | } else if(count<2) { |
444 | | /* too short for a surrogate pair */ |
445 | 0 | return nullptr; |
446 | 0 | } else if((uint32_t)c<=UCHAR_MAX_VALUE) { |
447 | | /* find supplementary code point as surrogate pair */ |
448 | 0 | const char16_t *limit=s+count-1; |
449 | 0 | char16_t lead=U16_LEAD(c), trail=U16_TRAIL(c); |
450 | |
|
451 | 0 | do { |
452 | 0 | if(*limit==trail && *(limit-1)==lead) { |
453 | 0 | return (char16_t *)(limit-1); |
454 | 0 | } |
455 | 0 | } while(s!=--limit); |
456 | 0 | return nullptr; |
457 | 0 | } else { |
458 | | /* not a Unicode code point, not findable */ |
459 | 0 | return nullptr; |
460 | 0 | } |
461 | 0 | } |
462 | | |
463 | | /* Tokenization functions --------------------------------------------------- */ |
464 | | |
465 | | /* |
466 | | * Match each code point in a string against each code point in the matchSet. |
467 | | * Return the index of the first string code point that |
468 | | * is (polarity==true) or is not (false) contained in the matchSet. |
469 | | * Return -(string length)-1 if there is no such code point. |
470 | | */ |
471 | | static int32_t |
472 | 0 | _matchFromSet(const char16_t *string, const char16_t *matchSet, UBool polarity) { |
473 | 0 | int32_t matchLen, matchBMPLen, strItr, matchItr; |
474 | 0 | UChar32 stringCh, matchCh; |
475 | 0 | char16_t c, c2; |
476 | | |
477 | | /* first part of matchSet contains only BMP code points */ |
478 | 0 | matchBMPLen = 0; |
479 | 0 | while((c = matchSet[matchBMPLen]) != 0 && U16_IS_SINGLE(c)) { |
480 | 0 | ++matchBMPLen; |
481 | 0 | } |
482 | | |
483 | | /* second part of matchSet contains BMP and supplementary code points */ |
484 | 0 | matchLen = matchBMPLen; |
485 | 0 | while(matchSet[matchLen] != 0) { |
486 | 0 | ++matchLen; |
487 | 0 | } |
488 | |
|
489 | 0 | for(strItr = 0; (c = string[strItr]) != 0;) { |
490 | 0 | ++strItr; |
491 | 0 | if(U16_IS_SINGLE(c)) { |
492 | 0 | if(polarity) { |
493 | 0 | for(matchItr = 0; matchItr < matchLen; ++matchItr) { |
494 | 0 | if(c == matchSet[matchItr]) { |
495 | 0 | return strItr - 1; /* one matches */ |
496 | 0 | } |
497 | 0 | } |
498 | 0 | } else { |
499 | 0 | for(matchItr = 0; matchItr < matchLen; ++matchItr) { |
500 | 0 | if(c == matchSet[matchItr]) { |
501 | 0 | goto endloop; |
502 | 0 | } |
503 | 0 | } |
504 | 0 | return strItr - 1; /* none matches */ |
505 | 0 | } |
506 | 0 | } else { |
507 | | /* |
508 | | * No need to check for string length before U16_IS_TRAIL |
509 | | * because c2 could at worst be the terminating NUL. |
510 | | */ |
511 | 0 | if(U16_IS_SURROGATE_LEAD(c) && U16_IS_TRAIL(c2 = string[strItr])) { |
512 | 0 | ++strItr; |
513 | 0 | stringCh = U16_GET_SUPPLEMENTARY(c, c2); |
514 | 0 | } else { |
515 | 0 | stringCh = c; /* unpaired trail surrogate */ |
516 | 0 | } |
517 | |
|
518 | 0 | if(polarity) { |
519 | 0 | for(matchItr = matchBMPLen; matchItr < matchLen;) { |
520 | 0 | U16_NEXT(matchSet, matchItr, matchLen, matchCh); |
521 | 0 | if(stringCh == matchCh) { |
522 | 0 | return strItr - U16_LENGTH(stringCh); /* one matches */ |
523 | 0 | } |
524 | 0 | } |
525 | 0 | } else { |
526 | 0 | for(matchItr = matchBMPLen; matchItr < matchLen;) { |
527 | 0 | U16_NEXT(matchSet, matchItr, matchLen, matchCh); |
528 | 0 | if(stringCh == matchCh) { |
529 | 0 | goto endloop; |
530 | 0 | } |
531 | 0 | } |
532 | 0 | return strItr - U16_LENGTH(stringCh); /* none matches */ |
533 | 0 | } |
534 | 0 | } |
535 | 0 | endloop: |
536 | 0 | /* wish C had continue with labels like Java... */; |
537 | 0 | } |
538 | | |
539 | | /* Didn't find it. */ |
540 | 0 | return -strItr-1; |
541 | 0 | } |
542 | | |
543 | | /* Search for a codepoint in a string that matches one of the matchSet codepoints. */ |
544 | | U_CAPI char16_t* U_EXPORT2 |
545 | | u_strpbrk(const char16_t* string U_LIFETIME_BOUND, const char16_t* matchSet) |
546 | 0 | { |
547 | 0 | int32_t idx = _matchFromSet(string, matchSet, true); |
548 | 0 | if(idx >= 0) { |
549 | 0 | return (char16_t *)string + idx; |
550 | 0 | } else { |
551 | 0 | return nullptr; |
552 | 0 | } |
553 | 0 | } |
554 | | |
555 | | /* Search for a codepoint in a string that matches one of the matchSet codepoints. */ |
556 | | U_CAPI int32_t U_EXPORT2 |
557 | | u_strcspn(const char16_t *string, const char16_t *matchSet) |
558 | 0 | { |
559 | 0 | int32_t idx = _matchFromSet(string, matchSet, true); |
560 | 0 | if(idx >= 0) { |
561 | 0 | return idx; |
562 | 0 | } else { |
563 | 0 | return -idx - 1; /* == u_strlen(string) */ |
564 | 0 | } |
565 | 0 | } |
566 | | |
567 | | /* Search for a codepoint in a string that does not match one of the matchSet codepoints. */ |
568 | | U_CAPI int32_t U_EXPORT2 |
569 | | u_strspn(const char16_t *string, const char16_t *matchSet) |
570 | 0 | { |
571 | 0 | int32_t idx = _matchFromSet(string, matchSet, false); |
572 | 0 | if(idx >= 0) { |
573 | 0 | return idx; |
574 | 0 | } else { |
575 | 0 | return -idx - 1; /* == u_strlen(string) */ |
576 | 0 | } |
577 | 0 | } |
578 | | |
579 | | /* ----- Text manipulation functions --- */ |
580 | | |
581 | | U_CAPI char16_t* U_EXPORT2 |
582 | | u_strtok_r(char16_t* src U_LIFETIME_BOUND, const char16_t* delim, char16_t** saveState) |
583 | 0 | { |
584 | 0 | char16_t *tokSource; |
585 | 0 | char16_t *nextToken; |
586 | 0 | uint32_t nonDelimIdx; |
587 | | |
588 | | /* If saveState is nullptr, the user messed up. */ |
589 | 0 | if (src != nullptr) { |
590 | 0 | tokSource = src; |
591 | 0 | *saveState = src; /* Set to "src" in case there are no delimiters */ |
592 | 0 | } |
593 | 0 | else if (*saveState) { |
594 | 0 | tokSource = *saveState; |
595 | 0 | } |
596 | 0 | else { |
597 | | /* src == nullptr && *saveState == nullptr */ |
598 | | /* This shouldn't happen. We already finished tokenizing. */ |
599 | 0 | return nullptr; |
600 | 0 | } |
601 | | |
602 | | /* Skip initial delimiters */ |
603 | 0 | nonDelimIdx = u_strspn(tokSource, delim); |
604 | 0 | tokSource = &tokSource[nonDelimIdx]; |
605 | |
|
606 | 0 | if (*tokSource) { |
607 | 0 | nextToken = u_strpbrk(tokSource, delim); |
608 | 0 | if (nextToken != nullptr) { |
609 | | /* Create a token */ |
610 | 0 | *(nextToken++) = 0; |
611 | 0 | *saveState = nextToken; |
612 | 0 | return tokSource; |
613 | 0 | } |
614 | 0 | else if (*saveState) { |
615 | | /* Return the last token */ |
616 | 0 | *saveState = nullptr; |
617 | 0 | return tokSource; |
618 | 0 | } |
619 | 0 | } |
620 | 0 | else { |
621 | | /* No tokens were found. Only delimiters were left. */ |
622 | 0 | *saveState = nullptr; |
623 | 0 | } |
624 | 0 | return nullptr; |
625 | 0 | } |
626 | | |
627 | | /* Miscellaneous functions -------------------------------------------------- */ |
628 | | |
629 | | U_CAPI char16_t* U_EXPORT2 |
630 | | u_strcat(char16_t* dst U_LIFETIME_BOUND, const char16_t* src) |
631 | 0 | { |
632 | 0 | char16_t *anchor = dst; /* save a pointer to start of dst */ |
633 | |
|
634 | 0 | while(*dst != 0) { /* To end of first string */ |
635 | 0 | ++dst; |
636 | 0 | } |
637 | 0 | while((*(dst++) = *(src++)) != 0) { /* copy string 2 over */ |
638 | 0 | } |
639 | |
|
640 | 0 | return anchor; |
641 | 0 | } |
642 | | |
643 | | U_CAPI char16_t* U_EXPORT2 |
644 | | u_strncat(char16_t* dst U_LIFETIME_BOUND, const char16_t* src, int32_t n) |
645 | 0 | { |
646 | 0 | if(n > 0) { |
647 | 0 | char16_t *anchor = dst; /* save a pointer to start of dst */ |
648 | |
|
649 | 0 | while(*dst != 0) { /* To end of first string */ |
650 | 0 | ++dst; |
651 | 0 | } |
652 | 0 | while((*dst = *src) != 0) { /* copy string 2 over */ |
653 | 0 | ++dst; |
654 | 0 | if(--n == 0) { |
655 | 0 | *dst = 0; |
656 | 0 | break; |
657 | 0 | } |
658 | 0 | ++src; |
659 | 0 | } |
660 | |
|
661 | 0 | return anchor; |
662 | 0 | } else { |
663 | 0 | return dst; |
664 | 0 | } |
665 | 0 | } |
666 | | |
667 | | /* ----- Text property functions --- */ |
668 | | |
669 | | U_CAPI int32_t U_EXPORT2 |
670 | | u_strcmp(const char16_t *s1, |
671 | | const char16_t *s2) |
672 | 131k | { |
673 | 131k | char16_t c1, c2; |
674 | | |
675 | 196k | for(;;) { |
676 | 196k | c1=*s1++; |
677 | 196k | c2=*s2++; |
678 | 196k | if (c1 != c2 || c1 == 0) { |
679 | 131k | break; |
680 | 131k | } |
681 | 196k | } |
682 | 131k | return (int32_t)c1 - (int32_t)c2; |
683 | 131k | } |
684 | | |
685 | | U_CFUNC int32_t U_EXPORT2 |
686 | | uprv_strCompare(const char16_t *s1, int32_t length1, |
687 | | const char16_t *s2, int32_t length2, |
688 | 5.66k | UBool strncmpStyle, UBool codePointOrder) { |
689 | 5.66k | const char16_t *start1, *start2, *limit1, *limit2; |
690 | 5.66k | char16_t c1, c2; |
691 | | |
692 | | /* setup for fix-up */ |
693 | 5.66k | start1=s1; |
694 | 5.66k | start2=s2; |
695 | | |
696 | | /* compare identical prefixes - they do not need to be fixed up */ |
697 | 5.66k | if(length1<0 && length2<0) { |
698 | | /* strcmp style, both NUL-terminated */ |
699 | 0 | if(s1==s2) { |
700 | 0 | return 0; |
701 | 0 | } |
702 | | |
703 | 0 | for(;;) { |
704 | 0 | c1=*s1; |
705 | 0 | c2=*s2; |
706 | 0 | if(c1!=c2) { |
707 | 0 | break; |
708 | 0 | } |
709 | 0 | if(c1==0) { |
710 | 0 | return 0; |
711 | 0 | } |
712 | 0 | ++s1; |
713 | 0 | ++s2; |
714 | 0 | } |
715 | | |
716 | | /* setup for fix-up */ |
717 | 0 | limit1=limit2=nullptr; |
718 | 5.66k | } else if(strncmpStyle) { |
719 | | /* special handling for strncmp, assume length1==length2>=0 but also check for NUL */ |
720 | 0 | if(s1==s2) { |
721 | 0 | return 0; |
722 | 0 | } |
723 | | |
724 | 0 | limit1=start1+length1; |
725 | |
|
726 | 0 | for(;;) { |
727 | | /* both lengths are same, check only one limit */ |
728 | 0 | if(s1==limit1) { |
729 | 0 | return 0; |
730 | 0 | } |
731 | | |
732 | 0 | c1=*s1; |
733 | 0 | c2=*s2; |
734 | 0 | if(c1!=c2) { |
735 | 0 | break; |
736 | 0 | } |
737 | 0 | if(c1==0) { |
738 | 0 | return 0; |
739 | 0 | } |
740 | 0 | ++s1; |
741 | 0 | ++s2; |
742 | 0 | } |
743 | | |
744 | | /* setup for fix-up */ |
745 | 0 | limit2=start2+length1; /* use length1 here, too, to enforce assumption */ |
746 | 5.66k | } else { |
747 | | /* memcmp/UnicodeString style, both length-specified */ |
748 | 5.66k | int32_t lengthResult; |
749 | | |
750 | 5.66k | if(length1<0) { |
751 | 0 | length1=u_strlen(s1); |
752 | 0 | } |
753 | 5.66k | if(length2<0) { |
754 | 0 | length2=u_strlen(s2); |
755 | 0 | } |
756 | | |
757 | | /* limit1=start1+min(length1, length2) */ |
758 | 5.66k | if(length1<length2) { |
759 | 0 | lengthResult=-1; |
760 | 0 | limit1=start1+length1; |
761 | 5.66k | } else if(length1==length2) { |
762 | 5.66k | lengthResult=0; |
763 | 5.66k | limit1=start1+length1; |
764 | 5.66k | } else /* length1>length2 */ { |
765 | 0 | lengthResult=1; |
766 | 0 | limit1=start1+length2; |
767 | 0 | } |
768 | | |
769 | 5.66k | if(s1==s2) { |
770 | 0 | return lengthResult; |
771 | 0 | } |
772 | | |
773 | 16.6k | for(;;) { |
774 | | /* check pseudo-limit */ |
775 | 16.6k | if(s1==limit1) { |
776 | 4.38k | return lengthResult; |
777 | 4.38k | } |
778 | | |
779 | 12.2k | c1=*s1; |
780 | 12.2k | c2=*s2; |
781 | 12.2k | if(c1!=c2) { |
782 | 1.28k | break; |
783 | 1.28k | } |
784 | 10.9k | ++s1; |
785 | 10.9k | ++s2; |
786 | 10.9k | } |
787 | | |
788 | | /* setup for fix-up */ |
789 | 1.28k | limit1=start1+length1; |
790 | 1.28k | limit2=start2+length2; |
791 | 1.28k | } |
792 | | |
793 | | /* if both values are in or above the surrogate range, fix them up */ |
794 | 1.28k | if(c1>=0xd800 && c2>=0xd800 && codePointOrder) { |
795 | | /* subtract 0x2800 from BMP code points to make them smaller than supplementary ones */ |
796 | 0 | if( |
797 | 0 | (c1<=0xdbff && (s1+1)!=limit1 && U16_IS_TRAIL(*(s1+1))) || |
798 | 0 | (U16_IS_TRAIL(c1) && start1!=s1 && U16_IS_LEAD(*(s1-1))) |
799 | 0 | ) { |
800 | | /* part of a surrogate pair, leave >=d800 */ |
801 | 0 | } else { |
802 | | /* BMP code point - may be surrogate code point - make <d800 */ |
803 | 0 | c1-=0x2800; |
804 | 0 | } |
805 | |
|
806 | 0 | if( |
807 | 0 | (c2<=0xdbff && (s2+1)!=limit2 && U16_IS_TRAIL(*(s2+1))) || |
808 | 0 | (U16_IS_TRAIL(c2) && start2!=s2 && U16_IS_LEAD(*(s2-1))) |
809 | 0 | ) { |
810 | | /* part of a surrogate pair, leave >=d800 */ |
811 | 0 | } else { |
812 | | /* BMP code point - may be surrogate code point - make <d800 */ |
813 | 0 | c2-=0x2800; |
814 | 0 | } |
815 | 0 | } |
816 | | |
817 | | /* now c1 and c2 are in the requested (code unit or code point) order */ |
818 | 1.28k | return (int32_t)c1-(int32_t)c2; |
819 | 5.66k | } |
820 | | |
821 | | /* |
822 | | * Compare two strings as presented by UCharIterators. |
823 | | * Use code unit or code point order. |
824 | | * When the function returns, it is undefined where the iterators |
825 | | * have stopped. |
826 | | */ |
827 | | U_CAPI int32_t U_EXPORT2 |
828 | 0 | u_strCompareIter(UCharIterator *iter1, UCharIterator *iter2, UBool codePointOrder) { |
829 | 0 | UChar32 c1, c2; |
830 | | |
831 | | /* argument checking */ |
832 | 0 | if(iter1==nullptr || iter2==nullptr) { |
833 | 0 | return 0; /* bad arguments */ |
834 | 0 | } |
835 | 0 | if(iter1==iter2) { |
836 | 0 | return 0; /* identical iterators */ |
837 | 0 | } |
838 | | |
839 | | /* reset iterators to start? */ |
840 | 0 | iter1->move(iter1, 0, UITER_START); |
841 | 0 | iter2->move(iter2, 0, UITER_START); |
842 | | |
843 | | /* compare identical prefixes - they do not need to be fixed up */ |
844 | 0 | for(;;) { |
845 | 0 | c1=iter1->next(iter1); |
846 | 0 | c2=iter2->next(iter2); |
847 | 0 | if(c1!=c2) { |
848 | 0 | break; |
849 | 0 | } |
850 | 0 | if(c1==-1) { |
851 | 0 | return 0; |
852 | 0 | } |
853 | 0 | } |
854 | | |
855 | | /* if both values are in or above the surrogate range, fix them up */ |
856 | 0 | if(c1>=0xd800 && c2>=0xd800 && codePointOrder) { |
857 | | /* subtract 0x2800 from BMP code points to make them smaller than supplementary ones */ |
858 | 0 | if( |
859 | 0 | (c1<=0xdbff && U16_IS_TRAIL(iter1->current(iter1))) || |
860 | 0 | (U16_IS_TRAIL(c1) && (iter1->previous(iter1), U16_IS_LEAD(iter1->previous(iter1)))) |
861 | 0 | ) { |
862 | | /* part of a surrogate pair, leave >=d800 */ |
863 | 0 | } else { |
864 | | /* BMP code point - may be surrogate code point - make <d800 */ |
865 | 0 | c1-=0x2800; |
866 | 0 | } |
867 | |
|
868 | 0 | if( |
869 | 0 | (c2<=0xdbff && U16_IS_TRAIL(iter2->current(iter2))) || |
870 | 0 | (U16_IS_TRAIL(c2) && (iter2->previous(iter2), U16_IS_LEAD(iter2->previous(iter2)))) |
871 | 0 | ) { |
872 | | /* part of a surrogate pair, leave >=d800 */ |
873 | 0 | } else { |
874 | | /* BMP code point - may be surrogate code point - make <d800 */ |
875 | 0 | c2-=0x2800; |
876 | 0 | } |
877 | 0 | } |
878 | | |
879 | | /* now c1 and c2 are in the requested (code unit or code point) order */ |
880 | 0 | return (int32_t)c1-(int32_t)c2; |
881 | 0 | } |
882 | | |
883 | | #if 0 |
884 | | /* |
885 | | * u_strCompareIter() does not leave the iterators _on_ the different units. |
886 | | * This is possible but would cost a few extra indirect function calls to back |
887 | | * up if the last unit (c1 or c2 respectively) was >=0. |
888 | | * |
889 | | * Consistently leaving them _behind_ the different units is not an option |
890 | | * because the current "unit" is the end of the string if that is reached, |
891 | | * and in such a case the iterator does not move. |
892 | | * For example, when comparing "ab" with "abc", both iterators rest _on_ the end |
893 | | * of their strings. Calling previous() on each does not move them to where |
894 | | * the comparison fails. |
895 | | * |
896 | | * So the simplest semantics is to not define where the iterators end up. |
897 | | * |
898 | | * The following fragment is part of what would need to be done for backing up. |
899 | | */ |
900 | | void fragment { |
901 | | /* iff a surrogate is part of a surrogate pair, leave >=d800 */ |
902 | | if(c1<=0xdbff) { |
903 | | if(!U16_IS_TRAIL(iter1->current(iter1))) { |
904 | | /* lead surrogate code point - make <d800 */ |
905 | | c1-=0x2800; |
906 | | } |
907 | | } else if(c1<=0xdfff) { |
908 | | int32_t idx=iter1->getIndex(iter1, UITER_CURRENT); |
909 | | iter1->previous(iter1); /* ==c1 */ |
910 | | if(!U16_IS_LEAD(iter1->previous(iter1))) { |
911 | | /* trail surrogate code point - make <d800 */ |
912 | | c1-=0x2800; |
913 | | } |
914 | | /* go back to behind where the difference is */ |
915 | | iter1->move(iter1, idx, UITER_ZERO); |
916 | | } else /* 0xe000<=c1<=0xffff */ { |
917 | | /* BMP code point - make <d800 */ |
918 | | c1-=0x2800; |
919 | | } |
920 | | } |
921 | | #endif |
922 | | |
923 | | U_CAPI int32_t U_EXPORT2 |
924 | | u_strCompare(const char16_t *s1, int32_t length1, |
925 | | const char16_t *s2, int32_t length2, |
926 | 5.66k | UBool codePointOrder) { |
927 | | /* argument checking */ |
928 | 5.66k | if(s1==nullptr || length1<-1 || s2==nullptr || length2<-1) { |
929 | 0 | return 0; |
930 | 0 | } |
931 | 5.66k | return uprv_strCompare(s1, length1, s2, length2, false, codePointOrder); |
932 | 5.66k | } |
933 | | |
934 | | /* String compare in code point order - u_strcmp() compares in code unit order. */ |
935 | | U_CAPI int32_t U_EXPORT2 |
936 | 0 | u_strcmpCodePointOrder(const char16_t *s1, const char16_t *s2) { |
937 | 0 | return uprv_strCompare(s1, -1, s2, -1, false, true); |
938 | 0 | } |
939 | | |
940 | | U_CAPI int32_t U_EXPORT2 |
941 | | u_strncmp(const char16_t *s1, |
942 | | const char16_t *s2, |
943 | | int32_t n) |
944 | 7.31k | { |
945 | 7.31k | if(n > 0) { |
946 | 7.31k | int32_t rc; |
947 | 21.8k | for(;;) { |
948 | 21.8k | rc = (int32_t)*s1 - (int32_t)*s2; |
949 | 21.8k | if(rc != 0 || *s1 == 0 || --n == 0) { |
950 | 7.31k | return rc; |
951 | 7.31k | } |
952 | 14.5k | ++s1; |
953 | 14.5k | ++s2; |
954 | 14.5k | } |
955 | 7.31k | } else { |
956 | 0 | return 0; |
957 | 0 | } |
958 | 7.31k | } |
959 | | |
960 | | U_CAPI int32_t U_EXPORT2 |
961 | 0 | u_strncmpCodePointOrder(const char16_t *s1, const char16_t *s2, int32_t n) { |
962 | 0 | return uprv_strCompare(s1, n, s2, n, true, true); |
963 | 0 | } |
964 | | |
965 | | U_CAPI char16_t* U_EXPORT2 |
966 | | u_strcpy(char16_t* dst U_LIFETIME_BOUND, const char16_t* src) |
967 | 2.16M | { |
968 | 2.16M | char16_t *anchor = dst; /* save a pointer to start of dst */ |
969 | | |
970 | 8.66M | while((*(dst++) = *(src++)) != 0) { /* copy string 2 over */ |
971 | 6.49M | } |
972 | | |
973 | 2.16M | return anchor; |
974 | 2.16M | } |
975 | | |
976 | | U_CAPI char16_t* U_EXPORT2 |
977 | 570k | u_strncpy(char16_t* dst U_LIFETIME_BOUND, const char16_t* src, int32_t n) { |
978 | 570k | char16_t *anchor = dst; /* save a pointer to start of dst */ |
979 | | |
980 | | /* copy string 2 over */ |
981 | 2.08M | while(n > 0 && (*(dst++) = *(src++)) != 0) { |
982 | 1.50M | --n; |
983 | 1.50M | } |
984 | | |
985 | 570k | return anchor; |
986 | 570k | } |
987 | | |
988 | | U_CAPI int32_t U_EXPORT2 |
989 | | u_strlen(const char16_t *s) |
990 | 22.3M | { |
991 | | #if U_SIZEOF_WCHAR_T == U_SIZEOF_UCHAR |
992 | | return (int32_t)uprv_wcslen((const wchar_t *)s); |
993 | | #else |
994 | 22.3M | const char16_t *t = s; |
995 | 216M | while(*t != 0) { |
996 | 193M | ++t; |
997 | 193M | } |
998 | 22.3M | return t - s; |
999 | 22.3M | #endif |
1000 | 22.3M | } |
1001 | | |
1002 | | U_CAPI int32_t U_EXPORT2 |
1003 | 6.51M | u_countChar32(const char16_t *s, int32_t length) { |
1004 | 6.51M | int32_t count; |
1005 | | |
1006 | 6.51M | if(s==nullptr || length<-1) { |
1007 | 0 | return 0; |
1008 | 0 | } |
1009 | | |
1010 | 6.51M | count=0; |
1011 | 6.51M | if(length>=0) { |
1012 | 253M | while(length>0) { |
1013 | 246M | ++count; |
1014 | 246M | if(U16_IS_LEAD(*s) && length>=2 && U16_IS_TRAIL(*(s+1))) { |
1015 | 4.45M | s+=2; |
1016 | 4.45M | length-=2; |
1017 | 242M | } else { |
1018 | 242M | ++s; |
1019 | 242M | --length; |
1020 | 242M | } |
1021 | 246M | } |
1022 | 6.51M | } else /* length==-1 */ { |
1023 | 0 | char16_t c; |
1024 | |
|
1025 | 0 | for(;;) { |
1026 | 0 | if((c=*s++)==0) { |
1027 | 0 | break; |
1028 | 0 | } |
1029 | 0 | ++count; |
1030 | | |
1031 | | /* |
1032 | | * sufficient to look ahead one because of UTF-16; |
1033 | | * safe to look ahead one because at worst that would be the terminating NUL |
1034 | | */ |
1035 | 0 | if(U16_IS_LEAD(c) && U16_IS_TRAIL(*s)) { |
1036 | 0 | ++s; |
1037 | 0 | } |
1038 | 0 | } |
1039 | 0 | } |
1040 | 6.51M | return count; |
1041 | 6.51M | } |
1042 | | |
1043 | | U_CAPI UBool U_EXPORT2 |
1044 | 0 | u_strHasMoreChar32Than(const char16_t *s, int32_t length, int32_t number) { |
1045 | |
|
1046 | 0 | if(number<0) { |
1047 | 0 | return true; |
1048 | 0 | } |
1049 | 0 | if(s==nullptr || length<-1) { |
1050 | 0 | return false; |
1051 | 0 | } |
1052 | | |
1053 | 0 | if(length==-1) { |
1054 | | /* s is NUL-terminated */ |
1055 | 0 | char16_t c; |
1056 | | |
1057 | | /* count code points until they exceed */ |
1058 | 0 | for(;;) { |
1059 | 0 | if((c=*s++)==0) { |
1060 | 0 | return false; |
1061 | 0 | } |
1062 | 0 | if(number==0) { |
1063 | 0 | return true; |
1064 | 0 | } |
1065 | 0 | if(U16_IS_LEAD(c) && U16_IS_TRAIL(*s)) { |
1066 | 0 | ++s; |
1067 | 0 | } |
1068 | 0 | --number; |
1069 | 0 | } |
1070 | 0 | } else { |
1071 | | /* length>=0 known */ |
1072 | 0 | const char16_t *limit; |
1073 | 0 | int32_t maxSupplementary; |
1074 | | |
1075 | | /* s contains at least (length+1)/2 code points: <=2 UChars per cp */ |
1076 | 0 | if(((length+1)/2)>number) { |
1077 | 0 | return true; |
1078 | 0 | } |
1079 | | |
1080 | | /* check if s does not even contain enough UChars */ |
1081 | 0 | maxSupplementary=length-number; |
1082 | 0 | if(maxSupplementary<=0) { |
1083 | 0 | return false; |
1084 | 0 | } |
1085 | | /* there are maxSupplementary=length-number more UChars than asked-for code points */ |
1086 | | |
1087 | | /* |
1088 | | * count code points until they exceed and also check that there are |
1089 | | * no more than maxSupplementary supplementary code points (char16_t pairs) |
1090 | | */ |
1091 | 0 | limit=s+length; |
1092 | 0 | for(;;) { |
1093 | 0 | if(s==limit) { |
1094 | 0 | return false; |
1095 | 0 | } |
1096 | 0 | if(number==0) { |
1097 | 0 | return true; |
1098 | 0 | } |
1099 | 0 | if(U16_IS_LEAD(*s++) && s!=limit && U16_IS_TRAIL(*s)) { |
1100 | 0 | ++s; |
1101 | 0 | if(--maxSupplementary<=0) { |
1102 | | /* too many pairs - too few code points */ |
1103 | 0 | return false; |
1104 | 0 | } |
1105 | 0 | } |
1106 | 0 | --number; |
1107 | 0 | } |
1108 | 0 | } |
1109 | 0 | } |
1110 | | |
1111 | | U_CAPI char16_t* U_EXPORT2 |
1112 | 66.8M | u_memcpy(char16_t* dest U_LIFETIME_BOUND, const char16_t* src, int32_t count) { |
1113 | 66.8M | if(count > 0) { |
1114 | 66.8M | uprv_memcpy(dest, src, (size_t)count*U_SIZEOF_UCHAR); |
1115 | 66.8M | } |
1116 | 66.8M | return dest; |
1117 | 66.8M | } |
1118 | | |
1119 | | U_CAPI char16_t* U_EXPORT2 |
1120 | 0 | u_memmove(char16_t* dest U_LIFETIME_BOUND, const char16_t* src, int32_t count) { |
1121 | 0 | if(count > 0) { |
1122 | 0 | uprv_memmove(dest, src, (size_t)count*U_SIZEOF_UCHAR); |
1123 | 0 | } |
1124 | 0 | return dest; |
1125 | 0 | } |
1126 | | |
1127 | | U_CAPI char16_t* U_EXPORT2 |
1128 | 0 | u_memset(char16_t* dest U_LIFETIME_BOUND, char16_t c, int32_t count) { |
1129 | 0 | if(count > 0) { |
1130 | 0 | char16_t *ptr = dest; |
1131 | 0 | char16_t *limit = dest + count; |
1132 | |
|
1133 | 0 | while (ptr < limit) { |
1134 | 0 | *(ptr++) = c; |
1135 | 0 | } |
1136 | 0 | } |
1137 | 0 | return dest; |
1138 | 0 | } |
1139 | | |
1140 | | U_CAPI int32_t U_EXPORT2 |
1141 | 1.04M | u_memcmp(const char16_t *buf1, const char16_t *buf2, int32_t count) { |
1142 | 1.04M | if(count > 0) { |
1143 | 1.04M | const char16_t *limit = buf1 + count; |
1144 | 1.04M | int32_t result; |
1145 | | |
1146 | 3.36M | while (buf1 < limit) { |
1147 | 3.22M | result = (int32_t)(uint16_t)*buf1 - (int32_t)(uint16_t)*buf2; |
1148 | 3.22M | if (result != 0) { |
1149 | 900k | return result; |
1150 | 900k | } |
1151 | 2.32M | buf1++; |
1152 | 2.32M | buf2++; |
1153 | 2.32M | } |
1154 | 1.04M | } |
1155 | 143k | return 0; |
1156 | 1.04M | } |
1157 | | |
1158 | | U_CAPI int32_t U_EXPORT2 |
1159 | 0 | u_memcmpCodePointOrder(const char16_t *s1, const char16_t *s2, int32_t count) { |
1160 | 0 | return uprv_strCompare(s1, count, s2, count, false, true); |
1161 | 0 | } |
1162 | | |
1163 | | /* u_unescape & support fns ------------------------------------------------- */ |
1164 | | |
1165 | | /* This map must be in ASCENDING ORDER OF THE ESCAPE CODE */ |
1166 | | static const char16_t UNESCAPE_MAP[] = { |
1167 | | /*" 0x22, 0x22 */ |
1168 | | /*' 0x27, 0x27 */ |
1169 | | /*? 0x3F, 0x3F */ |
1170 | | /*\ 0x5C, 0x5C */ |
1171 | | /*a*/ 0x61, 0x07, |
1172 | | /*b*/ 0x62, 0x08, |
1173 | | /*e*/ 0x65, 0x1b, |
1174 | | /*f*/ 0x66, 0x0c, |
1175 | | /*n*/ 0x6E, 0x0a, |
1176 | | /*r*/ 0x72, 0x0d, |
1177 | | /*t*/ 0x74, 0x09, |
1178 | | /*v*/ 0x76, 0x0b |
1179 | | }; |
1180 | | enum { UNESCAPE_MAP_LENGTH = UPRV_LENGTHOF(UNESCAPE_MAP) }; |
1181 | | |
1182 | | /* Convert one octal digit to a numeric value 0..7, or -1 on failure */ |
1183 | 244k | static int32_t _digit8(char16_t c) { |
1184 | 244k | if (c >= u'0' && c <= u'7') { |
1185 | 10.7k | return c - u'0'; |
1186 | 10.7k | } |
1187 | 234k | return -1; |
1188 | 244k | } |
1189 | | |
1190 | | /* Convert one hex digit to a numeric value 0..F, or -1 on failure */ |
1191 | 1.05M | static int32_t _digit16(char16_t c) { |
1192 | 1.05M | if (c >= u'0' && c <= u'9') { |
1193 | 747k | return c - u'0'; |
1194 | 747k | } |
1195 | 310k | if (c >= u'A' && c <= u'F') { |
1196 | 21.8k | return c - (u'A' - 10); |
1197 | 21.8k | } |
1198 | 289k | if (c >= u'a' && c <= u'f') { |
1199 | 283k | return c - (u'a' - 10); |
1200 | 283k | } |
1201 | 6.08k | return -1; |
1202 | 289k | } |
1203 | | |
1204 | | /* Parse a single escape sequence. Although this method deals in |
1205 | | * UChars, it does not use C++ or UnicodeString. This allows it to |
1206 | | * be used from C contexts. */ |
1207 | | U_CAPI UChar32 U_EXPORT2 |
1208 | | u_unescapeAt(UNESCAPE_CHAR_AT charAt, |
1209 | | int32_t *offset, |
1210 | | int32_t length, |
1211 | 503k | void *context) { |
1212 | | |
1213 | 503k | int32_t start = *offset; |
1214 | 503k | UChar32 c; |
1215 | 503k | UChar32 result = 0; |
1216 | 503k | int8_t n = 0; |
1217 | 503k | int8_t minDig = 0; |
1218 | 503k | int8_t maxDig = 0; |
1219 | 503k | int8_t bitsPerDigit = 4; |
1220 | 503k | int32_t dig; |
1221 | 503k | UBool braces = false; |
1222 | | |
1223 | | /* Check that offset is in range */ |
1224 | 503k | if (*offset < 0 || *offset >= length) { |
1225 | 84 | goto err; |
1226 | 84 | } |
1227 | | |
1228 | | /* Fetch first char16_t after '\\' */ |
1229 | 502k | c = charAt((*offset)++, context); |
1230 | | |
1231 | | /* Convert hexadecimal and octal escapes */ |
1232 | 502k | switch (c) { |
1233 | 257k | case u'u': |
1234 | 257k | minDig = maxDig = 4; |
1235 | 257k | break; |
1236 | 2.27k | case u'U': |
1237 | 2.27k | minDig = maxDig = 8; |
1238 | 2.27k | break; |
1239 | 8.67k | case u'x': |
1240 | 8.67k | minDig = 1; |
1241 | 8.67k | if (*offset < length && charAt(*offset, context) == u'{') { |
1242 | 854 | ++(*offset); |
1243 | 854 | braces = true; |
1244 | 854 | maxDig = 8; |
1245 | 7.82k | } else { |
1246 | 7.82k | maxDig = 2; |
1247 | 7.82k | } |
1248 | 8.67k | break; |
1249 | 234k | default: |
1250 | 234k | dig = _digit8(c); |
1251 | 234k | if (dig >= 0) { |
1252 | 8.37k | minDig = 1; |
1253 | 8.37k | maxDig = 3; |
1254 | 8.37k | n = 1; /* Already have first octal digit */ |
1255 | 8.37k | bitsPerDigit = 3; |
1256 | 8.37k | result = dig; |
1257 | 8.37k | } |
1258 | 234k | break; |
1259 | 502k | } |
1260 | 502k | if (minDig != 0) { |
1261 | 1.33M | while (*offset < length && n < maxDig) { |
1262 | 1.06M | c = charAt(*offset, context); |
1263 | 1.06M | dig = (bitsPerDigit == 3) ? _digit8(c) : _digit16(c); |
1264 | 1.06M | if (dig < 0) { |
1265 | 13.7k | break; |
1266 | 13.7k | } |
1267 | 1.05M | result = (result << bitsPerDigit) | dig; |
1268 | 1.05M | ++(*offset); |
1269 | 1.05M | ++n; |
1270 | 1.05M | } |
1271 | 276k | if (n < minDig) { |
1272 | 1.20k | goto err; |
1273 | 1.20k | } |
1274 | 275k | if (braces) { |
1275 | 841 | if (c != u'}') { |
1276 | 84 | goto err; |
1277 | 84 | } |
1278 | 757 | ++(*offset); |
1279 | 757 | } |
1280 | 275k | if (result < 0 || result >= 0x110000) { |
1281 | 484 | goto err; |
1282 | 484 | } |
1283 | | /* If an escape sequence specifies a lead surrogate, see if |
1284 | | * there is a trail surrogate after it, either as an escape or |
1285 | | * as a literal. If so, join them up into a supplementary. |
1286 | | */ |
1287 | 274k | if (*offset < length && U16_IS_LEAD(result)) { |
1288 | 175k | int32_t ahead = *offset + 1; |
1289 | 175k | c = charAt(*offset, context); |
1290 | 175k | if (c == u'\\' && ahead < length) { |
1291 | | // Calling ourselves recursively may cause a stack overflow if |
1292 | | // we have repeated escaped lead surrogates. |
1293 | | // Limit the length to 11 ("x{0000DFFF}") after ahead. |
1294 | 170k | int32_t tailLimit = ahead + 11; |
1295 | 170k | if (tailLimit > length) { |
1296 | 87.8k | tailLimit = length; |
1297 | 87.8k | } |
1298 | 170k | c = u_unescapeAt(charAt, &ahead, tailLimit, context); |
1299 | 170k | } |
1300 | 175k | if (U16_IS_TRAIL(c)) { |
1301 | 644 | *offset = ahead; |
1302 | 644 | result = U16_GET_SUPPLEMENTARY(result, c); |
1303 | 644 | } |
1304 | 175k | } |
1305 | 274k | return result; |
1306 | 275k | } |
1307 | | |
1308 | | /* Convert C-style escapes in table */ |
1309 | 590k | for (int32_t i=0; i<UNESCAPE_MAP_LENGTH; i+=2) { |
1310 | 568k | if (c == UNESCAPE_MAP[i]) { |
1311 | 2.55k | return UNESCAPE_MAP[i+1]; |
1312 | 565k | } else if (c < UNESCAPE_MAP[i]) { |
1313 | 201k | break; |
1314 | 201k | } |
1315 | 568k | } |
1316 | | |
1317 | | /* Map \cX to control-X: X & 0x1F */ |
1318 | 223k | if (c == u'c' && *offset < length) { |
1319 | 86.7k | c = charAt((*offset)++, context); |
1320 | 86.7k | if (U16_IS_LEAD(c) && *offset < length) { |
1321 | 74.1k | char16_t c2 = charAt(*offset, context); |
1322 | 74.1k | if (U16_IS_TRAIL(c2)) { |
1323 | 713 | ++(*offset); |
1324 | 713 | c = U16_GET_SUPPLEMENTARY(c, c2); |
1325 | 713 | } |
1326 | 74.1k | } |
1327 | 86.7k | return 0x1F & c; |
1328 | 86.7k | } |
1329 | | |
1330 | | /* If no special forms are recognized, then consider |
1331 | | * the backslash to generically escape the next character. |
1332 | | * Deal with surrogate pairs. */ |
1333 | 137k | if (U16_IS_LEAD(c) && *offset < length) { |
1334 | 5.95k | char16_t c2 = charAt(*offset, context); |
1335 | 5.95k | if (U16_IS_TRAIL(c2)) { |
1336 | 2.47k | ++(*offset); |
1337 | 2.47k | return U16_GET_SUPPLEMENTARY(c, c2); |
1338 | 2.47k | } |
1339 | 5.95k | } |
1340 | 134k | return c; |
1341 | | |
1342 | 1.85k | err: |
1343 | | /* Invalid escape sequence */ |
1344 | 1.85k | *offset = start; /* Reset to initial value */ |
1345 | 1.85k | return (UChar32)0xFFFFFFFF; |
1346 | 137k | } |
1347 | | |
1348 | | /* u_unescapeAt() callback to return a char16_t from a char* */ |
1349 | | static char16_t U_CALLCONV |
1350 | 0 | _charPtr_charAt(int32_t offset, void *context) { |
1351 | 0 | char16_t c16; |
1352 | | /* It would be more efficient to access the invariant tables |
1353 | | * directly but there is no API for that. */ |
1354 | 0 | u_charsToUChars(static_cast<char*>(context) + offset, &c16, 1); |
1355 | 0 | return c16; |
1356 | 0 | } |
1357 | | |
1358 | | /* Append an escape-free segment of the text; used by u_unescape() */ |
1359 | | static void _appendUChars(char16_t *dest, int32_t destCapacity, |
1360 | 0 | const char *src, int32_t srcLen) { |
1361 | 0 | if (destCapacity < 0) { |
1362 | 0 | destCapacity = 0; |
1363 | 0 | } |
1364 | 0 | if (srcLen > destCapacity) { |
1365 | 0 | srcLen = destCapacity; |
1366 | 0 | } |
1367 | 0 | u_charsToUChars(src, dest, srcLen); |
1368 | 0 | } |
1369 | | |
1370 | | /* Do an invariant conversion of char* -> char16_t*, with escape parsing */ |
1371 | | U_CAPI int32_t U_EXPORT2 |
1372 | 0 | u_unescape(const char *src, char16_t *dest, int32_t destCapacity) { |
1373 | 0 | const char *segment = src; |
1374 | 0 | int32_t i = 0; |
1375 | 0 | char c; |
1376 | |
|
1377 | 0 | while ((c=*src) != 0) { |
1378 | | /* '\\' intentionally written as compiler-specific |
1379 | | * character constant to correspond to compiler-specific |
1380 | | * char* constants. */ |
1381 | 0 | if (c == '\\') { |
1382 | 0 | int32_t lenParsed = 0; |
1383 | 0 | UChar32 c32; |
1384 | 0 | if (src != segment) { |
1385 | 0 | if (dest != nullptr) { |
1386 | 0 | _appendUChars(dest + i, destCapacity - i, |
1387 | 0 | segment, (int32_t)(src - segment)); |
1388 | 0 | } |
1389 | 0 | i += (int32_t)(src - segment); |
1390 | 0 | } |
1391 | 0 | ++src; /* advance past '\\' */ |
1392 | 0 | c32 = u_unescapeAt(_charPtr_charAt, &lenParsed, (int32_t)uprv_strlen(src), const_cast<char*>(src)); |
1393 | 0 | if (lenParsed == 0) { |
1394 | 0 | goto err; |
1395 | 0 | } |
1396 | 0 | src += lenParsed; /* advance past escape seq. */ |
1397 | 0 | if (dest != nullptr && U16_LENGTH(c32) <= (destCapacity - i)) { |
1398 | 0 | U16_APPEND_UNSAFE(dest, i, c32); |
1399 | 0 | } else { |
1400 | 0 | i += U16_LENGTH(c32); |
1401 | 0 | } |
1402 | 0 | segment = src; |
1403 | 0 | } else { |
1404 | 0 | ++src; |
1405 | 0 | } |
1406 | 0 | } |
1407 | 0 | if (src != segment) { |
1408 | 0 | if (dest != nullptr) { |
1409 | 0 | _appendUChars(dest + i, destCapacity - i, |
1410 | 0 | segment, (int32_t)(src - segment)); |
1411 | 0 | } |
1412 | 0 | i += (int32_t)(src - segment); |
1413 | 0 | } |
1414 | 0 | if (dest != nullptr && i < destCapacity) { |
1415 | 0 | dest[i] = 0; |
1416 | 0 | } |
1417 | 0 | return i; |
1418 | | |
1419 | 0 | err: |
1420 | 0 | if (dest != nullptr && destCapacity > 0) { |
1421 | 0 | *dest = 0; |
1422 | 0 | } |
1423 | 0 | return 0; |
1424 | 0 | } |
1425 | | |
1426 | | /* NUL-termination of strings ----------------------------------------------- */ |
1427 | | |
1428 | | /** |
1429 | | * NUL-terminate a string no matter what its type. |
1430 | | * Set warning and error codes accordingly. |
1431 | | */ |
1432 | 3.60M | #define __TERMINATE_STRING(dest, destCapacity, length, pErrorCode) UPRV_BLOCK_MACRO_BEGIN { \ |
1433 | 3.60M | if(pErrorCode!=nullptr && U_SUCCESS(*pErrorCode)) { \ |
1434 | 3.59M | /* not a public function, so no complete argument checking */ \ |
1435 | 3.59M | \ |
1436 | 3.59M | if(length<0) { \ |
1437 | 0 | /* assume that the caller handles this */ \ |
1438 | 3.59M | } else if(length<destCapacity) { \ |
1439 | 3.50M | /* NUL-terminate the string, the NUL fits */ \ |
1440 | 3.50M | dest[length]=0; \ |
1441 | 3.50M | /* unset the not-terminated warning but leave all others */ \ |
1442 | 3.50M | if(*pErrorCode==U_STRING_NOT_TERMINATED_WARNING) { \ |
1443 | 30 | *pErrorCode=U_ZERO_ERROR; \ |
1444 | 30 | } \ |
1445 | 3.50M | } else if(length==destCapacity) { \ |
1446 | 80.4k | /* unable to NUL-terminate, but the string itself fit - set a warning code */ \ |
1447 | 80.4k | *pErrorCode=U_STRING_NOT_TERMINATED_WARNING; \ |
1448 | 80.4k | } else /* length>destCapacity */ { \ |
1449 | 9.34k | /* even the string itself did not fit - set an error code */ \ |
1450 | 9.34k | *pErrorCode=U_BUFFER_OVERFLOW_ERROR; \ |
1451 | 9.34k | } \ |
1452 | 3.59M | } \ |
1453 | 3.60M | } UPRV_BLOCK_MACRO_END |
1454 | | |
1455 | | U_CAPI char16_t U_EXPORT2 |
1456 | 0 | u_asciiToUpper(char16_t c) { |
1457 | 0 | if (u'a' <= c && c <= u'z') { |
1458 | 0 | c = c + u'A' - u'a'; |
1459 | 0 | } |
1460 | 0 | return c; |
1461 | 0 | } |
1462 | | |
1463 | | U_CAPI int32_t U_EXPORT2 |
1464 | 1.33M | u_terminateUChars(char16_t *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) { |
1465 | 1.33M | __TERMINATE_STRING(dest, destCapacity, length, pErrorCode); |
1466 | 1.33M | return length; |
1467 | 1.33M | } |
1468 | | |
1469 | | U_CAPI int32_t U_EXPORT2 |
1470 | 2.27M | u_terminateChars(char *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) { |
1471 | 2.27M | __TERMINATE_STRING(dest, destCapacity, length, pErrorCode); |
1472 | 2.27M | return length; |
1473 | 2.27M | } |
1474 | | |
1475 | | U_CAPI int32_t U_EXPORT2 |
1476 | 0 | u_terminateUChar32s(UChar32 *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) { |
1477 | 0 | __TERMINATE_STRING(dest, destCapacity, length, pErrorCode); |
1478 | 0 | return length; |
1479 | 0 | } |
1480 | | |
1481 | | U_CAPI int32_t U_EXPORT2 |
1482 | 0 | u_terminateWChars(wchar_t *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) { |
1483 | 0 | __TERMINATE_STRING(dest, destCapacity, length, pErrorCode); |
1484 | 0 | return length; |
1485 | 0 | } |
1486 | | |
1487 | | // Compute the hash code for a string -------------------------------------- *** |
1488 | | |
1489 | | // Moved here from uhash.c so that UnicodeString::hashCode() does not depend |
1490 | | // on UHashtable code. |
1491 | | |
1492 | | /* |
1493 | | Compute the hash by iterating sparsely over about 32 (up to 63) |
1494 | | characters spaced evenly through the string. For each character, |
1495 | | multiply the previous hash value by a prime number and add the new |
1496 | | character in, like a linear congruential random number generator, |
1497 | | producing a pseudorandom deterministic value well distributed over |
1498 | | the output range. [LIU] |
1499 | | */ |
1500 | | |
1501 | 53.6M | #define STRING_HASH(TYPE, STR, STRLEN, DEREF) UPRV_BLOCK_MACRO_BEGIN { \ |
1502 | 53.6M | uint32_t hash = 0; \ |
1503 | 53.6M | const TYPE *p = (const TYPE*) STR; \ |
1504 | 53.6M | if (p != nullptr) { \ |
1505 | 53.6M | int32_t len = (int32_t)(STRLEN); \ |
1506 | 53.6M | int32_t inc = ((len - 32) / 32) + 1; \ |
1507 | 53.6M | const TYPE *limit = p + len; \ |
1508 | 353M | while (p<limit) { \ |
1509 | 300M | hash = (hash * 37) + DEREF; \ |
1510 | 300M | p += inc; \ |
1511 | 300M | } \ |
1512 | 53.6M | } \ |
1513 | 53.6M | return static_cast<int32_t>(hash); \ |
1514 | 53.6M | } UPRV_BLOCK_MACRO_END |
1515 | | |
1516 | | /* Used by UnicodeString to compute its hashcode - Not public API. */ |
1517 | | U_CAPI int32_t U_EXPORT2 |
1518 | 10.0M | ustr_hashUCharsN(const char16_t *str, int32_t length) { |
1519 | 10.0M | STRING_HASH(char16_t, str, length, *p); |
1520 | 10.0M | } |
1521 | | |
1522 | | U_CAPI int32_t U_EXPORT2 |
1523 | 10.7M | ustr_hashCharsN(const char *str, int32_t length) { |
1524 | 10.7M | STRING_HASH(uint8_t, str, length, *p); |
1525 | 10.7M | } |
1526 | | |
1527 | | U_CAPI int32_t U_EXPORT2 |
1528 | 32.8M | ustr_hashICharsN(const char *str, int32_t length) { |
1529 | 32.8M | STRING_HASH(char, str, length, (uint8_t)uprv_tolower(*p)); |
1530 | 32.8M | } |