Coverage Report

Created: 2026-09-28 06:10

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/icu/icu4c/source/common/ustring.cpp
Line
Count
Source
1
// © 2016 and later: Unicode, Inc. and others.
2
// License & terms of use: http://www.unicode.org/copyright.html
3
/*
4
******************************************************************************
5
*
6
*   Copyright (C) 1998-2016, International Business Machines
7
*   Corporation and others.  All Rights Reserved.
8
*
9
******************************************************************************
10
*
11
* File ustring.cpp
12
*
13
* Modification History:
14
*
15
*   Date        Name        Description
16
*   12/07/98    bertrand    Creation.
17
******************************************************************************
18
*/
19
20
#include "unicode/utypes.h"
21
#include "unicode/putil.h"
22
#include "unicode/uchar.h"
23
#include "unicode/ustring.h"
24
#include "unicode/utf16.h"
25
#include "cstring.h"
26
#include "cwchar.h"
27
#include "cmemory.h"
28
#include "ustr_imp.h"
29
30
/* ANSI string.h - style functions ------------------------------------------ */
31
32
/* U+ffff is the highest BMP code point, the highest one that fits into a 16-bit char16_t */
33
0
#define U_BMP_MAX 0xffff
34
35
/* Forward binary string search functions ----------------------------------- */
36
37
/*
38
 * Test if a substring match inside a string is at code point boundaries.
39
 * All pointers refer to the same buffer.
40
 * The limit pointer may be nullptr, all others must be real pointers.
41
 */
42
static inline UBool
43
235k
isMatchAtCPBoundary(const char16_t *start, const char16_t *match, const char16_t *matchLimit, const char16_t *limit) {
44
235k
    if(U16_IS_TRAIL(*match) && start!=match && U16_IS_LEAD(*(match-1))) {
45
        /* the leading edge of the match is in the middle of a surrogate pair */
46
0
        return false;
47
0
    }
48
235k
    if(U16_IS_LEAD(*(matchLimit-1)) && matchLimit!=limit && U16_IS_TRAIL(*matchLimit)) {
49
        /* the trailing edge of the match is in the middle of a surrogate pair */
50
0
        return false;
51
0
    }
52
235k
    return true;
53
235k
}
54
55
U_CAPI char16_t* U_EXPORT2
56
u_strFindFirst(const char16_t* s U_LIFETIME_BOUND,
57
               int32_t length,
58
               const char16_t* sub,
59
2.94M
               int32_t subLength) {
60
2.94M
    const char16_t *start, *p, *q, *subLimit;
61
2.94M
    char16_t c, cs, cq;
62
63
2.94M
    if(sub==nullptr || subLength<-1) {
64
0
        return (char16_t *)s;
65
0
    }
66
2.94M
    if(s==nullptr || length<-1) {
67
0
        return nullptr;
68
0
    }
69
70
2.94M
    start=s;
71
72
2.94M
    if(length<0 && subLength<0) {
73
        /* both strings are NUL-terminated */
74
0
        if((cs=*sub++)==0) {
75
0
            return (char16_t *)s;
76
0
        }
77
0
        if(*sub==0 && !U16_IS_SURROGATE(cs)) {
78
            /* the substring consists of a single, non-surrogate BMP code point */
79
0
            return u_strchr(s, cs);
80
0
        }
81
82
0
        while((c=*s++)!=0) {
83
0
            if(c==cs) {
84
                /* found first substring char16_t, compare rest */
85
0
                p=s;
86
0
                q=sub;
87
0
                for(;;) {
88
0
                    if((cq=*q)==0) {
89
0
                        if(isMatchAtCPBoundary(start, s-1, p, nullptr)) {
90
0
                            return (char16_t *)(s-1); /* well-formed match */
91
0
                        } else {
92
0
                            break; /* no match because surrogate pair is split */
93
0
                        }
94
0
                    }
95
0
                    if((c=*p)==0) {
96
0
                        return nullptr; /* no match, and none possible after s */
97
0
                    }
98
0
                    if(c!=cq) {
99
0
                        break; /* no match */
100
0
                    }
101
0
                    ++p;
102
0
                    ++q;
103
0
                }
104
0
            }
105
0
        }
106
107
        /* not found */
108
0
        return nullptr;
109
0
    }
110
111
2.94M
    if(subLength<0) {
112
1.62M
        subLength=u_strlen(sub);
113
1.62M
    }
114
2.94M
    if(subLength==0) {
115
0
        return (char16_t *)s;
116
0
    }
117
118
    /* get sub[0] to search for it fast */
119
2.94M
    cs=*sub++;
120
2.94M
    --subLength;
121
2.94M
    subLimit=sub+subLength;
122
123
2.94M
    if(subLength==0 && !U16_IS_SURROGATE(cs)) {
124
        /* the substring consists of a single, non-surrogate BMP code point */
125
106k
        return length<0 ? u_strchr(s, cs) : u_memchr(s, cs, length);
126
106k
    }
127
128
2.83M
    if(length<0) {
129
        /* s is NUL-terminated */
130
0
        while((c=*s++)!=0) {
131
0
            if(c==cs) {
132
                /* found first substring char16_t, compare rest */
133
0
                p=s;
134
0
                q=sub;
135
0
                for(;;) {
136
0
                    if(q==subLimit) {
137
0
                        if(isMatchAtCPBoundary(start, s-1, p, nullptr)) {
138
0
                            return (char16_t *)(s-1); /* well-formed match */
139
0
                        } else {
140
0
                            break; /* no match because surrogate pair is split */
141
0
                        }
142
0
                    }
143
0
                    if((c=*p)==0) {
144
0
                        return nullptr; /* no match, and none possible after s */
145
0
                    }
146
0
                    if(c!=*q) {
147
0
                        break; /* no match */
148
0
                    }
149
0
                    ++p;
150
0
                    ++q;
151
0
                }
152
0
            }
153
0
        }
154
2.83M
    } else {
155
2.83M
        const char16_t *limit, *preLimit;
156
157
        /* subLength was decremented above */
158
2.83M
        if(length<=subLength) {
159
949k
            return nullptr; /* s is shorter than sub */
160
949k
        }
161
162
1.88M
        limit=s+length;
163
164
        /* the substring must start before preLimit */
165
1.88M
        preLimit=limit-subLength;
166
167
59.9M
        while(s!=preLimit) {
168
58.2M
            c=*s++;
169
58.2M
            if(c==cs) {
170
                /* found first substring char16_t, compare rest */
171
2.77M
                p=s;
172
2.77M
                q=sub;
173
3.12M
                for(;;) {
174
3.12M
                    if(q==subLimit) {
175
218k
                        if(isMatchAtCPBoundary(start, s-1, p, limit)) {
176
218k
                            return (char16_t *)(s-1); /* well-formed match */
177
218k
                        } else {
178
0
                            break; /* no match because surrogate pair is split */
179
0
                        }
180
218k
                    }
181
2.90M
                    if(*p!=*q) {
182
2.55M
                        break; /* no match */
183
2.55M
                    }
184
349k
                    ++p;
185
349k
                    ++q;
186
349k
                }
187
2.77M
            }
188
58.2M
        }
189
1.88M
    }
190
191
    /* not found */
192
1.66M
    return nullptr;
193
2.83M
}
194
195
U_CAPI char16_t* U_EXPORT2
196
0
u_strstr(const char16_t* s U_LIFETIME_BOUND, const char16_t* substring) {
197
0
    return u_strFindFirst(s, -1, substring, -1);
198
0
}
199
200
U_CAPI char16_t* U_EXPORT2
201
77.3k
u_strchr(const char16_t* s U_LIFETIME_BOUND, char16_t c) {
202
77.3k
    if(U16_IS_SURROGATE(c)) {
203
        /* make sure to not find half of a surrogate pair */
204
0
        return u_strFindFirst(s, -1, &c, 1);
205
77.3k
    } else {
206
77.3k
        char16_t cs;
207
208
        /* trivial search for a BMP code point */
209
942k
        for(;;) {
210
942k
            if((cs=*s)==c) {
211
37.2k
                return (char16_t *)s;
212
37.2k
            }
213
904k
            if(cs==0) {
214
40.0k
                return nullptr;
215
40.0k
            }
216
864k
            ++s;
217
864k
        }
218
77.3k
    }
219
77.3k
}
220
221
U_CAPI char16_t* U_EXPORT2
222
0
u_strchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c) {
223
0
    if((uint32_t)c<=U_BMP_MAX) {
224
        /* find BMP code point */
225
0
        return u_strchr(s, (char16_t)c);
226
0
    } else if((uint32_t)c<=UCHAR_MAX_VALUE) {
227
        /* find supplementary code point as surrogate pair */
228
0
        char16_t cs, lead=U16_LEAD(c), trail=U16_TRAIL(c);
229
230
0
        while((cs=*s++)!=0) {
231
0
            if(cs==lead && *s==trail) {
232
0
                return (char16_t *)(s-1);
233
0
            }
234
0
        }
235
0
        return nullptr;
236
0
    } else {
237
        /* not a Unicode code point, not findable */
238
0
        return nullptr;
239
0
    }
240
0
}
241
242
U_CAPI char16_t* U_EXPORT2
243
32.7M
u_memchr(const char16_t* s U_LIFETIME_BOUND, char16_t c, int32_t count) {
244
32.7M
    if(count<=0) {
245
626k
        return nullptr; /* no string */
246
32.0M
    } else if(U16_IS_SURROGATE(c)) {
247
        /* make sure to not find half of a surrogate pair */
248
0
        return u_strFindFirst(s, count, &c, 1);
249
32.0M
    } else {
250
        /* trivial search for a BMP code point */
251
32.0M
        const char16_t *limit=s+count;
252
159M
        do {
253
159M
            if(*s==c) {
254
8.40M
                return (char16_t *)s;
255
8.40M
            }
256
159M
        } while(++s!=limit);
257
23.6M
        return nullptr;
258
32.0M
    }
259
32.7M
}
260
261
U_CAPI char16_t* U_EXPORT2
262
0
u_memchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c, int32_t count) {
263
0
    if((uint32_t)c<=U_BMP_MAX) {
264
        /* find BMP code point */
265
0
        return u_memchr(s, (char16_t)c, count);
266
0
    } else if(count<2) {
267
        /* too short for a surrogate pair */
268
0
        return nullptr;
269
0
    } else if((uint32_t)c<=UCHAR_MAX_VALUE) {
270
        /* find supplementary code point as surrogate pair */
271
0
        const char16_t *limit=s+count-1; /* -1 so that we do not need a separate check for the trail unit */
272
0
        char16_t lead=U16_LEAD(c), trail=U16_TRAIL(c);
273
274
0
        do {
275
0
            if(*s==lead && *(s+1)==trail) {
276
0
                return (char16_t *)s;
277
0
            }
278
0
        } while(++s!=limit);
279
0
        return nullptr;
280
0
    } else {
281
        /* not a Unicode code point, not findable */
282
0
        return nullptr;
283
0
    }
284
0
}
285
286
/* Backward binary string search functions ---------------------------------- */
287
288
U_CAPI char16_t* U_EXPORT2
289
u_strFindLast(const char16_t* s U_LIFETIME_BOUND,
290
              int32_t length,
291
              const char16_t* sub,
292
16.2k
              int32_t subLength) {
293
16.2k
    const char16_t *start, *limit, *p, *q, *subLimit;
294
16.2k
    char16_t c, cs;
295
296
16.2k
    if(sub==nullptr || subLength<-1) {
297
0
        return (char16_t *)s;
298
0
    }
299
16.2k
    if(s==nullptr || length<-1) {
300
0
        return nullptr;
301
0
    }
302
303
    /*
304
     * This implementation is more lazy than the one for u_strFindFirst():
305
     * There is no special search code for NUL-terminated strings.
306
     * It does not seem to be worth it for searching substrings to
307
     * search forward and find all matches like in u_strrchr() and similar.
308
     * Therefore, we simply get both string lengths and search backward.
309
     *
310
     * markus 2002oct23
311
     */
312
313
16.2k
    if(subLength<0) {
314
0
        subLength=u_strlen(sub);
315
0
    }
316
16.2k
    if(subLength==0) {
317
0
        return (char16_t *)s;
318
0
    }
319
320
    /* get sub[subLength-1] to search for it fast */
321
16.2k
    subLimit=sub+subLength;
322
16.2k
    cs=*(--subLimit);
323
16.2k
    --subLength;
324
325
16.2k
    if(subLength==0 && !U16_IS_SURROGATE(cs)) {
326
        /* the substring consists of a single, non-surrogate BMP code point */
327
0
        return length<0 ? u_strrchr(s, cs) : u_memrchr(s, cs, length);
328
0
    }
329
330
16.2k
    if(length<0) {
331
0
        length=u_strlen(s);
332
0
    }
333
334
    /* subLength was decremented above */
335
16.2k
    if(length<=subLength) {
336
0
        return nullptr; /* s is shorter than sub */
337
0
    }
338
339
16.2k
    start=s;
340
16.2k
    limit=s+length;
341
342
    /* the substring must start no later than s+subLength */
343
16.2k
    s+=subLength;
344
345
32.5k
    while(s!=limit) {
346
32.4k
        c=*(--limit);
347
32.4k
        if(c==cs) {
348
            /* found last substring char16_t, compare rest */
349
16.2k
            p=limit;
350
16.2k
            q=subLimit;
351
32.4k
            for(;;) {
352
32.4k
                if(q==sub) {
353
16.2k
                    if(isMatchAtCPBoundary(start, p, limit+1, start+length)) {
354
16.2k
                        return (char16_t *)p; /* well-formed match */
355
16.2k
                    } else {
356
0
                        break; /* no match because surrogate pair is split */
357
0
                    }
358
16.2k
                }
359
16.2k
                if(*(--p)!=*(--q)) {
360
42
                    break; /* no match */
361
42
                }
362
16.2k
            }
363
16.2k
        }
364
32.4k
    }
365
366
    /* not found */
367
42
    return nullptr;
368
16.2k
}
369
370
U_CAPI char16_t* U_EXPORT2
371
0
u_strrstr(const char16_t* s U_LIFETIME_BOUND, const char16_t* substring) {
372
0
    return u_strFindLast(s, -1, substring, -1);
373
0
}
374
375
U_CAPI char16_t* U_EXPORT2
376
0
u_strrchr(const char16_t* s U_LIFETIME_BOUND, char16_t c) {
377
0
    if(U16_IS_SURROGATE(c)) {
378
        /* make sure to not find half of a surrogate pair */
379
0
        return u_strFindLast(s, -1, &c, 1);
380
0
    } else {
381
0
        const char16_t *result=nullptr;
382
0
        char16_t cs;
383
384
        /* trivial search for a BMP code point */
385
0
        for(;;) {
386
0
            if((cs=*s)==c) {
387
0
                result=s;
388
0
            }
389
0
            if(cs==0) {
390
0
                return (char16_t *)result;
391
0
            }
392
0
            ++s;
393
0
        }
394
0
    }
395
0
}
396
397
U_CAPI char16_t* U_EXPORT2
398
0
u_strrchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c) {
399
0
    if((uint32_t)c<=U_BMP_MAX) {
400
        /* find BMP code point */
401
0
        return u_strrchr(s, (char16_t)c);
402
0
    } else if((uint32_t)c<=UCHAR_MAX_VALUE) {
403
        /* find supplementary code point as surrogate pair */
404
0
        const char16_t *result=nullptr;
405
0
        char16_t cs, lead=U16_LEAD(c), trail=U16_TRAIL(c);
406
407
0
        while((cs=*s++)!=0) {
408
0
            if(cs==lead && *s==trail) {
409
0
                result=s-1;
410
0
            }
411
0
        }
412
0
        return (char16_t *)result;
413
0
    } else {
414
        /* not a Unicode code point, not findable */
415
0
        return nullptr;
416
0
    }
417
0
}
418
419
U_CAPI char16_t* U_EXPORT2
420
127k
u_memrchr(const char16_t* s U_LIFETIME_BOUND, char16_t c, int32_t count) {
421
127k
    if(count<=0) {
422
0
        return nullptr; /* no string */
423
127k
    } else if(U16_IS_SURROGATE(c)) {
424
        /* make sure to not find half of a surrogate pair */
425
0
        return u_strFindLast(s, count, &c, 1);
426
127k
    } else {
427
        /* trivial search for a BMP code point */
428
127k
        const char16_t *limit=s+count;
429
922k
        do {
430
922k
            if(*(--limit)==c) {
431
110k
                return (char16_t *)limit;
432
110k
            }
433
922k
        } while(s!=limit);
434
16.9k
        return nullptr;
435
127k
    }
436
127k
}
437
438
U_CAPI char16_t* U_EXPORT2
439
0
u_memrchr32(const char16_t* s U_LIFETIME_BOUND, UChar32 c, int32_t count) {
440
0
    if((uint32_t)c<=U_BMP_MAX) {
441
        /* find BMP code point */
442
0
        return u_memrchr(s, (char16_t)c, count);
443
0
    } else if(count<2) {
444
        /* too short for a surrogate pair */
445
0
        return nullptr;
446
0
    } else if((uint32_t)c<=UCHAR_MAX_VALUE) {
447
        /* find supplementary code point as surrogate pair */
448
0
        const char16_t *limit=s+count-1;
449
0
        char16_t lead=U16_LEAD(c), trail=U16_TRAIL(c);
450
451
0
        do {
452
0
            if(*limit==trail && *(limit-1)==lead) {
453
0
                return (char16_t *)(limit-1);
454
0
            }
455
0
        } while(s!=--limit);
456
0
        return nullptr;
457
0
    } else {
458
        /* not a Unicode code point, not findable */
459
0
        return nullptr;
460
0
    }
461
0
}
462
463
/* Tokenization functions --------------------------------------------------- */
464
465
/*
466
 * Match each code point in a string against each code point in the matchSet.
467
 * Return the index of the first string code point that
468
 * is (polarity==true) or is not (false) contained in the matchSet.
469
 * Return -(string length)-1 if there is no such code point.
470
 */
471
static int32_t
472
0
_matchFromSet(const char16_t *string, const char16_t *matchSet, UBool polarity) {
473
0
    int32_t matchLen, matchBMPLen, strItr, matchItr;
474
0
    UChar32 stringCh, matchCh;
475
0
    char16_t c, c2;
476
477
    /* first part of matchSet contains only BMP code points */
478
0
    matchBMPLen = 0;
479
0
    while((c = matchSet[matchBMPLen]) != 0 && U16_IS_SINGLE(c)) {
480
0
        ++matchBMPLen;
481
0
    }
482
483
    /* second part of matchSet contains BMP and supplementary code points */
484
0
    matchLen = matchBMPLen;
485
0
    while(matchSet[matchLen] != 0) {
486
0
        ++matchLen;
487
0
    }
488
489
0
    for(strItr = 0; (c = string[strItr]) != 0;) {
490
0
        ++strItr;
491
0
        if(U16_IS_SINGLE(c)) {
492
0
            if(polarity) {
493
0
                for(matchItr = 0; matchItr < matchLen; ++matchItr) {
494
0
                    if(c == matchSet[matchItr]) {
495
0
                        return strItr - 1; /* one matches */
496
0
                    }
497
0
                }
498
0
            } else {
499
0
                for(matchItr = 0; matchItr < matchLen; ++matchItr) {
500
0
                    if(c == matchSet[matchItr]) {
501
0
                        goto endloop;
502
0
                    }
503
0
                }
504
0
                return strItr - 1; /* none matches */
505
0
            }
506
0
        } else {
507
            /*
508
             * No need to check for string length before U16_IS_TRAIL
509
             * because c2 could at worst be the terminating NUL.
510
             */
511
0
            if(U16_IS_SURROGATE_LEAD(c) && U16_IS_TRAIL(c2 = string[strItr])) {
512
0
                ++strItr;
513
0
                stringCh = U16_GET_SUPPLEMENTARY(c, c2);
514
0
            } else {
515
0
                stringCh = c; /* unpaired trail surrogate */
516
0
            }
517
518
0
            if(polarity) {
519
0
                for(matchItr = matchBMPLen; matchItr < matchLen;) {
520
0
                    U16_NEXT(matchSet, matchItr, matchLen, matchCh);
521
0
                    if(stringCh == matchCh) {
522
0
                        return strItr - U16_LENGTH(stringCh); /* one matches */
523
0
                    }
524
0
                }
525
0
            } else {
526
0
                for(matchItr = matchBMPLen; matchItr < matchLen;) {
527
0
                    U16_NEXT(matchSet, matchItr, matchLen, matchCh);
528
0
                    if(stringCh == matchCh) {
529
0
                        goto endloop;
530
0
                    }
531
0
                }
532
0
                return strItr - U16_LENGTH(stringCh); /* none matches */
533
0
            }
534
0
        }
535
0
endloop:
536
0
        /* wish C had continue with labels like Java... */;
537
0
    }
538
539
    /* Didn't find it. */
540
0
    return -strItr-1;
541
0
}
542
543
/* Search for a codepoint in a string that matches one of the matchSet codepoints. */
544
U_CAPI char16_t* U_EXPORT2
545
u_strpbrk(const char16_t* string U_LIFETIME_BOUND, const char16_t* matchSet)
546
0
{
547
0
    int32_t idx = _matchFromSet(string, matchSet, true);
548
0
    if(idx >= 0) {
549
0
        return (char16_t *)string + idx;
550
0
    } else {
551
0
        return nullptr;
552
0
    }
553
0
}
554
555
/* Search for a codepoint in a string that matches one of the matchSet codepoints. */
556
U_CAPI int32_t U_EXPORT2
557
u_strcspn(const char16_t *string, const char16_t *matchSet)
558
0
{
559
0
    int32_t idx = _matchFromSet(string, matchSet, true);
560
0
    if(idx >= 0) {
561
0
        return idx;
562
0
    } else {
563
0
        return -idx - 1; /* == u_strlen(string) */
564
0
    }
565
0
}
566
567
/* Search for a codepoint in a string that does not match one of the matchSet codepoints. */
568
U_CAPI int32_t U_EXPORT2
569
u_strspn(const char16_t *string, const char16_t *matchSet)
570
0
{
571
0
    int32_t idx = _matchFromSet(string, matchSet, false);
572
0
    if(idx >= 0) {
573
0
        return idx;
574
0
    } else {
575
0
        return -idx - 1; /* == u_strlen(string) */
576
0
    }
577
0
}
578
579
/* ----- Text manipulation functions --- */
580
581
U_CAPI char16_t* U_EXPORT2
582
u_strtok_r(char16_t* src U_LIFETIME_BOUND, const char16_t* delim, char16_t** saveState)
583
0
{
584
0
    char16_t *tokSource;
585
0
    char16_t *nextToken;
586
0
    uint32_t nonDelimIdx;
587
588
    /* If saveState is nullptr, the user messed up. */
589
0
    if (src != nullptr) {
590
0
        tokSource = src;
591
0
        *saveState = src; /* Set to "src" in case there are no delimiters */
592
0
    }
593
0
    else if (*saveState) {
594
0
        tokSource = *saveState;
595
0
    }
596
0
    else {
597
        /* src == nullptr && *saveState == nullptr */
598
        /* This shouldn't happen. We already finished tokenizing. */
599
0
        return nullptr;
600
0
    }
601
602
    /* Skip initial delimiters */
603
0
    nonDelimIdx = u_strspn(tokSource, delim);
604
0
    tokSource = &tokSource[nonDelimIdx];
605
606
0
    if (*tokSource) {
607
0
        nextToken = u_strpbrk(tokSource, delim);
608
0
        if (nextToken != nullptr) {
609
            /* Create a token */
610
0
            *(nextToken++) = 0;
611
0
            *saveState = nextToken;
612
0
            return tokSource;
613
0
        }
614
0
        else if (*saveState) {
615
            /* Return the last token */
616
0
            *saveState = nullptr;
617
0
            return tokSource;
618
0
        }
619
0
    }
620
0
    else {
621
        /* No tokens were found. Only delimiters were left. */
622
0
        *saveState = nullptr;
623
0
    }
624
0
    return nullptr;
625
0
}
626
627
/* Miscellaneous functions -------------------------------------------------- */
628
629
U_CAPI char16_t* U_EXPORT2
630
u_strcat(char16_t* dst U_LIFETIME_BOUND, const char16_t* src)
631
0
{
632
0
    char16_t *anchor = dst;            /* save a pointer to start of dst */
633
634
0
    while(*dst != 0) {              /* To end of first string          */
635
0
        ++dst;
636
0
    }
637
0
    while((*(dst++) = *(src++)) != 0) {     /* copy string 2 over              */
638
0
    }
639
640
0
    return anchor;
641
0
}
642
643
U_CAPI char16_t* U_EXPORT2
644
u_strncat(char16_t* dst U_LIFETIME_BOUND, const char16_t* src, int32_t n)
645
0
{
646
0
    if(n > 0) {
647
0
        char16_t *anchor = dst;            /* save a pointer to start of dst */
648
649
0
        while(*dst != 0) {              /* To end of first string          */
650
0
            ++dst;
651
0
        }
652
0
        while((*dst = *src) != 0) {     /* copy string 2 over              */
653
0
            ++dst;
654
0
            if(--n == 0) {
655
0
                *dst = 0;
656
0
                break;
657
0
            }
658
0
            ++src;
659
0
        }
660
661
0
        return anchor;
662
0
    } else {
663
0
        return dst;
664
0
    }
665
0
}
666
667
/* ----- Text property functions --- */
668
669
U_CAPI int32_t   U_EXPORT2
670
u_strcmp(const char16_t *s1,
671
    const char16_t *s2)
672
131k
{
673
131k
    char16_t  c1, c2;
674
675
196k
    for(;;) {
676
196k
        c1=*s1++;
677
196k
        c2=*s2++;
678
196k
        if (c1 != c2 || c1 == 0) {
679
131k
            break;
680
131k
        }
681
196k
    }
682
131k
    return (int32_t)c1 - (int32_t)c2;
683
131k
}
684
685
U_CFUNC int32_t U_EXPORT2
686
uprv_strCompare(const char16_t *s1, int32_t length1,
687
                const char16_t *s2, int32_t length2,
688
5.66k
                UBool strncmpStyle, UBool codePointOrder) {
689
5.66k
    const char16_t *start1, *start2, *limit1, *limit2;
690
5.66k
    char16_t c1, c2;
691
692
    /* setup for fix-up */
693
5.66k
    start1=s1;
694
5.66k
    start2=s2;
695
696
    /* compare identical prefixes - they do not need to be fixed up */
697
5.66k
    if(length1<0 && length2<0) {
698
        /* strcmp style, both NUL-terminated */
699
0
        if(s1==s2) {
700
0
            return 0;
701
0
        }
702
703
0
        for(;;) {
704
0
            c1=*s1;
705
0
            c2=*s2;
706
0
            if(c1!=c2) {
707
0
                break;
708
0
            }
709
0
            if(c1==0) {
710
0
                return 0;
711
0
            }
712
0
            ++s1;
713
0
            ++s2;
714
0
        }
715
716
        /* setup for fix-up */
717
0
        limit1=limit2=nullptr;
718
5.66k
    } else if(strncmpStyle) {
719
        /* special handling for strncmp, assume length1==length2>=0 but also check for NUL */
720
0
        if(s1==s2) {
721
0
            return 0;
722
0
        }
723
724
0
        limit1=start1+length1;
725
726
0
        for(;;) {
727
            /* both lengths are same, check only one limit */
728
0
            if(s1==limit1) {
729
0
                return 0;
730
0
            }
731
732
0
            c1=*s1;
733
0
            c2=*s2;
734
0
            if(c1!=c2) {
735
0
                break;
736
0
            }
737
0
            if(c1==0) {
738
0
                return 0;
739
0
            }
740
0
            ++s1;
741
0
            ++s2;
742
0
        }
743
744
        /* setup for fix-up */
745
0
        limit2=start2+length1; /* use length1 here, too, to enforce assumption */
746
5.66k
    } else {
747
        /* memcmp/UnicodeString style, both length-specified */
748
5.66k
        int32_t lengthResult;
749
750
5.66k
        if(length1<0) {
751
0
            length1=u_strlen(s1);
752
0
        }
753
5.66k
        if(length2<0) {
754
0
            length2=u_strlen(s2);
755
0
        }
756
757
        /* limit1=start1+min(length1, length2) */
758
5.66k
        if(length1<length2) {
759
0
            lengthResult=-1;
760
0
            limit1=start1+length1;
761
5.66k
        } else if(length1==length2) {
762
5.66k
            lengthResult=0;
763
5.66k
            limit1=start1+length1;
764
5.66k
        } else /* length1>length2 */ {
765
0
            lengthResult=1;
766
0
            limit1=start1+length2;
767
0
        }
768
769
5.66k
        if(s1==s2) {
770
0
            return lengthResult;
771
0
        }
772
773
16.6k
        for(;;) {
774
            /* check pseudo-limit */
775
16.6k
            if(s1==limit1) {
776
4.38k
                return lengthResult;
777
4.38k
            }
778
779
12.2k
            c1=*s1;
780
12.2k
            c2=*s2;
781
12.2k
            if(c1!=c2) {
782
1.28k
                break;
783
1.28k
            }
784
10.9k
            ++s1;
785
10.9k
            ++s2;
786
10.9k
        }
787
788
        /* setup for fix-up */
789
1.28k
        limit1=start1+length1;
790
1.28k
        limit2=start2+length2;
791
1.28k
    }
792
793
    /* if both values are in or above the surrogate range, fix them up */
794
1.28k
    if(c1>=0xd800 && c2>=0xd800 && codePointOrder) {
795
        /* subtract 0x2800 from BMP code points to make them smaller than supplementary ones */
796
0
        if(
797
0
            (c1<=0xdbff && (s1+1)!=limit1 && U16_IS_TRAIL(*(s1+1))) ||
798
0
            (U16_IS_TRAIL(c1) && start1!=s1 && U16_IS_LEAD(*(s1-1)))
799
0
        ) {
800
            /* part of a surrogate pair, leave >=d800 */
801
0
        } else {
802
            /* BMP code point - may be surrogate code point - make <d800 */
803
0
            c1-=0x2800;
804
0
        }
805
806
0
        if(
807
0
            (c2<=0xdbff && (s2+1)!=limit2 && U16_IS_TRAIL(*(s2+1))) ||
808
0
            (U16_IS_TRAIL(c2) && start2!=s2 && U16_IS_LEAD(*(s2-1)))
809
0
        ) {
810
            /* part of a surrogate pair, leave >=d800 */
811
0
        } else {
812
            /* BMP code point - may be surrogate code point - make <d800 */
813
0
            c2-=0x2800;
814
0
        }
815
0
    }
816
817
    /* now c1 and c2 are in the requested (code unit or code point) order */
818
1.28k
    return (int32_t)c1-(int32_t)c2;
819
5.66k
}
820
821
/*
822
 * Compare two strings as presented by UCharIterators.
823
 * Use code unit or code point order.
824
 * When the function returns, it is undefined where the iterators
825
 * have stopped.
826
 */
827
U_CAPI int32_t U_EXPORT2
828
0
u_strCompareIter(UCharIterator *iter1, UCharIterator *iter2, UBool codePointOrder) {
829
0
    UChar32 c1, c2;
830
831
    /* argument checking */
832
0
    if(iter1==nullptr || iter2==nullptr) {
833
0
        return 0; /* bad arguments */
834
0
    }
835
0
    if(iter1==iter2) {
836
0
        return 0; /* identical iterators */
837
0
    }
838
839
    /* reset iterators to start? */
840
0
    iter1->move(iter1, 0, UITER_START);
841
0
    iter2->move(iter2, 0, UITER_START);
842
843
    /* compare identical prefixes - they do not need to be fixed up */
844
0
    for(;;) {
845
0
        c1=iter1->next(iter1);
846
0
        c2=iter2->next(iter2);
847
0
        if(c1!=c2) {
848
0
            break;
849
0
        }
850
0
        if(c1==-1) {
851
0
            return 0;
852
0
        }
853
0
    }
854
855
    /* if both values are in or above the surrogate range, fix them up */
856
0
    if(c1>=0xd800 && c2>=0xd800 && codePointOrder) {
857
        /* subtract 0x2800 from BMP code points to make them smaller than supplementary ones */
858
0
        if(
859
0
            (c1<=0xdbff && U16_IS_TRAIL(iter1->current(iter1))) ||
860
0
            (U16_IS_TRAIL(c1) && (iter1->previous(iter1), U16_IS_LEAD(iter1->previous(iter1))))
861
0
        ) {
862
            /* part of a surrogate pair, leave >=d800 */
863
0
        } else {
864
            /* BMP code point - may be surrogate code point - make <d800 */
865
0
            c1-=0x2800;
866
0
        }
867
868
0
        if(
869
0
            (c2<=0xdbff && U16_IS_TRAIL(iter2->current(iter2))) ||
870
0
            (U16_IS_TRAIL(c2) && (iter2->previous(iter2), U16_IS_LEAD(iter2->previous(iter2))))
871
0
        ) {
872
            /* part of a surrogate pair, leave >=d800 */
873
0
        } else {
874
            /* BMP code point - may be surrogate code point - make <d800 */
875
0
            c2-=0x2800;
876
0
        }
877
0
    }
878
879
    /* now c1 and c2 are in the requested (code unit or code point) order */
880
0
    return (int32_t)c1-(int32_t)c2;
881
0
}
882
883
#if 0
884
/*
885
 * u_strCompareIter() does not leave the iterators _on_ the different units.
886
 * This is possible but would cost a few extra indirect function calls to back
887
 * up if the last unit (c1 or c2 respectively) was >=0.
888
 *
889
 * Consistently leaving them _behind_ the different units is not an option
890
 * because the current "unit" is the end of the string if that is reached,
891
 * and in such a case the iterator does not move.
892
 * For example, when comparing "ab" with "abc", both iterators rest _on_ the end
893
 * of their strings. Calling previous() on each does not move them to where
894
 * the comparison fails.
895
 *
896
 * So the simplest semantics is to not define where the iterators end up.
897
 *
898
 * The following fragment is part of what would need to be done for backing up.
899
 */
900
void fragment {
901
        /* iff a surrogate is part of a surrogate pair, leave >=d800 */
902
        if(c1<=0xdbff) {
903
            if(!U16_IS_TRAIL(iter1->current(iter1))) {
904
                /* lead surrogate code point - make <d800 */
905
                c1-=0x2800;
906
            }
907
        } else if(c1<=0xdfff) {
908
            int32_t idx=iter1->getIndex(iter1, UITER_CURRENT);
909
            iter1->previous(iter1); /* ==c1 */
910
            if(!U16_IS_LEAD(iter1->previous(iter1))) {
911
                /* trail surrogate code point - make <d800 */
912
                c1-=0x2800;
913
            }
914
            /* go back to behind where the difference is */
915
            iter1->move(iter1, idx, UITER_ZERO);
916
        } else /* 0xe000<=c1<=0xffff */ {
917
            /* BMP code point - make <d800 */
918
            c1-=0x2800;
919
        }
920
}
921
#endif
922
923
U_CAPI int32_t U_EXPORT2
924
u_strCompare(const char16_t *s1, int32_t length1,
925
             const char16_t *s2, int32_t length2,
926
5.66k
             UBool codePointOrder) {
927
    /* argument checking */
928
5.66k
    if(s1==nullptr || length1<-1 || s2==nullptr || length2<-1) {
929
0
        return 0;
930
0
    }
931
5.66k
    return uprv_strCompare(s1, length1, s2, length2, false, codePointOrder);
932
5.66k
}
933
934
/* String compare in code point order - u_strcmp() compares in code unit order. */
935
U_CAPI int32_t U_EXPORT2
936
0
u_strcmpCodePointOrder(const char16_t *s1, const char16_t *s2) {
937
0
    return uprv_strCompare(s1, -1, s2, -1, false, true);
938
0
}
939
940
U_CAPI int32_t   U_EXPORT2
941
u_strncmp(const char16_t  *s1,
942
     const char16_t  *s2,
943
     int32_t     n) 
944
7.31k
{
945
7.31k
    if(n > 0) {
946
7.31k
        int32_t rc;
947
21.8k
        for(;;) {
948
21.8k
            rc = (int32_t)*s1 - (int32_t)*s2;
949
21.8k
            if(rc != 0 || *s1 == 0 || --n == 0) {
950
7.31k
                return rc;
951
7.31k
            }
952
14.5k
            ++s1;
953
14.5k
            ++s2;
954
14.5k
        }
955
7.31k
    } else {
956
0
        return 0;
957
0
    }
958
7.31k
}
959
960
U_CAPI int32_t U_EXPORT2
961
0
u_strncmpCodePointOrder(const char16_t *s1, const char16_t *s2, int32_t n) {
962
0
    return uprv_strCompare(s1, n, s2, n, true, true);
963
0
}
964
965
U_CAPI char16_t* U_EXPORT2
966
u_strcpy(char16_t* dst U_LIFETIME_BOUND, const char16_t* src)
967
2.16M
{
968
2.16M
    char16_t *anchor = dst;            /* save a pointer to start of dst */
969
970
8.66M
    while((*(dst++) = *(src++)) != 0) {     /* copy string 2 over              */
971
6.49M
    }
972
973
2.16M
    return anchor;
974
2.16M
}
975
976
U_CAPI char16_t* U_EXPORT2
977
570k
u_strncpy(char16_t* dst U_LIFETIME_BOUND, const char16_t* src, int32_t n) {
978
570k
    char16_t *anchor = dst;            /* save a pointer to start of dst */
979
980
    /* copy string 2 over */
981
2.08M
    while(n > 0 && (*(dst++) = *(src++)) != 0) {
982
1.50M
        --n;
983
1.50M
    }
984
985
570k
    return anchor;
986
570k
}
987
988
U_CAPI int32_t   U_EXPORT2
989
u_strlen(const char16_t *s)
990
22.3M
{
991
#if U_SIZEOF_WCHAR_T == U_SIZEOF_UCHAR
992
    return (int32_t)uprv_wcslen((const wchar_t *)s);
993
#else
994
22.3M
    const char16_t *t = s;
995
216M
    while(*t != 0) {
996
193M
      ++t;
997
193M
    }
998
22.3M
    return t - s;
999
22.3M
#endif
1000
22.3M
}
1001
1002
U_CAPI int32_t U_EXPORT2
1003
6.51M
u_countChar32(const char16_t *s, int32_t length) {
1004
6.51M
    int32_t count;
1005
1006
6.51M
    if(s==nullptr || length<-1) {
1007
0
        return 0;
1008
0
    }
1009
1010
6.51M
    count=0;
1011
6.51M
    if(length>=0) {
1012
253M
        while(length>0) {
1013
246M
            ++count;
1014
246M
            if(U16_IS_LEAD(*s) && length>=2 && U16_IS_TRAIL(*(s+1))) {
1015
4.45M
                s+=2;
1016
4.45M
                length-=2;
1017
242M
            } else {
1018
242M
                ++s;
1019
242M
                --length;
1020
242M
            }
1021
246M
        }
1022
6.51M
    } else /* length==-1 */ {
1023
0
        char16_t c;
1024
1025
0
        for(;;) {
1026
0
            if((c=*s++)==0) {
1027
0
                break;
1028
0
            }
1029
0
            ++count;
1030
1031
            /*
1032
             * sufficient to look ahead one because of UTF-16;
1033
             * safe to look ahead one because at worst that would be the terminating NUL
1034
             */
1035
0
            if(U16_IS_LEAD(c) && U16_IS_TRAIL(*s)) {
1036
0
                ++s;
1037
0
            }
1038
0
        }
1039
0
    }
1040
6.51M
    return count;
1041
6.51M
}
1042
1043
U_CAPI UBool U_EXPORT2
1044
0
u_strHasMoreChar32Than(const char16_t *s, int32_t length, int32_t number) {
1045
1046
0
    if(number<0) {
1047
0
        return true;
1048
0
    }
1049
0
    if(s==nullptr || length<-1) {
1050
0
        return false;
1051
0
    }
1052
1053
0
    if(length==-1) {
1054
        /* s is NUL-terminated */
1055
0
        char16_t c;
1056
1057
        /* count code points until they exceed */
1058
0
        for(;;) {
1059
0
            if((c=*s++)==0) {
1060
0
                return false;
1061
0
            }
1062
0
            if(number==0) {
1063
0
                return true;
1064
0
            }
1065
0
            if(U16_IS_LEAD(c) && U16_IS_TRAIL(*s)) {
1066
0
                ++s;
1067
0
            }
1068
0
            --number;
1069
0
        }
1070
0
    } else {
1071
        /* length>=0 known */
1072
0
        const char16_t *limit;
1073
0
        int32_t maxSupplementary;
1074
1075
        /* s contains at least (length+1)/2 code points: <=2 UChars per cp */
1076
0
        if(((length+1)/2)>number) {
1077
0
            return true;
1078
0
        }
1079
1080
        /* check if s does not even contain enough UChars */
1081
0
        maxSupplementary=length-number;
1082
0
        if(maxSupplementary<=0) {
1083
0
            return false;
1084
0
        }
1085
        /* there are maxSupplementary=length-number more UChars than asked-for code points */
1086
1087
        /*
1088
         * count code points until they exceed and also check that there are
1089
         * no more than maxSupplementary supplementary code points (char16_t pairs)
1090
         */
1091
0
        limit=s+length;
1092
0
        for(;;) {
1093
0
            if(s==limit) {
1094
0
                return false;
1095
0
            }
1096
0
            if(number==0) {
1097
0
                return true;
1098
0
            }
1099
0
            if(U16_IS_LEAD(*s++) && s!=limit && U16_IS_TRAIL(*s)) {
1100
0
                ++s;
1101
0
                if(--maxSupplementary<=0) {
1102
                    /* too many pairs - too few code points */
1103
0
                    return false;
1104
0
                }
1105
0
            }
1106
0
            --number;
1107
0
        }
1108
0
    }
1109
0
}
1110
1111
U_CAPI char16_t* U_EXPORT2
1112
66.8M
u_memcpy(char16_t* dest U_LIFETIME_BOUND, const char16_t* src, int32_t count) {
1113
66.8M
    if(count > 0) {
1114
66.8M
        uprv_memcpy(dest, src, (size_t)count*U_SIZEOF_UCHAR);
1115
66.8M
    }
1116
66.8M
    return dest;
1117
66.8M
}
1118
1119
U_CAPI char16_t* U_EXPORT2
1120
0
u_memmove(char16_t* dest U_LIFETIME_BOUND, const char16_t* src, int32_t count) {
1121
0
    if(count > 0) {
1122
0
        uprv_memmove(dest, src, (size_t)count*U_SIZEOF_UCHAR);
1123
0
    }
1124
0
    return dest;
1125
0
}
1126
1127
U_CAPI char16_t* U_EXPORT2
1128
0
u_memset(char16_t* dest U_LIFETIME_BOUND, char16_t c, int32_t count) {
1129
0
    if(count > 0) {
1130
0
        char16_t *ptr = dest;
1131
0
        char16_t *limit = dest + count;
1132
1133
0
        while (ptr < limit) {
1134
0
            *(ptr++) = c;
1135
0
        }
1136
0
    }
1137
0
    return dest;
1138
0
}
1139
1140
U_CAPI int32_t U_EXPORT2
1141
1.04M
u_memcmp(const char16_t *buf1, const char16_t *buf2, int32_t count) {
1142
1.04M
    if(count > 0) {
1143
1.04M
        const char16_t *limit = buf1 + count;
1144
1.04M
        int32_t result;
1145
1146
3.36M
        while (buf1 < limit) {
1147
3.22M
            result = (int32_t)(uint16_t)*buf1 - (int32_t)(uint16_t)*buf2;
1148
3.22M
            if (result != 0) {
1149
900k
                return result;
1150
900k
            }
1151
2.32M
            buf1++;
1152
2.32M
            buf2++;
1153
2.32M
        }
1154
1.04M
    }
1155
143k
    return 0;
1156
1.04M
}
1157
1158
U_CAPI int32_t U_EXPORT2
1159
0
u_memcmpCodePointOrder(const char16_t *s1, const char16_t *s2, int32_t count) {
1160
0
    return uprv_strCompare(s1, count, s2, count, false, true);
1161
0
}
1162
1163
/* u_unescape & support fns ------------------------------------------------- */
1164
1165
/* This map must be in ASCENDING ORDER OF THE ESCAPE CODE */
1166
static const char16_t UNESCAPE_MAP[] = {
1167
    /*"   0x22, 0x22 */
1168
    /*'   0x27, 0x27 */
1169
    /*?   0x3F, 0x3F */
1170
    /*\   0x5C, 0x5C */
1171
    /*a*/ 0x61, 0x07,
1172
    /*b*/ 0x62, 0x08,
1173
    /*e*/ 0x65, 0x1b,
1174
    /*f*/ 0x66, 0x0c,
1175
    /*n*/ 0x6E, 0x0a,
1176
    /*r*/ 0x72, 0x0d,
1177
    /*t*/ 0x74, 0x09,
1178
    /*v*/ 0x76, 0x0b
1179
};
1180
enum { UNESCAPE_MAP_LENGTH = UPRV_LENGTHOF(UNESCAPE_MAP) };
1181
1182
/* Convert one octal digit to a numeric value 0..7, or -1 on failure */
1183
244k
static int32_t _digit8(char16_t c) {
1184
244k
    if (c >= u'0' && c <= u'7') {
1185
10.7k
        return c - u'0';
1186
10.7k
    }
1187
234k
    return -1;
1188
244k
}
1189
1190
/* Convert one hex digit to a numeric value 0..F, or -1 on failure */
1191
1.05M
static int32_t _digit16(char16_t c) {
1192
1.05M
    if (c >= u'0' && c <= u'9') {
1193
747k
        return c - u'0';
1194
747k
    }
1195
310k
    if (c >= u'A' && c <= u'F') {
1196
21.8k
        return c - (u'A' - 10);
1197
21.8k
    }
1198
289k
    if (c >= u'a' && c <= u'f') {
1199
283k
        return c - (u'a' - 10);
1200
283k
    }
1201
6.08k
    return -1;
1202
289k
}
1203
1204
/* Parse a single escape sequence.  Although this method deals in
1205
 * UChars, it does not use C++ or UnicodeString.  This allows it to
1206
 * be used from C contexts. */
1207
U_CAPI UChar32 U_EXPORT2
1208
u_unescapeAt(UNESCAPE_CHAR_AT charAt,
1209
             int32_t *offset,
1210
             int32_t length,
1211
503k
             void *context) {
1212
1213
503k
    int32_t start = *offset;
1214
503k
    UChar32 c;
1215
503k
    UChar32 result = 0;
1216
503k
    int8_t n = 0;
1217
503k
    int8_t minDig = 0;
1218
503k
    int8_t maxDig = 0;
1219
503k
    int8_t bitsPerDigit = 4; 
1220
503k
    int32_t dig;
1221
503k
    UBool braces = false;
1222
1223
    /* Check that offset is in range */
1224
503k
    if (*offset < 0 || *offset >= length) {
1225
84
        goto err;
1226
84
    }
1227
1228
    /* Fetch first char16_t after '\\' */
1229
502k
    c = charAt((*offset)++, context);
1230
1231
    /* Convert hexadecimal and octal escapes */
1232
502k
    switch (c) {
1233
257k
    case u'u':
1234
257k
        minDig = maxDig = 4;
1235
257k
        break;
1236
2.27k
    case u'U':
1237
2.27k
        minDig = maxDig = 8;
1238
2.27k
        break;
1239
8.67k
    case u'x':
1240
8.67k
        minDig = 1;
1241
8.67k
        if (*offset < length && charAt(*offset, context) == u'{') {
1242
854
            ++(*offset);
1243
854
            braces = true;
1244
854
            maxDig = 8;
1245
7.82k
        } else {
1246
7.82k
            maxDig = 2;
1247
7.82k
        }
1248
8.67k
        break;
1249
234k
    default:
1250
234k
        dig = _digit8(c);
1251
234k
        if (dig >= 0) {
1252
8.37k
            minDig = 1;
1253
8.37k
            maxDig = 3;
1254
8.37k
            n = 1; /* Already have first octal digit */
1255
8.37k
            bitsPerDigit = 3;
1256
8.37k
            result = dig;
1257
8.37k
        }
1258
234k
        break;
1259
502k
    }
1260
502k
    if (minDig != 0) {
1261
1.33M
        while (*offset < length && n < maxDig) {
1262
1.06M
            c = charAt(*offset, context);
1263
1.06M
            dig = (bitsPerDigit == 3) ? _digit8(c) : _digit16(c);
1264
1.06M
            if (dig < 0) {
1265
13.7k
                break;
1266
13.7k
            }
1267
1.05M
            result = (result << bitsPerDigit) | dig;
1268
1.05M
            ++(*offset);
1269
1.05M
            ++n;
1270
1.05M
        }
1271
276k
        if (n < minDig) {
1272
1.20k
            goto err;
1273
1.20k
        }
1274
275k
        if (braces) {
1275
841
            if (c != u'}') {
1276
84
                goto err;
1277
84
            }
1278
757
            ++(*offset);
1279
757
        }
1280
275k
        if (result < 0 || result >= 0x110000) {
1281
484
            goto err;
1282
484
        }
1283
        /* If an escape sequence specifies a lead surrogate, see if
1284
         * there is a trail surrogate after it, either as an escape or
1285
         * as a literal.  If so, join them up into a supplementary.
1286
         */
1287
274k
        if (*offset < length && U16_IS_LEAD(result)) {
1288
175k
            int32_t ahead = *offset + 1;
1289
175k
            c = charAt(*offset, context);
1290
175k
            if (c == u'\\' && ahead < length) {
1291
                // Calling ourselves recursively may cause a stack overflow if
1292
                // we have repeated escaped lead surrogates.
1293
                // Limit the length to 11 ("x{0000DFFF}") after ahead.
1294
170k
                int32_t tailLimit = ahead + 11;
1295
170k
                if (tailLimit > length) {
1296
87.8k
                    tailLimit = length;
1297
87.8k
                }
1298
170k
                c = u_unescapeAt(charAt, &ahead, tailLimit, context);
1299
170k
            }
1300
175k
            if (U16_IS_TRAIL(c)) {
1301
644
                *offset = ahead;
1302
644
                result = U16_GET_SUPPLEMENTARY(result, c);
1303
644
            }
1304
175k
        }
1305
274k
        return result;
1306
275k
    }
1307
1308
    /* Convert C-style escapes in table */
1309
590k
    for (int32_t i=0; i<UNESCAPE_MAP_LENGTH; i+=2) {
1310
568k
        if (c == UNESCAPE_MAP[i]) {
1311
2.55k
            return UNESCAPE_MAP[i+1];
1312
565k
        } else if (c < UNESCAPE_MAP[i]) {
1313
201k
            break;
1314
201k
        }
1315
568k
    }
1316
1317
    /* Map \cX to control-X: X & 0x1F */
1318
223k
    if (c == u'c' && *offset < length) {
1319
86.7k
        c = charAt((*offset)++, context);
1320
86.7k
        if (U16_IS_LEAD(c) && *offset < length) {
1321
74.1k
            char16_t c2 = charAt(*offset, context);
1322
74.1k
            if (U16_IS_TRAIL(c2)) {
1323
713
                ++(*offset);
1324
713
                c = U16_GET_SUPPLEMENTARY(c, c2);
1325
713
            }
1326
74.1k
        }
1327
86.7k
        return 0x1F & c;
1328
86.7k
    }
1329
1330
    /* If no special forms are recognized, then consider
1331
     * the backslash to generically escape the next character.
1332
     * Deal with surrogate pairs. */
1333
137k
    if (U16_IS_LEAD(c) && *offset < length) {
1334
5.95k
        char16_t c2 = charAt(*offset, context);
1335
5.95k
        if (U16_IS_TRAIL(c2)) {
1336
2.47k
            ++(*offset);
1337
2.47k
            return U16_GET_SUPPLEMENTARY(c, c2);
1338
2.47k
        }
1339
5.95k
    }
1340
134k
    return c;
1341
1342
1.85k
 err:
1343
    /* Invalid escape sequence */
1344
1.85k
    *offset = start; /* Reset to initial value */
1345
1.85k
    return (UChar32)0xFFFFFFFF;
1346
137k
}
1347
1348
/* u_unescapeAt() callback to return a char16_t from a char* */
1349
static char16_t U_CALLCONV
1350
0
_charPtr_charAt(int32_t offset, void *context) {
1351
0
    char16_t c16;
1352
    /* It would be more efficient to access the invariant tables
1353
     * directly but there is no API for that. */
1354
0
    u_charsToUChars(static_cast<char*>(context) + offset, &c16, 1);
1355
0
    return c16;
1356
0
}
1357
1358
/* Append an escape-free segment of the text; used by u_unescape() */
1359
static void _appendUChars(char16_t *dest, int32_t destCapacity,
1360
0
                          const char *src, int32_t srcLen) {
1361
0
    if (destCapacity < 0) {
1362
0
        destCapacity = 0;
1363
0
    }
1364
0
    if (srcLen > destCapacity) {
1365
0
        srcLen = destCapacity;
1366
0
    }
1367
0
    u_charsToUChars(src, dest, srcLen);
1368
0
}
1369
1370
/* Do an invariant conversion of char* -> char16_t*, with escape parsing */
1371
U_CAPI int32_t U_EXPORT2
1372
0
u_unescape(const char *src, char16_t *dest, int32_t destCapacity) {
1373
0
    const char *segment = src;
1374
0
    int32_t i = 0;
1375
0
    char c;
1376
1377
0
    while ((c=*src) != 0) {
1378
        /* '\\' intentionally written as compiler-specific
1379
         * character constant to correspond to compiler-specific
1380
         * char* constants. */
1381
0
        if (c == '\\') {
1382
0
            int32_t lenParsed = 0;
1383
0
            UChar32 c32;
1384
0
            if (src != segment) {
1385
0
                if (dest != nullptr) {
1386
0
                    _appendUChars(dest + i, destCapacity - i,
1387
0
                                  segment, (int32_t)(src - segment));
1388
0
                }
1389
0
                i += (int32_t)(src - segment);
1390
0
            }
1391
0
            ++src; /* advance past '\\' */
1392
0
            c32 = u_unescapeAt(_charPtr_charAt, &lenParsed, (int32_t)uprv_strlen(src), const_cast<char*>(src));
1393
0
            if (lenParsed == 0) {
1394
0
                goto err;
1395
0
            }
1396
0
            src += lenParsed; /* advance past escape seq. */
1397
0
            if (dest != nullptr && U16_LENGTH(c32) <= (destCapacity - i)) {
1398
0
                U16_APPEND_UNSAFE(dest, i, c32);
1399
0
            } else {
1400
0
                i += U16_LENGTH(c32);
1401
0
            }
1402
0
            segment = src;
1403
0
        } else {
1404
0
            ++src;
1405
0
        }
1406
0
    }
1407
0
    if (src != segment) {
1408
0
        if (dest != nullptr) {
1409
0
            _appendUChars(dest + i, destCapacity - i,
1410
0
                          segment, (int32_t)(src - segment));
1411
0
        }
1412
0
        i += (int32_t)(src - segment);
1413
0
    }
1414
0
    if (dest != nullptr && i < destCapacity) {
1415
0
        dest[i] = 0;
1416
0
    }
1417
0
    return i;
1418
1419
0
 err:
1420
0
    if (dest != nullptr && destCapacity > 0) {
1421
0
        *dest = 0;
1422
0
    }
1423
0
    return 0;
1424
0
}
1425
1426
/* NUL-termination of strings ----------------------------------------------- */
1427
1428
/**
1429
 * NUL-terminate a string no matter what its type.
1430
 * Set warning and error codes accordingly.
1431
 */
1432
3.60M
#define __TERMINATE_STRING(dest, destCapacity, length, pErrorCode) UPRV_BLOCK_MACRO_BEGIN { \
1433
3.60M
    if(pErrorCode!=nullptr && U_SUCCESS(*pErrorCode)) {                    \
1434
3.59M
        /* not a public function, so no complete argument checking */   \
1435
3.59M
                                                                        \
1436
3.59M
        if(length<0) {                                                  \
1437
0
            /* assume that the caller handles this */                   \
1438
3.59M
        } else if(length<destCapacity) {                                \
1439
3.50M
            /* NUL-terminate the string, the NUL fits */                \
1440
3.50M
            dest[length]=0;                                             \
1441
3.50M
            /* unset the not-terminated warning but leave all others */ \
1442
3.50M
            if(*pErrorCode==U_STRING_NOT_TERMINATED_WARNING) {          \
1443
30
                *pErrorCode=U_ZERO_ERROR;                               \
1444
30
            }                                                           \
1445
3.50M
        } else if(length==destCapacity) {                               \
1446
80.4k
            /* unable to NUL-terminate, but the string itself fit - set a warning code */ \
1447
80.4k
            *pErrorCode=U_STRING_NOT_TERMINATED_WARNING;                \
1448
80.4k
        } else /* length>destCapacity */ {                              \
1449
9.34k
            /* even the string itself did not fit - set an error code */ \
1450
9.34k
            *pErrorCode=U_BUFFER_OVERFLOW_ERROR;                        \
1451
9.34k
        }                                                               \
1452
3.59M
    } \
1453
3.60M
} UPRV_BLOCK_MACRO_END
1454
1455
U_CAPI char16_t U_EXPORT2
1456
0
u_asciiToUpper(char16_t c) {
1457
0
    if (u'a' <= c && c <= u'z') {
1458
0
        c = c + u'A' - u'a';
1459
0
    }
1460
0
    return c;
1461
0
}
1462
1463
U_CAPI int32_t U_EXPORT2
1464
1.33M
u_terminateUChars(char16_t *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) {
1465
1.33M
    __TERMINATE_STRING(dest, destCapacity, length, pErrorCode);
1466
1.33M
    return length;
1467
1.33M
}
1468
1469
U_CAPI int32_t U_EXPORT2
1470
2.27M
u_terminateChars(char *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) {
1471
2.27M
    __TERMINATE_STRING(dest, destCapacity, length, pErrorCode);
1472
2.27M
    return length;
1473
2.27M
}
1474
1475
U_CAPI int32_t U_EXPORT2
1476
0
u_terminateUChar32s(UChar32 *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) {
1477
0
    __TERMINATE_STRING(dest, destCapacity, length, pErrorCode);
1478
0
    return length;
1479
0
}
1480
1481
U_CAPI int32_t U_EXPORT2
1482
0
u_terminateWChars(wchar_t *dest, int32_t destCapacity, int32_t length, UErrorCode *pErrorCode) {
1483
0
    __TERMINATE_STRING(dest, destCapacity, length, pErrorCode);
1484
0
    return length;
1485
0
}
1486
1487
// Compute the hash code for a string -------------------------------------- ***
1488
1489
// Moved here from uhash.c so that UnicodeString::hashCode() does not depend
1490
// on UHashtable code.
1491
1492
/*
1493
  Compute the hash by iterating sparsely over about 32 (up to 63)
1494
  characters spaced evenly through the string.  For each character,
1495
  multiply the previous hash value by a prime number and add the new
1496
  character in, like a linear congruential random number generator,
1497
  producing a pseudorandom deterministic value well distributed over
1498
  the output range. [LIU]
1499
*/
1500
1501
53.6M
#define STRING_HASH(TYPE, STR, STRLEN, DEREF) UPRV_BLOCK_MACRO_BEGIN { \
1502
53.6M
    uint32_t hash = 0;                        \
1503
53.6M
    const TYPE *p = (const TYPE*) STR;        \
1504
53.6M
    if (p != nullptr) {                          \
1505
53.6M
        int32_t len = (int32_t)(STRLEN);      \
1506
53.6M
        int32_t inc = ((len - 32) / 32) + 1;  \
1507
53.6M
        const TYPE *limit = p + len;          \
1508
353M
        while (p<limit) {                     \
1509
300M
            hash = (hash * 37) + DEREF;       \
1510
300M
            p += inc;                         \
1511
300M
        }                                     \
1512
53.6M
    }                                         \
1513
53.6M
    return static_cast<int32_t>(hash);        \
1514
53.6M
} UPRV_BLOCK_MACRO_END
1515
1516
/* Used by UnicodeString to compute its hashcode - Not public API. */
1517
U_CAPI int32_t U_EXPORT2
1518
10.0M
ustr_hashUCharsN(const char16_t *str, int32_t length) {
1519
10.0M
    STRING_HASH(char16_t, str, length, *p);
1520
10.0M
}
1521
1522
U_CAPI int32_t U_EXPORT2
1523
10.7M
ustr_hashCharsN(const char *str, int32_t length) {
1524
10.7M
    STRING_HASH(uint8_t, str, length, *p);
1525
10.7M
}
1526
1527
U_CAPI int32_t U_EXPORT2
1528
32.8M
ustr_hashICharsN(const char *str, int32_t length) {
1529
32.8M
    STRING_HASH(char, str, length, (uint8_t)uprv_tolower(*p));
1530
32.8M
}