Coverage Report

Created: 2026-09-28 06:10

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/icu/icu4c/source/common/unames.cpp
Line
Count
Source
1
// © 2016 and later: Unicode, Inc. and others.
2
// License & terms of use: http://www.unicode.org/copyright.html
3
/*
4
******************************************************************************
5
*
6
*   Copyright (C) 1999-2014, International Business Machines
7
*   Corporation and others.  All Rights Reserved.
8
*
9
******************************************************************************
10
*   file name:  unames.c
11
*   encoding:   UTF-8
12
*   tab size:   8 (not used)
13
*   indentation:4
14
*
15
*   created on: 1999oct04
16
*   created by: Markus W. Scherer
17
*/
18
19
#include "unicode/utypes.h"
20
#include "unicode/putil.h"
21
#include "unicode/uchar.h"
22
#include "unicode/udata.h"
23
#include "unicode/utf.h"
24
#include "unicode/utf16.h"
25
#include "uassert.h"
26
#include "ustr_imp.h"
27
#include "umutex.h"
28
#include "cmemory.h"
29
#include "cstring.h"
30
#include "ucln_cmn.h"
31
#include "udataswp.h"
32
#include "uprops.h"
33
34
U_NAMESPACE_BEGIN
35
36
/* prototypes ------------------------------------------------------------- */
37
38
static const char DATA_NAME[] = "unames";
39
static const char DATA_TYPE[] = "icu";
40
41
1.64G
#define GROUP_SHIFT 5
42
1.62G
#define LINES_PER_GROUP (1L<<GROUP_SHIFT)
43
1.25G
#define GROUP_MASK (LINES_PER_GROUP-1)
44
45
/*
46
 * This struct was replaced by explicitly accessing equivalent
47
 * fields from triples of uint16_t.
48
 * The Group struct was padded to 8 bytes on compilers for early ARM CPUs,
49
 * which broke the assumption that sizeof(Group)==6 and that the ++ operator
50
 * would advance by 6 bytes (3 uint16_t).
51
 *
52
 * We can't just change the data structure because it's loaded from a data file,
53
 * and we don't want to make it less compact, so we changed the access code.
54
 *
55
 * For details see ICU tickets 6331 and 6008.
56
typedef struct {
57
    uint16_t groupMSB,
58
             offsetHigh, offsetLow; / * avoid padding * /
59
} Group;
60
 */
61
enum {
62
    GROUP_MSB,
63
    GROUP_OFFSET_HIGH,
64
    GROUP_OFFSET_LOW,
65
    GROUP_LENGTH
66
};
67
68
/*
69
 * Get the 32-bit group offset.
70
 * @param group (const uint16_t *) pointer to a Group triple of uint16_t
71
 * @return group offset (int32_t)
72
 */
73
19.6M
#define GET_GROUP_OFFSET(group) ((int32_t)(group)[GROUP_OFFSET_HIGH]<<16|(group)[GROUP_OFFSET_LOW])
74
75
19.6M
#define NEXT_GROUP(group) ((group)+GROUP_LENGTH)
76
541
#define PREV_GROUP(group) ((group)-GROUP_LENGTH)
77
78
typedef struct {
79
    uint32_t start, end;
80
    uint8_t type, variant;
81
    uint16_t size;
82
} AlgorithmicRange;
83
84
typedef struct {
85
    uint32_t tokenStringOffset, groupsOffset, groupStringOffset, algNamesOffset;
86
} UCharNames;
87
88
namespace {
89
// For a character name `name`, a query string `query` matches `name` under UAX44-LM2 if and only if
90
// the following returns true:
91
//     auto matcher = CharacterNameQuery(query).matcher();
92
//     for (const char c : name) {
93
//         if (!matcher.consistentWith(c)) {
94
//             return false;
95
//         }
96
//     }
97
//     return matcher.matches();
98
// `name` must be an exact character name, in particular, all uppercase, with no underscores, no
99
// double hyphens, no isolated hyphens.  There are no constraints on `query`; for instance, this
100
// class will match name="ZANABAZAR SQUARE LETTER -A" with query="Zanabazar-square_letter_-A".
101
// `Matcher::consistentWith` is not retryable: once it returns false, the `Matcher` should not be
102
// used again.
103
class CharacterNameQuery {
104
  public:
105
    class Matcher {
106
      public:
107
        // `query` must outlive the constructed object.
108
        Matcher(const CharacterNameQuery& query)
109
628M
            : query(query), skeletonIterator(query.skeleton.begin()) {}
110
111
574M
        bool consistentWith(char c) {
112
            // Instead of constructing a skeleton from the name as in the constructor of
113
            // CharacterNameQuery, we check character-by-character if the skeleton we would
114
            // construct from the name would be consistent with `query->skeleton`.
115
            // We require that this be called with characters from an actual character name, so we
116
            // need not worry about case or underscores here.
117
118
574M
            if (skeletonIterator == query.skeleton.end()) {
119
468k
                return false;
120
468k
            }
121
573M
            if (c == ' ') {
122
                // The last hyphen was word-final; check that there is a corresponding significant
123
                // hyphen in the skeleton.  A hyphen in a character name cannot be both word-final
124
                // and word-initial by rule R3 in
125
                // https://unicode.org/versions/Unicode17.0.0/core-spec/chapter-4/#G135165, so we do
126
                // not have to worry about checking the skeleton for the same significant hyphen
127
                // twice.
128
3.24k
                if (lastChar == '-') {
129
0
                    if (*skeletonIterator++ != lastChar) {
130
0
                        return false;
131
0
                    }
132
0
                }
133
3.24k
                lastChar = c;
134
3.24k
                return true;
135
573M
            } else if (c == '-') {
136
6.15k
                if (lastChar == ' ' ||
137
6.15k
                    (query.is1180 && skeletonIterator == query.skeleton.end() - 2)) {
138
                    // If lastChar == ' ', this is a word-initial hyphen, so we know it is
139
                    // significant.  Check that we are expecting it.
140
                    // If we are looking for U+1180 and we matched everything but the trailing -E,
141
                    // this could be that hyphen; move past it.  This could turn out to be a
142
                    // different character name if there is something other than E afterwards, in
143
                    // which case we should in principle have ignored the hyphen; but then since the
144
                    // suffixes will differ, we will return false anyway.
145
                    // For example, when searching for HANGUL JUNGSEONG O-E and checking consistency
146
                    // with HANGUL JUNGSEONG O-U, if we computed both skeleta and compared them,
147
                    // the skeleta would be HANGULJUNGSEONGO-E and HANGULJUNGSEONGOU, and the
148
                    // comparison would fail on - vs. U.
149
                    // Here, while feeding HANGUL JUNGSEONG O-U character-by-character, we assume
150
                    // its skeleton would be HANGULJUNGSEONGO-, and fail on E vs. U.
151
0
                    if (*skeletonIterator++ != c) {
152
0
                        return false;
153
0
                    }
154
0
                }
155
                // If lastChar is not ' ', we do not know whether this hyphen is word-final, so we
156
                // cannot check against the skeleton.
157
6.15k
                lastChar = c;
158
6.15k
                return true;
159
573M
            } else {
160
573M
                if (*skeletonIterator++ != c) {
161
552M
                    return false;
162
552M
                }
163
20.7M
                lastChar = c;
164
20.7M
                return true;
165
573M
            }
166
573M
        }
167
168
544M
        bool consistentWith(const std::string_view substring) {
169
564M
            for (const char c : substring) {
170
564M
                if (!consistentWith(c)) {
171
544M
                    return false;
172
544M
                }
173
564M
            }
174
4.61k
            return true;
175
544M
        }
176
177
75.5M
        bool matches() {
178
            // If a character name could end with a HYPHEN-MINUS, that HYPHEN-MINUS would be
179
            // significant: we would need to check that if lastChar == '-',
180
            // *skeletonIterator == '-', and then advance skeletonIterator.
181
            // However, this is disallowed by rule R3 in
182
            // https://unicode.org/versions/Unicode17.0.0/core-spec/chapter-4/#G135165.
183
75.5M
            U_ASSERT(lastChar != '-');
184
75.5M
            return skeletonIterator == query.skeleton.end();
185
75.5M
        }
186
187
0
        std::string_view remainingSignificantCharacters() const {
188
0
            return query.skeleton.substr(skeletonIterator - query.skeleton.begin());
189
0
        }
190
191
      private:
192
        const CharacterNameQuery& query;
193
        std::string_view::const_iterator skeletonIterator;
194
        // Initialized to ' ' so that a leading hyphen is treated like a hyphen that follows a
195
        // space (non-medial and thus not ignorable).  There are in fact no leading hyphens in
196
        // character names by rule R3 in
197
        // https://unicode.org/versions/Unicode17.0.0/core-spec/chapter-4/#G135165, and technically
198
        // we do not read the value of this variable before writing to it because the first call to
199
        // consistentWith never has c=' ' either by R4, but it seems cleanest to initialize lastChar
200
        // nonetheless.
201
        char lastChar = ' ';
202
    };
203
204
17.6k
    explicit CharacterNameQuery(std::string_view query) {
205
        // Construct a skeleton obtained by
206
        // 1. removing medial hyphens (except the one in the name of U+1180);
207
        // 2. removing spaces and underscores;
208
        // 3. uppercasing,
209
        // as described in https://www.unicode.org/reports/tr44/#UAX44-LM2.
210
        // We do all three in a single pass.
211
17.6k
        char *skeletonLimit = skeletonData;
212
67.4k
        for (std::size_t i = 0; i < query.length(); ++i) {
213
49.8k
            U_ASSERT(skeletonLimit < skeletonData + sizeof(skeletonData));
214
49.8k
            if (skeletonLimit >= skeletonData + sizeof(skeletonData)) {
215
                // The caller should limit the query length appropriately; if they do not, assert,
216
                // and if assertions are disabled, query for the empty string (which will quickly
217
                // find nothing).
218
0
                skeletonLimit = skeletonData;
219
0
                break;
220
0
            }
221
49.8k
            if (query[i] == ' ' || query[i] == '_') {
222
3.27k
                continue;
223
3.27k
            }
224
46.5k
            if (query[i] == '-') {
225
4.31k
                bool isMedial;
226
4.31k
                bool is1180MedialHyphen = false;
227
4.31k
                if (i == 0 || i == query.length() - 1) {
228
183
                    isMedial = false;
229
4.13k
                } else {
230
4.13k
                    isMedial = isASCIILetterOrDigit(query[i - 1]) && isASCIILetterOrDigit(query[i + 1]);
231
4.13k
                }
232
                // A medial hyphen is the hyphen in the name of U+1180 HANGUL JUNGSEONG O-E if what
233
                // comes before skeletonizes to HANGULJUNGSEONGO and what comes after skeletonizes
234
                // to E.
235
4.31k
                if (isMedial && uprv_toupper(query[i + 1]) == 'E' &&
236
439
                    std::string_view(skeletonData, skeletonLimit - skeletonData) == "HANGULJUNGSEONGO") {
237
0
                    is1180MedialHyphen = true;
238
                    // If there is anything significant after the E, the part of the name after the
239
                    // hyphen does not skeletonize to E, and thus this is not U+1180.
240
                    // There can be no hyphens there: if there is one, it is either non-medial, or
241
                    // there is another letter beyond the E.  The insignificant characters are thus
242
                    // only spaces and underscores.
243
0
                    for (std::size_t j = i + 2; j < query.length(); ++j) {
244
0
                        if (query[j] != ' ' && query[j] != '_') {
245
0
                            is1180MedialHyphen = false;
246
0
                        }
247
0
                    }
248
0
                }
249
4.31k
                if (!isMedial || is1180MedialHyphen) {
250
1.99k
                    *skeletonLimit++ = query[i];
251
1.99k
                }
252
42.2k
            } else {
253
42.2k
                *skeletonLimit++ = uprv_toupper(query[i]);
254
42.2k
            }
255
46.5k
        }
256
17.6k
        skeleton = std::string_view(skeletonData, skeletonLimit - skeletonData);
257
        // This can be true even if we never went through the is1180MedialHyphen path, e.g., if the
258
        // query was "HANGUL JUNGSEONG O -E".
259
17.6k
        is1180 = skeleton == "HANGULJUNGSEONGO-E";
260
17.6k
    }
261
262
628M
    Matcher matcher() const {
263
628M
        return Matcher(*this);
264
628M
    }
265
266
private:
267
6.97k
    static bool isASCIILetterOrDigit(const char c) {
268
6.97k
        return uprv_isASCIILetter(c) || (c >= '0' && c <= '9');
269
6.97k
    }
270
271
    char skeletonData[120];
272
    std::string_view skeleton;
273
    bool is1180;
274
};
275
} // namespace
276
277
/*
278
 * Get the groups table from a UCharNames struct.
279
 * The groups table consists of one uint16_t groupCount followed by
280
 * groupCount groups. Each group is a triple of uint16_t, see GROUP_LENGTH
281
 * and the comment for the old struct Group above.
282
 *
283
 * @param names (const UCharNames *) pointer to the UCharNames indexes
284
 * @return (const uint16_t *) pointer to the groups table
285
 */
286
35.2k
#define GET_GROUPS(names) (const uint16_t *)((const char *)names+names->groupsOffset)
287
288
typedef struct {
289
    const CharacterNameQuery *otherName;
290
    UChar32 code;
291
} FindName;
292
293
20.1M
#define DO_FIND_NAME nullptr
294
295
static UDataMemory *uCharNamesData=nullptr;
296
static UCharNames *uCharNames=nullptr;
297
static icu::UInitOnce gCharNamesInitOnce {};
298
299
/*
300
 * Maximum length of character names (regular & 1.0).
301
 */
302
static int32_t gMaxNameLength=0;
303
304
/*
305
 * Set of chars used in character names (regular & 1.0).
306
 * Chars are platform-dependent (can be EBCDIC).
307
 */
308
static uint32_t gNameSet[8]={ 0 };
309
310
15
#define U_NONCHARACTER_CODE_POINT U_CHAR_CATEGORY_COUNT
311
11
#define U_LEAD_SURROGATE U_CHAR_CATEGORY_COUNT + 1
312
15
#define U_TRAIL_SURROGATE U_CHAR_CATEGORY_COUNT + 2
313
314
#define U_CHAR_EXTENDED_CATEGORY_COUNT (U_CHAR_CATEGORY_COUNT + 3)
315
316
static const char * const charCatNames[U_CHAR_EXTENDED_CATEGORY_COUNT] = {
317
    "unassigned",
318
    "uppercase letter",
319
    "lowercase letter",
320
    "titlecase letter",
321
    "modifier letter",
322
    "other letter",
323
    "non spacing mark",
324
    "enclosing mark",
325
    "combining spacing mark",
326
    "decimal digit number",
327
    "letter number",
328
    "other number",
329
    "space separator",
330
    "line separator",
331
    "paragraph separator",
332
    "control",
333
    "format",
334
    "private use area",
335
    "surrogate",
336
    "dash punctuation",   
337
    "start punctuation",
338
    "end punctuation",
339
    "connector punctuation",
340
    "other punctuation",
341
    "math symbol",
342
    "currency symbol",
343
    "modifier symbol",
344
    "other symbol",
345
    "initial punctuation",
346
    "final punctuation",
347
    "noncharacter",
348
    "lead surrogate",
349
    "trail surrogate"
350
};
351
352
/* implementation ----------------------------------------------------------- */
353
354
static UBool U_CALLCONV unames_cleanup()
355
0
{
356
0
    if(uCharNamesData) {
357
0
        udata_close(uCharNamesData);
358
0
        uCharNamesData = nullptr;
359
0
    }
360
0
    if(uCharNames) {
361
0
        uCharNames = nullptr;
362
0
    }
363
0
    gCharNamesInitOnce.reset();
364
0
    gMaxNameLength=0;
365
0
    return true;
366
0
}
367
368
static UBool U_CALLCONV
369
isAcceptable(void * /*context*/,
370
             const char * /*type*/, const char * /*name*/,
371
3
             const UDataInfo *pInfo) {
372
3
    return
373
3
        pInfo->size>=20 &&
374
3
        pInfo->isBigEndian==U_IS_BIG_ENDIAN &&
375
3
        pInfo->charsetFamily==U_CHARSET_FAMILY &&
376
3
        pInfo->dataFormat[0]==0x75 &&   /* dataFormat="unam" */
377
3
        pInfo->dataFormat[1]==0x6e &&
378
3
        pInfo->dataFormat[2]==0x61 &&
379
3
        pInfo->dataFormat[3]==0x6d &&
380
3
        pInfo->formatVersion[0]==1;
381
3
}
382
383
static void U_CALLCONV
384
3
loadCharNames(UErrorCode &status) {
385
3
    U_ASSERT(uCharNamesData == nullptr);
386
3
    U_ASSERT(uCharNames == nullptr);
387
388
3
    uCharNamesData = udata_openChoice(nullptr, DATA_TYPE, DATA_NAME, isAcceptable, nullptr, &status);
389
3
    if(U_FAILURE(status)) {
390
0
        uCharNamesData = nullptr;
391
3
    } else {
392
3
        uCharNames = (UCharNames *)udata_getMemory(uCharNamesData);
393
3
    }
394
3
    ucln_common_registerCleanup(UCLN_COMMON_UNAMES, unames_cleanup);
395
3
}
396
397
398
static UBool
399
19.7k
isDataLoaded(UErrorCode *pErrorCode) {
400
19.7k
    umtx_initOnce(gCharNamesInitOnce, &loadCharNames, *pErrorCode);
401
19.7k
    return U_SUCCESS(*pErrorCode);
402
19.7k
}
403
404
0
#define WRITE_CHAR(buffer, bufferLength, bufferPos, c) UPRV_BLOCK_MACRO_BEGIN { \
405
0
    if((bufferLength)>0) { \
406
0
        *(buffer)++=c; \
407
0
        --(bufferLength); \
408
0
    } \
409
0
    ++(bufferPos); \
410
0
} UPRV_BLOCK_MACRO_END
411
412
24.3M
#define U_ISO_COMMENT U_CHAR_NAME_CHOICE_COUNT
413
414
/*
415
 * Important: expandName() and compareName() are almost the same -
416
 * apply fixes to both.
417
 *
418
 * UnicodeData.txt uses ';' as a field separator, so no
419
 * field can contain ';' as part of its contents.
420
 * In unames.dat, it is marked as token[';']==-1 only if the
421
 * semicolon is used in the data file - which is iff we
422
 * have Unicode 1.0 names or ISO comments or aliases.
423
 * So, it will be token[';']==-1 if we store U1.0 names/ISO comments/aliases
424
 * although we know that it will never be part of a name.
425
 */
426
static uint16_t
427
expandName(UCharNames *names,
428
           const uint8_t *name, uint16_t nameLength, UCharNameChoice nameChoice,
429
0
           char *buffer, uint16_t bufferLength) {
430
0
    uint16_t* tokens = reinterpret_cast<uint16_t*>(names) + 8;
431
0
    uint16_t token, tokenCount=*tokens++, bufferPos=0;
432
0
    uint8_t* tokenStrings = reinterpret_cast<uint8_t*>(names) + names->tokenStringOffset;
433
0
    uint8_t c;
434
435
0
    if(nameChoice!=U_UNICODE_CHAR_NAME && nameChoice!=U_EXTENDED_CHAR_NAME) {
436
        /*
437
         * skip the modern name if it is not requested _and_
438
         * if the semicolon byte value is a character, not a token number
439
         */
440
0
        if (static_cast<uint8_t>(';') >= tokenCount || tokens[static_cast<uint8_t>(';')] == static_cast<uint16_t>(-1)) {
441
0
            int fieldIndex= nameChoice==U_ISO_COMMENT ? 2 : nameChoice;
442
0
            do {
443
0
                while(nameLength>0) {
444
0
                    --nameLength;
445
0
                    if(*name++==';') {
446
0
                        break;
447
0
                    }
448
0
                }
449
0
            } while(--fieldIndex>0);
450
0
        } else {
451
            /*
452
             * the semicolon byte value is a token number, therefore
453
             * only modern names are stored in unames.dat and there is no
454
             * such requested alternate name here
455
             */
456
0
            nameLength=0;
457
0
        }
458
0
    }
459
460
    /* write each letter directly, and write a token word per token */
461
0
    while(nameLength>0) {
462
0
        --nameLength;
463
0
        c=*name++;
464
465
0
        if(c>=tokenCount) {
466
0
            if(c!=';') {
467
                /* implicit letter */
468
0
                WRITE_CHAR(buffer, bufferLength, bufferPos, c);
469
0
            } else {
470
                /* finished */
471
0
                break;
472
0
            }
473
0
        } else {
474
0
            token=tokens[c];
475
0
            if (token == static_cast<uint16_t>(-2)) {
476
                /* this is a lead byte for a double-byte token */
477
0
                token=tokens[c<<8|*name++];
478
0
                --nameLength;
479
0
            }
480
0
            if (token == static_cast<uint16_t>(-1)) {
481
0
                if(c!=';') {
482
                    /* explicit letter */
483
0
                    WRITE_CHAR(buffer, bufferLength, bufferPos, c);
484
0
                } else {
485
                    /* finished */
486
0
                    break;
487
0
                }
488
0
            } else {
489
                /* write token word */
490
0
                uint8_t *tokenString=tokenStrings+token;
491
0
                while((c=*tokenString++)!=0) {
492
0
                    WRITE_CHAR(buffer, bufferLength, bufferPos, c);
493
0
                }
494
0
            }
495
0
        }
496
0
    }
497
498
    /* zero-terminate */
499
0
    if(bufferLength>0) {
500
0
        *buffer=0;
501
0
    }
502
503
0
    return bufferPos;
504
0
}
505
506
/*
507
 * compareName() is almost the same as expandName() except that it compares
508
 * the currently expanded name to an input name.
509
 * It returns the match/no match result as soon as possible.
510
 */
511
static UBool
512
compareName(UCharNames *names,
513
            const uint8_t *name, uint16_t nameLength, UCharNameChoice nameChoice,
514
628M
            const CharacterNameQuery& otherName) {
515
628M
    uint16_t* tokens = reinterpret_cast<uint16_t*>(names) + 8;
516
628M
    uint16_t token, tokenCount=*tokens++;
517
628M
    uint8_t* tokenStrings = reinterpret_cast<uint8_t*>(names) + names->tokenStringOffset;
518
628M
    uint8_t c;
519
628M
    auto matcher = otherName.matcher();
520
521
628M
    if(nameChoice!=U_UNICODE_CHAR_NAME && nameChoice!=U_EXTENDED_CHAR_NAME) {
522
        /*
523
         * skip the modern name if it is not requested _and_
524
         * if the semicolon byte value is a character, not a token number
525
         */
526
24.3M
        if (static_cast<uint8_t>(';') >= tokenCount || tokens[static_cast<uint8_t>(';')] == static_cast<uint16_t>(-1)) {
527
24.3M
            int fieldIndex= nameChoice==U_ISO_COMMENT ? 2 : nameChoice;
528
72.9M
            do {
529
228M
                while(nameLength>0) {
530
155M
                    --nameLength;
531
155M
                    if(*name++==';') {
532
63.2k
                        break;
533
63.2k
                    }
534
155M
                }
535
72.9M
            } while(--fieldIndex>0);
536
24.3M
        } else {
537
            /*
538
             * the semicolon byte value is a token number, therefore
539
             * only modern names are stored in unames.dat and there is no
540
             * such requested alternate name here
541
             */
542
0
            nameLength=0;
543
0
        }
544
24.3M
    }
545
546
    /* compare each letter directly, and compare a token word per token */
547
629M
    while(nameLength>0) {
548
553M
        --nameLength;
549
553M
        c=*name++;
550
551
553M
        if(c>=tokenCount) {
552
0
            if(c!=';') {
553
                /* implicit letter */
554
0
                if (!matcher.consistentWith(c)) {
555
0
                    return false;
556
0
                }
557
0
            } else {
558
                /* finished */
559
0
                break;
560
0
            }
561
553M
        } else {
562
553M
            token=tokens[c];
563
553M
            if (token == static_cast<uint16_t>(-2)) {
564
                /* this is a lead byte for a double-byte token */
565
108M
                token=tokens[c<<8|*name++];
566
108M
                --nameLength;
567
108M
            }
568
553M
            if (token == static_cast<uint16_t>(-1)) {
569
9.40M
                if(c!=';') {
570
                    /* explicit letter */
571
9.40M
                    if (!matcher.consistentWith(c)) {
572
9.06M
                        return false;
573
9.06M
                    }
574
9.40M
                } else {
575
                    /* finished */
576
0
                    break;
577
0
                }
578
544M
            } else {
579
                /* write token word */
580
544M
                if (!matcher.consistentWith(reinterpret_cast<const char *>(tokenStrings + token))) {
581
544M
                    return false;
582
544M
                }
583
544M
            }
584
553M
        }
585
553M
    }
586
587
    /* complete match? */
588
75.5M
    return matcher.matches();
589
628M
}
590
591
1.57k
static uint8_t getCharCat(UChar32 cp) {
592
1.57k
    uint8_t cat;
593
594
1.57k
    if (U_IS_UNICODE_NONCHAR(cp)) {
595
15
        return U_NONCHARACTER_CODE_POINT;
596
15
    }
597
598
1.56k
    if ((cat = u_charType(cp)) == U_SURROGATE) {
599
13
        cat = U_IS_LEAD(cp) ? U_LEAD_SURROGATE : U_TRAIL_SURROGATE;
600
13
    }
601
602
1.56k
    return cat;
603
1.57k
}
604
605
0
static const char *getCharCatName(UChar32 cp) {
606
0
    uint8_t cat = getCharCat(cp);
607
608
    /* Return unknown if the table of names above is not up to
609
       date. */
610
611
0
    if (cat >= UPRV_LENGTHOF(charCatNames)) {
612
0
        return "unknown";
613
0
    } else {
614
0
        return charCatNames[cat];
615
0
    }
616
0
}
617
618
0
static uint16_t getExtName(uint32_t code, char *buffer, uint16_t bufferLength) {
619
0
    const char *catname = getCharCatName(code);
620
0
    uint16_t length = 0;
621
622
0
    UChar32 cp;
623
0
    int ndigits, i;
624
    
625
0
    WRITE_CHAR(buffer, bufferLength, length, '<');
626
0
    while (catname[length - 1]) {
627
0
        WRITE_CHAR(buffer, bufferLength, length, catname[length - 1]);
628
0
    }
629
0
    WRITE_CHAR(buffer, bufferLength, length, '-');
630
0
    for (cp = code, ndigits = 0; cp; ++ndigits, cp >>= 4)
631
0
        ;
632
0
    if (ndigits < 4)
633
0
        ndigits = 4;
634
0
    for (cp = code, i = ndigits; (cp || i > 0) && bufferLength; cp >>= 4, bufferLength--) {
635
0
        uint8_t v = static_cast<uint8_t>(cp & 0xf);
636
0
        buffer[--i] = (v < 10 ? '0' + v : 'A' + v - 10);
637
0
    }
638
0
    buffer += ndigits;
639
0
    length += static_cast<uint16_t>(ndigits);
640
0
    WRITE_CHAR(buffer, bufferLength, length, '>');
641
642
0
    return length;
643
0
}
644
645
/*
646
 * getGroup() does a binary search for the group that contains the
647
 * Unicode code point "code".
648
 * The return value is always a valid Group* that may contain "code"
649
 * or else is the highest group before "code".
650
 * If the lowest group is after "code", then that one is returned.
651
 */
652
static const uint16_t *
653
17.6k
getGroup(UCharNames *names, uint32_t code) {
654
17.6k
    const uint16_t *groups=GET_GROUPS(names);
655
17.6k
    uint16_t groupMSB = static_cast<uint16_t>(code >> GROUP_SHIFT),
656
17.6k
             start=0,
657
17.6k
             limit=*groups++,
658
17.6k
             number;
659
660
    /* binary search for the group of names that contains the one for code */
661
193k
    while(start<limit-1) {
662
176k
        number = static_cast<uint16_t>((start + limit) / 2);
663
176k
        if(groupMSB<groups[number*GROUP_LENGTH+GROUP_MSB]) {
664
176k
            limit=number;
665
176k
        } else {
666
0
            start=number;
667
0
        }
668
176k
    }
669
670
    /* return this regardless of whether it is an exact match */
671
17.6k
    return groups+start*GROUP_LENGTH;
672
17.6k
}
673
674
/*
675
 * expandGroupLengths() reads a block of compressed lengths of 32 strings and
676
 * expands them into offsets and lengths for each string.
677
 * Lengths are stored with a variable-width encoding in consecutive nibbles:
678
 * If a nibble<0xc, then it is the length itself (0=empty string).
679
 * If a nibble>=0xc, then it forms a length value with the following nibble.
680
 * Calculation see below.
681
 * The offsets and lengths arrays must be at least 33 (one more) long because
682
 * there is no check here at the end if the last nibble is still used.
683
 */
684
static const uint8_t *
685
expandGroupLengths(const uint8_t *s,
686
19.6M
                   uint16_t offsets[LINES_PER_GROUP+1], uint16_t lengths[LINES_PER_GROUP+1]) {
687
    /* read the lengths of the 32 strings in this group and get each string's offset */
688
19.6M
    uint16_t i=0, offset=0, length=0;
689
19.6M
    uint8_t lengthByte;
690
691
    /* all 32 lengths must be read to get the offset of the first group string */
692
352M
    while(i<LINES_PER_GROUP) {
693
332M
        lengthByte=*s++;
694
695
        /* read even nibble - MSBs of lengthByte */
696
332M
        if(length>=12) {
697
            /* double-nibble length spread across two bytes */
698
13.8M
            length = static_cast<uint16_t>(((length & 0x3) << 4 | lengthByte >> 4) + 12);
699
13.8M
            lengthByte&=0xf;
700
318M
        } else if((lengthByte /* &0xf0 */)>=0xc0) {
701
            /* double-nibble length spread across this one byte */
702
18.2M
            length = static_cast<uint16_t>((lengthByte & 0x3f) + 12);
703
300M
        } else {
704
            /* single-nibble length in MSBs */
705
300M
            length = static_cast<uint16_t>(lengthByte >> 4);
706
300M
            lengthByte&=0xf;
707
300M
        }
708
709
332M
        *offsets++=offset;
710
332M
        *lengths++=length;
711
712
332M
        offset+=length;
713
332M
        ++i;
714
715
        /* read odd nibble - LSBs of lengthByte */
716
332M
        if((lengthByte&0xf0)==0) {
717
            /* this nibble was not consumed for a double-nibble length above */
718
314M
            length=lengthByte;
719
314M
            if(length<12) {
720
                /* single-nibble length in LSBs */
721
300M
                *offsets++=offset;
722
300M
                *lengths++=length;
723
724
300M
                offset+=length;
725
300M
                ++i;
726
300M
            }
727
314M
        } else {
728
18.2M
            length=0;   /* prevent double-nibble detection in the next iteration */
729
18.2M
        }
730
332M
    }
731
732
    /* now, s is at the first group string */
733
19.6M
    return s;
734
19.6M
}
735
736
static uint16_t
737
expandGroupName(UCharNames *names, const uint16_t *group,
738
                uint16_t lineNumber, UCharNameChoice nameChoice,
739
0
                char *buffer, uint16_t bufferLength) {
740
0
    uint16_t offsets[LINES_PER_GROUP+2], lengths[LINES_PER_GROUP+2];
741
0
    const uint8_t* s = reinterpret_cast<uint8_t*>(names) + names->groupStringOffset + GET_GROUP_OFFSET(group);
742
0
    s=expandGroupLengths(s, offsets, lengths);
743
0
    return expandName(names, s+offsets[lineNumber], lengths[lineNumber], nameChoice,
744
0
                      buffer, bufferLength);
745
0
}
746
747
static uint16_t
748
getName(UCharNames *names, uint32_t code, UCharNameChoice nameChoice,
749
0
        char *buffer, uint16_t bufferLength) {
750
0
    const uint16_t *group=getGroup(names, code);
751
0
    if (static_cast<uint16_t>(code >> GROUP_SHIFT) == group[GROUP_MSB]) {
752
0
        return expandGroupName(names, group, static_cast<uint16_t>(code & GROUP_MASK), nameChoice,
753
0
                               buffer, bufferLength);
754
0
    } else {
755
        /* group not found */
756
        /* zero-terminate */
757
0
        if(bufferLength>0) {
758
0
            *buffer=0;
759
0
        }
760
0
        return 0;
761
0
    }
762
0
}
763
764
/*
765
 * enumGroupNames() enumerates all the names in a 32-group
766
 * and either calls the enumerator function or finds a given input name.
767
 */
768
static UBool
769
enumGroupNames(UCharNames *names, const uint16_t *group,
770
               UChar32 start, UChar32 end,
771
               UEnumCharNamesFn *fn, void *context,
772
19.6M
               UCharNameChoice nameChoice) {
773
19.6M
    uint16_t offsets[LINES_PER_GROUP+2], lengths[LINES_PER_GROUP+2];
774
19.6M
    const uint8_t* s = reinterpret_cast<uint8_t*>(names) + names->groupStringOffset + GET_GROUP_OFFSET(group);
775
776
19.6M
    s=expandGroupLengths(s, offsets, lengths);
777
19.6M
    if(fn!=DO_FIND_NAME) {
778
0
        char buffer[200];
779
0
        uint16_t length;
780
781
0
        while(start<=end) {
782
0
            length=expandName(names, s+offsets[start&GROUP_MASK], lengths[start&GROUP_MASK], nameChoice, buffer, sizeof(buffer));
783
0
            if (!length && nameChoice == U_EXTENDED_CHAR_NAME) {
784
0
                buffer[length = getExtName(start, buffer, sizeof(buffer))] = 0;
785
0
            }
786
            /* here, we assume that the buffer is large enough */
787
0
            if(length>0) {
788
0
                if(!fn(context, start, nameChoice, buffer, length)) {
789
0
                    return false;
790
0
                }
791
0
            }
792
0
            ++start;
793
0
        }
794
19.6M
    } else {
795
19.6M
        const CharacterNameQuery& otherName = *static_cast<FindName*>(context)->otherName;
796
648M
        while(start<=end) {
797
628M
            if(compareName(names, s+offsets[start&GROUP_MASK], lengths[start&GROUP_MASK], nameChoice, otherName)) {
798
16.4k
                static_cast<FindName*>(context)->code = start;
799
16.4k
                return false;
800
16.4k
            }
801
628M
            ++start;
802
628M
        }
803
19.6M
    }
804
19.6M
    return true;
805
19.6M
}
806
807
/*
808
 * enumExtNames enumerate extended names.
809
 * It only needs to do it if it is called with a real function and not
810
 * with the dummy DO_FIND_NAME, because u_charFromName() does a check
811
 * for extended names by itself.
812
 */ 
813
static UBool
814
enumExtNames(UChar32 start, UChar32 end,
815
             UEnumCharNamesFn *fn, void *context)
816
493k
{
817
493k
    if(fn!=DO_FIND_NAME) {
818
0
        char buffer[200];
819
0
        uint16_t length;
820
        
821
0
        while(start<=end) {
822
0
            buffer[length = getExtName(start, buffer, sizeof(buffer))] = 0;
823
            /* here, we assume that the buffer is large enough */
824
0
            if(length>0) {
825
0
                if(!fn(context, start, U_EXTENDED_CHAR_NAME, buffer, length)) {
826
0
                    return false;
827
0
                }
828
0
            }
829
0
            ++start;
830
0
        }
831
0
    }
832
833
493k
    return true;
834
493k
}
835
836
static UBool
837
enumNames(UCharNames *names,
838
          UChar32 start, UChar32 limit,
839
          UEnumCharNamesFn *fn, void *context,
840
17.6k
          UCharNameChoice nameChoice) {
841
17.6k
    uint16_t startGroupMSB, endGroupMSB, groupCount;
842
17.6k
    const uint16_t *group, *groupLimit;
843
844
17.6k
    startGroupMSB = static_cast<uint16_t>(start >> GROUP_SHIFT);
845
17.6k
    endGroupMSB = static_cast<uint16_t>((limit - 1) >> GROUP_SHIFT);
846
847
    /* find the group that contains start, or the highest before it */
848
17.6k
    group=getGroup(names, start);
849
850
17.6k
    if(startGroupMSB<group[GROUP_MSB] && nameChoice==U_EXTENDED_CHAR_NAME) {
851
        /* enumerate synthetic names between start and the group start */
852
9.16k
        UChar32 extLimit = static_cast<UChar32>(group[GROUP_MSB]) << GROUP_SHIFT;
853
9.16k
        if(extLimit>limit) {
854
0
            extLimit=limit;
855
0
        }
856
9.16k
        if(!enumExtNames(start, extLimit-1, fn, context)) {
857
0
            return false;
858
0
        }
859
9.16k
        start=extLimit;
860
9.16k
    }
861
862
17.6k
    if(startGroupMSB==endGroupMSB) {
863
0
        if(startGroupMSB==group[GROUP_MSB]) {
864
            /* if start and limit-1 are in the same group, then enumerate only in that one */
865
0
            return enumGroupNames(names, group, start, limit-1, fn, context, nameChoice);
866
0
        }
867
17.6k
    } else {
868
17.6k
        const uint16_t *groups=GET_GROUPS(names);
869
17.6k
        groupCount=*groups++;
870
17.6k
        groupLimit=groups+groupCount*GROUP_LENGTH;
871
872
17.6k
        if(startGroupMSB==group[GROUP_MSB]) {
873
            /* enumerate characters in the partial start group */
874
0
            if((start&GROUP_MASK)!=0) {
875
0
                if(!enumGroupNames(names, group,
876
0
                                   start, (static_cast<UChar32>(startGroupMSB) << GROUP_SHIFT) + LINES_PER_GROUP - 1,
877
0
                                   fn, context, nameChoice)) {
878
0
                    return false;
879
0
                }
880
0
                group=NEXT_GROUP(group); /* continue with the next group */
881
0
            }
882
17.6k
        } else if(startGroupMSB>group[GROUP_MSB]) {
883
            /* make sure that we start enumerating with the first group after start */
884
0
            const uint16_t *nextGroup=NEXT_GROUP(group);
885
0
            if (nextGroup < groupLimit && nextGroup[GROUP_MSB] > startGroupMSB && nameChoice == U_EXTENDED_CHAR_NAME) {
886
0
                UChar32 end = nextGroup[GROUP_MSB] << GROUP_SHIFT;
887
0
                if (end > limit) {
888
0
                    end = limit;
889
0
                }
890
0
                if (!enumExtNames(start, end - 1, fn, context)) {
891
0
                    return false;
892
0
                }
893
0
            }
894
0
            group=nextGroup;
895
0
        }
896
897
        /* enumerate entire groups between the start- and end-groups */
898
19.6M
        while(group<groupLimit && group[GROUP_MSB]<endGroupMSB) {
899
19.6M
            const uint16_t *nextGroup;
900
19.6M
            start = static_cast<UChar32>(group[GROUP_MSB]) << GROUP_SHIFT;
901
19.6M
            if(!enumGroupNames(names, group, start, start+LINES_PER_GROUP-1, fn, context, nameChoice)) {
902
16.4k
                return false;
903
16.4k
            }
904
19.6M
            nextGroup=NEXT_GROUP(group);
905
19.6M
            if (nextGroup < groupLimit && nextGroup[GROUP_MSB] > group[GROUP_MSB] + 1 && nameChoice == U_EXTENDED_CHAR_NAME) {
906
484k
                UChar32 end = nextGroup[GROUP_MSB] << GROUP_SHIFT;
907
484k
                if (end > limit) {
908
0
                    end = limit;
909
0
                }
910
484k
                if (!enumExtNames((group[GROUP_MSB] + 1) << GROUP_SHIFT, end - 1, fn, context)) {
911
0
                    return false;
912
0
                }
913
484k
            }
914
19.6M
            group=nextGroup;
915
19.6M
        }
916
917
        /* enumerate within the end group (group[GROUP_MSB]==endGroupMSB) */
918
1.18k
        if(group<groupLimit && group[GROUP_MSB]==endGroupMSB) {
919
0
            return enumGroupNames(names, group, (limit-1)&~GROUP_MASK, limit-1, fn, context, nameChoice);
920
1.18k
        } else if (nameChoice == U_EXTENDED_CHAR_NAME && group == groupLimit) {
921
541
            UChar32 next = (PREV_GROUP(group)[GROUP_MSB] + 1) << GROUP_SHIFT;
922
541
            if (next > start) {
923
541
                start = next;
924
541
            }
925
643
        } else {
926
643
            return true;
927
643
        }
928
1.18k
    }
929
930
    /* we have not found a group, which means everything is made of
931
       extended names. */
932
541
    if (nameChoice == U_EXTENDED_CHAR_NAME) {
933
541
        if (limit > UCHAR_MAX_VALUE + 1) {
934
0
            limit = UCHAR_MAX_VALUE + 1;
935
0
        }
936
541
        return enumExtNames(start, limit - 1, fn, context);
937
541
    }
938
    
939
0
    return true;
940
541
}
941
942
static uint16_t
943
writeFactorSuffix(const uint16_t *factors, uint16_t count,
944
                  const char *s, /* suffix elements */
945
                  uint32_t code,
946
                  uint16_t indexes[8], /* output fields from here */
947
                  const char *elementBases[8], const char *elements[8],
948
0
                  char *buffer, uint16_t bufferLength) {
949
0
    uint16_t i, factor, bufferPos=0;
950
0
    char c;
951
952
    /* write elements according to the factors */
953
954
    /*
955
     * the factorized elements are determined by modulo arithmetic
956
     * with the factors of this algorithm
957
     *
958
     * note that for fewer operations, count is decremented here
959
     */
960
0
    --count;
961
0
    for(i=count; i>0; --i) {
962
0
        factor=factors[i];
963
0
        indexes[i] = static_cast<uint16_t>(code % factor);
964
0
        code/=factor;
965
0
    }
966
    /*
967
     * we don't need to calculate the last modulus because start<=code<=end
968
     * guarantees here that code<=factors[0]
969
     */
970
0
    indexes[0] = static_cast<uint16_t>(code);
971
972
    /* write each element */
973
0
    for(;;) {
974
0
        if(elementBases!=nullptr) {
975
0
            *elementBases++=s;
976
0
        }
977
978
        /* skip indexes[i] strings */
979
0
        factor=indexes[i];
980
0
        while(factor>0) {
981
0
            while(*s++!=0) {}
982
0
            --factor;
983
0
        }
984
0
        if(elements!=nullptr) {
985
0
            *elements++=s;
986
0
        }
987
988
        /* write element */
989
0
        while((c=*s++)!=0) {
990
0
            WRITE_CHAR(buffer, bufferLength, bufferPos, c);
991
0
        }
992
993
        /* we do not need to perform the rest of this loop for i==count - break here */
994
0
        if(i>=count) {
995
0
            break;
996
0
        }
997
998
        /* skip the rest of the strings for this factors[i] */
999
0
        factor = static_cast<uint16_t>(factors[i] - indexes[i] - 1);
1000
0
        while(factor>0) {
1001
0
            while(*s++!=0) {}
1002
0
            --factor;
1003
0
        }
1004
1005
0
        ++i;
1006
0
    }
1007
1008
    /* zero-terminate */
1009
0
    if(bufferLength>0) {
1010
0
        *buffer=0;
1011
0
    }
1012
1013
0
    return bufferPos;
1014
0
}
1015
1016
/*
1017
 * Important:
1018
 * Parts of findAlgName() are almost the same as some of getAlgName().
1019
 * Fixes must be applied to both.
1020
 */
1021
static uint16_t
1022
getAlgName(AlgorithmicRange *range, uint32_t code, UCharNameChoice nameChoice,
1023
0
        char *buffer, uint16_t bufferLength) {
1024
0
    uint16_t bufferPos=0;
1025
1026
    /* Only the normative character name can be algorithmic. */
1027
0
    if(nameChoice!=U_UNICODE_CHAR_NAME && nameChoice!=U_EXTENDED_CHAR_NAME) {
1028
        /* zero-terminate */
1029
0
        if(bufferLength>0) {
1030
0
            *buffer=0;
1031
0
        }
1032
0
        return 0;
1033
0
    }
1034
1035
0
    switch(range->type) {
1036
0
    case 0: {
1037
        /* name = prefix hex-digits */
1038
0
        const char* s = reinterpret_cast<const char*>(range + 1);
1039
0
        char c;
1040
1041
0
        uint16_t i, count;
1042
1043
        /* copy prefix */
1044
0
        while((c=*s++)!=0) {
1045
0
            WRITE_CHAR(buffer, bufferLength, bufferPos, c);
1046
0
        }
1047
1048
        /* write hexadecimal code point value */
1049
0
        count=range->variant;
1050
1051
        /* zero-terminate */
1052
0
        if(count<bufferLength) {
1053
0
            buffer[count]=0;
1054
0
        }
1055
1056
0
        for(i=count; i>0;) {
1057
0
            if(--i<bufferLength) {
1058
0
                c = static_cast<char>(code & 0xf);
1059
0
                if(c<10) {
1060
0
                    c+='0';
1061
0
                } else {
1062
0
                    c+='A'-10;
1063
0
                }
1064
0
                buffer[i]=c;
1065
0
            }
1066
0
            code>>=4;
1067
0
        }
1068
1069
0
        bufferPos+=count;
1070
0
        break;
1071
0
    }
1072
0
    case 1: {
1073
        /* name = prefix factorized-elements */
1074
0
        uint16_t indexes[8];
1075
0
        const uint16_t* factors = reinterpret_cast<const uint16_t*>(range + 1);
1076
0
        uint16_t count=range->variant;
1077
0
        const char* s = reinterpret_cast<const char*>(factors + count);
1078
0
        char c;
1079
1080
        /* copy prefix */
1081
0
        while((c=*s++)!=0) {
1082
0
            WRITE_CHAR(buffer, bufferLength, bufferPos, c);
1083
0
        }
1084
1085
0
        bufferPos+=writeFactorSuffix(factors, count,
1086
0
                                     s, code-range->start, indexes, nullptr, nullptr, buffer, bufferLength);
1087
0
        break;
1088
0
    }
1089
0
    default:
1090
        /* undefined type */
1091
        /* zero-terminate */
1092
0
        if(bufferLength>0) {
1093
0
            *buffer=0;
1094
0
        }
1095
0
        break;
1096
0
    }
1097
1098
0
    return bufferPos;
1099
0
}
1100
1101
/*
1102
 * Important: enumAlgNames() and findAlgName() are almost the same.
1103
 * Any fix must be applied to both.
1104
 */
1105
static UBool
1106
enumAlgNames(AlgorithmicRange *range,
1107
             UChar32 start, UChar32 limit,
1108
             UEnumCharNamesFn *fn, void *context,
1109
0
             UCharNameChoice nameChoice) {
1110
0
    char buffer[200];
1111
0
    uint16_t length;
1112
1113
0
    if(nameChoice!=U_UNICODE_CHAR_NAME && nameChoice!=U_EXTENDED_CHAR_NAME) {
1114
0
        return true;
1115
0
    }
1116
1117
0
    switch(range->type) {
1118
0
    case 0: {
1119
0
        char *s, *end;
1120
0
        char c;
1121
1122
        /* get the full name of the start character */
1123
0
        length = getAlgName(range, static_cast<uint32_t>(start), nameChoice, buffer, sizeof(buffer));
1124
0
        if(length<=0) {
1125
0
            return true;
1126
0
        }
1127
1128
        /* call the enumerator function with this first character */
1129
0
        if(!fn(context, start, nameChoice, buffer, length)) {
1130
0
            return false;
1131
0
        }
1132
1133
        /* go to the end of the name; all these names have the same length */
1134
0
        end=buffer;
1135
0
        while(*end!=0) {
1136
0
            ++end;
1137
0
        }
1138
1139
        /* enumerate the rest of the names */
1140
0
        while(++start<limit) {
1141
            /* increment the hexadecimal number on a character-basis */
1142
0
            s=end;
1143
0
            for (;;) {
1144
0
                c=*--s;
1145
0
                if(('0'<=c && c<'9') || ('A'<=c && c<'F')) {
1146
0
                    *s = static_cast<char>(c + 1);
1147
0
                    break;
1148
0
                } else if(c=='9') {
1149
0
                    *s='A';
1150
0
                    break;
1151
0
                } else if(c=='F') {
1152
0
                    *s='0';
1153
0
                }
1154
0
            }
1155
1156
0
            if(!fn(context, start, nameChoice, buffer, length)) {
1157
0
                return false;
1158
0
            }
1159
0
        }
1160
0
        break;
1161
0
    }
1162
0
    case 1: {
1163
0
        uint16_t indexes[8];
1164
0
        const char *elementBases[8], *elements[8];
1165
0
        const uint16_t* factors = reinterpret_cast<const uint16_t*>(range + 1);
1166
0
        uint16_t count=range->variant;
1167
0
        const char* s = reinterpret_cast<const char*>(factors + count);
1168
0
        char *suffix, *t;
1169
0
        uint16_t prefixLength, i, idx;
1170
1171
0
        char c;
1172
1173
        /* name = prefix factorized-elements */
1174
1175
        /* copy prefix */
1176
0
        suffix=buffer;
1177
0
        prefixLength=0;
1178
0
        while((c=*s++)!=0) {
1179
0
            *suffix++=c;
1180
0
            ++prefixLength;
1181
0
        }
1182
1183
        /* append the suffix of the start character */
1184
0
        length = static_cast<uint16_t>(prefixLength + writeFactorSuffix(factors, count,
1185
0
                                              s, static_cast<uint32_t>(start) - range->start,
1186
0
                                              indexes, elementBases, elements,
1187
0
                                              suffix, static_cast<uint16_t>(sizeof(buffer) - prefixLength)));
1188
1189
        /* call the enumerator function with this first character */
1190
0
        if(!fn(context, start, nameChoice, buffer, length)) {
1191
0
            return false;
1192
0
        }
1193
1194
        /* enumerate the rest of the names */
1195
0
        while(++start<limit) {
1196
            /* increment the indexes in lexical order bound by the factors */
1197
0
            i=count;
1198
0
            for (;;) {
1199
0
                idx = static_cast<uint16_t>(indexes[--i] + 1);
1200
0
                if(idx<factors[i]) {
1201
                    /* skip one index and its element string */
1202
0
                    indexes[i]=idx;
1203
0
                    s=elements[i];
1204
0
                    while(*s++!=0) {
1205
0
                    }
1206
0
                    elements[i]=s;
1207
0
                    break;
1208
0
                } else {
1209
                    /* reset this index to 0 and its element string to the first one */
1210
0
                    indexes[i]=0;
1211
0
                    elements[i]=elementBases[i];
1212
0
                }
1213
0
            }
1214
1215
            /* to make matters a little easier, just append all elements to the suffix */
1216
0
            t=suffix;
1217
0
            length=prefixLength;
1218
0
            for(i=0; i<count; ++i) {
1219
0
                s=elements[i];
1220
0
                while((c=*s++)!=0) {
1221
0
                    *t++=c;
1222
0
                    ++length;
1223
0
                }
1224
0
            }
1225
            /* zero-terminate */
1226
0
            *t=0;
1227
1228
0
            if(!fn(context, start, nameChoice, buffer, length)) {
1229
0
                return false;
1230
0
            }
1231
0
        }
1232
0
        break;
1233
0
    }
1234
0
    default:
1235
        /* undefined type */
1236
0
        break;
1237
0
    }
1238
1239
0
    return true;
1240
0
}
1241
1242
/*
1243
 * findAlgName() is almost the same as enumAlgNames() except that it
1244
 * returns the code point for a name if it fits into the range.
1245
 * It returns 0xffff otherwise.
1246
 */
1247
static UChar32
1248
246k
findAlgName(AlgorithmicRange *range, UCharNameChoice nameChoice, CharacterNameQuery& query) {
1249
246k
    UChar32 code;
1250
246k
    auto matcher = query.matcher();
1251
1252
246k
    if(nameChoice!=U_UNICODE_CHAR_NAME && nameChoice!=U_EXTENDED_CHAR_NAME) {
1253
7.57k
        return 0xffff;
1254
7.57k
    }
1255
1256
239k
    switch(range->type) {
1257
221k
    case 0: {
1258
        /* name = prefix hex-digits */
1259
221k
        const char* s = reinterpret_cast<const char*>(range + 1);
1260
1261
        /* compare prefix */
1262
221k
        if (!matcher.consistentWith(s)) {
1263
221k
            return 0xffff;
1264
221k
        }
1265
1266
        /* read hexadecimal code point value */
1267
0
        code=0;
1268
0
        for(char c : matcher.remainingSignificantCharacters()) {
1269
0
            if('0'<=c && c<='9') {
1270
0
                code=(code<<4)|(c-'0');
1271
0
            } else if('A'<=c && c<='F') {
1272
0
                code=(code<<4)|(c-'A'+10);
1273
0
            } else {
1274
0
                return 0xffff;
1275
0
            }
1276
0
        }
1277
1278
        /* does it fit into the range? */
1279
0
        if (range->start <= static_cast<uint32_t>(code) && static_cast<uint32_t>(code) <= range->end) {
1280
0
            return code;
1281
0
        }
1282
0
        break;
1283
0
    }
1284
17.0k
    case 1: {
1285
17.0k
        char buffer[64];
1286
17.0k
        uint16_t indexes[8];
1287
17.0k
        const char *elementBases[8], *elements[8];
1288
17.0k
        const uint16_t* factors = reinterpret_cast<const uint16_t*>(range + 1);
1289
17.0k
        uint16_t count=range->variant;
1290
17.0k
        const char *s = reinterpret_cast<const char*>(factors + count);
1291
17.0k
        UChar32 start, limit;
1292
17.0k
        uint16_t i, idx;
1293
1294
        /* name = prefix factorized-elements */
1295
1296
        /* compare prefix */
1297
17.0k
        std::string_view prefix = s;
1298
17.0k
        if (!matcher.consistentWith(prefix)) {
1299
17.0k
            return 0xffff;
1300
17.0k
        }
1301
1302
0
        s += prefix.length() + 1;
1303
1304
0
        start = static_cast<UChar32>(range->start);
1305
0
        limit = static_cast<UChar32>(range->end + 1);
1306
        
1307
        // This implementation assumes that the suffix does not contain SPACEs nor HYPHEN-MINUSes,
1308
        // which is the case in practice for the names of Hangul syllables.
1309
0
        std::string_view querySuffix = matcher.remainingSignificantCharacters();
1310
        /* initialize the suffix elements for enumeration; indexes should all be set to 0 */
1311
0
        writeFactorSuffix(factors, count, s, 0,
1312
0
                          indexes, elementBases, elements, buffer, sizeof(buffer));
1313
1314
        /* compare the first suffix */
1315
0
        if(querySuffix == buffer) {
1316
0
            return start;
1317
0
        }
1318
1319
        /* enumerate and compare the rest of the suffixes */
1320
0
        while(++start<limit) {
1321
            /* increment the indexes in lexical order bound by the factors */
1322
0
            i=count;
1323
0
            for (;;) {
1324
0
                idx = static_cast<uint16_t>(indexes[--i] + 1);
1325
0
                if(idx<factors[i]) {
1326
                    /* skip one index and its element string */
1327
0
                    indexes[i]=idx;
1328
0
                    s=elements[i];
1329
0
                    while(*s++!=0) {}
1330
0
                    elements[i]=s;
1331
0
                    break;
1332
0
                } else {
1333
                    /* reset this index to 0 and its element string to the first one */
1334
0
                    indexes[i]=0;
1335
0
                    elements[i]=elementBases[i];
1336
0
                }
1337
0
            }
1338
1339
            /* to make matters a little easier, just compare all elements of the suffix */
1340
0
            auto it = querySuffix.begin();
1341
0
            for(i=0; i<count; ++i) {
1342
0
                s=elements[i];
1343
0
                char c;
1344
0
                while((c=*s++)!=0) {
1345
0
                    if(it == querySuffix.end() || c!=*it++) {
1346
0
                        s=""; /* does not match */
1347
0
                        i=99;
1348
0
                    }
1349
0
                }
1350
0
            }
1351
0
            if(i<99 && it == querySuffix.end()) {
1352
0
                return start;
1353
0
            }
1354
0
        }
1355
0
        break;
1356
0
    }
1357
0
    default:
1358
        /* undefined type */
1359
0
        break;
1360
239k
    }
1361
1362
0
    return 0xffff;
1363
239k
}
1364
1365
/* sets of name characters, maximum name lengths ---------------------------- */
1366
1367
0
#define SET_ADD(set, c) ((set)[(uint8_t)c>>5]|=((uint32_t)1<<((uint8_t)c&0x1f)))
1368
0
#define SET_CONTAINS(set, c) (((set)[(uint8_t)c>>5]&((uint32_t)1<<((uint8_t)c&0x1f)))!=0)
1369
1370
static int32_t
1371
0
calcStringSetLength(uint32_t set[8], const char *s) {
1372
0
    int32_t length=0;
1373
0
    char c;
1374
1375
0
    while((c=*s++)!=0) {
1376
0
        SET_ADD(set, c);
1377
0
        ++length;
1378
0
    }
1379
0
    return length;
1380
0
}
1381
1382
static int32_t
1383
0
calcAlgNameSetsLengths(int32_t maxNameLength) {
1384
0
    AlgorithmicRange *range;
1385
0
    uint32_t *p;
1386
0
    uint32_t rangeCount;
1387
0
    int32_t length;
1388
1389
    /* enumerate algorithmic ranges */
1390
0
    p = reinterpret_cast<uint32_t*>(reinterpret_cast<uint8_t*>(uCharNames) + uCharNames->algNamesOffset);
1391
0
    rangeCount=*p;
1392
0
    range = reinterpret_cast<AlgorithmicRange*>(p + 1);
1393
0
    while(rangeCount>0) {
1394
0
        switch(range->type) {
1395
0
        case 0:
1396
            /* name = prefix + (range->variant times) hex-digits */
1397
            /* prefix */
1398
0
            length = calcStringSetLength(gNameSet, reinterpret_cast<const char*>(range + 1)) + range->variant;
1399
0
            if(length>maxNameLength) {
1400
0
                maxNameLength=length;
1401
0
            }
1402
0
            break;
1403
0
        case 1: {
1404
            /* name = prefix factorized-elements */
1405
0
            const uint16_t* factors = reinterpret_cast<const uint16_t*>(range + 1);
1406
0
            const char *s;
1407
0
            int32_t i, count=range->variant, factor, factorLength, maxFactorLength;
1408
1409
            /* prefix length */
1410
0
            s = reinterpret_cast<const char*>(factors + count);
1411
0
            length=calcStringSetLength(gNameSet, s);
1412
0
            s+=length+1; /* start of factor suffixes */
1413
1414
            /* get the set and maximum factor suffix length for each factor */
1415
0
            for(i=0; i<count; ++i) {
1416
0
                maxFactorLength=0;
1417
0
                for(factor=factors[i]; factor>0; --factor) {
1418
0
                    factorLength=calcStringSetLength(gNameSet, s);
1419
0
                    s+=factorLength+1;
1420
0
                    if(factorLength>maxFactorLength) {
1421
0
                        maxFactorLength=factorLength;
1422
0
                    }
1423
0
                }
1424
0
                length+=maxFactorLength;
1425
0
            }
1426
1427
0
            if(length>maxNameLength) {
1428
0
                maxNameLength=length;
1429
0
            }
1430
0
            break;
1431
0
        }
1432
0
        default:
1433
            /* unknown type */
1434
0
            break;
1435
0
        }
1436
1437
0
        range = reinterpret_cast<AlgorithmicRange*>(reinterpret_cast<uint8_t*>(range) + range->size);
1438
0
        --rangeCount;
1439
0
    }
1440
0
    return maxNameLength;
1441
0
}
1442
1443
static int32_t
1444
0
calcExtNameSetsLengths(int32_t maxNameLength) {
1445
0
    int32_t i, length;
1446
1447
0
    for(i=0; i<UPRV_LENGTHOF(charCatNames); ++i) {
1448
        /*
1449
         * for each category, count the length of the category name
1450
         * plus 9=
1451
         * 2 for <>
1452
         * 1 for -
1453
         * 6 for most hex digits per code point
1454
         */
1455
0
        length=9+calcStringSetLength(gNameSet, charCatNames[i]);
1456
0
        if(length>maxNameLength) {
1457
0
            maxNameLength=length;
1458
0
        }
1459
0
    }
1460
0
    return maxNameLength;
1461
0
}
1462
1463
static int32_t
1464
calcNameSetLength(const uint16_t *tokens, uint16_t tokenCount, const uint8_t *tokenStrings, int8_t *tokenLengths,
1465
                  uint32_t set[8],
1466
0
                  const uint8_t **pLine, const uint8_t *lineLimit) {
1467
0
    const uint8_t *line=*pLine;
1468
0
    int32_t length=0, tokenLength;
1469
0
    uint16_t c, token;
1470
1471
0
    while (line != lineLimit && (c = *line++) != static_cast<uint8_t>(';')) {
1472
0
        if(c>=tokenCount) {
1473
            /* implicit letter */
1474
0
            SET_ADD(set, c);
1475
0
            ++length;
1476
0
        } else {
1477
0
            token=tokens[c];
1478
0
            if (token == static_cast<uint16_t>(-2)) {
1479
                /* this is a lead byte for a double-byte token */
1480
0
                c=c<<8|*line++;
1481
0
                token=tokens[c];
1482
0
            }
1483
0
            if (token == static_cast<uint16_t>(-1)) {
1484
                /* explicit letter */
1485
0
                SET_ADD(set, c);
1486
0
                ++length;
1487
0
            } else {
1488
                /* count token word */
1489
0
                if(tokenLengths!=nullptr) {
1490
                    /* use cached token length */
1491
0
                    tokenLength=tokenLengths[c];
1492
0
                    if(tokenLength==0) {
1493
0
                        tokenLength = calcStringSetLength(set, reinterpret_cast<const char*>(tokenStrings) + token);
1494
0
                        tokenLengths[c] = static_cast<int8_t>(tokenLength);
1495
0
                    }
1496
0
                } else {
1497
0
                    tokenLength = calcStringSetLength(set, reinterpret_cast<const char*>(tokenStrings) + token);
1498
0
                }
1499
0
                length+=tokenLength;
1500
0
            }
1501
0
        }
1502
0
    }
1503
1504
0
    *pLine=line;
1505
0
    return length;
1506
0
}
1507
1508
static void
1509
0
calcGroupNameSetsLengths(int32_t maxNameLength) {
1510
0
    uint16_t offsets[LINES_PER_GROUP+2], lengths[LINES_PER_GROUP+2];
1511
1512
0
    uint16_t* tokens = reinterpret_cast<uint16_t*>(uCharNames) + 8;
1513
0
    uint16_t tokenCount=*tokens++;
1514
0
    uint8_t* tokenStrings = reinterpret_cast<uint8_t*>(uCharNames) + uCharNames->tokenStringOffset;
1515
1516
0
    int8_t *tokenLengths;
1517
1518
0
    const uint16_t *group;
1519
0
    const uint8_t *s, *line, *lineLimit;
1520
1521
0
    int32_t groupCount, lineNumber, length;
1522
1523
0
    tokenLengths = static_cast<int8_t*>(uprv_malloc(tokenCount));
1524
0
    if(tokenLengths!=nullptr) {
1525
0
        uprv_memset(tokenLengths, 0, tokenCount);
1526
0
    }
1527
1528
0
    group=GET_GROUPS(uCharNames);
1529
0
    groupCount=*group++;
1530
1531
    /* enumerate all groups */
1532
0
    while(groupCount>0) {
1533
0
        s = reinterpret_cast<uint8_t*>(uCharNames) + uCharNames->groupStringOffset + GET_GROUP_OFFSET(group);
1534
0
        s=expandGroupLengths(s, offsets, lengths);
1535
1536
        /* enumerate all lines in each group */
1537
0
        for(lineNumber=0; lineNumber<LINES_PER_GROUP; ++lineNumber) {
1538
0
            line=s+offsets[lineNumber];
1539
0
            length=lengths[lineNumber];
1540
0
            if(length==0) {
1541
0
                continue;
1542
0
            }
1543
1544
0
            lineLimit=line+length;
1545
1546
            /* read regular name */
1547
0
            length=calcNameSetLength(tokens, tokenCount, tokenStrings, tokenLengths, gNameSet, &line, lineLimit);
1548
0
            if(length>maxNameLength) {
1549
0
                maxNameLength=length;
1550
0
            }
1551
0
            if(line==lineLimit) {
1552
0
                continue;
1553
0
            }
1554
1555
            /* read Unicode 1.0 name */
1556
0
            length=calcNameSetLength(tokens, tokenCount, tokenStrings, tokenLengths, gNameSet, &line, lineLimit);
1557
0
            if(length>maxNameLength) {
1558
0
                maxNameLength=length;
1559
0
            }
1560
0
            if(line==lineLimit) {
1561
0
                continue;
1562
0
            }
1563
1564
            /* read ISO comment */
1565
            /*length=calcNameSetLength(tokens, tokenCount, tokenStrings, tokenLengths, gISOCommentSet, &line, lineLimit);*/
1566
0
        }
1567
1568
0
        group=NEXT_GROUP(group);
1569
0
        --groupCount;
1570
0
    }
1571
1572
0
    if(tokenLengths!=nullptr) {
1573
0
        uprv_free(tokenLengths);
1574
0
    }
1575
1576
    /* set gMax... - name length last for threading */
1577
0
    gMaxNameLength=maxNameLength;
1578
0
}
1579
1580
static UBool
1581
0
calcNameSetsLengths(UErrorCode *pErrorCode) {
1582
0
    static const char extChars[]="0123456789ABCDEF<>-";
1583
0
    int32_t i, maxNameLength;
1584
1585
0
    if(gMaxNameLength!=0) {
1586
0
        return true;
1587
0
    }
1588
1589
0
    if(!isDataLoaded(pErrorCode)) {
1590
0
        return false;
1591
0
    }
1592
1593
    /* set hex digits, used in various names, and <>-, used in extended names */
1594
0
    for (i = 0; i < static_cast<int32_t>(sizeof(extChars)) - 1; ++i) {
1595
0
        SET_ADD(gNameSet, extChars[i]);
1596
0
    }
1597
1598
    /* set sets and lengths from algorithmic names */
1599
0
    maxNameLength=calcAlgNameSetsLengths(0);
1600
1601
    /* set sets and lengths from extended names */
1602
0
    maxNameLength=calcExtNameSetsLengths(maxNameLength);
1603
1604
    /* set sets and lengths from group names, set global maximum values */
1605
0
    calcGroupNameSetsLengths(maxNameLength);
1606
1607
0
    return true;
1608
0
}
1609
1610
U_NAMESPACE_END
1611
1612
/* public API --------------------------------------------------------------- */
1613
1614
U_NAMESPACE_USE
1615
1616
U_CAPI int32_t U_EXPORT2
1617
u_charName(UChar32 code, UCharNameChoice nameChoice,
1618
           char *buffer, int32_t bufferLength,
1619
0
           UErrorCode *pErrorCode) {
1620
0
     AlgorithmicRange *algRange;
1621
0
    uint32_t *p;
1622
0
    uint32_t i;
1623
0
    int32_t length;
1624
1625
    /* check the argument values */
1626
0
    if(pErrorCode==nullptr || U_FAILURE(*pErrorCode)) {
1627
0
        return 0;
1628
0
    } else if(nameChoice>=U_CHAR_NAME_CHOICE_COUNT ||
1629
0
              bufferLength<0 || (bufferLength>0 && buffer==nullptr)
1630
0
    ) {
1631
0
        *pErrorCode=U_ILLEGAL_ARGUMENT_ERROR;
1632
0
        return 0;
1633
0
    }
1634
1635
0
    if((uint32_t)code>UCHAR_MAX_VALUE || !isDataLoaded(pErrorCode)) {
1636
0
        return u_terminateChars(buffer, bufferLength, 0, pErrorCode);
1637
0
    }
1638
1639
0
    length=0;
1640
1641
    /* try algorithmic names first */
1642
0
    p=(uint32_t *)((uint8_t *)uCharNames+uCharNames->algNamesOffset);
1643
0
    i=*p;
1644
0
    algRange=(AlgorithmicRange *)(p+1);
1645
0
    while(i>0) {
1646
0
        if(algRange->start<=(uint32_t)code && (uint32_t)code<=algRange->end) {
1647
0
            length=getAlgName(algRange, (uint32_t)code, nameChoice, buffer, (uint16_t)bufferLength);
1648
0
            break;
1649
0
        }
1650
0
        algRange=(AlgorithmicRange *)((uint8_t *)algRange+algRange->size);
1651
0
        --i;
1652
0
    }
1653
1654
0
    if(i==0) {
1655
0
        if (nameChoice == U_EXTENDED_CHAR_NAME) {
1656
0
            length = getName(uCharNames, (uint32_t )code, U_EXTENDED_CHAR_NAME, buffer, (uint16_t) bufferLength);
1657
0
            if (!length) {
1658
                /* extended character name */
1659
0
                length = getExtName((uint32_t) code, buffer, (uint16_t) bufferLength);
1660
0
            }
1661
0
        } else {
1662
            /* normal character name */
1663
0
            length=getName(uCharNames, (uint32_t)code, nameChoice, buffer, (uint16_t)bufferLength);
1664
0
        }
1665
0
    }
1666
1667
0
    return u_terminateChars(buffer, bufferLength, length, pErrorCode);
1668
0
}
1669
1670
U_CAPI int32_t U_EXPORT2
1671
u_getISOComment(UChar32 /*c*/,
1672
                char *dest, int32_t destCapacity,
1673
0
                UErrorCode *pErrorCode) {
1674
    /* check the argument values */
1675
0
    if(pErrorCode==nullptr || U_FAILURE(*pErrorCode)) {
1676
0
        return 0;
1677
0
    } else if(destCapacity<0 || (destCapacity>0 && dest==nullptr)) {
1678
0
        *pErrorCode=U_ILLEGAL_ARGUMENT_ERROR;
1679
0
        return 0;
1680
0
    }
1681
1682
0
    return u_terminateChars(dest, destCapacity, 0, pErrorCode);
1683
0
}
1684
1685
U_CAPI UChar32 U_EXPORT2
1686
u_charFromName(UCharNameChoice nameChoice,
1687
               const char *const name,
1688
20.0k
               UErrorCode *pErrorCode) {
1689
20.0k
    char lower[120] = {0};
1690
20.0k
    FindName findName;
1691
20.0k
    AlgorithmicRange *algRange;
1692
20.0k
    uint32_t *p;
1693
20.0k
    uint32_t i;
1694
20.0k
    UChar32 cp = 0;
1695
20.0k
    char c0;
1696
20.0k
    static constexpr UChar32 error = 0xffff;     /* Undefined, but use this for backwards compatibility. */
1697
1698
20.0k
    if(pErrorCode==nullptr || U_FAILURE(*pErrorCode)) {
1699
1
        return error;
1700
1
    }
1701
1702
20.0k
    if(nameChoice>=U_CHAR_NAME_CHOICE_COUNT || name==nullptr || *name==0) {
1703
230
        *pErrorCode=U_ILLEGAL_ARGUMENT_ERROR;
1704
230
        return error;
1705
230
    }
1706
1707
19.7k
    if(!isDataLoaded(pErrorCode)) {
1708
0
        return error;
1709
0
    }
1710
1711
    /* construct the lowercase of the name first */
1712
19.7k
    const char* nameChar = name;
1713
99.2k
    for(i=0; i<sizeof(lower); ++i) {
1714
99.2k
        if((c0=*nameChar++)!=0) {
1715
79.5k
            lower[i]=uprv_tolower(c0);
1716
79.5k
        } else {
1717
19.7k
            break;
1718
19.7k
        }
1719
99.2k
    }
1720
19.7k
    if(i==sizeof(lower)) {
1721
        /* name too long, there is no such character */
1722
26
        *pErrorCode = U_ILLEGAL_CHAR_FOUND;
1723
26
        return error;
1724
26
    }
1725
    // i==strlen(name)==strlen(lower)
1726
1727
    /* try extended names first */
1728
    // The ICU extended names are not consistent with the UAX44 code point labels, e.g.,
1729
    // "unassigned" vs. UAX44 "reserved", "private use area" vs. UAX44 "private-use", so we do not
1730
    // apply UAX44-LM2 matching to them.
1731
19.7k
    if (lower[0] == '<') {
1732
2.15k
        if (nameChoice == U_EXTENDED_CHAR_NAME && lower[--i] == '>') {
1733
            // Parse a string like "<category-HHHH>" where HHHH is a hex code point.
1734
1.66k
            uint32_t limit = i;
1735
6.95k
            while (i >= 3 && lower[--i] != '-') {}
1736
1737
            // There should be 1 to 8 hex digits.
1738
1.66k
            int32_t hexLength = limit - (i + 1);
1739
1.66k
            if (i >= 2 && lower[i] == '-' && 1 <= hexLength && hexLength <= 8) {
1740
1.60k
                uint32_t cIdx;
1741
1742
1.60k
                lower[i] = 0;
1743
1744
6.51k
                for (++i; i < limit; ++i) {
1745
4.93k
                    if (lower[i] >= '0' && lower[i] <= '9') {
1746
4.30k
                        cp = (cp << 4) + lower[i] - '0';
1747
4.30k
                    } else if (lower[i] >= 'a' && lower[i] <= 'f') {
1748
616
                        cp = (cp << 4) + lower[i] - 'a' + 10;
1749
616
                    } else {
1750
21
                        *pErrorCode = U_ILLEGAL_CHAR_FOUND;
1751
21
                        return error;
1752
21
                    }
1753
                    // Prevent signed-integer overflow and out-of-range code points.
1754
4.91k
                    if (cp > UCHAR_MAX_VALUE) {
1755
8
                        *pErrorCode = U_ILLEGAL_CHAR_FOUND;
1756
8
                        return error;
1757
8
                    }
1758
4.91k
                }
1759
1760
                /* Now validate the category name.
1761
                   We could use a binary search, or a trie, if
1762
                   we really wanted to. */
1763
1.57k
                uint8_t cat = getCharCat(cp);
1764
31.2k
                for (lower[i] = 0, cIdx = 0; cIdx < UPRV_LENGTHOF(charCatNames); ++cIdx) {
1765
1766
30.8k
                    if (!uprv_strcmp(lower + 1, charCatNames[cIdx])) {
1767
1.24k
                        if (cat == cIdx) {
1768
1.23k
                            return cp;
1769
1.23k
                        }
1770
4
                        break;
1771
1.24k
                    }
1772
30.8k
                }
1773
1.57k
            }
1774
1.66k
        }
1775
1776
884
        *pErrorCode = U_ILLEGAL_CHAR_FOUND;
1777
884
        return error;
1778
2.15k
    }
1779
1780
    // From now on we do UAX44 matching.
1781
17.6k
    CharacterNameQuery query(name);
1782
1783
    /* try algorithmic names now */
1784
17.6k
    p=(uint32_t *)((uint8_t *)uCharNames+uCharNames->algNamesOffset);
1785
17.6k
    i=*p;
1786
17.6k
    algRange=(AlgorithmicRange *)(p+1);
1787
264k
    while(i>0) {
1788
246k
        if((cp=findAlgName(algRange, nameChoice, query))!=0xffff) {
1789
0
            return cp;
1790
0
        }
1791
246k
        algRange=(AlgorithmicRange *)((uint8_t *)algRange+algRange->size);
1792
246k
        --i;
1793
246k
    }
1794
1795
    /* normal character name */
1796
17.6k
    findName.otherName=&query;
1797
17.6k
    findName.code=error;
1798
17.6k
    enumNames(uCharNames, 0, UCHAR_MAX_VALUE + 1, DO_FIND_NAME, &findName, nameChoice);
1799
17.6k
    if (findName.code == error) {
1800
1.18k
         *pErrorCode = U_ILLEGAL_CHAR_FOUND;
1801
1.18k
    }
1802
17.6k
    return findName.code;
1803
17.6k
}
1804
1805
U_CAPI void U_EXPORT2
1806
u_enumCharNames(UChar32 start, UChar32 limit,
1807
                UEnumCharNamesFn *fn,
1808
                void *context,
1809
                UCharNameChoice nameChoice,
1810
0
                UErrorCode *pErrorCode) {
1811
0
    AlgorithmicRange *algRange;
1812
0
    uint32_t *p;
1813
0
    uint32_t i;
1814
1815
0
    if(pErrorCode==nullptr || U_FAILURE(*pErrorCode)) {
1816
0
        return;
1817
0
    }
1818
1819
0
    if(nameChoice>=U_CHAR_NAME_CHOICE_COUNT || fn==nullptr) {
1820
0
        *pErrorCode=U_ILLEGAL_ARGUMENT_ERROR;
1821
0
        return;
1822
0
    }
1823
1824
0
    if((uint32_t) limit > UCHAR_MAX_VALUE + 1) {
1825
0
        limit = UCHAR_MAX_VALUE + 1;
1826
0
    }
1827
0
    if((uint32_t)start>=(uint32_t)limit) {
1828
0
        return;
1829
0
    }
1830
1831
0
    if(!isDataLoaded(pErrorCode)) {
1832
0
        return;
1833
0
    }
1834
1835
    /* interleave the data-driven ones with the algorithmic ones */
1836
    /* iterate over all algorithmic ranges; assume that they are in ascending order */
1837
0
    p=(uint32_t *)((uint8_t *)uCharNames+uCharNames->algNamesOffset);
1838
0
    i=*p;
1839
0
    algRange=(AlgorithmicRange *)(p+1);
1840
0
    while(i>0) {
1841
        /* enumerate the character names before the current algorithmic range */
1842
        /* here: start<limit */
1843
0
        if((uint32_t)start<algRange->start) {
1844
0
            if((uint32_t)limit<=algRange->start) {
1845
0
                enumNames(uCharNames, start, limit, fn, context, nameChoice);
1846
0
                return;
1847
0
            }
1848
0
            if(!enumNames(uCharNames, start, (UChar32)algRange->start, fn, context, nameChoice)) {
1849
0
                return;
1850
0
            }
1851
0
            start=(UChar32)algRange->start;
1852
0
        }
1853
        /* enumerate the character names in the current algorithmic range */
1854
        /* here: algRange->start<=start<limit */
1855
0
        if((uint32_t)start<=algRange->end) {
1856
0
            if((uint32_t)limit<=(algRange->end+1)) {
1857
0
                enumAlgNames(algRange, start, limit, fn, context, nameChoice);
1858
0
                return;
1859
0
            }
1860
0
            if(!enumAlgNames(algRange, start, (UChar32)algRange->end+1, fn, context, nameChoice)) {
1861
0
                return;
1862
0
            }
1863
0
            start=(UChar32)algRange->end+1;
1864
0
        }
1865
        /* continue to the next algorithmic range (here: start<limit) */
1866
0
        algRange=(AlgorithmicRange *)((uint8_t *)algRange+algRange->size);
1867
0
        --i;
1868
0
    }
1869
    /* enumerate the character names after the last algorithmic range */
1870
0
    enumNames(uCharNames, start, limit, fn, context, nameChoice);
1871
0
}
1872
1873
U_CAPI int32_t U_EXPORT2
1874
0
uprv_getMaxCharNameLength() {
1875
0
    UErrorCode errorCode=U_ZERO_ERROR;
1876
0
    if(calcNameSetsLengths(&errorCode)) {
1877
0
        return gMaxNameLength;
1878
0
    } else {
1879
0
        return 0;
1880
0
    }
1881
0
}
1882
1883
/**
1884
 * Converts the char set cset into a Unicode set uset.
1885
 * @param cset Set of 256 bit flags corresponding to a set of chars.
1886
 * @param uset USet to receive characters. Existing contents are deleted.
1887
 */
1888
static void
1889
0
charSetToUSet(uint32_t cset[8], const USetAdder *sa) {
1890
0
    char16_t us[256];
1891
0
    char cs[256];
1892
1893
0
    int32_t i, length;
1894
0
    UErrorCode errorCode;
1895
1896
0
    errorCode=U_ZERO_ERROR;
1897
1898
0
    if(!calcNameSetsLengths(&errorCode)) {
1899
0
        return;
1900
0
    }
1901
1902
    /* build a char string with all chars that are used in character names */
1903
0
    length=0;
1904
0
    for(i=0; i<256; ++i) {
1905
0
        if(SET_CONTAINS(cset, i)) {
1906
0
            cs[length++] = static_cast<char>(i);
1907
0
        }
1908
0
    }
1909
1910
    /* convert the char string to a char16_t string */
1911
0
    u_charsToUChars(cs, us, length);
1912
1913
    /* add each char16_t to the USet */
1914
0
    for(i=0; i<length; ++i) {
1915
0
        if(us[i]!=0 || cs[i]==0) { /* non-invariant chars become (char16_t)0 */
1916
0
            sa->add(sa->set, us[i]);
1917
0
        }
1918
0
    }
1919
0
}
1920
1921
/**
1922
 * Fills set with characters that are used in Unicode character names.
1923
 * @param set USet to receive characters.
1924
 */
1925
U_CAPI void U_EXPORT2
1926
0
uprv_getCharNameCharacters(const USetAdder *sa) {
1927
0
    charSetToUSet(gNameSet, sa);
1928
0
}
1929
1930
/* data swapping ------------------------------------------------------------ */
1931
1932
/*
1933
 * The token table contains non-negative entries for token bytes,
1934
 * and -1 for bytes that represent themselves in the data file's charset.
1935
 * -2 entries are used for lead bytes.
1936
 *
1937
 * Direct bytes (-1 entries) must be translated from the input charset family
1938
 * to the output charset family.
1939
 * makeTokenMap() writes a permutation mapping for this.
1940
 * Use it once for single-/lead-byte tokens and once more for all trail byte
1941
 * tokens. (';' is an unused trail byte marked with -1.)
1942
 */
1943
static void
1944
makeTokenMap(const UDataSwapper *ds,
1945
             int16_t tokens[], uint16_t tokenCount,
1946
             uint8_t map[256],
1947
0
             UErrorCode *pErrorCode) {
1948
0
    UBool usedOutChar[256];
1949
0
    uint16_t i, j;
1950
0
    uint8_t c1, c2;
1951
1952
0
    if(U_FAILURE(*pErrorCode)) {
1953
0
        return;
1954
0
    }
1955
1956
0
    if(ds->inCharset==ds->outCharset) {
1957
        /* Same charset family: identity permutation */
1958
0
        for(i=0; i<256; ++i) {
1959
0
            map[i] = static_cast<uint8_t>(i);
1960
0
        }
1961
0
    } else {
1962
0
        uprv_memset(map, 0, 256);
1963
0
        uprv_memset(usedOutChar, 0, 256);
1964
1965
0
        if(tokenCount>256) {
1966
0
            tokenCount=256;
1967
0
        }
1968
1969
        /* set the direct bytes (byte 0 always maps to itself) */
1970
0
        for(i=1; i<tokenCount; ++i) {
1971
0
            if(tokens[i]==-1) {
1972
                /* convert the direct byte character */
1973
0
                c1 = static_cast<uint8_t>(i);
1974
0
                ds->swapInvChars(ds, &c1, 1, &c2, pErrorCode);
1975
0
                if(U_FAILURE(*pErrorCode)) {
1976
0
                    udata_printError(ds, "unames/makeTokenMap() finds variant character 0x%02x used (input charset family %d)\n",
1977
0
                                     i, ds->inCharset);
1978
0
                    return;
1979
0
                }
1980
1981
                /* enter the converted character into the map and mark it used */
1982
0
                map[c1]=c2;
1983
0
                usedOutChar[c2]=true;
1984
0
            }
1985
0
        }
1986
1987
        /* set the mappings for the rest of the permutation */
1988
0
        for(i=j=1; i<tokenCount; ++i) {
1989
            /* set mappings that were not set for direct bytes */
1990
0
            if(map[i]==0) {
1991
                /* set an output byte value that was not used as an output byte above */
1992
0
                while(usedOutChar[j]) {
1993
0
                    ++j;
1994
0
                }
1995
0
                map[i] = static_cast<uint8_t>(j++);
1996
0
            }
1997
0
        }
1998
1999
        /*
2000
         * leave mappings at tokenCount and above unset if tokenCount<256
2001
         * because they won't be used
2002
         */
2003
0
    }
2004
0
}
2005
2006
U_CAPI int32_t U_EXPORT2
2007
uchar_swapNames(const UDataSwapper *ds,
2008
                const void *inData, int32_t length, void *outData,
2009
0
                UErrorCode *pErrorCode) {
2010
0
    const UDataInfo *pInfo;
2011
0
    int32_t headerSize;
2012
2013
0
    const uint8_t *inBytes;
2014
0
    uint8_t *outBytes;
2015
2016
0
    uint32_t tokenStringOffset, groupsOffset, groupStringOffset, algNamesOffset,
2017
0
             offset, i, count, stringsCount;
2018
2019
0
    const AlgorithmicRange *inRange;
2020
0
    AlgorithmicRange *outRange;
2021
2022
    /* udata_swapDataHeader checks the arguments */
2023
0
    headerSize=udata_swapDataHeader(ds, inData, length, outData, pErrorCode);
2024
0
    if(pErrorCode==nullptr || U_FAILURE(*pErrorCode)) {
2025
0
        return 0;
2026
0
    }
2027
2028
    /* check data format and format version */
2029
0
    pInfo=(const UDataInfo *)((const char *)inData+4);
2030
0
    if(!(
2031
0
        pInfo->dataFormat[0]==0x75 &&   /* dataFormat="unam" */
2032
0
        pInfo->dataFormat[1]==0x6e &&
2033
0
        pInfo->dataFormat[2]==0x61 &&
2034
0
        pInfo->dataFormat[3]==0x6d &&
2035
0
        pInfo->formatVersion[0]==1
2036
0
    )) {
2037
0
        udata_printError(ds, "uchar_swapNames(): data format %02x.%02x.%02x.%02x (format version %02x) is not recognized as unames.icu\n",
2038
0
                         pInfo->dataFormat[0], pInfo->dataFormat[1],
2039
0
                         pInfo->dataFormat[2], pInfo->dataFormat[3],
2040
0
                         pInfo->formatVersion[0]);
2041
0
        *pErrorCode=U_UNSUPPORTED_ERROR;
2042
0
        return 0;
2043
0
    }
2044
2045
0
    inBytes=(const uint8_t *)inData+headerSize;
2046
0
    outBytes=(outData == nullptr) ? nullptr : (uint8_t *)outData+headerSize;
2047
0
    if(length<0) {
2048
0
        algNamesOffset=ds->readUInt32(((const uint32_t *)inBytes)[3]);
2049
0
    } else {
2050
0
        length-=headerSize;
2051
0
        if( length<20 ||
2052
0
            (uint32_t)length<(algNamesOffset=ds->readUInt32(((const uint32_t *)inBytes)[3]))
2053
0
        ) {
2054
0
            udata_printError(ds, "uchar_swapNames(): too few bytes (%d after header) for unames.icu\n",
2055
0
                             length);
2056
0
            *pErrorCode=U_INDEX_OUTOFBOUNDS_ERROR;
2057
0
            return 0;
2058
0
        }
2059
0
    }
2060
2061
0
    if(length<0) {
2062
        /* preflighting: iterate through algorithmic ranges */
2063
0
        offset=algNamesOffset;
2064
0
        count=ds->readUInt32(*((const uint32_t *)(inBytes+offset)));
2065
0
        offset+=4;
2066
2067
0
        for(i=0; i<count; ++i) {
2068
0
            inRange=(const AlgorithmicRange *)(inBytes+offset);
2069
0
            offset+=ds->readUInt16(inRange->size);
2070
0
        }
2071
0
    } else {
2072
        /* swap data */
2073
0
        const uint16_t *p;
2074
0
        uint16_t *q, *temp;
2075
2076
0
        int16_t tokens[512];
2077
0
        uint16_t tokenCount;
2078
2079
0
        uint8_t map[256], trailMap[256];
2080
2081
        /* copy the data for inaccessible bytes */
2082
0
        if(inBytes!=outBytes) {
2083
0
            uprv_memcpy(outBytes, inBytes, length);
2084
0
        }
2085
2086
        /* the initial 4 offsets first */
2087
0
        tokenStringOffset=ds->readUInt32(((const uint32_t *)inBytes)[0]);
2088
0
        groupsOffset=ds->readUInt32(((const uint32_t *)inBytes)[1]);
2089
0
        groupStringOffset=ds->readUInt32(((const uint32_t *)inBytes)[2]);
2090
0
        ds->swapArray32(ds, inBytes, 16, outBytes, pErrorCode);
2091
2092
        /*
2093
         * now the tokens table
2094
         * it needs to be permutated along with the compressed name strings
2095
         */
2096
0
        p=(const uint16_t *)(inBytes+16);
2097
0
        q=(uint16_t *)(outBytes+16);
2098
2099
        /* read and swap the tokenCount */
2100
0
        tokenCount=ds->readUInt16(*p);
2101
0
        ds->swapArray16(ds, p, 2, q, pErrorCode);
2102
0
        ++p;
2103
0
        ++q;
2104
2105
        /* read the first 512 tokens and make the token maps */
2106
0
        if(tokenCount<=512) {
2107
0
            count=tokenCount;
2108
0
        } else {
2109
0
            count=512;
2110
0
        }
2111
0
        for(i=0; i<count; ++i) {
2112
0
            tokens[i]=udata_readInt16(ds, p[i]);
2113
0
        }
2114
0
        for(; i<512; ++i) {
2115
0
            tokens[i]=0; /* fill the rest of the tokens array if tokenCount<512 */
2116
0
        }
2117
0
        makeTokenMap(ds, tokens, tokenCount, map, pErrorCode);
2118
0
        makeTokenMap(ds, tokens+256, (uint16_t)(tokenCount>256 ? tokenCount-256 : 0), trailMap, pErrorCode);
2119
0
        if(U_FAILURE(*pErrorCode)) {
2120
0
            return 0;
2121
0
        }
2122
2123
        /*
2124
         * swap and permutate the tokens
2125
         * go through a temporary array to support in-place swapping
2126
         */
2127
0
        temp=(uint16_t *)uprv_malloc(tokenCount*2);
2128
0
        if(temp==nullptr) {
2129
0
            udata_printError(ds, "out of memory swapping %u unames.icu tokens\n",
2130
0
                             tokenCount);
2131
0
            *pErrorCode=U_MEMORY_ALLOCATION_ERROR;
2132
0
            return 0;
2133
0
        }
2134
2135
        /* swap and permutate single-/lead-byte tokens */
2136
0
        for(i=0; i<tokenCount && i<256; ++i) {
2137
0
            ds->swapArray16(ds, p+i, 2, temp+map[i], pErrorCode);
2138
0
        }
2139
2140
        /* swap and permutate trail-byte tokens */
2141
0
        for(; i<tokenCount; ++i) {
2142
0
            ds->swapArray16(ds, p+i, 2, temp+(i&0xffffff00)+trailMap[i&0xff], pErrorCode);
2143
0
        }
2144
2145
        /* copy the result into the output and free the temporary array */
2146
0
        uprv_memcpy(q, temp, tokenCount*2);
2147
0
        uprv_free(temp);
2148
2149
        /*
2150
         * swap the token strings but not a possible padding byte after
2151
         * the terminating NUL of the last string
2152
         */
2153
0
        udata_swapInvStringBlock(ds, inBytes+tokenStringOffset, (int32_t)(groupsOffset-tokenStringOffset),
2154
0
                                    outBytes+tokenStringOffset, pErrorCode);
2155
0
        if(U_FAILURE(*pErrorCode)) {
2156
0
            udata_printError(ds, "uchar_swapNames(token strings) failed\n");
2157
0
            return 0;
2158
0
        }
2159
2160
        /* swap the group table */
2161
0
        count=ds->readUInt16(*((const uint16_t *)(inBytes+groupsOffset)));
2162
0
        ds->swapArray16(ds, inBytes+groupsOffset, (int32_t)((1+count*3)*2),
2163
0
                           outBytes+groupsOffset, pErrorCode);
2164
2165
        /*
2166
         * swap the group strings
2167
         * swap the string bytes but not the nibble-encoded string lengths
2168
         */
2169
0
        if(ds->inCharset!=ds->outCharset) {
2170
0
            uint16_t offsets[LINES_PER_GROUP+1], lengths[LINES_PER_GROUP+1];
2171
2172
0
            const uint8_t *inStrings, *nextInStrings;
2173
0
            uint8_t *outStrings;
2174
2175
0
            uint8_t c;
2176
2177
0
            inStrings=inBytes+groupStringOffset;
2178
0
            outStrings=outBytes+groupStringOffset;
2179
2180
0
            stringsCount=algNamesOffset-groupStringOffset;
2181
2182
            /* iterate through string groups until only a few padding bytes are left */
2183
0
            while(stringsCount>32) {
2184
0
                nextInStrings=expandGroupLengths(inStrings, offsets, lengths);
2185
2186
                /* move past the length bytes */
2187
0
                stringsCount-=(uint32_t)(nextInStrings-inStrings);
2188
0
                outStrings+=nextInStrings-inStrings;
2189
0
                inStrings=nextInStrings;
2190
2191
0
                count=offsets[31]+lengths[31]; /* total number of string bytes in this group */
2192
0
                stringsCount-=count;
2193
2194
                /* swap the string bytes using map[] and trailMap[] */
2195
0
                while(count>0) {
2196
0
                    c=*inStrings++;
2197
0
                    *outStrings++=map[c];
2198
0
                    if(tokens[c]!=-2) {
2199
0
                        --count;
2200
0
                    } else {
2201
                        /* token lead byte: swap the trail byte, too */
2202
0
                        *outStrings++=trailMap[*inStrings++];
2203
0
                        count-=2;
2204
0
                    }
2205
0
                }
2206
0
            }
2207
0
        }
2208
2209
        /* swap the algorithmic ranges */
2210
0
        offset=algNamesOffset;
2211
0
        count=ds->readUInt32(*((const uint32_t *)(inBytes+offset)));
2212
0
        ds->swapArray32(ds, inBytes+offset, 4, outBytes+offset, pErrorCode);
2213
0
        offset+=4;
2214
2215
0
        for(i=0; i<count; ++i) {
2216
0
            if(offset>(uint32_t)length) {
2217
0
                udata_printError(ds, "uchar_swapNames(): too few bytes (%d after header) for unames.icu algorithmic range %u\n",
2218
0
                                 length, i);
2219
0
                *pErrorCode=U_INDEX_OUTOFBOUNDS_ERROR;
2220
0
                return 0;
2221
0
            }
2222
2223
0
            inRange=(const AlgorithmicRange *)(inBytes+offset);
2224
0
            outRange=(AlgorithmicRange *)(outBytes+offset);
2225
0
            offset+=ds->readUInt16(inRange->size);
2226
2227
0
            ds->swapArray32(ds, inRange, 8, outRange, pErrorCode);
2228
0
            ds->swapArray16(ds, &inRange->size, 2, &outRange->size, pErrorCode);
2229
0
            switch(inRange->type) {
2230
0
            case 0:
2231
                /* swap prefix string */
2232
0
                ds->swapInvChars(ds, inRange+1, (int32_t)uprv_strlen((const char *)(inRange+1)),
2233
0
                                    outRange+1, pErrorCode);
2234
0
                if(U_FAILURE(*pErrorCode)) {
2235
0
                    udata_printError(ds, "uchar_swapNames(prefix string of algorithmic range %u) failed\n",
2236
0
                                     i);
2237
0
                    return 0;
2238
0
                }
2239
0
                break;
2240
0
            case 1:
2241
0
                {
2242
                    /* swap factors and the prefix and factor strings */
2243
0
                    uint32_t factorsCount;
2244
2245
0
                    factorsCount=inRange->variant;
2246
0
                    p=(const uint16_t *)(inRange+1);
2247
0
                    q=(uint16_t *)(outRange+1);
2248
0
                    ds->swapArray16(ds, p, (int32_t)(factorsCount*2), q, pErrorCode);
2249
2250
                    /* swap the strings, up to the last terminating NUL */
2251
0
                    p+=factorsCount;
2252
0
                    q+=factorsCount;
2253
0
                    stringsCount=(uint32_t)((inBytes+offset)-(const uint8_t *)p);
2254
0
                    while(stringsCount>0 && ((const uint8_t *)p)[stringsCount-1]!=0) {
2255
0
                        --stringsCount;
2256
0
                    }
2257
0
                    ds->swapInvChars(ds, p, (int32_t)stringsCount, q, pErrorCode);
2258
0
                }
2259
0
                break;
2260
0
            default:
2261
0
                udata_printError(ds, "uchar_swapNames(): unknown type %u of algorithmic range %u\n",
2262
0
                                 inRange->type, i);
2263
0
                *pErrorCode=U_UNSUPPORTED_ERROR;
2264
0
                return 0;
2265
0
            }
2266
0
        }
2267
0
    }
2268
2269
0
    return headerSize+(int32_t)offset;
2270
0
}
2271
2272
/*
2273
 * Hey, Emacs, please set the following:
2274
 *
2275
 * Local Variables:
2276
 * indent-tabs-mode: nil
2277
 * End:
2278
 *
2279
 */