Coverage Report

Created: 2026-07-30 07:11

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/FreeRDP/winpr/libwinpr/crt/unicode_builtin.c
Line
Count
Source
1
/*
2
 * Copyright 2001-2004 Unicode, Inc.
3
 *
4
 * Disclaimer
5
 *
6
 * This source code is provided as is by Unicode, Inc. No claims are
7
 * made as to fitness for any particular purpose. No warranties of any
8
 * kind are expressed or implied. The recipient agrees to determine
9
 * applicability of information provided. If this file has been
10
 * purchased on magnetic or optical media from Unicode, Inc., the
11
 * sole remedy for any claim will be exchange of defective media
12
 * within 90 days of receipt.
13
 *
14
 * Limitations on Rights to Redistribute This Code
15
 *
16
 * Unicode, Inc. hereby grants the right to freely use the information
17
 * supplied in this file in the creation of products supporting the
18
 * Unicode Standard, and to make copies of this file in any form
19
 * for internal or external distribution as long as this notice
20
 * remains attached.
21
 */
22
23
/* ---------------------------------------------------------------------
24
25
Conversions between UTF32, UTF-16, and UTF-8. Source code file.
26
Author: Mark E. Davis, 1994.
27
Rev History: Rick McGowan, fixes & updates May 2001.
28
Sept 2001: fixed const & error conditions per
29
mods suggested by S. Parent & A. Lillich.
30
June 2002: Tim Dodd added detection and handling of incomplete
31
source sequences, enhanced error detection, added casts
32
to eliminate compiler warnings.
33
July 2003: slight mods to back out aggressive FFFE detection.
34
Jan 2004: updated switches in from-UTF8 conversions.
35
Oct 2004: updated to use UNI_MAX_LEGAL_UTF32 in UTF-32 conversions.
36
37
See the header file "utf.h" for complete documentation.
38
39
------------------------------------------------------------------------ */
40
41
#include <winpr/wtypes.h>
42
#include <winpr/string.h>
43
#include <winpr/assert.h>
44
#include <winpr/cast.h>
45
46
#include "unicode.h"
47
48
#include "../log.h"
49
#define TAG WINPR_TAG("unicode")
50
51
/*
52
 * Character Types:
53
 *
54
 * UTF8:    uint8_t   8 bits
55
 * UTF16: uint16_t  16 bits
56
 * UTF32: uint32_t  32 bits
57
 */
58
59
/* Some fundamental constants */
60
0
#define UNI_REPLACEMENT_CHAR (uint32_t)0x0000FFFD
61
17.8M
#define UNI_MAX_BMP (uint32_t)0x0000FFFF
62
1.49k
#define UNI_MAX_UTF16 (uint32_t)0x0010FFFF
63
#define UNI_MAX_UTF32 (uint32_t)0x7FFFFFFF
64
#define UNI_MAX_LEGAL_UTF32 (uint32_t)0x0010FFFF
65
66
typedef enum
67
{
68
  conversionOK,    /* conversion successful */
69
  sourceExhausted, /* partial character in source, but hit end */
70
  targetExhausted, /* insuff. room in target for conversion */
71
  sourceIllegal    /* source sequence is illegal/malformed */
72
} ConversionResult;
73
74
typedef enum
75
{
76
  strictConversion = 0,
77
  lenientConversion
78
} ConversionFlags;
79
80
static const int halfShift = 10; /* used for shifting by 10 bits */
81
82
static const uint32_t halfBase = 0x0010000UL;
83
static const uint32_t halfMask = 0x3FFUL;
84
85
79.9M
#define UNI_SUR_HIGH_START (uint32_t)0xD800
86
1.74M
#define UNI_SUR_HIGH_END (uint32_t)0xDBFF
87
44.3M
#define UNI_SUR_LOW_START (uint32_t)0xDC00
88
1.73M
#define UNI_SUR_LOW_END (uint32_t)0xDFFF
89
90
/* --------------------------------------------------------------------- */
91
92
/*
93
 * Index into the table below with the first byte of a UTF-8 sequence to
94
 * get the number of trailing bytes that are supposed to follow it.
95
 * Note that *legal* UTF-8 values can't have 4 or 5-bytes. The table is
96
 * left as-is for anyone who may want to do such conversion, which was
97
 * allowed in earlier algorithms.
98
 */
99
static const char trailingBytesForUTF8[256] = {
100
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
101
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
102
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
103
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
104
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
105
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
106
  1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
107
  2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5
108
};
109
110
/*
111
 * Magic values subtracted from a buffer value during UTF8 conversion.
112
 * This table contains as many values as there might be trailing bytes
113
 * in a UTF-8 sequence.
114
 */
115
static const uint32_t offsetsFromUTF8[6] = { 0x00000000UL, 0x00003080UL, 0x000E2080UL,
116
                                           0x03C82080UL, 0xFA082080UL, 0x82082080UL };
117
118
/*
119
 * Once the bits are split out into bytes of UTF-8, this is a mask OR-ed
120
 * into the first byte, depending on how many bytes follow.  There are
121
 * as many entries in this table as there are UTF-8 sequence types.
122
 * (I.e., one byte sequence, two byte... etc.). Remember that sequence
123
 * for *legal* UTF-8 will be 4 or fewer bytes total.
124
 */
125
static const uint8_t firstByteMark[7] = { 0x00, 0x00, 0xC0, 0xE0, 0xF0, 0xF8, 0xFC };
126
127
/* We always need UTF-16LE, even on big endian systems! */
128
static WCHAR setWcharFrom(WCHAR w)
129
35.4M
{
130
#if defined(__BIG_ENDIAN__)
131
  union
132
  {
133
    WCHAR w;
134
    char c[2];
135
  } cnv;
136
137
  cnv.w = w;
138
  const char c = cnv.c[0];
139
  cnv.c[0] = cnv.c[1];
140
  cnv.c[1] = c;
141
  return cnv.w;
142
#else
143
35.4M
  return w;
144
35.4M
#endif
145
35.4M
}
146
147
/* --------------------------------------------------------------------- */
148
149
/* The interface converts a whole buffer to avoid function-call overhead.
150
 * Constants have been gathered. Loops & conditionals have been removed as
151
 * much as possible for efficiency, in favor of drop-through switches.
152
 * (See "Note A" at the bottom of the file for equivalent code.)
153
 * If your compiler supports it, the "isLegalUTF8" call can be turned
154
 * into an inline function.
155
 */
156
157
/* --------------------------------------------------------------------- */
158
159
static ConversionResult winpr_ConvertUTF16toUTF8_Internal(const uint16_t** sourceStart,
160
                                                          const uint16_t* sourceEnd,
161
                                                          uint8_t** targetStart,
162
                                                          const uint8_t* targetEnd,
163
                                                          ConversionFlags flags)
164
745k
{
165
745k
  bool computeLength = (!targetEnd) ? true : false;
166
745k
  const uint16_t* source = *sourceStart;
167
745k
  uint8_t* target = *targetStart;
168
745k
  ConversionResult result = conversionOK;
169
170
22.9M
  while (source < sourceEnd)
171
22.1M
  {
172
22.1M
    uint32_t ch = 0;
173
22.1M
    unsigned short bytesToWrite = 0;
174
22.1M
    const uint32_t byteMask = 0xBF;
175
22.1M
    const uint32_t byteMark = 0x80;
176
22.1M
    const uint16_t* oldSource =
177
22.1M
        source; /* In case we have to back up because of target overflow. */
178
179
22.1M
    ch = setWcharFrom(*source++);
180
181
    /* If we have a surrogate pair, convert to UTF32 first. */
182
22.1M
    if (ch >= UNI_SUR_HIGH_START && ch <= UNI_SUR_HIGH_END)
183
14.6k
    {
184
      /* If the 16 bits following the high surrogate are in the source buffer... */
185
14.6k
      if (source < sourceEnd)
186
14.6k
      {
187
14.6k
        uint32_t ch2 = setWcharFrom(*source);
188
189
        /* If it's a low surrogate, convert to UTF32. */
190
14.6k
        if (ch2 >= UNI_SUR_LOW_START && ch2 <= UNI_SUR_LOW_END)
191
2.58k
        {
192
2.58k
          ch = ((ch - UNI_SUR_HIGH_START) << halfShift) + (ch2 - UNI_SUR_LOW_START) +
193
2.58k
               halfBase;
194
2.58k
          ++source;
195
2.58k
        }
196
12.0k
        else if (flags == strictConversion)
197
12.0k
        {
198
          /* it's an unpaired high surrogate */
199
12.0k
          --source; /* return to the illegal value itself */
200
12.0k
          result = sourceIllegal;
201
12.0k
          break;
202
12.0k
        }
203
14.6k
      }
204
40
      else
205
40
      {
206
        /* We don't have the 16 bits following the high surrogate. */
207
40
        --source; /* return to the high surrogate */
208
40
        result = sourceExhausted;
209
40
        break;
210
40
      }
211
14.6k
    }
212
22.1M
    else if (flags == strictConversion)
213
22.1M
    {
214
      /* UTF-16 surrogate values are illegal in UTF-32 */
215
22.1M
      if (ch >= UNI_SUR_LOW_START && ch <= UNI_SUR_LOW_END)
216
286
      {
217
286
        --source; /* return to the illegal value itself */
218
286
        result = sourceIllegal;
219
286
        break;
220
286
      }
221
22.1M
    }
222
223
    /* Figure out how many bytes the result will require */
224
22.1M
    if (ch < (uint32_t)0x80)
225
9.47M
    {
226
9.47M
      bytesToWrite = 1;
227
9.47M
    }
228
12.6M
    else if (ch < (uint32_t)0x800)
229
7.33M
    {
230
7.33M
      bytesToWrite = 2;
231
7.33M
    }
232
5.34M
    else if (ch < (uint32_t)0x10000)
233
5.34M
    {
234
5.34M
      bytesToWrite = 3;
235
5.34M
    }
236
2.58k
    else if (ch < (uint32_t)0x110000)
237
2.58k
    {
238
2.58k
      bytesToWrite = 4;
239
2.58k
    }
240
0
    else
241
0
    {
242
0
      bytesToWrite = 3;
243
0
      ch = UNI_REPLACEMENT_CHAR;
244
0
    }
245
246
22.1M
    target += bytesToWrite;
247
248
22.1M
    if ((target > targetEnd) && (!computeLength))
249
1.57k
    {
250
1.57k
      source = oldSource; /* Back up source pointer! */
251
1.57k
      target -= bytesToWrite;
252
1.57k
      result = targetExhausted;
253
1.57k
      break;
254
1.57k
    }
255
256
22.1M
    if (!computeLength)
257
14.3M
    {
258
14.3M
      switch (bytesToWrite)
259
14.3M
      {
260
          /* note: everything falls through. */
261
1.16k
        case 4:
262
1.16k
          *--target = (uint8_t)((ch | byteMark) & byteMask);
263
1.16k
          ch >>= 6;
264
          /* fallthrough */
265
1.16k
          WINPR_FALLTHROUGH
266
2.78M
        case 3:
267
2.78M
          *--target = (uint8_t)((ch | byteMark) & byteMask);
268
2.78M
          ch >>= 6;
269
          /* fallthrough */
270
2.78M
          WINPR_FALLTHROUGH
271
272
6.43M
        case 2:
273
6.43M
          *--target = (uint8_t)((ch | byteMark) & byteMask);
274
6.43M
          ch >>= 6;
275
          /* fallthrough */
276
6.43M
          WINPR_FALLTHROUGH
277
278
14.3M
        case 1:
279
14.3M
          *--target = (uint8_t)(ch | firstByteMark[bytesToWrite]);
280
14.3M
          break;
281
0
        default:
282
0
          return sourceIllegal;
283
14.3M
      }
284
14.3M
    }
285
7.80M
    else
286
7.80M
    {
287
7.80M
      switch (bytesToWrite)
288
7.80M
      {
289
          /* note: everything falls through. */
290
1.41k
        case 4:
291
1.41k
          --target;
292
          /* fallthrough */
293
1.41k
          WINPR_FALLTHROUGH
294
295
2.55M
        case 3:
296
2.55M
          --target;
297
          /* fallthrough */
298
2.55M
          WINPR_FALLTHROUGH
299
300
6.24M
        case 2:
301
6.24M
          --target;
302
          /* fallthrough */
303
6.24M
          WINPR_FALLTHROUGH
304
305
7.80M
        case 1:
306
7.80M
          --target;
307
7.80M
          break;
308
0
        default:
309
0
          return sourceIllegal;
310
7.80M
      }
311
7.80M
    }
312
313
22.1M
    target += bytesToWrite;
314
22.1M
  }
315
316
745k
  *sourceStart = source;
317
745k
  *targetStart = target;
318
745k
  return result;
319
745k
}
320
321
/* --------------------------------------------------------------------- */
322
323
/*
324
 * Utility routine to tell whether a sequence of bytes is legal UTF-8.
325
 * This must be called with the length pre-determined by the first byte.
326
 * If not calling this from ConvertUTF8to*, then the length can be set by:
327
 *  length = trailingBytesForUTF8[*source]+1;
328
 * and the sequence is illegal right away if there aren't that many bytes
329
 * available.
330
 * If presented with a length > 4, this returns false.  The Unicode
331
 * definition of UTF-8 goes up to 4-byte sequences.
332
 */
333
334
static bool isLegalUTF8(const uint8_t* source, int length)
335
17.8M
{
336
17.8M
  uint8_t a = 0;
337
17.8M
  const uint8_t* srcptr = source + length;
338
339
17.8M
  switch (length)
340
17.8M
  {
341
47
    default:
342
47
      return false;
343
344
      /* Everything else falls through when "true"... */
345
1.53k
    case 4:
346
1.53k
      if ((a = (*--srcptr)) < 0x80 || a > 0xBF)
347
27
        return false;
348
      /* fallthrough */
349
1.50k
      WINPR_FALLTHROUGH
350
351
3.03k
    case 3:
352
3.03k
      if ((a = (*--srcptr)) < 0x80 || a > 0xBF)
353
24
        return false;
354
      /* fallthrough */
355
3.00k
      WINPR_FALLTHROUGH
356
357
3.66k
    case 2:
358
3.66k
      if ((a = (*--srcptr)) > 0xBF)
359
6
        return false;
360
361
3.65k
      switch (*source)
362
3.65k
      {
363
          /* no fall-through in this inner switch */
364
331
        case 0xE0:
365
331
          if (a < 0xA0)
366
19
            return false;
367
368
312
          break;
369
370
344
        case 0xED:
371
344
          if (a > 0x9F)
372
2
            return false;
373
374
342
          break;
375
376
342
        case 0xF0:
377
311
          if (a < 0x90)
378
3
            return false;
379
380
308
          break;
381
382
665
        case 0xF4:
383
665
          if (a > 0x8F)
384
2
            return false;
385
386
663
          break;
387
388
2.00k
        default:
389
2.00k
          if (a < 0x80)
390
34
            return false;
391
1.97k
          break;
392
3.65k
      }
393
      /* fallthrough */
394
3.59k
      WINPR_FALLTHROUGH
395
396
17.8M
    case 1:
397
17.8M
      if (*source >= 0x80 && *source < 0xC2)
398
81
        return false;
399
17.8M
  }
400
401
17.8M
  if (*source > 0xF4)
402
4
    return false;
403
404
17.8M
  return true;
405
17.8M
}
406
407
/* --------------------------------------------------------------------- */
408
409
static ConversionResult winpr_ConvertUTF8toUTF16_Internal(const uint8_t** sourceStart,
410
                                                          const uint8_t* sourceEnd,
411
                                                          uint16_t** targetStart,
412
                                                          const uint16_t* targetEnd,
413
                                                          ConversionFlags flags)
414
642k
{
415
642k
  bool computeLength = (!targetEnd) ? true : false;
416
642k
  ConversionResult result = conversionOK;
417
642k
  const uint8_t* source = *sourceStart;
418
642k
  uint16_t* target = *targetStart;
419
420
18.4M
  while (source < sourceEnd)
421
17.8M
  {
422
17.8M
    uint32_t ch = 0;
423
17.8M
    unsigned short extraBytesToRead =
424
17.8M
        WINPR_ASSERTING_INT_CAST(unsigned short, trailingBytesForUTF8[*source]);
425
426
17.8M
    if ((source + extraBytesToRead) >= sourceEnd)
427
21
    {
428
21
      result = sourceExhausted;
429
21
      break;
430
21
    }
431
432
    /* Do this check whether lenient or strict */
433
17.8M
    if (!isLegalUTF8(source, extraBytesToRead + 1))
434
249
    {
435
249
      result = sourceIllegal;
436
249
      break;
437
249
    }
438
439
    /*
440
     * The cases all fall through. See "Note A" below.
441
     */
442
17.8M
    switch (extraBytesToRead)
443
17.8M
    {
444
0
      case 5:
445
0
        ch += *source++;
446
0
        ch <<= 6; /* remember, illegal UTF-8 */
447
                  /* fallthrough */
448
0
        WINPR_FALLTHROUGH
449
450
0
      case 4:
451
0
        ch += *source++;
452
0
        ch <<= 6; /* remember, illegal UTF-8 */
453
                  /* fallthrough */
454
0
        WINPR_FALLTHROUGH
455
456
1.49k
      case 3:
457
1.49k
        ch += *source++;
458
1.49k
        ch <<= 6;
459
        /* fallthrough */
460
1.49k
        WINPR_FALLTHROUGH
461
462
2.97k
      case 2:
463
2.97k
        ch += *source++;
464
2.97k
        ch <<= 6;
465
        /* fallthrough */
466
2.97k
        WINPR_FALLTHROUGH
467
468
3.58k
      case 1:
469
3.58k
        ch += *source++;
470
3.58k
        ch <<= 6;
471
        /* fallthrough */
472
3.58k
        WINPR_FALLTHROUGH
473
474
17.8M
      case 0:
475
17.8M
        ch += *source++;
476
17.8M
        break;
477
0
      default:
478
0
        return sourceIllegal;
479
17.8M
    }
480
481
17.8M
    ch -= offsetsFromUTF8[extraBytesToRead];
482
483
17.8M
    if ((target >= targetEnd) && (!computeLength))
484
0
    {
485
0
      source -= (extraBytesToRead + 1); /* Back up source pointer! */
486
0
      result = targetExhausted;
487
0
      break;
488
0
    }
489
490
17.8M
    if (ch <= UNI_MAX_BMP)
491
17.8M
    {
492
      /* Target is a character <= 0xFFFF */
493
      /* UTF-16 surrogate values are illegal in UTF-32 */
494
17.8M
      if (ch >= UNI_SUR_HIGH_START && ch <= UNI_SUR_LOW_END)
495
0
      {
496
0
        if (flags == strictConversion)
497
0
        {
498
0
          source -= (extraBytesToRead + 1); /* return to the illegal value itself */
499
0
          result = sourceIllegal;
500
0
          break;
501
0
        }
502
0
        else
503
0
        {
504
0
          if (!computeLength)
505
0
            *target++ = setWcharFrom(UNI_REPLACEMENT_CHAR);
506
0
          else
507
0
            target++;
508
0
        }
509
0
      }
510
17.8M
      else
511
17.8M
      {
512
17.8M
        if (!computeLength)
513
13.3M
          *target++ = setWcharFrom((WCHAR)ch); /* normal case */
514
4.48M
        else
515
4.48M
          target++;
516
17.8M
      }
517
17.8M
    }
518
1.49k
    else if (ch > UNI_MAX_UTF16)
519
0
    {
520
0
      if (flags == strictConversion)
521
0
      {
522
0
        result = sourceIllegal;
523
0
        source -= (extraBytesToRead + 1); /* return to the start */
524
0
        break;                            /* Bail out; shouldn't continue */
525
0
      }
526
0
      else
527
0
      {
528
0
        if (!computeLength)
529
0
          *target++ = setWcharFrom(UNI_REPLACEMENT_CHAR);
530
0
        else
531
0
          target++;
532
0
      }
533
0
    }
534
1.49k
    else
535
1.49k
    {
536
      /* target is a character in range 0xFFFF - 0x10FFFF. */
537
1.49k
      if ((target + 1 >= targetEnd) && (!computeLength))
538
0
      {
539
0
        source -= (extraBytesToRead + 1); /* Back up source pointer! */
540
0
        result = targetExhausted;
541
0
        break;
542
0
      }
543
544
1.49k
      ch -= halfBase;
545
546
1.49k
      if (!computeLength)
547
730
      {
548
730
        *target++ = setWcharFrom((WCHAR)((ch >> halfShift) + UNI_SUR_HIGH_START));
549
730
        *target++ = setWcharFrom((WCHAR)((ch & halfMask) + UNI_SUR_LOW_START));
550
730
      }
551
765
      else
552
765
      {
553
765
        target++;
554
765
        target++;
555
765
      }
556
1.49k
    }
557
17.8M
  }
558
559
642k
  *sourceStart = source;
560
642k
  *targetStart = target;
561
642k
  return result;
562
642k
}
563
564
/**
565
 * WinPR built-in Unicode API
566
 */
567
568
static int winpr_ConvertUTF8toUTF16(const uint8_t* src, int cchSrc, uint16_t* dst, int cchDst)
569
642k
{
570
642k
  size_t length = 0;
571
642k
  uint16_t* dstBeg = nullptr;
572
642k
  uint16_t* dstEnd = nullptr;
573
642k
  const uint8_t* srcBeg = nullptr;
574
642k
  const uint8_t* srcEnd = nullptr;
575
642k
  ConversionResult result = sourceIllegal;
576
577
642k
  if (cchSrc == -1)
578
0
    cchSrc = (int)strnlen((const char*)src, INT32_MAX - 1) + 1;
579
580
642k
  srcBeg = src;
581
642k
  srcEnd = &src[cchSrc];
582
583
642k
  if (cchDst == 0)
584
18.3k
  {
585
18.3k
    result =
586
18.3k
        winpr_ConvertUTF8toUTF16_Internal(&srcBeg, srcEnd, &dstBeg, dstEnd, strictConversion);
587
588
18.3k
    length = dstBeg - (uint16_t*)nullptr;
589
18.3k
  }
590
623k
  else
591
623k
  {
592
623k
    dstBeg = dst;
593
623k
    dstEnd = &dst[cchDst];
594
595
623k
    result =
596
623k
        winpr_ConvertUTF8toUTF16_Internal(&srcBeg, srcEnd, &dstBeg, dstEnd, strictConversion);
597
598
623k
    length = dstBeg - dst;
599
623k
  }
600
601
642k
  if (result == targetExhausted)
602
0
  {
603
0
    SetLastError(ERROR_INSUFFICIENT_BUFFER);
604
0
    return 0;
605
0
  }
606
607
642k
  return (result == conversionOK) ? WINPR_ASSERTING_INT_CAST(int, length) : 0;
608
642k
}
609
610
static int winpr_ConvertUTF16toUTF8(const uint16_t* src, int cchSrc, uint8_t* dst, int cchDst)
611
745k
{
612
745k
  size_t length = 0;
613
745k
  uint8_t* dstBeg = nullptr;
614
745k
  uint8_t* dstEnd = nullptr;
615
745k
  const uint16_t* srcBeg = nullptr;
616
745k
  const uint16_t* srcEnd = nullptr;
617
745k
  ConversionResult result = sourceIllegal;
618
619
745k
  if (cchSrc == -1)
620
0
    cchSrc = (int)_wcsnlen((const WCHAR*)src, INT32_MAX - 1) + 1;
621
622
745k
  srcBeg = src;
623
745k
  srcEnd = &src[cchSrc];
624
625
745k
  if (cchDst == 0)
626
196k
  {
627
196k
    result =
628
196k
        winpr_ConvertUTF16toUTF8_Internal(&srcBeg, srcEnd, &dstBeg, dstEnd, strictConversion);
629
630
196k
    length = dstBeg - ((uint8_t*)nullptr);
631
196k
  }
632
548k
  else
633
548k
  {
634
548k
    dstBeg = dst;
635
548k
    dstEnd = &dst[cchDst];
636
637
548k
    result =
638
548k
        winpr_ConvertUTF16toUTF8_Internal(&srcBeg, srcEnd, &dstBeg, dstEnd, strictConversion);
639
640
548k
    length = dstBeg - dst;
641
548k
  }
642
643
745k
  if (result == targetExhausted)
644
1.57k
  {
645
1.57k
    SetLastError(ERROR_INSUFFICIENT_BUFFER);
646
1.57k
    return 0;
647
1.57k
  }
648
649
743k
  return (result == conversionOK) ? WINPR_ASSERTING_INT_CAST(int, length) : 0;
650
745k
}
651
652
/* --------------------------------------------------------------------- */
653
654
int int_MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int cbMultiByte,
655
                            LPWSTR lpWideCharStr, int cchWideChar)
656
642k
{
657
642k
  size_t cbCharLen = (size_t)cbMultiByte;
658
659
642k
  WINPR_UNUSED(dwFlags);
660
661
  /* If cbMultiByte is 0, the function fails */
662
642k
  if ((cbMultiByte == 0) || (cbMultiByte < -1))
663
0
    return 0;
664
665
642k
  if (cchWideChar < 0)
666
0
    return -1;
667
668
642k
  if (cbMultiByte < 0)
669
0
  {
670
0
    const size_t len = strlen(lpMultiByteStr);
671
0
    if (len >= INT32_MAX)
672
0
      return 0;
673
0
    cbCharLen = (int)len + 1;
674
0
  }
675
642k
  else
676
642k
    cbCharLen = cbMultiByte;
677
678
642k
  WINPR_ASSERT(lpMultiByteStr);
679
642k
  switch (CodePage)
680
642k
  {
681
0
    case CP_ACP:
682
642k
    case CP_UTF8:
683
642k
      break;
684
685
0
    default:
686
0
      WLog_ERR(TAG, "Unsupported encoding %u", CodePage);
687
0
      return 0;
688
642k
  }
689
690
642k
  return winpr_ConvertUTF8toUTF16((const uint8_t*)lpMultiByteStr,
691
642k
                                  WINPR_ASSERTING_INT_CAST(int, cbCharLen),
692
642k
                                  (uint16_t*)lpWideCharStr, cchWideChar);
693
642k
}
694
695
int int_WideCharToMultiByte(UINT CodePage, DWORD dwFlags, LPCWSTR lpWideCharStr, int cchWideChar,
696
                            LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar,
697
                            LPBOOL lpUsedDefaultChar)
698
745k
{
699
745k
  size_t cbCharLen = (size_t)cchWideChar;
700
701
745k
  WINPR_UNUSED(dwFlags);
702
  /* If cchWideChar is 0, the function fails */
703
745k
  if ((cchWideChar == 0) || (cchWideChar < -1))
704
0
    return 0;
705
706
745k
  if (cbMultiByte < 0)
707
0
    return -1;
708
709
745k
  WINPR_ASSERT(lpWideCharStr);
710
  /* If cchWideChar is -1, the string is null-terminated */
711
745k
  if (cchWideChar == -1)
712
0
  {
713
0
    const size_t len = _wcslen(lpWideCharStr);
714
0
    if (len >= INT32_MAX)
715
0
      return 0;
716
0
    cbCharLen = (int)len + 1;
717
0
  }
718
745k
  else
719
745k
    cbCharLen = cchWideChar;
720
721
  /*
722
   * if cbMultiByte is 0, the function returns the required buffer size
723
   * in bytes for lpMultiByteStr and makes no use of the output parameter itself.
724
   */
725
726
745k
  return winpr_ConvertUTF16toUTF8((const uint16_t*)lpWideCharStr,
727
745k
                                  WINPR_ASSERTING_INT_CAST(int, cbCharLen),
728
745k
                                  (uint8_t*)lpMultiByteStr, cbMultiByte);
729
745k
}