Coverage Report

Created: 2026-09-28 06:47

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/wget2/libwget/encoding.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2012-2015 Tim Ruehsen
3
 * Copyright (c) 2015-2026 Free Software Foundation, Inc.
4
 *
5
 * This file is part of libwget.
6
 *
7
 * Libwget is free software: you can redistribute it and/or modify
8
 * it under the terms of the GNU Lesser General Public License as published by
9
 * the Free Software Foundation, either version 3 of the License, or
10
 * (at your option) any later version.
11
 *
12
 * Libwget is distributed in the hope that it will be useful,
13
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
15
 * GNU Lesser General Public License for more details.
16
 *
17
 * You should have received a copy of the GNU Lesser General Public License
18
 * along with libwget.  If not, see <https://www.gnu.org/licenses/>.
19
 *
20
 *
21
 * a collection of charset encoding routines
22
 *
23
 * Changelog
24
 * 02.10.2013  Tim Ruehsen  created
25
 *
26
 */
27
28
#include <config.h>
29
30
#include <string.h>
31
#include <errno.h>
32
33
#ifdef HAVE_ICONV
34
# include <iconv.h>
35
#endif
36
37
#include <langinfo.h>
38
39
#if defined HAVE_IDN2_H && defined WITH_LIBIDN2
40
# include <idn2.h>
41
#elif defined HAVE_IDNA_H && defined WITH_LIBIDN
42
# include <idna.h>
43
# ifdef _WIN32
44
#   include <idn-free.h>
45
# endif
46
#elif defined HAVE_IDN_IDNA_H && defined WITH_LIBIDN
47
// OpenSolaris uses the idn subdir
48
# include <idn/idna.h>
49
#endif
50
51
#include <wget.h>
52
#include "private.h"
53
54
const char *wget_local_charset_encoding(void)
55
9.01k
{
56
9.01k
  const char *encoding = nl_langinfo(CODESET);
57
58
9.01k
  if (encoding && *encoding)
59
9.01k
    return wget_strdup(encoding);
60
61
0
  return wget_strdup("ASCII");
62
9.01k
}
63
64
// void *wget_memiconv(const void *src, size_t length, const char *src_encoding, const char *dst_encoding)
65
int wget_memiconv(const char *src_encoding, const void *src, size_t srclen, const char *dst_encoding, char **out, size_t *outlen)
66
60.2k
{
67
60.2k
  if (!src)
68
0
    return WGET_E_INVALID;
69
70
60.2k
#ifdef HAVE_ICONV
71
60.2k
  if (!src_encoding)
72
12.6k
    src_encoding = "iso-8859-1"; // default character-set for most browsers
73
60.2k
  if (!dst_encoding)
74
0
    dst_encoding = "iso-8859-1"; // default character-set for most browsers
75
76
60.2k
  if (wget_strcasecmp_ascii(src_encoding, dst_encoding)) {
77
50.7k
    int ret = WGET_E_UNKNOWN;
78
50.7k
    iconv_t cd = iconv_open(dst_encoding, src_encoding);
79
80
50.7k
    if (cd != (iconv_t)-1) {
81
49.0k
      char *tmp = (char *) src; // iconv won't change where src points to, but changes tmp itself
82
49.0k
      size_t tmp_len = srclen;
83
49.0k
      size_t dst_len = tmp_len * 6, dst_len_tmp = dst_len;
84
49.0k
      char *dst = wget_malloc(dst_len + 1), *dst_tmp = dst;
85
86
49.0k
      if (!dst) {
87
0
        iconv_close(cd);
88
0
        return WGET_E_MEMORY;
89
0
      }
90
91
49.0k
      errno = 0;
92
49.0k
      if (iconv(cd, (ICONV_CONST char **)&tmp, &tmp_len, &dst_tmp, &dst_len_tmp) == 0
93
30.6k
        && iconv(cd, NULL, NULL, &dst_tmp, &dst_len_tmp) == 0)
94
30.6k
      {
95
30.6k
        debug_printf("transcoded %zu bytes from '%s' to '%s'\n", srclen, src_encoding, dst_encoding);
96
30.6k
        if (out) {
97
          // here we reduce the allocated memory size, if it fails we use the original memory chunk
98
30.6k
          tmp = wget_realloc(dst, dst_len - dst_len_tmp + 1);
99
30.6k
          if (!tmp)
100
0
            tmp = dst;
101
30.6k
          tmp[dst_len - dst_len_tmp] = 0;
102
30.6k
          *out = tmp;
103
30.6k
        } else
104
0
          xfree(dst);
105
106
30.6k
        if (outlen)
107
0
          *outlen = dst_len - dst_len_tmp;
108
109
30.6k
        ret = WGET_E_SUCCESS;
110
30.6k
      } else {
111
        // erno == 0 means some codepoints were encoded non-reversible, treat as error
112
18.3k
        error_printf(_("Failed to transcode '%s' string into '%s' (%d)\n"), src_encoding, dst_encoding, errno);
113
18.3k
        xfree(dst);
114
115
18.3k
        if (out)
116
18.3k
          *out = NULL;
117
118
18.3k
        if (outlen)
119
0
          *outlen = 0;
120
18.3k
      }
121
122
49.0k
      iconv_close(cd);
123
49.0k
    } else
124
1.74k
      error_printf(_("Failed to prepare transcoding '%s' into '%s' (%d)\n"), src_encoding, dst_encoding, errno);
125
126
50.7k
    return ret;
127
50.7k
  }
128
9.44k
#endif
129
130
9.44k
  if (out)
131
9.44k
    *out = wget_strmemdup(src, srclen);
132
133
9.44k
  if (outlen)
134
0
    *outlen = srclen;
135
136
9.44k
  return WGET_E_SUCCESS;
137
60.2k
}
138
139
// src must be a ASCII compatible C string
140
char *wget_striconv(const char *src, const char *src_encoding, const char *dst_encoding)
141
60.2k
{
142
60.2k
  if (!src)
143
0
    return NULL;
144
145
60.2k
  char *dst;
146
60.2k
  if (wget_memiconv(src_encoding, src, strlen(src), dst_encoding, &dst, NULL))
147
20.0k
    return NULL;
148
149
40.1k
  return dst;
150
60.2k
}
151
152
bool wget_str_needs_encoding(const char *s)
153
248k
{
154
248k
  if (!s)
155
0
    return false;
156
157
938k
  while (*s && (*s & ~0x7f) == 0) s++;
158
159
248k
  return *s != 0;
160
248k
}
161
162
bool wget_str_is_valid_utf8(const char *utf8)
163
0
{
164
0
  const unsigned char *s = (const unsigned char *) utf8;
165
166
0
  if (!s)
167
0
    return 0;
168
169
0
  while (*s) {
170
0
    if ((*s & 0x80) == 0) /* 0xxxxxxx ASCII char */
171
0
      s++;
172
0
    else if ((*s & 0xE0) == 0xC0) /* 110xxxxx 10xxxxxx */ {
173
0
      if ((s[1] & 0xC0) != 0x80)
174
0
        return 0;
175
0
      s += 2;
176
0
    } else if ((*s & 0xF0) == 0xE0) /* 1110xxxx 10xxxxxx 10xxxxxx */ {
177
0
      if ((s[1] & 0xC0) != 0x80 || (s[2] & 0xC0) != 0x80)
178
0
        return 0;
179
0
      s += 3;
180
0
    } else if ((*s & 0xF8) == 0xF0) /* 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx */ {
181
0
      if ((s[1] & 0xC0) != 0x80 || (s[2] & 0xC0) != 0x80 || (s[3] & 0xC0) != 0x80)
182
0
        return 0;
183
0
      s += 4;
184
0
    } else
185
0
      return 0;
186
0
  }
187
188
0
  return 1;
189
0
}
190
191
char *wget_str_to_utf8(const char *src, const char *encoding)
192
58.3k
{
193
58.3k
  return wget_striconv(src, encoding, "utf-8");
194
58.3k
}
195
196
char *wget_utf8_to_str(const char *src, const char *encoding)
197
1.83k
{
198
1.83k
  return wget_striconv(src, "utf-8", encoding);
199
1.83k
}
200
201
#ifdef WITH_LIBIDN
202
/*
203
 * Work around a libidn <= 1.30 vulnerability.
204
 *
205
 * The function checks for a valid UTF-8 character sequence before
206
 * passing it to idna_to_ascii_8z().
207
 *
208
 * [1] https://lists.gnu.org/archive/html/help-libidn/2015-05/msg00002.html
209
 * [2] https://lists.gnu.org/archive/html/bug-wget/2015-06/msg00002.html
210
 * [3] https://curl.haxx.se/mail/lib-2015-06/0143.html
211
 */
212
static int WGET_GCC_PURE _utf8_is_valid(const char *utf8)
213
{
214
  const unsigned char *s = (const unsigned char *) utf8;
215
216
  while (*s) {
217
    if ((*s & 0x80) == 0) /* 0xxxxxxx ASCII char */
218
      s++;
219
    else if ((*s & 0xE0) == 0xC0) /* 110xxxxx 10xxxxxx */ {
220
      if ((s[1] & 0xC0) != 0x80)
221
        return 0;
222
      s += 2;
223
    } else if ((*s & 0xF0) == 0xE0) /* 1110xxxx 10xxxxxx 10xxxxxx */ {
224
      if ((s[1] & 0xC0) != 0x80 || (s[2] & 0xC0) != 0x80)
225
        return 0;
226
      s += 3;
227
    } else if ((*s & 0xF8) == 0xF0) /* 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx */ {
228
      if ((s[1] & 0xC0) != 0x80 || (s[2] & 0xC0) != 0x80 || (s[3] & 0xC0) != 0x80)
229
        return 0;
230
      s += 4;
231
    } else
232
      return 0;
233
  }
234
235
  return 1;
236
}
237
#endif
238
239
/* We convert hostnames and thus have to apply IDN2_USE_STD3_ASCII_RULES.
240
 * If we don't do, the result could contain any ascii characters,
241
 * e.g. 'evil.c\u2100.example.com' will be converted into
242
 * 'evil.ca/c.example.com', which seems no good idea. */
243
const char *wget_str_to_ascii(const char *src)
244
106k
{
245
106k
#ifdef WITH_LIBIDN2
246
106k
  if (wget_str_needs_encoding(src)) {
247
49.1k
    char *asc = NULL;
248
49.1k
    int rc;
249
250
49.1k
    debug_printf("toASCII: src='%s', strlen=%zu\n", src, strlen(src));
251
1.89M
    for (const unsigned char *p = (const unsigned char *)src; *p; ++p)
252
1.84M
        debug_printf("%02x ", *p);
253
49.1k
    debug_printf("\n");
254
255
49.1k
    if ((rc = idn2_lookup_u8((uint8_t *)src, (uint8_t **)&asc, IDN2_NONTRANSITIONAL|IDN2_USE_STD3_ASCII_RULES)) != IDN2_OK)
256
40.9k
      rc = idn2_lookup_u8((uint8_t *)src, (uint8_t **)&asc, IDN2_TRANSITIONAL|IDN2_USE_STD3_ASCII_RULES);
257
49.1k
    if (rc == IDN2_OK)
258
23.1k
    {
259
23.1k
      debug_printf("idn2 '%s' -> '%s'\n", src, asc);
260
#  ifdef _WIN32
261
        src = wget_strdup(asc);
262
        idn2_free(asc);
263
#  else
264
23.1k
        src = asc;
265
23.1k
#  endif
266
23.1k
    } else
267
25.9k
      error_printf(_("toASCII(%s) failed (%d): %s\n"), src, rc, idn2_strerror(rc));
268
49.1k
  }
269
#elif defined WITH_LIBIDN
270
  if (wget_str_needs_encoding(src)) {
271
    char *asc = NULL;
272
273
    if (_utf8_is_valid(src)) {
274
      int rc;
275
276
      // idna_to_ascii_8z() automatically converts UTF-8 to lowercase
277
      if ((rc = idna_to_ascii_8z(src, &asc, IDNA_USE_STD3_ASCII_RULES)) == IDNA_SUCCESS) {
278
        // debug_printf("toASCII '%s' -> '%s'\n", src, asc);
279
# ifdef _WIN32
280
        src = wget_strdup(asc);
281
        idn_free(asc);
282
# else
283
        src = asc;
284
# endif
285
      } else
286
        error_printf(_("toASCII failed (%d): %s\n"), rc, idna_strerror(rc));
287
    }
288
    else
289
      error_printf(_("Invalid UTF-8 sequence not converted: '%s'\n"), src);
290
  }
291
#else
292
  if (wget_str_needs_encoding(src)) {
293
    error_printf(_("toASCII not available: '%s'\n"), src);
294
  }
295
#endif
296
297
106k
  return src;
298
106k
}