Coverage Report

Created: 2026-09-14 07:05

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/curl/lib/urlapi.c
Line
Count
Source
1
/***************************************************************************
2
 *                                  _   _ ____  _
3
 *  Project                     ___| | | |  _ \| |
4
 *                             / __| | | | |_) | |
5
 *                            | (__| |_| |  _ <| |___
6
 *                             \___|\___/|_| \_\_____|
7
 *
8
 * Copyright (C) Daniel Stenberg, <daniel@haxx.se>, et al.
9
 *
10
 * This software is licensed as described in the file COPYING, which
11
 * you should have received as part of this distribution. The terms
12
 * are also available at https://curl.se/docs/copyright.html.
13
 *
14
 * You may opt to use, copy, modify, merge, publish, distribute and/or sell
15
 * copies of the Software, and permit persons to whom the Software is
16
 * furnished to do so, under the terms of the COPYING file.
17
 *
18
 * This software is distributed on an "AS IS" basis, WITHOUT WARRANTY OF ANY
19
 * KIND, either express or implied.
20
 *
21
 * SPDX-License-Identifier: curl
22
 *
23
 ***************************************************************************/
24
#include "curl_setup.h"
25
26
#include "urldata.h"
27
#include "urlapi-int.h"
28
#include "strcase.h"
29
#include "url.h"
30
#include "escape.h"
31
#include "curlx/inet_pton.h"
32
#include "curlx/inet_ntop.h"
33
#include "curlx/strdup.h"
34
#include "idn.h"
35
#include "curlx/strparse.h"
36
#include "curl_memrchr.h"
37
38
#ifdef _WIN32
39
/* MS-DOS/Windows style drive prefix, eg c: in c:foo */
40
#define STARTS_WITH_DRIVE_PREFIX(str)        \
41
  ((('a' <= (str)[0] && (str)[0] <= 'z') ||  \
42
    ('A' <= (str)[0] && (str)[0] <= 'Z')) && \
43
   ((str)[1] == ':'))
44
#endif
45
46
/* MS-DOS/Windows style drive prefix, optionally with
47
 * a '|' instead of ':', followed by a slash or NUL */
48
#define STARTS_WITH_URL_DRIVE_PREFIX(str)                  \
49
56
  ((('a' <= (str)[0] && (str)[0] <= 'z') ||                \
50
56
    ('A' <= (str)[0] && (str)[0] <= 'Z')) &&               \
51
56
   ((str)[1] == ':' || (str)[1] == '|') &&                 \
52
56
   ((str)[2] == '/' || (str)[2] == '\\' || (str)[2] == 0))
53
54
/* scheme is not URL encoded, the longest libcurl supported ones are... */
55
115k
#define MAX_SCHEME_LEN 40
56
63
#define MAX_ZONEID_LEN 16
57
58
/*
59
 * If USE_IPV6 is disabled, we still want to parse IPv6 addresses, so make
60
 * sure we have _some_ value for AF_INET6 without polluting our fake value
61
 * everywhere.
62
 */
63
#if !defined(USE_IPV6) && !defined(AF_INET6)
64
#define AF_INET6 (AF_INET + 1)
65
#endif
66
67
0
#define DEFAULT_SCHEME "https"
68
69
static void free_urlhandle(struct Curl_URL *u)
70
27.4k
{
71
27.4k
  curlx_free(u->scheme);
72
27.4k
  curlx_free(u->user);
73
27.4k
  curlx_strzero(u->password);
74
27.4k
  curlx_free(u->password);
75
27.4k
  curlx_free(u->options);
76
27.4k
  curlx_free(u->host);
77
27.4k
  curlx_free(u->zoneid);
78
27.4k
  curlx_free(u->path);
79
27.4k
  curlx_free(u->query);
80
27.4k
  curlx_free(u->fragment);
81
27.4k
}
82
83
/*
84
 * Find the separator at the end of the hostname, or the '?' in cases like
85
 * http://www.example.com?id=2380
86
 */
87
static const char *find_host_sep(const char *url)
88
0
{
89
  /* Find the start of the hostname */
90
0
  const char *sep = strstr(url, "//");
91
0
  if(!sep)
92
0
    sep = url;
93
0
  else
94
0
    sep += 2;
95
96
  /* Find first / or ? */
97
0
  while(*sep && *sep != '/' && *sep != '?')
98
0
    sep++;
99
100
0
  return sep;
101
0
}
102
103
/* convert CURLcode to CURLUcode */
104
#define cc2cu(x) \
105
0
  ((x) == CURLE_TOO_LARGE ? CURLUE_TOO_LARGE : CURLUE_OUT_OF_MEMORY)
106
107
/* urlencode_str() writes data into an output dynbuf and URL-encodes the
108
 * spaces in the source URL accordingly.
109
 *
110
 * This function re-encodes the string, meaning that it leaves already encoded
111
 * bytes as-is and works by encoding only what *has* to be encoded - unless it
112
 * has to uppercase the hex to normalize.
113
 *
114
 * Illegal percent-encoding sequences are left as-is.
115
 *
116
 * URL encoding should be skipped for hostnames, otherwise IDN resolution
117
 * will fail.
118
 *
119
 * 'query' tells if it is a query part or not, or if it is allowed to
120
 * "transition" into a query part with a question mark.
121
 *
122
 * @unittest 1675
123
 */
124
UNITTEST CURLUcode urlencode_str(struct dynbuf *o, const char *url,
125
                                 size_t len, bool relative,
126
                                 unsigned int query);
127
UNITTEST CURLUcode urlencode_str(struct dynbuf *o, const char *url,
128
                                 size_t len, bool relative,
129
                                 unsigned int query)
130
12.9k
{
131
  /* we must add this with whitespace-replacing */
132
12.9k
  const unsigned char *iptr;
133
12.9k
  const unsigned char *host_sep = (const unsigned char *)url;
134
12.9k
  CURLcode result = CURLE_OK;
135
136
12.9k
  DEBUGASSERT((query >= QUERY_NO) && (query <= QUERY_YES));
137
138
12.9k
  if(!relative) {
139
0
    size_t n;
140
0
    host_sep = (const unsigned char *)find_host_sep(url);
141
142
    /* output the first piece as-is */
143
0
    n = (const char *)host_sep - url;
144
0
    result = curlx_dyn_addn(o, url, n);
145
0
    len -= n;
146
0
  }
147
148
4.88M
  for(iptr = host_sep; len && !result;) {
149
4.87M
    if(*iptr == ' ') {
150
277
      if(query != QUERY_YES)
151
220
        result = curlx_dyn_addn(o, "%20", 3);
152
57
      else
153
57
        result = curlx_dyn_addn(o, "+", 1);
154
277
      iptr++;
155
277
      len--;
156
277
    }
157
4.87M
    else if((*iptr < ' ') || (*iptr >= 0x7f)) {
158
4.76M
      unsigned char out[3] = { '%' };
159
4.76M
      Curl_hexbyte(&out[1], *iptr);
160
4.76M
      result = curlx_dyn_addn(o, out, 3);
161
4.76M
      iptr++;
162
4.76M
      len--;
163
4.76M
    }
164
110k
    else if(*iptr == '%' && (len >= 3) &&
165
7.02k
            ISXDIGIT(iptr[1]) && ISXDIGIT(iptr[2]) &&
166
3.61k
            (ISLOWER(iptr[1]) || ISLOWER(iptr[2]))) {
167
      /* uppercase it */
168
3.54k
      unsigned char hex = (unsigned char)((curlx_hexval(iptr[1]) << 4) |
169
3.54k
                                          curlx_hexval(iptr[2]));
170
3.54k
      unsigned char out[3] = { '%' };
171
3.54k
      Curl_hexbyte(&out[1], hex);
172
3.54k
      result = curlx_dyn_addn(o, out, 3);
173
3.54k
      iptr += 3;
174
3.54k
      len -= 3;
175
3.54k
    }
176
106k
    else {
177
106k
      const unsigned char *start = iptr;
178
976k
      while(len) {
179
963k
        if(*iptr == ' ' || *iptr < ' ' || *iptr >= 0x7f)
180
90.6k
          break;
181
873k
        if(*iptr == '%' && (len >= 3) &&
182
38.0k
           ISXDIGIT(iptr[1]) && ISXDIGIT(iptr[2]) &&
183
4.33k
           (ISLOWER(iptr[1]) || ISLOWER(iptr[2])))
184
3.37k
          break;
185
869k
        if(*iptr == '?') {
186
546
          if(query == QUERY_NOT_YET) {
187
9
            iptr++;
188
9
            len--;
189
9
            query = QUERY_YES;
190
9
            break;
191
9
          }
192
546
        }
193
869k
        iptr++;
194
869k
        len--;
195
869k
      }
196
106k
      result = curlx_dyn_addn(o, (const char *)start, (size_t)(iptr - start));
197
106k
    }
198
4.87M
  }
199
200
12.9k
  if(result)
201
0
    return cc2cu(result);
202
12.9k
  return CURLUE_OK;
203
12.9k
}
204
205
/*
206
 * Returns the length of the scheme if the given URL is absolute (as opposed
207
 * to relative). Stores the scheme in the buffer if TRUE and 'buf' is
208
 * non-NULL. The buflen must be larger than MAX_SCHEME_LEN if buf is set.
209
 *
210
 * If 'guess_scheme' is TRUE, it means the URL might be provided without
211
 * scheme.
212
 */
213
size_t Curl_is_absolute_url(const char *url, char *buf, size_t buflen,
214
                            bool guess_scheme)
215
28.5k
{
216
28.5k
  size_t i = 0;
217
28.5k
  DEBUGASSERT(!buf || (buflen > MAX_SCHEME_LEN));
218
28.5k
  (void)buflen; /* only used in debug-builds */
219
28.5k
  if(buf)
220
13.8k
    buf[0] = 0; /* always leave a defined value in buf */
221
#ifdef _WIN32
222
  if(guess_scheme && STARTS_WITH_DRIVE_PREFIX(url))
223
    return 0;
224
#endif
225
28.5k
  if(ISALPHA(url[0])) {
226
28.4k
    if(buf)
227
13.8k
      buf[0] = Curl_raw_tolower(url[0]);
228
115k
    for(i = 1; i < MAX_SCHEME_LEN; ++i) {
229
115k
      char s = url[i];
230
115k
      if(s && (ISALNUM(s) || (s == '+') || (s == '-') || (s == '.'))) {
231
87.0k
        if(buf)
232
42.0k
          buf[i] = Curl_raw_tolower(s);
233
87.0k
      }
234
28.4k
      else {
235
28.4k
        break;
236
28.4k
      }
237
115k
    }
238
28.4k
  }
239
28.5k
  if(i && (url[i] == ':') && ((url[i + 1] == '/') || !guess_scheme)) {
240
    /* If this does not guess scheme, the scheme always ends with the colon so
241
       that this also detects data: URLs etc. In guessing mode, data: could
242
       be the hostname "data" with a specified port number. */
243
244
    /* the length of the scheme is the name part only */
245
28.3k
    size_t len = i;
246
28.3k
    if(buf)
247
13.8k
      buf[i] = 0;
248
28.3k
    return len;
249
28.3k
  }
250
232
  if(buf)
251
0
    buf[0] = 0;
252
232
  return 0;
253
28.5k
}
254
255
/* scan for byte values <= 31, 127 and maybe space */
256
static bool badoctets(const char *input, size_t n, int flags)
257
2.09k
{
258
2.09k
  const uint8_t *p = (const unsigned char *)input;
259
2.09k
  const uint8_t control = flags & CURLU_ALLOW_SPACE ? 0x1f : 0x20;
260
7.10M
  while(n--) {
261
7.10M
    if(*p <= control || *p == 127)
262
37
      return TRUE;
263
7.10M
    p++;
264
7.10M
  }
265
2.05k
  return FALSE;
266
2.09k
}
267
268
/*
269
 * parse_hostname_login()
270
 *
271
 * Parse the login details (username, password and options) from the URL and
272
 * strip them out of the hostname
273
 *
274
 * @unittest 1675
275
 */
276
UNITTEST CURLUcode parse_hostname_login(struct Curl_URL *u,
277
                                        const char *login,
278
                                        size_t len,
279
                                        unsigned int flags,
280
                                        size_t *hostname_offset);
281
UNITTEST CURLUcode parse_hostname_login(struct Curl_URL *u,
282
                                        const char *login,
283
                                        size_t len,
284
                                        unsigned int flags,
285
                                        size_t *hostname_offset)
286
13.6k
{
287
13.6k
  CURLUcode ures = CURLUE_OK;
288
13.6k
  CURLcode result;
289
13.6k
  char *userp = NULL;
290
13.6k
  char *passwdp = NULL;
291
13.6k
  char *optionsp = NULL;
292
13.6k
  const struct Curl_scheme *h = NULL;
293
294
  /* At this point, we assume all the other special cases have been taken
295
   * care of, so the host is at most
296
   *
297
   *   [user[:password][;options]]@]hostname
298
   *
299
   * We need somewhere to put the embedded details, so do that first.
300
   */
301
13.6k
  const char *ptr;
302
303
13.6k
  DEBUGASSERT(login);
304
305
13.6k
  *hostname_offset = 0;
306
13.6k
  ptr = memchr(login, '@', len);
307
13.6k
  if(!ptr)
308
13.4k
    goto out;
309
310
  /* We will now try to extract the
311
   * possible login information in a string like:
312
   * ftp://user:password@ftp.site.example:8021/README */
313
209
  ptr++;
314
315
  /* if this is a known scheme, get some details */
316
209
  if(u->scheme)
317
209
    h = Curl_get_scheme(u->scheme);
318
319
  /* We could use the login information in the URL so extract it. Only parse
320
     options if the handler says we should. Note that 'h' might be NULL! */
321
209
  result = Curl_parse_login_details(login, ptr - login - 1,
322
209
                                    &userp, &passwdp,
323
209
                                    (h && (h->flags & PROTOPT_URLOPTIONS)) ?
324
209
                                    &optionsp : NULL);
325
209
  if(result) {
326
    /* the only possible error from Curl_parse_login_details is out of
327
       memory: */
328
0
    ures = CURLUE_OUT_OF_MEMORY;
329
0
    goto out;
330
0
  }
331
332
209
  if(userp) {
333
209
    if(flags & CURLU_DISALLOW_USER) {
334
      /* Option DISALLOW_USER is set and URL contains username. */
335
0
      ures = CURLUE_USER_NOT_ALLOWED;
336
0
      goto out;
337
0
    }
338
209
    curlx_free(u->user);
339
209
    u->user = userp;
340
209
  }
341
342
209
  if(passwdp) {
343
25
    curlx_strzero(u->password);
344
25
    curlx_free(u->password);
345
25
    u->password = passwdp;
346
25
  }
347
348
209
  if(optionsp) {
349
5
    curlx_free(u->options);
350
5
    u->options = optionsp;
351
5
  }
352
353
209
  if(userp && badoctets(userp, strlen(userp), flags))
354
0
    ures = CURLUE_BAD_USER;
355
209
  else if(passwdp && badoctets(passwdp, strlen(passwdp), flags))
356
0
    ures = CURLUE_BAD_PASSWORD;
357
209
  else if(optionsp && badoctets(optionsp, strlen(optionsp), flags))
358
0
    ures = CURLUE_MALFORMED_INPUT;
359
360
209
  userp = passwdp = optionsp = NULL;
361
362
209
  if(!ures) {
363
    /* the hostname starts at this offset */
364
209
    *hostname_offset = ptr - login;
365
209
    return CURLUE_OK;
366
209
  }
367
368
13.4k
out:
369
370
13.4k
  curlx_free(userp);
371
13.4k
  curlx_strzero(passwdp);
372
13.4k
  curlx_free(passwdp);
373
13.4k
  curlx_free(optionsp);
374
13.4k
  curlx_safefree(u->user);
375
13.4k
  curlx_strzero(u->password);
376
13.4k
  curlx_safefree(u->password);
377
13.4k
  curlx_safefree(u->options);
378
379
13.4k
  return ures;
380
209
}
381
382
/* @unittest 1653 */
383
UNITTEST CURLUcode parse_port(struct Curl_URL *u, struct dynbuf *host,
384
                              bool has_scheme);
385
UNITTEST CURLUcode parse_port(struct Curl_URL *u, struct dynbuf *host,
386
                              bool has_scheme)
387
13.6k
{
388
13.6k
  const char *portptr;
389
13.6k
  const char *hostname = curlx_dyn_ptr(host);
390
  /*
391
   * Find the end of an IPv6 address on the ']' ending bracket.
392
   */
393
13.6k
  u->portnum = 0;
394
13.6k
  u->port_present = FALSE;
395
13.6k
  if(hostname[0] == '[') {
396
59
    portptr = memchr(hostname + 1, ']', curlx_dyn_len(host) - 1);
397
59
    if(!portptr)
398
1
      return CURLUE_BAD_IPV6;
399
58
    portptr++;
400
    /* this is a RFC2732-style specified IP-address */
401
58
    if(*portptr) {
402
11
      if(*portptr != ':')
403
7
        return CURLUE_BAD_PORT_NUMBER;
404
11
    }
405
47
    else
406
47
      portptr = NULL;
407
58
  }
408
13.5k
  else
409
13.5k
    portptr = memchr(hostname, ':', curlx_dyn_len(host));
410
411
13.6k
  if(portptr) {
412
151
    curl_off_t port;
413
151
    size_t keep = portptr - hostname;
414
151
    int rc;
415
416
    /* Browser behavior adaptation. If there is a colon with no digits after,
417
       cut off the name there which makes us ignore the colon and use the
418
       default port. Firefox, Chrome and Safari all do that.
419
420
       Do not do it if the URL has no scheme, to make something that looks like
421
       a scheme not work! */
422
151
    curlx_dyn_setlen(host, keep);
423
151
    portptr++;
424
151
    if(!*portptr)
425
16
      return has_scheme ? CURLUE_OK : CURLUE_BAD_PORT_NUMBER;
426
135
    if(*portptr == '\\')
427
1
      return CURLUE_BACKSLASH;
428
134
    rc = curlx_str_number(&portptr, &port, 0xffff);
429
134
    if(rc)
430
9
      return CURLUE_BAD_PORT_NUMBER;
431
125
    else if(*portptr == '\\')
432
1
      return CURLUE_BACKSLASH;
433
124
    else if(*portptr)
434
6
      return CURLUE_BAD_PORT_NUMBER;
435
436
118
    u->portnum = (uint16_t)port;
437
118
    u->port_present = TRUE;
438
118
  }
439
440
13.6k
  return CURLUE_OK;
441
13.6k
}
442
443
/* This function assumes 'hostname' now starts with [. It trims 'hostname' in
444
 * place and it sets u->zoneid if present.
445
 *
446
 * @unittest 1675
447
 */
448
UNITTEST CURLUcode ipv6_parse(struct Curl_URL *u, char *hostname,
449
                              size_t hlen);
450
UNITTEST CURLUcode ipv6_parse(struct Curl_URL *u, char *hostname,
451
                              size_t hlen) /* length of hostname */
452
51
{
453
51
  size_t len;
454
51
  DEBUGASSERT(*hostname == '[');
455
51
  if(hlen < 4) /* '[::]' is the shortest possible valid string */
456
1
    return CURLUE_BAD_IPV6;
457
50
  hostname++;
458
50
  hlen -= 2;
459
460
  /* only valid IPv6 letters are ok */
461
50
  len = strspn(hostname, "0123456789abcdefABCDEF:.");
462
463
50
  if(hlen != len) {
464
19
    hlen = len;
465
19
    if(hostname[len] == '%') {
466
      /* this could now be '%[zone id]' */
467
14
      char zoneid[MAX_ZONEID_LEN];
468
14
      int i = 0;
469
14
      char *h = &hostname[len + 1];
470
      /* pass '25' if present and is a URL encoded percent sign */
471
14
      if(!strncmp(h, "25", 2) && h[2] && (h[2] != ']'))
472
0
        h += 2;
473
73
      while(*h && (*h != ']') && (i < (MAX_ZONEID_LEN - 1)) &&
474
63
            (*h != ' '))
475
59
        zoneid[i++] = *h++;
476
14
      if(!i || (']' != *h))
477
5
        return CURLUE_BAD_IPV6;
478
9
      zoneid[i] = 0;
479
9
      u->zoneid = curlx_strdup(zoneid);
480
9
      if(!u->zoneid)
481
0
        return CURLUE_OUT_OF_MEMORY;
482
9
      hostname[len] = ']'; /* insert end bracket */
483
9
      hostname[len + 1] = 0; /* terminate the hostname */
484
9
    }
485
5
    else
486
5
      return CURLUE_BAD_IPV6;
487
    /* hostname is fine */
488
19
  }
489
490
  /* Normalize the IPv6 address */
491
40
  {
492
40
    char dest[16]; /* fits a binary IPv6 address */
493
40
    hostname[hlen] = 0; /* end the address there */
494
40
    if(curlx_inet_pton(AF_INET6, hostname, dest) != 1)
495
10
      return CURLUE_BAD_IPV6;
496
30
    if(!curlx_inet_ntop(AF_INET6, dest, hostname, hlen + 1)) {
497
26
      hlen = strlen(hostname); /* might be shorter now */
498
26
      hostname[hlen + 1] = 0;
499
26
    }
500
30
    hostname[hlen] = ']'; /* restore ending bracket */
501
30
  }
502
0
  return CURLUE_OK;
503
40
}
504
505
/* characters not allowed in hostnames:
506
   " \r\n\t/:#?!@{}[]\\$\'\"^`*<>=;,+&()%|" */
507
508
static const bool invalid_host_char[256] = {
509
  1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x00-0x0F */
510
  1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x10-0x1F */
511
  1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 1, /* 0x20-0x2F */
512
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, /* 0x30-0x3F */
513
  1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x40-0x4F */
514
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, /* 0x50-0x5F */
515
  1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x60-0x6F */
516
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 0, 1  /* 0x70-0x7F */
517
};
518
519
/* the input is a confirmed hostname, never an IPv6 address */
520
static CURLUcode hostname_check(char *hostname, size_t hlen)
521
13.4k
{
522
13.4k
  size_t i;
523
134k
  for(i = 0; i < hlen; i++) {
524
121k
    if(invalid_host_char[(unsigned char)hostname[i]])
525
73
      return CURLUE_BAD_HOSTNAME;
526
121k
  }
527
13.3k
  if((hlen >= 2) &&
528
13.3k
     (hostname[hlen - 1] == '.') && (hostname[hlen - 2] == '.'))
529
    /* more than one trailing dot is not allowed */
530
1
    return CURLUE_BAD_HOSTNAME;
531
13.3k
  else if((hlen == 1) && (hostname[0] == '.'))
532
    /* a single dot alone is not allowed */
533
1
    return CURLUE_BAD_HOSTNAME;
534
13.3k
  return CURLUE_OK;
535
13.3k
}
536
537
/* the input is a hostname or perhaps an IPv6 address */
538
static CURLUcode hostname_check6(struct Curl_URL *u, char *hostname,
539
                                size_t hlen) /* length of hostname */
540
0
{
541
0
  DEBUGASSERT(hostname);
542
543
0
  if(!hlen)
544
0
    return CURLUE_NO_HOST;
545
0
  else if(hostname[0] == '[')
546
0
    return ipv6_parse(u, hostname, hlen);
547
548
0
  return hostname_check(hostname, hlen);
549
0
}
550
551
/*
552
 * Handle partial IPv4 numerical addresses and different bases, like
553
 * '16843009', '0x7f', '0x7f.1' '0177.1.1.1' etc.
554
 *
555
 * If the given input string is syntactically wrong IPv4 or any part for
556
 * example is too big, this function returns HOST_NAME.
557
 *
558
 * Output the "normalized" version of that input string in plain quad decimal
559
 * integers.
560
 *
561
 * A single dot following the numerical address is accepted and "swallowed" as
562
 * if it was never there.
563
 *
564
 * Returns the host type.
565
 *
566
 * @unittest 1675
567
 */
568
UNITTEST int ipv4_normalize(struct dynbuf *host);
569
UNITTEST int ipv4_normalize(struct dynbuf *host)
570
13.5k
{
571
13.5k
  bool done = FALSE;
572
13.5k
  int n = 0;
573
13.5k
  const char *c = curlx_dyn_ptr(host);
574
13.5k
  unsigned int parts[4] = { 0, 0, 0, 0 };
575
13.5k
  CURLcode result = CURLE_OK;
576
577
13.5k
  if(!ISDIGIT(*c))
578
13.3k
    return HOST_NAME;
579
580
429
  while(!done) {
581
293
    int rc;
582
293
    curl_off_t l;
583
293
    if(*c == '0') {
584
103
      if((c[1] | 0x20) == 'x') {
585
3
        c += 2; /* skip the prefix */
586
3
        rc = curlx_str_hex(&c, &l, UINT_MAX);
587
3
        if(rc)
588
1
          return HOST_NAME;
589
3
      }
590
100
      else
591
100
        rc = curlx_str_octal(&c, &l, UINT_MAX);
592
103
    }
593
190
    else
594
190
      rc = curlx_str_number(&c, &l, UINT_MAX);
595
596
292
    if(rc) {
597
17
      if(!n || (rc != STRE_NO_NUM) || *c)
598
7
        return HOST_NAME;
599
10
      n--;
600
10
    }
601
275
    else
602
275
      parts[n] = (unsigned int)l;
603
604
285
    switch(*c) {
605
113
    case '.':
606
113
      if(n == 3) {
607
3
        if(c[1])
608
          /* something follows this dot */
609
2
          return HOST_NAME;
610
1
        done = TRUE;
611
1
      }
612
110
      else {
613
110
        n++;
614
110
        c++;
615
110
      }
616
111
      break;
617
618
135
    case '\0':
619
135
      done = TRUE;
620
135
      break;
621
622
37
    default:
623
37
      return HOST_NAME;
624
285
    }
625
285
  }
626
627
136
  switch(n) {
628
94
  case 0: /* a -- 32 bits */
629
94
    curlx_dyn_reset(host);
630
631
94
    result = curlx_dyn_addf(host, "%u.%u.%u.%u",
632
94
                            (parts[0] >> 24),
633
94
                            ((parts[0] >> 16) & 0xff),
634
94
                            ((parts[0] >> 8) & 0xff),
635
94
                            (parts[0] & 0xff));
636
94
    break;
637
10
  case 1: /* a.b -- 8.24 bits */
638
10
    if((parts[0] > 0xff) || (parts[1] > 0xffffff))
639
8
      return HOST_NAME;
640
2
    curlx_dyn_reset(host);
641
2
    result = curlx_dyn_addf(host, "%u.%u.%u.%u",
642
2
                            parts[0],
643
2
                            ((parts[1] >> 16) & 0xff),
644
2
                            ((parts[1] >> 8) & 0xff),
645
2
                            (parts[1] & 0xff));
646
2
    break;
647
21
  case 2: /* a.b.c -- 8.8.16 bits */
648
21
    if((parts[0] > 0xff) || (parts[1] > 0xff) || (parts[2] > 0xffff))
649
13
      return HOST_NAME;
650
8
    curlx_dyn_reset(host);
651
8
    result = curlx_dyn_addf(host, "%u.%u.%u.%u",
652
8
                            parts[0],
653
8
                            parts[1],
654
8
                            ((parts[2] >> 8) & 0xff),
655
8
                            (parts[2] & 0xff));
656
8
    break;
657
11
  case 3: /* a.b.c.d -- 8.8.8.8 bits */
658
11
    if((parts[0] > 0xff) || (parts[1] > 0xff) || (parts[2] > 0xff) ||
659
5
       (parts[3] > 0xff))
660
9
      return HOST_NAME;
661
2
    curlx_dyn_reset(host);
662
2
    result = curlx_dyn_addf(host, "%u.%u.%u.%u",
663
2
                            parts[0],
664
2
                            parts[1],
665
2
                            parts[2],
666
2
                            parts[3]);
667
2
    break;
668
136
  }
669
106
  if(result)
670
0
    return HOST_ERROR;
671
106
  return HOST_IPV4;
672
106
}
673
674
/* if necessary, replace the host content with a URL decoded version */
675
static CURLUcode urldecode_host(struct dynbuf *host)
676
13.5k
{
677
13.5k
  const char *per;
678
13.5k
  const char *hostname = curlx_dyn_ptr(host);
679
13.5k
  per = memchr(hostname, '%', curlx_dyn_len(host));
680
13.5k
  if(!per)
681
    /* nothing to decode */
682
13.5k
    return CURLUE_OK;
683
69
  else {
684
    /* encoded */
685
69
    size_t dlen;
686
69
    char *decoded;
687
69
    CURLcode result = Curl_urldecode(hostname, 0, &decoded, &dlen,
688
69
                                     REJECT_CTRL);
689
69
    if(result)
690
3
      return CURLUE_BAD_HOSTNAME;
691
66
    curlx_dyn_reset(host);
692
66
    result = curlx_dyn_addn(host, decoded, dlen);
693
66
    curlx_free(decoded);
694
66
    if(result)
695
0
      return cc2cu(result);
696
66
  }
697
698
66
  return CURLUE_OK;
699
13.5k
}
700
701
static CURLUcode parse_authority(struct Curl_URL *u,
702
                                 const char *auth, size_t authlen,
703
                                 unsigned int flags,
704
                                 struct dynbuf *host,
705
                                 bool has_scheme)
706
13.6k
{
707
13.6k
  size_t offset;
708
13.6k
  CURLUcode uc;
709
13.6k
  CURLcode result;
710
711
  /*
712
   * Parse the login details and strip them out of the hostname.
713
   */
714
13.6k
  uc = parse_hostname_login(u, auth, authlen, flags, &offset);
715
13.6k
  if(uc)
716
0
    return uc;
717
718
13.6k
  result = curlx_dyn_addn(host, auth + offset, authlen - offset);
719
13.6k
  if(result) {
720
0
    uc = cc2cu(result);
721
0
    return uc;
722
0
  }
723
724
  /* parse_port() also sets the hostname length correctly */
725
13.6k
  uc = parse_port(u, host, has_scheme);
726
727
13.6k
  if(!curlx_dyn_len(host))
728
    /* this makes no-host errors override port number problems */
729
39
    uc = CURLUE_NO_HOST;
730
13.6k
  if(!uc)
731
13.5k
    uc = urldecode_host(host);
732
13.6k
  if(uc)
733
61
    ;
734
13.5k
  else if(auth[offset] == '[')
735
51
    uc = ipv6_parse(u, curlx_dyn_ptr(host), curlx_dyn_len(host));
736
13.5k
  else {
737
    /* ipv4_normalize() returns *NAME, *IPV4 or *ERROR */
738
13.5k
    int type = ipv4_normalize(host);
739
740
13.5k
    if(type == HOST_NAME)
741
13.4k
      uc = hostname_check(curlx_dyn_ptr(host), curlx_dyn_len(host));
742
106
    else if(type == HOST_ERROR)
743
0
      uc = CURLUE_OUT_OF_MEMORY;
744
13.5k
  }
745
746
13.6k
  return uc;
747
13.6k
}
748
749
/* used for HTTP/2 server push */
750
CURLUcode Curl_url_set_authority(CURLU *u, const char *authority)
751
0
{
752
0
  CURLUcode ures;
753
0
  struct dynbuf host;
754
755
0
  DEBUGASSERT(authority);
756
0
  curlx_dyn_init(&host, CURL_MAX_INPUT_LENGTH);
757
758
0
  ures = parse_authority(u, authority, strlen(authority),
759
0
                         CURLU_DISALLOW_USER, &host, !!u->scheme);
760
0
  if(ures)
761
0
    curlx_dyn_free(&host);
762
0
  else {
763
0
    curlx_free(u->host);
764
0
    u->host = curlx_dyn_ptr(&host);
765
0
  }
766
0
  return ures;
767
0
}
768
769
/*
770
 * "Remove Dot Segments"
771
 * https://datatracker.ietf.org/doc/html/rfc3986#section-5.2.4
772
 */
773
774
static bool is_dot(const char **str, size_t *clen)
775
4.20M
{
776
4.20M
  const char *p = *str;
777
4.20M
  if(*p == '.') {
778
240k
    (*str)++;
779
240k
    (*clen)--;
780
240k
    return TRUE;
781
240k
  }
782
3.96M
  else if((*clen >= 3) &&
783
3.96M
          (p[0] == '%') && (p[1] == '2') && ((p[2] | 0x20) == 'e')) {
784
1.30k
    *str += 3;
785
1.30k
    *clen -= 3;
786
1.30k
    return TRUE;
787
1.30k
  }
788
3.96M
  return FALSE;
789
4.20M
}
790
791
1.46M
#define ISSLASH(x) ((x) == '/')
792
793
/* prescan the string to see if it needs work */
794
static bool needs_dedotdot(const char *p, size_t pn)
795
1.57k
{
796
  /* a single byte path cannot be cleaned up */
797
1.57k
  if(pn < 2)
798
0
    return FALSE;
799
1.57k
  if(!memchr(p, '.', pn) && !memchr(p, '%', pn))
800
274
    return FALSE;
801
4.04M
  while(pn) {
802
4.04M
    if(is_dot(&p, &pn)) {
803
      /* "./" or dot before end of string */
804
118k
      if(!pn || ISSLASH(*p))
805
477
        return TRUE;
806
      /* "../" or ".." before end of string */
807
118k
      else if(is_dot(&p, &pn) && (!pn || ISSLASH(*p)))
808
135
        return TRUE;
809
118k
    }
810
3.92M
    else {
811
3.92M
      p++;
812
3.92M
      pn--;
813
3.92M
    }
814
4.04M
  }
815
690
  return FALSE;
816
1.30k
}
817
818
/*
819
 * dedotdotify()
820
 *
821
 * This function gets a null-terminated path with dot and dotdot sequences
822
 * passed in and strips them off according to the rules in RFC 3986 section
823
 * 5.2.4.
824
 *
825
 * The function handles a path. It should not contain the query nor fragment.
826
 *
827
 * RETURNS
828
 *
829
 * Zero for success and 'out' set to an allocated string (or NULL if there's
830
 * nothing to do).
831
 *
832
 * @unittest 1395
833
 */
834
UNITTEST int dedotdotify(const char *input, size_t clen, char **outp);
835
UNITTEST int dedotdotify(const char *input, size_t clen, char **outp)
836
1.57k
{
837
1.57k
  struct dynbuf out;
838
1.57k
  CURLcode result = CURLE_OK;
839
840
  /* variables for leading dot checks */
841
1.57k
  const char *dinput = input;
842
1.57k
  size_t dlen = clen;
843
844
1.57k
  *outp = NULL;
845
1.57k
  if(!needs_dedotdot(input, clen))
846
964
    return 0;
847
848
612
  curlx_dyn_init(&out, clen + 1);
849
850
  /* if the input buffer begins with a prefix of "../" or "./", then remove
851
     that prefix from the input buffer; otherwise, */
852
612
  if(is_dot(&dinput, &dlen)) {
853
0
    if(ISSLASH(*dinput)) {
854
      /* one dot followed by a slash */
855
0
      input = dinput + 1;
856
0
      clen = dlen - 1;
857
0
    }
858
859
    /* if the input buffer consists only of "." or "..", then remove
860
       that from the input buffer; otherwise, */
861
0
    else if(is_dot(&dinput, &dlen)) {
862
0
      if(!dlen)
863
        /* .. [end] */
864
0
        goto end;
865
0
      else if(ISSLASH(*dinput)) {
866
        /* ../ */
867
0
        input = dinput + 1;
868
0
        clen = dlen - 1;
869
0
      }
870
0
    }
871
0
  }
872
873
1.12M
  while(clen && !result) { /* until end of path content */
874
1.12M
    if(ISSLASH(*input)) {
875
29.3k
      const char *p = &input[1];
876
29.3k
      size_t blen = clen - 1;
877
      /* if the input buffer begins with a prefix of "/./" or "/.", where "."
878
         is a complete path segment, then replace that prefix with "/" in the
879
         input buffer; otherwise, */
880
29.3k
      if(is_dot(&p, &blen)) {
881
12.2k
        if(!blen) { /* /. */
882
25
          result = curlx_dyn_addn(&out, "/", 1);
883
25
          break;
884
25
        }
885
12.2k
        else if(ISSLASH(*p)) { /* /./ */
886
1.95k
          input = p;
887
1.95k
          clen = blen;
888
1.95k
          continue;
889
1.95k
        }
890
891
        /* if the input buffer begins with a prefix of "/../" or "/..", where
892
           ".." is a complete path segment, then replace that prefix with "/"
893
           in the input buffer and remove the last segment and its preceding
894
           "/" (if any) from the output buffer; otherwise, */
895
10.2k
        else if(is_dot(&p, &blen) && (ISSLASH(*p) || !blen)) {
896
          /* remove the last segment from the output buffer */
897
1.48k
          size_t len = curlx_dyn_len(&out);
898
1.48k
          if(len) {
899
1.33k
            const char *ptr = curlx_dyn_ptr(&out);
900
1.33k
            const char *last = memrchr(ptr, '/', len);
901
1.33k
            if(last)
902
              /* trim the output at the slash */
903
1.33k
              curlx_dyn_setlen(&out, last - ptr);
904
1.33k
          }
905
906
1.48k
          if(blen) { /* /../ */
907
1.47k
            input = p;
908
1.47k
            clen = blen;
909
1.47k
            continue;
910
1.47k
          }
911
12
          result = curlx_dyn_addn(&out, "/", 1);
912
12
          break;
913
1.48k
        }
914
12.2k
      }
915
29.3k
    }
916
917
    /* move the first path segment in the input buffer to the end of the
918
       output buffer, including the initial "/" character (if any) and any
919
       subsequent characters up to, but not including, the next "/" character
920
       or the end of the input buffer. */
921
922
1.11M
    result = curlx_dyn_addn(&out, input, 1);
923
1.11M
    input++;
924
1.11M
    clen--;
925
1.11M
  }
926
612
end:
927
612
  if(!result) {
928
612
    if(curlx_dyn_len(&out))
929
612
      *outp = curlx_dyn_ptr(&out);
930
0
    else {
931
0
      *outp = curlx_strdup("");
932
0
      if(!*outp)
933
0
        return 1;
934
0
    }
935
612
  }
936
612
  return result ? 1 : 0; /* success */
937
612
}
938
939
/*
940
 * @unittest 1675
941
 */
942
UNITTEST CURLUcode parse_file(const char *url, size_t urllen, CURLU *u,
943
                              const char **pathp, size_t *pathlenp);
944
UNITTEST CURLUcode parse_file(const char *url, size_t urllen, CURLU *u,
945
                              const char **pathp, size_t *pathlenp)
946
42
{
947
42
  const char *path;
948
42
  size_t pathlen;
949
950
42
  *pathp = NULL;
951
42
  *pathlenp = 0;
952
42
  if(urllen <= 6)
953
    /* file:/ is not enough to actually be a complete file: URL */
954
1
    return CURLUE_BAD_FILE_URL;
955
956
  /* path has been allocated large enough to hold this */
957
41
  path = &url[5];
958
41
  pathlen = urllen - 5;
959
960
  /* RFC 8089: file-hier-part = ( "//" auth-path ) / local-path, where
961
     local-path also starts with a "/". So reject anything that does not
962
     start with at least one "/" */
963
41
  if(path[0] != '/')
964
7
    return CURLUE_BAD_FILE_URL;
965
966
  /* Extra handling URLs with an authority component (i.e. that start with
967
   * "file://")
968
   *
969
   * We allow omitted hostname (e.g. file:/<path>) -- valid according to
970
   * RFC 8089, but not the (current) WHAT-WG URL spec.
971
   */
972
34
  if(path[1] == '/') {
973
    /* swallow the two slashes */
974
17
    const char *ptr = &path[2];
975
976
    /*
977
     * According to RFC 8089, a file: URL can be reliably dereferenced if:
978
     *
979
     *  o it has no/blank hostname, or
980
     *
981
     *  o the hostname matches "localhost" (case-insensitively), or
982
     *
983
     *  o the hostname is a FQDN that resolves to this machine, or
984
     *
985
     * For brevity, we only consider URLs with empty, "localhost", or
986
     * "127.0.0.1" hostnames as local, otherwise as an UNC String.
987
     *
988
     * Additionally, there is an exception for URLs with a Windows drive
989
     * letter in the authority (which was accidentally omitted from RFC 8089
990
     * Appendix E, but believe me, it was meant to be there. --MK)
991
     */
992
17
    if(ptr[0] != '/' && !STARTS_WITH_URL_DRIVE_PREFIX(ptr)) {
993
      /* the URL includes a hostname, it must match "localhost" or
994
         "127.0.0.1" to be valid */
995
9
      if(checkprefix("localhost/", ptr) ||
996
9
         checkprefix("127.0.0.1/", ptr)) {
997
0
        ptr += 9; /* now points to the slash after the host */
998
0
      }
999
9
      else
1000
        /* Invalid file://hostname/, expected localhost or 127.0.0.1 or
1001
           none */
1002
9
        return CURLUE_BAD_FILE_URL;
1003
9
    }
1004
1005
8
    path = ptr;
1006
8
    pathlen = urllen - (ptr - url);
1007
8
  }
1008
1009
25
#if !defined(_WIN32) && !defined(MSDOS) && !defined(__CYGWIN__)
1010
  /* Do not allow Windows drive letters when not in Windows.
1011
   * This catches both "file:/c:" and "file:c:" */
1012
25
  if(('/' == path[0] && STARTS_WITH_URL_DRIVE_PREFIX(&path[1])) ||
1013
22
     STARTS_WITH_URL_DRIVE_PREFIX(path)) {
1014
    /* File drive letters are only accepted in MS-DOS/Windows */
1015
7
    return CURLUE_BAD_FILE_URL;
1016
7
  }
1017
#else
1018
  /* If the path starts with a slash and a drive letter, ditch the slash */
1019
  if('/' == path[0] && STARTS_WITH_URL_DRIVE_PREFIX(&path[1])) {
1020
    /* This cannot be done with strcpy, as the memory chunks overlap! */
1021
    path++;
1022
    pathlen--;
1023
  }
1024
#endif
1025
18
  u->scheme = curlx_strdup("file");
1026
18
  if(!u->scheme)
1027
0
    return CURLUE_OUT_OF_MEMORY;
1028
1029
18
  *pathp = path;
1030
18
  *pathlenp = pathlen;
1031
18
  return CURLUE_OK;
1032
18
}
1033
1034
static CURLUcode parse_scheme(const char *url, CURLU *u, char *schemebuf,
1035
                              size_t schemelen, unsigned int flags,
1036
                              const char **hostpp)
1037
13.8k
{
1038
  /* clear path */
1039
13.8k
  const char *schemep = NULL;
1040
1041
13.8k
  if(schemelen) {
1042
13.8k
    int num_slashes = 0;
1043
13.8k
    const char *p = &url[schemelen + 1];
1044
13.8k
    if(!Curl_getn_scheme(schemebuf, schemelen) &&
1045
511
       !(flags & CURLU_NON_SUPPORT_SCHEME))
1046
0
      return CURLUE_UNSUPPORTED_SCHEME;
1047
1048
13.8k
    if(!ISSLASH(*p))
1049
      /* less than one */
1050
155
      return CURLUE_BAD_SLASHES;
1051
13.6k
    if((flags & CURLU_NO_AUTHORITY)) {
1052
0
      while(ISSLASH(*p) && (num_slashes < 2)) {
1053
0
        p++;
1054
0
        num_slashes++;
1055
0
      }
1056
0
    }
1057
13.6k
    else {
1058
40.5k
      while(ISSLASH(*p) && (num_slashes < 4)) {
1059
26.9k
        p++;
1060
26.9k
        num_slashes++;
1061
26.9k
      }
1062
13.6k
      if(num_slashes > 3)
1063
2
        return CURLUE_BAD_SLASHES;
1064
13.6k
    }
1065
1066
13.6k
    schemep = schemebuf;
1067
13.6k
    *hostpp = p; /* hostname starts here */
1068
13.6k
  }
1069
0
  else {
1070
    /* no scheme! */
1071
1072
0
    if(!(flags & (CURLU_DEFAULT_SCHEME | CURLU_GUESS_SCHEME)))
1073
0
      return CURLUE_BAD_SCHEME;
1074
1075
0
    if(flags & CURLU_DEFAULT_SCHEME)
1076
0
      schemep = DEFAULT_SCHEME;
1077
1078
    /*
1079
     * The URL was badly formatted, let's try without scheme specified.
1080
     */
1081
0
    *hostpp = url;
1082
0
  }
1083
1084
13.6k
  if(schemep) {
1085
13.6k
    u->scheme = curlx_strdup(schemep);
1086
13.6k
    if(!u->scheme)
1087
0
      return CURLUE_OUT_OF_MEMORY;
1088
13.6k
  }
1089
13.6k
  return CURLUE_OK;
1090
13.6k
}
1091
1092
static CURLUcode guess_scheme(CURLU *u, struct dynbuf *host)
1093
0
{
1094
0
  const char *hostname = curlx_dyn_ptr(host);
1095
0
  const char *schemep = NULL;
1096
  /* legacy curl-style guess based on hostname */
1097
0
  if(checkprefix("ftp.", hostname))
1098
0
    schemep = "ftp";
1099
0
  else if(checkprefix("dict.", hostname))
1100
0
    schemep = "dict";
1101
0
  else if(checkprefix("ldap.", hostname))
1102
0
    schemep = "ldap";
1103
0
  else if(checkprefix("imap.", hostname))
1104
0
    schemep = "imap";
1105
0
  else if(checkprefix("smtp.", hostname))
1106
0
    schemep = "smtp";
1107
0
  else if(checkprefix("pop3.", hostname))
1108
0
    schemep = "pop3";
1109
0
  else
1110
0
    schemep = "http";
1111
1112
0
  u->scheme = curlx_strdup(schemep);
1113
0
  if(!u->scheme)
1114
0
    return CURLUE_OUT_OF_MEMORY;
1115
1116
0
  u->guessed_scheme = TRUE;
1117
0
  return CURLUE_OK;
1118
0
}
1119
1120
static CURLUcode handle_fragment(CURLU *u, const char *fragment,
1121
                                 size_t fraglen, unsigned int flags)
1122
97
{
1123
97
  CURLUcode ures;
1124
97
  u->fragment_present = TRUE;
1125
97
  if(fraglen > 1) {
1126
    /* skip the leading '#' in the copy but include the null-terminator */
1127
58
    if(flags & CURLU_URLENCODE) {
1128
7
      struct dynbuf enc;
1129
7
      curlx_dyn_init(&enc, CURL_MAX_INPUT_LENGTH);
1130
7
      ures = urlencode_str(&enc, fragment + 1, fraglen - 1, TRUE, QUERY_NO);
1131
7
      if(ures)
1132
0
        return ures;
1133
7
      u->fragment = curlx_dyn_ptr(&enc);
1134
7
    }
1135
51
    else {
1136
51
      if(badoctets(fragment, fraglen, flags))
1137
0
        return CURLUE_BAD_FRAGMENT;
1138
51
      u->fragment = curlx_memdup0(fragment + 1, fraglen - 1);
1139
51
      if(!u->fragment)
1140
0
        return CURLUE_OUT_OF_MEMORY;
1141
51
    }
1142
58
  }
1143
97
  return CURLUE_OK;
1144
97
}
1145
1146
static CURLUcode handle_query(CURLU *u, const char *query,
1147
                              size_t qlen, unsigned int flags)
1148
266
{
1149
266
  u->query_present = TRUE;
1150
266
  if(qlen > 1) {
1151
172
    if(flags & CURLU_URLENCODE) {
1152
19
      struct dynbuf enc;
1153
19
      CURLUcode ures;
1154
19
      curlx_dyn_init(&enc, CURL_MAX_INPUT_LENGTH);
1155
      /* skip the leading question mark */
1156
19
      ures = urlencode_str(&enc, query + 1, qlen - 1, TRUE, QUERY_YES);
1157
19
      if(ures)
1158
0
        return ures;
1159
19
      u->query = curlx_dyn_ptr(&enc);
1160
19
    }
1161
153
    else {
1162
153
      if(badoctets(query, qlen, flags))
1163
2
        return CURLUE_BAD_QUERY;
1164
1165
151
      u->query = curlx_memdup0(query + 1, qlen - 1);
1166
151
      if(!u->query)
1167
0
        return CURLUE_OUT_OF_MEMORY;
1168
151
    }
1169
172
  }
1170
94
  else {
1171
    /* single byte query */
1172
94
    u->query = curlx_strdup("");
1173
94
    if(!u->query)
1174
0
      return CURLUE_OUT_OF_MEMORY;
1175
94
  }
1176
264
  return CURLUE_OK;
1177
266
}
1178
1179
static CURLUcode handle_path(CURLU *u, const char *path,
1180
                             size_t pathlen, unsigned int flags,
1181
                             bool is_file)
1182
13.5k
{
1183
13.5k
  CURLUcode ures;
1184
13.5k
  if(pathlen && (flags & CURLU_URLENCODE)) {
1185
317
    struct dynbuf enc;
1186
317
    curlx_dyn_init(&enc, CURL_MAX_INPUT_LENGTH);
1187
317
    ures = urlencode_str(&enc, path, pathlen, TRUE, QUERY_NO);
1188
317
    if(ures)
1189
0
      return ures;
1190
317
    pathlen = curlx_dyn_len(&enc);
1191
317
    path = u->path = curlx_dyn_ptr(&enc);
1192
317
  }
1193
1194
13.5k
  if(pathlen >= (size_t)(1 + !is_file)) {
1195
1.64k
    if(badoctets(path, pathlen, flags))
1196
35
      return CURLUE_BAD_PATH;
1197
1198
    /* paths for file:// scheme can be one byte, others need to be two */
1199
1.61k
    if(!u->path) {
1200
1.43k
      u->path = curlx_memdup0(path, pathlen);
1201
1.43k
      if(!u->path)
1202
0
        return CURLUE_OUT_OF_MEMORY;
1203
1.43k
      path = u->path;
1204
1.43k
    }
1205
182
    else if(flags & CURLU_URLENCODE)
1206
      /* it might have encoded more than the path so cut it */
1207
182
      u->path[pathlen] = 0;
1208
1209
1.61k
    if(!(flags & CURLU_PATH_AS_IS)) {
1210
      /* remove ../ and ./ sequences according to RFC3986 */
1211
1.57k
      char *dedot;
1212
1.57k
      int err = dedotdotify(path, pathlen, &dedot);
1213
1.57k
      if(err)
1214
0
        return CURLUE_OUT_OF_MEMORY;
1215
1.57k
      if(dedot) {
1216
612
        curlx_free(u->path);
1217
612
        u->path = dedot;
1218
612
      }
1219
1.57k
    }
1220
1.61k
  }
1221
13.4k
  return CURLUE_OK;
1222
13.5k
}
1223
1224
static CURLUcode parseurl(const char *url, CURLU *u, unsigned int flags)
1225
13.8k
{
1226
13.8k
  const char *path;
1227
13.8k
  size_t pathlen;
1228
13.8k
  char schemebuf[MAX_SCHEME_LEN + 1];
1229
13.8k
  size_t schemelen = 0;
1230
13.8k
  size_t urllen;
1231
13.8k
  CURLUcode ures = CURLUE_OK;
1232
13.8k
  struct dynbuf host;
1233
13.8k
  bool is_file = FALSE;
1234
1235
13.8k
  DEBUGASSERT(url);
1236
1237
13.8k
  urllen = strlen(url);
1238
13.8k
  if(urllen > CURL_MAX_INPUT_LENGTH)
1239
0
    return CURLUE_MALFORMED_INPUT;
1240
1241
13.8k
  curlx_dyn_init(&host, CURL_MAX_INPUT_LENGTH);
1242
1243
13.8k
  schemelen = Curl_is_absolute_url(url, schemebuf, sizeof(schemebuf),
1244
13.8k
                                   flags & (CURLU_GUESS_SCHEME |
1245
13.8k
                                            CURLU_DEFAULT_SCHEME));
1246
1247
  /* handle the file: scheme */
1248
13.8k
  if(schemelen == 4 && !memcmp(schemebuf, "file", 4)) {
1249
42
    is_file = TRUE;
1250
42
    ures = parse_file(url, urllen, u, &path, &pathlen);
1251
42
  }
1252
13.8k
  else {
1253
13.8k
    const char *hostp = NULL;
1254
13.8k
    const char *p;
1255
13.8k
    size_t hostlen;
1256
13.8k
    ures = parse_scheme(url, u, schemebuf, schemelen, flags, &hostp);
1257
13.8k
    if(ures)
1258
157
      goto fail;
1259
1260
    /* find the end of the hostname + port number */
1261
13.6k
    p = hostp;
1262
156k
    while(*p && *p != '/' && *p != '?' && *p != '#')
1263
142k
      p++;
1264
13.6k
    hostlen = p - hostp;
1265
13.6k
    path = p;
1266
1267
    /* this pathlen also contains the query and the fragment */
1268
13.6k
    pathlen = urllen - (path - url);
1269
13.6k
    if(hostlen) {
1270
13.6k
      ures = parse_authority(u, hostp, hostlen, flags, &host, !!u->scheme);
1271
13.6k
      if(!ures && (flags & CURLU_GUESS_SCHEME) && !u->scheme)
1272
0
        ures = guess_scheme(u, &host);
1273
13.6k
    }
1274
15
    else if(flags & CURLU_NO_AUTHORITY) {
1275
      /* allowed to be empty. */
1276
0
      if(curlx_dyn_add(&host, ""))
1277
0
        ures = CURLUE_OUT_OF_MEMORY;
1278
0
    }
1279
15
    else
1280
15
      ures = CURLUE_NO_HOST;
1281
13.6k
  }
1282
13.7k
  if(!ures) {
1283
    /* The path might at this point contain a fragment and/or a query to
1284
       handle */
1285
13.5k
    const char *fragment = memchr(path, '#', pathlen);
1286
13.5k
    if(fragment) {
1287
97
      size_t fraglen = pathlen - (fragment - path);
1288
97
      ures = handle_fragment(u, fragment, fraglen, flags);
1289
      /* after this, pathlen still contains the query */
1290
97
      pathlen -= fraglen;
1291
97
    }
1292
13.5k
  }
1293
13.7k
  if(!ures) {
1294
13.5k
    const char *query = memchr(path, '?', pathlen);
1295
13.5k
    if(query) {
1296
266
      size_t qlen = pathlen - (query - path);
1297
266
      ures = handle_query(u, query, qlen, flags);
1298
266
      pathlen -= qlen;
1299
266
    }
1300
13.5k
  }
1301
13.7k
  if(!ures)
1302
    /* the fragment and query parts are trimmed off from the path */
1303
13.5k
    ures = handle_path(u, path, pathlen, flags, is_file);
1304
13.7k
  if(!ures) {
1305
13.4k
    u->host = curlx_dyn_ptr(&host);
1306
13.4k
    return CURLUE_OK;
1307
13.4k
  }
1308
390
fail:
1309
390
  curlx_dyn_free(&host);
1310
390
  free_urlhandle(u);
1311
390
  return ures;
1312
13.7k
}
1313
1314
/*
1315
 * Parse the URL and, if successful, replace everything in the Curl_URL struct.
1316
 */
1317
static CURLUcode parseurl_and_replace(const char *url, CURLU *u,
1318
                                      unsigned int flags)
1319
13.8k
{
1320
13.8k
  CURLUcode ures;
1321
13.8k
  CURLU tmpurl;
1322
13.8k
  memset(&tmpurl, 0, sizeof(tmpurl));
1323
13.8k
  ures = parseurl(url, &tmpurl, flags);
1324
13.8k
  if(!ures) {
1325
13.4k
    free_urlhandle(u);
1326
13.4k
    *u = tmpurl;
1327
13.4k
  }
1328
13.8k
  return ures;
1329
13.8k
}
1330
1331
/*
1332
 * Concatenate a relative URL onto a base URL making it absolute.
1333
 */
1334
static CURLUcode redirect_url(const char *base, const char *relurl,
1335
                              CURLU *u, unsigned int flags)
1336
41
{
1337
41
  struct dynbuf urlbuf;
1338
41
  bool host_changed = FALSE;
1339
41
  const char *useurl = relurl;
1340
41
  const char *cutoff = NULL;
1341
41
  size_t prelen;
1342
41
  CURLUcode uc;
1343
  /* this can get here with a NULL u->scheme only if asked to use the default
1344
     scheme, so allow fallback to that */
1345
41
  const char *scheme = u->scheme ? u->scheme : DEFAULT_SCHEME;
1346
1347
  /* protsep points to the start of the hostname, after [scheme]:// */
1348
41
  const char *protsep = base + strlen(scheme) + 3;
1349
41
  DEBUGASSERT(base && relurl && u); /* all set here */
1350
41
  if(!base)
1351
0
    return CURLUE_MALFORMED_INPUT; /* should never happen */
1352
1353
  /* handle different relative URL types */
1354
41
  switch(relurl[0]) {
1355
1
  case '/':
1356
1
    if(relurl[1] == '/') {
1357
      /* protocol-relative URL: //example.com/path */
1358
0
      cutoff = protsep;
1359
0
      useurl = &relurl[2];
1360
0
      host_changed = TRUE;
1361
0
    }
1362
1
    else
1363
      /* absolute /path */
1364
1
      cutoff = strchr(protsep, '/');
1365
1
    break;
1366
1367
0
  case '#':
1368
    /* fragment-only change */
1369
0
    if(u->fragment_present)
1370
0
      cutoff = strchr(protsep, '#');
1371
0
    break;
1372
1373
40
  default:
1374
    /* path or query-only change */
1375
40
    if(u->query_present)
1376
      /* remove existing query */
1377
2
      cutoff = strchr(protsep, '?');
1378
38
    else if(u->fragment_present)
1379
      /* Remove existing fragment */
1380
4
      cutoff = strchr(protsep, '#');
1381
1382
40
    if(relurl[0] != '?') {
1383
      /* append a relative path after the last slash */
1384
39
      cutoff = memrchr(protsep, '/',
1385
39
                       cutoff ? (size_t)(cutoff - protsep) : strlen(protsep));
1386
39
      if(cutoff)
1387
39
        cutoff++; /* truncate after last slash */
1388
39
    }
1389
40
    break;
1390
41
  }
1391
1392
41
  prelen = cutoff ? (size_t)(cutoff - base) : strlen(base);
1393
1394
  /* build new URL */
1395
41
  curlx_dyn_init(&urlbuf, CURL_MAX_INPUT_LENGTH);
1396
1397
41
  if(!curlx_dyn_addn(&urlbuf, base, prelen) &&
1398
41
     !urlencode_str(&urlbuf, useurl, strlen(useurl), !host_changed,
1399
41
                    QUERY_NOT_YET)) {
1400
41
    uc = parseurl_and_replace(curlx_dyn_ptr(&urlbuf), u,
1401
41
                              flags & ~U_CURLU_PATH_AS_IS);
1402
41
  }
1403
0
  else
1404
0
    uc = CURLUE_OUT_OF_MEMORY;
1405
1406
41
  curlx_dyn_free(&urlbuf);
1407
41
  return uc;
1408
41
}
1409
1410
/*
1411
 */
1412
CURLU *curl_url(void)
1413
13.5k
{
1414
13.5k
  return curlx_calloc(1, sizeof(struct Curl_URL));
1415
13.5k
}
1416
1417
void curl_url_cleanup(CURLU *u)
1418
46.1k
{
1419
46.1k
  if(u) {
1420
13.5k
    free_urlhandle(u);
1421
13.5k
    curlx_free(u);
1422
13.5k
  }
1423
46.1k
}
1424
1425
#define DUP(dest, src, name)                    \
1426
0
  do {                                          \
1427
0
    if((src)->name) {                           \
1428
0
      (dest)->name = curlx_strdup((src)->name); \
1429
0
      if(!(dest)->name)                         \
1430
0
        goto fail;                              \
1431
0
    }                                           \
1432
0
  } while(0)
1433
1434
CURLU *curl_url_dup(const CURLU *in)
1435
0
{
1436
0
  struct Curl_URL *u = curlx_calloc(1, sizeof(struct Curl_URL));
1437
0
  if(u) {
1438
0
    DUP(u, in, scheme);
1439
0
    DUP(u, in, user);
1440
0
    DUP(u, in, password);
1441
0
    DUP(u, in, options);
1442
0
    DUP(u, in, host);
1443
0
    DUP(u, in, path);
1444
0
    DUP(u, in, query);
1445
0
    DUP(u, in, fragment);
1446
0
    DUP(u, in, zoneid);
1447
0
    u->portnum = in->portnum;
1448
0
    u->port_present = in->port_present;
1449
0
    u->fragment_present = in->fragment_present;
1450
0
    u->query_present = in->query_present;
1451
0
  }
1452
0
  return u;
1453
0
fail:
1454
0
  curl_url_cleanup(u);
1455
0
  return NULL;
1456
0
}
1457
1458
#ifndef USE_IDN
1459
#define host_decode(x, y) CURLUE_LACKS_IDN
1460
#define host_encode(x, y) CURLUE_LACKS_IDN
1461
#else
1462
static CURLUcode host_decode(const char *host, char **allochost)
1463
0
{
1464
0
  CURLcode result = Curl_idn_decode(host, allochost);
1465
0
  if(result)
1466
0
    return (result == CURLE_OUT_OF_MEMORY) ?
1467
0
      CURLUE_OUT_OF_MEMORY : CURLUE_BAD_HOSTNAME;
1468
0
  return CURLUE_OK;
1469
0
}
1470
1471
static CURLUcode host_encode(const char *host, char **allochost)
1472
0
{
1473
0
  CURLcode result = Curl_idn_encode(host, allochost);
1474
0
  if(result)
1475
0
    return (result == CURLE_OUT_OF_MEMORY) ?
1476
0
      CURLUE_OUT_OF_MEMORY : CURLUE_BAD_HOSTNAME;
1477
0
  return CURLUE_OK;
1478
0
}
1479
#endif
1480
1481
static CURLUcode urlget_format(const CURLU *u, CURLUPart what,
1482
                               const char *ptr, char **partp,
1483
                               bool plusdecode, unsigned int flags)
1484
38.8k
{
1485
38.8k
  CURLUcode uc = CURLUE_OK;
1486
38.8k
  size_t partlen = strlen(ptr);
1487
38.8k
  bool urldecode = (flags & CURLU_URLDECODE) ? 1 : 0;
1488
38.8k
  bool urlencode = (flags & CURLU_URLENCODE) ? 1 : 0;
1489
38.8k
  bool punycode = (flags & CURLU_PUNYCODE) && (what == CURLUPART_HOST);
1490
38.8k
  bool depunyfy = (flags & CURLU_PUNY2IDN) && (what == CURLUPART_HOST);
1491
38.8k
  char *part = curlx_memdup0(ptr, partlen);
1492
38.8k
  *partp = NULL;
1493
38.8k
  if(!part)
1494
0
    return CURLUE_OUT_OF_MEMORY;
1495
38.8k
  if(plusdecode) {
1496
    /* convert + to space */
1497
0
    char *plus = part;
1498
0
    size_t i = 0;
1499
0
    for(i = 0; i < partlen; ++plus, i++) {
1500
0
      if(*plus == '+')
1501
0
        *plus = ' ';
1502
0
    }
1503
0
  }
1504
38.8k
  if(urldecode) {
1505
0
    char *decoded;
1506
0
    size_t dlen;
1507
    /* this unconditional rejection of control bytes is documented API
1508
       behavior */
1509
0
    CURLcode result = Curl_urldecode(part, partlen, &decoded, &dlen,
1510
0
                                     REJECT_CTRL);
1511
0
    curlx_free(part);
1512
0
    if(result)
1513
0
      return CURLUE_URLDECODE;
1514
0
    part = decoded;
1515
0
    partlen = dlen;
1516
0
  }
1517
38.8k
  if(urlencode) {
1518
12.5k
    struct dynbuf enc;
1519
12.5k
    curlx_dyn_init(&enc, CURL_MAX_INPUT_LENGTH);
1520
12.5k
    uc = urlencode_str(&enc, part, partlen, TRUE, what == CURLUPART_QUERY ?
1521
12.5k
                       QUERY_YES : QUERY_NO);
1522
12.5k
    curlx_free(part);
1523
12.5k
    if(uc)
1524
0
      return uc;
1525
12.5k
    part = curlx_dyn_ptr(&enc);
1526
12.5k
  }
1527
26.2k
  else if(punycode) {
1528
0
    if(!Curl_is_ASCII_name(u->host)) {
1529
0
      char *punyversion = NULL;
1530
0
      uc = host_decode(part, &punyversion);
1531
0
      curlx_free(part);
1532
0
      if(uc)
1533
0
        return uc;
1534
0
      part = punyversion;
1535
0
    }
1536
0
  }
1537
26.2k
  else if(depunyfy && Curl_is_ASCII_name(u->host)) {
1538
0
    char *unpunified = NULL;
1539
0
    uc = host_encode(part, &unpunified);
1540
0
    curlx_free(part);
1541
0
    if(uc)
1542
0
      return uc;
1543
0
    part = unpunified;
1544
0
  }
1545
38.8k
  *partp = part;
1546
38.8k
  return CURLUE_OK;
1547
38.8k
}
1548
1549
static CURLUcode file_url(const CURLU *u, char **part,
1550
                          const char *fragmentsep,
1551
                          const char *querysep)
1552
0
{
1553
0
  char *url = curl_maprintf("file://%s%s%s%s%s",
1554
0
                            u->path, querysep, u->query ? u->query : "",
1555
0
                            fragmentsep, u->fragment ? u->fragment : "");
1556
0
  if(!url)
1557
0
    return CURLUE_OUT_OF_MEMORY;
1558
1559
0
  *part = url;
1560
0
  return CURLUE_OK;
1561
0
}
1562
1563
static CURLUcode urlget_url(const CURLU *u, char **part, unsigned int flags)
1564
12.9k
{
1565
12.9k
  char *url;
1566
12.9k
  char *allochost = NULL;
1567
12.9k
  const char *fragmentsep =
1568
12.9k
    (u->fragment || (u->fragment_present && flags & CURLU_GET_EMPTY)) ?
1569
12.8k
    "#" : "";
1570
12.9k
  const char *querysep = ((u->query && u->query[0]) ||
1571
12.7k
                          (u->query_present && flags & CURLU_GET_EMPTY)) ?
1572
12.6k
    "?" : "";
1573
12.9k
  char portbuf[7];
1574
12.9k
  if(curl_strequal("file", u->scheme))
1575
0
    return file_url(u, part, fragmentsep, querysep);
1576
12.9k
  else if(!u->host)
1577
0
    return CURLUE_NO_HOST;
1578
12.9k
  else {
1579
12.9k
    const char *scheme;
1580
12.9k
    char *options = u->options;
1581
12.9k
    char *port = NULL;
1582
12.9k
    const struct Curl_scheme *h = NULL;
1583
12.9k
    char schemebuf[MAX_SCHEME_LEN + 5];
1584
12.9k
    if(u->scheme)
1585
12.9k
      scheme = u->scheme;
1586
0
    else if(flags & CURLU_DEFAULT_SCHEME)
1587
0
      scheme = DEFAULT_SCHEME;
1588
0
    else
1589
0
      return CURLUE_NO_SCHEME;
1590
1591
12.9k
    if(u->port_present) {
1592
0
      curl_msnprintf(portbuf, sizeof(portbuf), "%u", u->portnum);
1593
0
      port = portbuf;
1594
0
    }
1595
1596
12.9k
    h = Curl_get_scheme(scheme);
1597
12.9k
    if(h) {
1598
12.9k
      if(!u->port_present && (flags & CURLU_DEFAULT_PORT)) {
1599
        /* there is no stored port number, but asked to deliver a default one
1600
           for the scheme */
1601
0
        curl_msnprintf(portbuf, sizeof(portbuf), "%u", h->defport);
1602
0
        port = portbuf;
1603
0
      }
1604
12.9k
      else if(u->port_present && (h->defport == u->portnum) &&
1605
0
              (flags & CURLU_NO_DEFAULT_PORT)) {
1606
        /* there is a stored port number, but asked to inhibit if it matches
1607
           the default port for the scheme */
1608
0
        port = NULL;
1609
0
      }
1610
1611
12.9k
      if(!(h->flags & PROTOPT_URLOPTIONS))
1612
12.9k
        options = NULL;
1613
12.9k
    }
1614
1615
12.9k
    if(u->host[0] == '[') {
1616
0
      if(u->zoneid) {
1617
        /* make it '[ host %25 zoneid ]' */
1618
0
        struct dynbuf enc;
1619
0
        size_t hostlen = strlen(u->host);
1620
0
        curlx_dyn_init(&enc, CURL_MAX_INPUT_LENGTH);
1621
0
        if(curlx_dyn_addf(&enc, "%.*s%%25%s]", (int)hostlen - 1, u->host,
1622
0
                          u->zoneid))
1623
0
          return CURLUE_OUT_OF_MEMORY;
1624
0
        allochost = curlx_dyn_ptr(&enc);
1625
0
      }
1626
0
    }
1627
12.9k
    else if(flags & CURLU_URLENCODE) {
1628
0
      allochost = curl_easy_escape(NULL, u->host, 0);
1629
0
      if(!allochost)
1630
0
        return CURLUE_OUT_OF_MEMORY;
1631
0
    }
1632
12.9k
    else if(flags & CURLU_PUNYCODE) {
1633
0
      if(!Curl_is_ASCII_name(u->host)) {
1634
0
        CURLUcode ret = host_decode(u->host, &allochost);
1635
0
        if(ret)
1636
0
          return ret;
1637
0
      }
1638
0
    }
1639
12.9k
    else if(flags & CURLU_PUNY2IDN) {
1640
0
      if(Curl_is_ASCII_name(u->host)) {
1641
0
        CURLUcode ret = host_encode(u->host, &allochost);
1642
0
        if(ret)
1643
0
          return ret;
1644
0
      }
1645
0
    }
1646
1647
12.9k
    if(!(flags & CURLU_NO_GUESS_SCHEME) || !u->guessed_scheme)
1648
12.9k
      curl_msnprintf(schemebuf, sizeof(schemebuf), "%s://", scheme);
1649
0
    else
1650
0
      schemebuf[0] = 0;
1651
1652
12.9k
    url = curl_maprintf("%s%s%s%s%s%s%s%s%s%s%s%s%s%s%s",
1653
12.9k
                        schemebuf,
1654
12.9k
                        u->user ? u->user : "",
1655
12.9k
                        u->password ? ":" : "",
1656
12.9k
                        u->password ? u->password : "",
1657
12.9k
                        options ? ";" : "",
1658
12.9k
                        options ? options : "",
1659
12.9k
                        (u->user || u->password || options) ? "@" : "",
1660
12.9k
                        allochost ? allochost : u->host,
1661
12.9k
                        port ? ":" : "",
1662
12.9k
                        port ? port : "",
1663
12.9k
                        u->path ? u->path : "/",
1664
12.9k
                        querysep,
1665
12.9k
                        u->query ? u->query : "",
1666
12.9k
                        fragmentsep,
1667
12.9k
                        u->fragment ? u->fragment : "");
1668
12.9k
    curlx_free(allochost);
1669
12.9k
  }
1670
12.9k
  if(!url)
1671
0
    return CURLUE_OUT_OF_MEMORY;
1672
12.9k
  *part = url;
1673
12.9k
  return CURLUE_OK;
1674
12.9k
}
1675
1676
CURLUcode curl_url_get(const CURLU *u, CURLUPart what,
1677
                       char **part, unsigned int flags)
1678
114k
{
1679
114k
  const char *ptr;
1680
114k
  CURLUcode ifmissing = CURLUE_UNKNOWN_PART;
1681
114k
  char portbuf[7];
1682
114k
  bool plusdecode = FALSE;
1683
114k
  if(!u)
1684
0
    return CURLUE_BAD_HANDLE;
1685
114k
  if(!part)
1686
0
    return CURLUE_BAD_PARTPOINTER;
1687
114k
  *part = NULL;
1688
1689
114k
  switch(what) {
1690
12.8k
  case CURLUPART_SCHEME:
1691
12.8k
    ptr = u->scheme;
1692
12.8k
    ifmissing = CURLUE_NO_SCHEME;
1693
12.8k
    flags &= ~U_CURLU_URLDECODE; /* never for schemes */
1694
12.8k
    if((flags & CURLU_NO_GUESS_SCHEME) && u->guessed_scheme)
1695
0
      return CURLUE_NO_SCHEME;
1696
12.8k
    break;
1697
12.8k
  case CURLUPART_USER:
1698
12.5k
    ptr = u->user;
1699
12.5k
    ifmissing = CURLUE_NO_USER;
1700
12.5k
    break;
1701
12.5k
  case CURLUPART_PASSWORD:
1702
12.5k
    ptr = u->password;
1703
12.5k
    ifmissing = CURLUE_NO_PASSWORD;
1704
12.5k
    break;
1705
12.5k
  case CURLUPART_OPTIONS:
1706
12.5k
    ptr = u->options;
1707
12.5k
    ifmissing = CURLUE_NO_OPTIONS;
1708
12.5k
    break;
1709
12.8k
  case CURLUPART_HOST:
1710
12.8k
    ptr = u->host;
1711
12.8k
    ifmissing = CURLUE_NO_HOST;
1712
12.8k
    break;
1713
12.5k
  case CURLUPART_ZONEID:
1714
12.5k
    ptr = u->zoneid;
1715
12.5k
    ifmissing = CURLUE_NO_ZONEID;
1716
12.5k
    break;
1717
260
  case CURLUPART_PORT:
1718
260
    ptr = NULL;
1719
260
    ifmissing = CURLUE_NO_PORT;
1720
260
    flags &= ~U_CURLU_URLDECODE; /* never for port */
1721
260
    if(u->port_present) {
1722
113
      const struct Curl_scheme *h = u->scheme ?
1723
113
                                    Curl_get_scheme(u->scheme) : NULL;
1724
      /* there is a stored port number, but ask to inhibit if
1725
         it matches the default one for the scheme */
1726
113
      if(h && (h->defport == u->portnum) &&
1727
0
         (flags & CURLU_NO_DEFAULT_PORT)) {
1728
0
        ptr = NULL;
1729
0
      }
1730
113
      else {
1731
113
        curl_msnprintf(portbuf, sizeof(portbuf), "%u", u->portnum);
1732
113
        ptr = portbuf;
1733
113
      }
1734
113
    }
1735
147
    else if((flags & CURLU_DEFAULT_PORT) && u->scheme) {
1736
      /* there is no stored port number, but asked to deliver
1737
         a default one for the scheme */
1738
0
      const struct Curl_scheme *h = Curl_get_scheme(u->scheme);
1739
0
      if(h) {
1740
0
        curl_msnprintf(portbuf, sizeof(portbuf), "%u", h->defport);
1741
0
        ptr = portbuf;
1742
0
      }
1743
0
    }
1744
260
    break;
1745
12.8k
  case CURLUPART_PATH:
1746
12.8k
    ptr = u->path;
1747
12.8k
    if(!ptr)
1748
11.6k
      ptr = "/";
1749
12.8k
    break;
1750
12.8k
  case CURLUPART_QUERY:
1751
12.8k
    ptr = u->query;
1752
12.8k
    ifmissing = CURLUE_NO_QUERY;
1753
12.8k
    plusdecode = flags & CURLU_URLDECODE;
1754
12.8k
    if(ptr && !ptr[0] && !(flags & CURLU_GET_EMPTY))
1755
      /* there was a blank query and the user does not ask for it */
1756
5
      ptr = NULL;
1757
12.8k
    break;
1758
0
  case CURLUPART_FRAGMENT:
1759
0
    ptr = u->fragment;
1760
0
    ifmissing = CURLUE_NO_FRAGMENT;
1761
0
    if(!ptr && u->fragment_present && flags & CURLU_GET_EMPTY)
1762
      /* there was a blank fragment and the user asks for it */
1763
0
      ptr = "";
1764
0
    break;
1765
12.9k
  case CURLUPART_URL:
1766
12.9k
    return urlget_url(u, part, flags);
1767
0
  default:
1768
0
    ptr = NULL;
1769
0
    break;
1770
114k
  }
1771
101k
  if(ptr)
1772
38.8k
    return urlget_format(u, what, ptr, part, plusdecode, flags);
1773
1774
63.0k
  return ifmissing;
1775
101k
}
1776
1777
static CURLUcode set_url_scheme(CURLU *u, const char *scheme,
1778
                                unsigned int flags)
1779
0
{
1780
0
  size_t plen = strlen(scheme);
1781
0
  const struct Curl_scheme *h = NULL;
1782
0
  if((plen > MAX_SCHEME_LEN) || (plen < 1))
1783
    /* too long or too short */
1784
0
    return CURLUE_BAD_SCHEME;
1785
  /* verify that it is a fine scheme */
1786
0
  h = Curl_get_scheme(scheme);
1787
0
  if(!(flags & CURLU_NON_SUPPORT_SCHEME) && (!h || !h->run))
1788
0
    return CURLUE_UNSUPPORTED_SCHEME;
1789
0
  if(!h) {
1790
0
    const char *s = scheme;
1791
0
    if(ISALPHA(*s)) {
1792
      /* ALPHA *( ALPHA / DIGIT / "+" / "-" / "." ) */
1793
0
      s++;
1794
0
      while(--plen) {
1795
0
        if(ISALNUM(*s) || (*s == '+') || (*s == '-') || (*s == '.'))
1796
0
          s++; /* fine */
1797
0
        else
1798
0
          return CURLUE_BAD_SCHEME;
1799
0
      }
1800
0
    }
1801
0
    else
1802
0
      return CURLUE_BAD_SCHEME;
1803
0
  }
1804
0
  u->guessed_scheme = FALSE;
1805
0
  return CURLUE_OK;
1806
0
}
1807
1808
static CURLUcode set_url_port(CURLU *u, const char *provided_port)
1809
0
{
1810
0
  curl_off_t port;
1811
0
  if(!ISDIGIT(provided_port[0]))
1812
    /* not a number */
1813
0
    return CURLUE_BAD_PORT_NUMBER;
1814
0
  if(curlx_str_number(&provided_port, &port, 0xffff) || *provided_port)
1815
    /* weirdly provided number, not good! */
1816
0
    return CURLUE_BAD_PORT_NUMBER;
1817
0
  u->portnum = (uint16_t)port;
1818
0
  u->port_present = TRUE;
1819
0
  return CURLUE_OK;
1820
0
}
1821
1822
static CURLUcode set_url(CURLU *u, const char *url, size_t part_size,
1823
                         unsigned int flags)
1824
13.8k
{
1825
  /*
1826
   * Allow a new URL to replace the existing (if any) contents.
1827
   *
1828
   * If the existing contents is enough for a URL, allow a relative URL to
1829
   * replace it.
1830
   */
1831
13.8k
  CURLUcode uc;
1832
13.8k
  char *oldurl = NULL;
1833
1834
13.8k
  if(!part_size) {
1835
    /* a blank URL is not a valid URL unless we already have a complete one
1836
       and this is a redirect */
1837
0
    uc = curl_url_get(u, CURLUPART_URL, &oldurl, flags);
1838
0
    if(!uc) {
1839
      /* success, meaning the "" is a fine relative URL, and the new URL
1840
         inherits scheme/authority/path/query, but not fragment, from the
1841
         existing URL (RFC 3986 section 5.2.2) */
1842
0
      curlx_safefree(u->fragment);
1843
0
      u->fragment_present = FALSE;
1844
0
      curlx_free(oldurl);
1845
0
      return CURLUE_OK;
1846
0
    }
1847
0
    if(uc == CURLUE_OUT_OF_MEMORY)
1848
0
      return uc;
1849
0
    return CURLUE_MALFORMED_INPUT;
1850
0
  }
1851
1852
  /* if the new URL is absolute replace the existing with the new. */
1853
13.8k
  if(Curl_is_absolute_url(url, NULL, 0,
1854
13.8k
                          flags & (CURLU_GUESS_SCHEME | CURLU_DEFAULT_SCHEME)))
1855
13.8k
    return parseurl_and_replace(url, u, flags);
1856
1857
  /* if the old URL is incomplete (we cannot get an absolute URL in
1858
     'oldurl'), replace the existing with the new.
1859
     Always include "scheme://" to make the URL "complete" */
1860
  /* Preserve empty query/fragment separators: they affect where relative
1861
     references splice into the base URL. */
1862
41
  uc = curl_url_get(u, CURLUPART_URL, &oldurl,
1863
41
                    (flags & ~CURLU_NO_GUESS_SCHEME) | CURLU_GET_EMPTY);
1864
41
  if(uc == CURLUE_OUT_OF_MEMORY)
1865
0
    return uc;
1866
41
  else if(uc)
1867
0
    return parseurl_and_replace(url, u, flags);
1868
1869
41
  DEBUGASSERT(oldurl); /* it is set here */
1870
  /* apply the relative part to create a new URL */
1871
41
  uc = redirect_url(oldurl, url, u, flags);
1872
41
  curlx_free(oldurl);
1873
41
  return uc;
1874
41
}
1875
1876
static CURLUcode urlset_clear(CURLU *u, CURLUPart what)
1877
0
{
1878
0
  switch(what) {
1879
0
  case CURLUPART_URL:
1880
0
    free_urlhandle(u);
1881
0
    memset(u, 0, sizeof(struct Curl_URL));
1882
0
    break;
1883
0
  case CURLUPART_SCHEME:
1884
0
    curlx_safefree(u->scheme);
1885
0
    u->guessed_scheme = FALSE;
1886
0
    break;
1887
0
  case CURLUPART_USER:
1888
0
    curlx_safefree(u->user);
1889
0
    break;
1890
0
  case CURLUPART_PASSWORD:
1891
0
    curlx_strzero(u->password);
1892
0
    curlx_safefree(u->password);
1893
0
    break;
1894
0
  case CURLUPART_OPTIONS:
1895
0
    curlx_safefree(u->options);
1896
0
    break;
1897
0
  case CURLUPART_HOST:
1898
0
    curlx_safefree(u->host);
1899
0
    break;
1900
0
  case CURLUPART_ZONEID:
1901
0
    curlx_safefree(u->zoneid);
1902
0
    break;
1903
0
  case CURLUPART_PORT:
1904
0
    u->portnum = 0;
1905
0
    u->port_present = FALSE;
1906
0
    break;
1907
0
  case CURLUPART_PATH:
1908
0
    curlx_safefree(u->path);
1909
0
    break;
1910
0
  case CURLUPART_QUERY:
1911
0
    curlx_safefree(u->query);
1912
0
    u->query_present = FALSE;
1913
0
    break;
1914
0
  case CURLUPART_FRAGMENT:
1915
0
    curlx_safefree(u->fragment);
1916
0
    u->fragment_present = FALSE;
1917
0
    break;
1918
0
  default:
1919
0
    return CURLUE_UNKNOWN_PART;
1920
0
  }
1921
0
  return CURLUE_OK;
1922
0
}
1923
1924
static bool allowed_in_path(unsigned char x)
1925
0
{
1926
0
  switch(x) {
1927
0
  case '!':
1928
0
  case '$':
1929
0
  case '&':
1930
0
  case '\'':
1931
0
  case '(':
1932
0
  case ')':
1933
0
  case '{':
1934
0
  case '}':
1935
0
  case '[':
1936
0
  case ']':
1937
0
  case '*':
1938
0
  case '+':
1939
0
  case ',':
1940
0
  case ';':
1941
0
  case '=':
1942
0
  case ':':
1943
0
  case '@':
1944
0
  case '/':
1945
0
    return TRUE;
1946
0
  }
1947
0
  return FALSE;
1948
0
}
1949
1950
static CURLUcode url_encode_part(struct dynbuf *encp,
1951
                                 const char *part,
1952
                                 bool plusencode,
1953
                                 bool pathmode,
1954
                                 bool equalsencode)
1955
0
{
1956
0
  const unsigned char *i;
1957
1958
0
  for(i = (const unsigned char *)part; *i; i++) {
1959
0
    CURLcode result;
1960
0
    if((*i == ' ') && plusencode)
1961
0
      result = curlx_dyn_addn(encp, "+", 1);
1962
0
    else if(ISUNRESERVED(*i) ||
1963
0
            (pathmode && allowed_in_path(*i)) ||
1964
0
            ((*i == '=') && equalsencode)) {
1965
0
      if((*i == '=') && equalsencode)
1966
        /* only skip the first equals sign */
1967
0
        equalsencode = FALSE;
1968
0
      result = curlx_dyn_addn(encp, i, 1);
1969
0
    }
1970
0
    else {
1971
0
      unsigned char out[3] = { '%' };
1972
0
      Curl_hexbyte(&out[1], *i);
1973
0
      result = curlx_dyn_addn(encp, out, 3);
1974
0
    }
1975
0
    if(result)
1976
0
      return cc2cu(result);
1977
0
  }
1978
0
  return CURLUE_OK;
1979
0
}
1980
1981
static CURLUcode url_uppercasehex_part(struct dynbuf *encp,
1982
                                       const char *part)
1983
0
{
1984
0
  char *p;
1985
0
  CURLcode result = curlx_dyn_add(encp, part);
1986
0
  if(result)
1987
0
    return cc2cu(result);
1988
0
  p = curlx_dyn_ptr(encp);
1989
0
  while(*p) {
1990
    /* make sure percent encoded are upper case */
1991
0
    if((*p == '%') && ISXDIGIT(p[1]) && ISXDIGIT(p[2]) &&
1992
0
       (ISLOWER(p[1]) || ISLOWER(p[2]))) {
1993
0
      p[1] = Curl_raw_toupper(p[1]);
1994
0
      p[2] = Curl_raw_toupper(p[2]);
1995
0
      p += 3;
1996
0
    }
1997
0
    else
1998
0
      p++;
1999
0
  }
2000
0
  return CURLUE_OK;
2001
0
}
2002
2003
static CURLUcode url_append_query(CURLU *u, struct dynbuf *encp)
2004
0
{
2005
  /* Append the 'encp' string onto the old query. Add a '&' separator if none
2006
     is already present at the end of the existing query */
2007
2008
0
  size_t querylen = u->query ? strlen(u->query) : 0;
2009
0
  bool addamperand = querylen && (u->query[querylen - 1] != '&');
2010
0
  if(querylen) {
2011
0
    struct dynbuf qbuf;
2012
0
    CURLcode result;
2013
0
    const char *newp = curlx_dyn_ptr(encp);
2014
0
    curlx_dyn_init(&qbuf, CURL_MAX_INPUT_LENGTH);
2015
2016
    /* add original query */
2017
0
    result = curlx_dyn_addn(&qbuf, u->query, querylen);
2018
0
    if(!result && addamperand)
2019
      /* add ampersand */
2020
0
      result = curlx_dyn_addn(&qbuf, "&", 1);
2021
0
    if(!result)
2022
      /* add new query part */
2023
0
      result = curlx_dyn_add(&qbuf, newp);
2024
0
    if(result)
2025
0
      goto nomem;
2026
0
    curlx_dyn_free(encp);
2027
0
    curlx_free(u->query);
2028
0
    u->query = curlx_dyn_ptr(&qbuf);
2029
0
    return CURLUE_OK;
2030
0
nomem:
2031
0
    curlx_dyn_free(encp);
2032
0
    return cc2cu(result);
2033
0
  }
2034
0
  else {
2035
0
    curlx_free(u->query);
2036
0
    u->query = curlx_dyn_ptr(encp);
2037
0
  }
2038
0
  return CURLUE_OK;
2039
0
}
2040
2041
static CURLUcode url_sethost(CURLU *u, struct dynbuf *encp,
2042
                             bool urlencode,
2043
                             unsigned int flags)
2044
0
{
2045
0
  size_t n = curlx_dyn_len(encp);
2046
0
  bool bad = FALSE;
2047
0
  char *newp = curlx_dyn_ptr(encp);
2048
0
  if(!n)
2049
    /* an empty hostname is okay if told so */
2050
0
    bad = (flags & CURLU_NO_AUTHORITY) ? FALSE : TRUE;
2051
0
  else if(!urlencode) {
2052
    /* if the hostname part was not URL encoded here, it was set already URL
2053
       encoded so we need to decode it to check */
2054
0
    size_t dlen;
2055
0
    char *decoded = NULL;
2056
0
    CURLcode result = Curl_urldecode(newp, n, &decoded, &dlen, REJECT_CTRL);
2057
0
    if(result || hostname_check6(u, decoded, dlen))
2058
0
      bad = TRUE;
2059
0
    curlx_free(decoded);
2060
0
  }
2061
0
  else if(hostname_check6(u, newp, n))
2062
0
    bad = TRUE;
2063
0
  if(bad) {
2064
0
    curlx_dyn_free(encp);
2065
0
    return CURLUE_BAD_HOSTNAME;
2066
0
  }
2067
0
  return CURLUE_OK;
2068
0
}
2069
2070
CURLUcode curl_url_set(CURLU *u, CURLUPart what,
2071
                       const char *part, unsigned int flags)
2072
13.8k
{
2073
13.8k
  char **storep = NULL;
2074
13.8k
  bool urlencode = (flags & CURLU_URLENCODE) ? 1 : 0;
2075
13.8k
  bool plusencode = FALSE;
2076
13.8k
  bool pathmode = FALSE;
2077
13.8k
  bool leadingslash = FALSE;
2078
13.8k
  bool appendquery = FALSE;
2079
13.8k
  bool equalsencode = FALSE;
2080
13.8k
  size_t nalloc;
2081
2082
13.8k
  if(!u)
2083
0
    return CURLUE_BAD_HANDLE;
2084
13.8k
  if(!part)
2085
    /* setting a part to NULL clears it */
2086
0
    return urlset_clear(u, what);
2087
2088
13.8k
  nalloc = strlen(part);
2089
13.8k
  if(nalloc > CURL_MAX_INPUT_LENGTH)
2090
    /* excessive input length */
2091
0
    return CURLUE_MALFORMED_INPUT;
2092
2093
13.8k
  switch(what) {
2094
0
  case CURLUPART_SCHEME: {
2095
0
    CURLUcode status = set_url_scheme(u, part, flags);
2096
0
    if(status)
2097
0
      return status;
2098
0
    storep = &u->scheme;
2099
0
    urlencode = FALSE; /* never */
2100
0
    break;
2101
0
  }
2102
0
  case CURLUPART_USER:
2103
0
    storep = &u->user;
2104
0
    break;
2105
0
  case CURLUPART_PASSWORD:
2106
0
    storep = &u->password;
2107
0
    break;
2108
0
  case CURLUPART_OPTIONS:
2109
0
    storep = &u->options;
2110
0
    break;
2111
0
  case CURLUPART_HOST:
2112
0
    storep = &u->host;
2113
0
    curlx_safefree(u->zoneid);
2114
0
    break;
2115
0
  case CURLUPART_ZONEID:
2116
0
    storep = &u->zoneid;
2117
0
    break;
2118
0
  case CURLUPART_PORT:
2119
0
    return set_url_port(u, part);
2120
0
  case CURLUPART_PATH:
2121
0
    pathmode = TRUE;
2122
0
    leadingslash = TRUE; /* enforce */
2123
0
    storep = &u->path;
2124
0
    break;
2125
0
  case CURLUPART_QUERY:
2126
0
    plusencode = urlencode;
2127
0
    appendquery = (flags & CURLU_APPENDQUERY) ? 1 : 0;
2128
0
    equalsencode = appendquery;
2129
0
    storep = &u->query;
2130
0
    u->query_present = TRUE;
2131
0
    break;
2132
0
  case CURLUPART_FRAGMENT:
2133
0
    storep = &u->fragment;
2134
0
    u->fragment_present = TRUE;
2135
0
    break;
2136
13.8k
  case CURLUPART_URL:
2137
13.8k
    return set_url(u, part, nalloc, flags);
2138
0
  default:
2139
0
    return CURLUE_UNKNOWN_PART;
2140
13.8k
  }
2141
0
  DEBUGASSERT(storep);
2142
0
  {
2143
0
    const char *newp = NULL;
2144
0
    struct dynbuf enc;
2145
0
    CURLUcode status;
2146
0
    curlx_dyn_init(&enc, (nalloc * 3) + 1 + leadingslash);
2147
2148
0
    if(leadingslash && (part[0] != '/')) {
2149
0
      CURLcode result = curlx_dyn_addn(&enc, "/", 1);
2150
0
      if(result)
2151
0
        return cc2cu(result);
2152
0
    }
2153
0
    if(urlencode)
2154
0
      status = url_encode_part(&enc, part, plusencode, pathmode, equalsencode);
2155
0
    else
2156
0
      status = url_uppercasehex_part(&enc, part);
2157
0
    if(!status) {
2158
0
      newp = curlx_dyn_ptr(&enc);
2159
2160
0
      if(appendquery && newp)
2161
0
        return url_append_query(u, &enc);
2162
0
      else if(what == CURLUPART_HOST)
2163
0
        status = url_sethost(u, &enc, urlencode, flags);
2164
0
    }
2165
0
    if(status)
2166
0
      return status;
2167
2168
0
    if(what == CURLUPART_PASSWORD)
2169
0
      curlx_strzero(*storep);
2170
0
    curlx_free(*storep);
2171
0
    *storep = (char *)CURL_UNCONST(newp);
2172
0
  }
2173
0
  return CURLUE_OK;
2174
0
}
2175
2176
bool Curl_url_same_origin(CURLU *base, CURLU *href)
2177
317
{
2178
317
  const struct Curl_scheme *s = NULL;
2179
2180
  /* base must be an absolute URL */
2181
317
  if(!base->scheme || !base->host)
2182
0
    return FALSE;
2183
317
  if(href->scheme && !curl_strequal(base->scheme, href->scheme))
2184
0
    return FALSE;
2185
317
  if(href->host) {
2186
317
    if(!curl_strequal(base->host, href->host))
2187
0
      return FALSE;
2188
2189
317
    if(base->port_present != href->port_present) {
2190
      /* one is present, one is not */
2191
0
      s = Curl_get_scheme(base->scheme);
2192
0
      if(!s) /* Cannot match default port for unknown scheme */
2193
0
        return FALSE;
2194
      /* to match, the present one must be the default port */
2195
0
      if((base->port_present && (base->portnum != s->defport)) ||
2196
0
         (href->port_present && (href->portnum != s->defport)))
2197
0
        return FALSE;
2198
0
    }
2199
317
    else if(base->portnum != href->portnum) /* both present or missing */
2200
0
      return FALSE;
2201
2202
317
    if(!curl_strequal(base->zoneid ? base->zoneid : "",
2203
317
                      href->zoneid ? href->zoneid : ""))
2204
0
      return FALSE;
2205
317
  }
2206
0
  else if(href->port_present) /* no host in href, then there must be no port */
2207
0
    return FALSE;
2208
317
  return TRUE;
2209
317
}
2210
2211
CURLUcode Curl_url_get_port(CURLU *u, uint16_t *pport)
2212
12.5k
{
2213
12.5k
  if(u->port_present) {
2214
0
    *pport = u->portnum;
2215
0
    return CURLUE_OK;
2216
0
  }
2217
12.5k
  else if(u->scheme) {
2218
12.5k
    const struct Curl_scheme *s = Curl_get_scheme(u->scheme);
2219
12.5k
    if(s && s->defport) {
2220
12.5k
      *pport = s->defport;
2221
12.5k
      return CURLUE_OK;
2222
12.5k
    }
2223
12.5k
  }
2224
0
  *pport = 0;
2225
0
  return CURLUE_NO_PORT;
2226
12.5k
}