Coverage Report

Created: 2026-07-30 06:06

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libxml2/HTMLparser.c
Line
Count
Source
1
/*
2
 * HTMLparser.c : an HTML parser
3
 *
4
 * References:
5
 *   HTML Living Standard
6
 *     https://html.spec.whatwg.org/multipage/parsing.html
7
 *
8
 * Tokenization now conforms to HTML5. Tree construction still follows
9
 * a custom, non-standard implementation. See:
10
 *
11
 *     https://gitlab.gnome.org/GNOME/libxml2/-/issues/211
12
 *
13
 * See Copyright for the status of this software.
14
 *
15
 * Author: Daniel Veillard
16
 */
17
18
#define IN_LIBXML
19
#include "libxml.h"
20
#ifdef LIBXML_HTML_ENABLED
21
22
#include <string.h>
23
#include <ctype.h>
24
#include <stdlib.h>
25
26
#include <libxml/HTMLparser.h>
27
#include <libxml/xmlmemory.h>
28
#include <libxml/tree.h>
29
#include <libxml/parser.h>
30
#include <libxml/parserInternals.h>
31
#include <libxml/xmlerror.h>
32
#include <libxml/HTMLtree.h>
33
#include <libxml/entities.h>
34
#include <libxml/encoding.h>
35
#include <libxml/xmlIO.h>
36
#include <libxml/uri.h>
37
38
#include "private/buf.h"
39
#include "private/dict.h"
40
#include "private/enc.h"
41
#include "private/error.h"
42
#include "private/html.h"
43
#include "private/io.h"
44
#include "private/memory.h"
45
#include "private/parser.h"
46
#include "private/tree.h"
47
48
#define HTML_MAX_NAMELEN 1000
49
2.60k
#define HTML_MAX_ATTRS 100000000 /* 100 million */
50
0
#define HTML_PARSER_BIG_BUFFER_SIZE 1000
51
3.44M
#define HTML_PARSER_BUFFER_SIZE 100
52
53
#define IS_HEX_DIGIT(c) \
54
128
    ((IS_ASCII_DIGIT(c)) || \
55
128
     ((((c) | 0x20) >= 'a') && (((c) | 0x20) <= 'f')))
56
57
#define IS_UPPER(c) \
58
1.16M
    (((c) >= 'A') && ((c) <= 'Z'))
59
60
#define IS_ALNUM(c) \
61
109
    (IS_ASCII_LETTER(c) || IS_ASCII_DIGIT(c))
62
63
typedef enum {
64
    INSERT_INITIAL = 1,
65
    INSERT_IN_HEAD = 3,
66
    INSERT_IN_BODY = 10
67
} htmlInsertMode;
68
69
typedef const unsigned htmlAsciiMask[2];
70
71
static htmlAsciiMask MASK_DQ = {
72
    0,
73
    1u << ('"' - 32),
74
};
75
static htmlAsciiMask MASK_SQ = {
76
    0,
77
    1u << ('\'' - 32),
78
};
79
static htmlAsciiMask MASK_GT = {
80
    0,
81
    1u << ('>' - 32),
82
};
83
static htmlAsciiMask MASK_DASH = {
84
    0,
85
    1u << ('-' - 32),
86
};
87
static htmlAsciiMask MASK_WS_GT = {
88
    1u << 0x09 | 1u << 0x0A | 1u << 0x0C | 1u << 0x0D,
89
    1u << (' ' - 32) | 1u << ('>' - 32),
90
};
91
static htmlAsciiMask MASK_DQ_GT = {
92
    0,
93
    1u << ('"' - 32) | 1u << ('>' - 32),
94
};
95
static htmlAsciiMask MASK_SQ_GT = {
96
    0,
97
    1u << ('\'' - 32) | 1u << ('>' - 32),
98
};
99
100
static int htmlOmittedDefaultValue = 1;
101
102
static int
103
htmlParseElementInternal(htmlParserCtxtPtr ctxt);
104
105
/************************************************************************
106
 *                  *
107
 *    Some factorized error routines        *
108
 *                  *
109
 ************************************************************************/
110
111
/**
112
 * Handle an out-of-memory error
113
 *
114
 * @param ctxt  an HTML parser context
115
 */
116
static void
117
htmlErrMemory(xmlParserCtxtPtr ctxt)
118
10
{
119
10
    xmlCtxtErrMemory(ctxt);
120
10
}
121
122
/**
123
 * Handle a fatal parser error, i.e. violating Well-Formedness constraints
124
 *
125
 * @param ctxt  an HTML parser context
126
 * @param error  the error number
127
 * @param msg  the error message
128
 * @param str1  string infor
129
 * @param str2  string infor
130
 */
131
static void LIBXML_ATTR_FORMAT(3,0)
132
htmlParseErr(xmlParserCtxtPtr ctxt, xmlParserErrors error,
133
             const char *msg, const xmlChar *str1, const xmlChar *str2)
134
2.31k
{
135
2.31k
    xmlCtxtErr(ctxt, NULL, XML_FROM_HTML, error, XML_ERR_ERROR,
136
2.31k
               str1, str2, NULL, 0, msg, str1, str2);
137
2.31k
}
138
139
/************************************************************************
140
 *                  *
141
 *  Parser stacks related functions and macros    *
142
 *                  *
143
 ************************************************************************/
144
145
/**
146
 * Pushes a new element name on top of the name stack
147
 *
148
 * @param ctxt  an HTML parser context
149
 * @param value  the element name
150
 * @returns -1 in case of error, the index in the stack otherwise
151
 */
152
static int
153
htmlnamePush(htmlParserCtxtPtr ctxt, const xmlChar * value)
154
17.5k
{
155
17.5k
    if ((ctxt->html < INSERT_IN_HEAD) && (xmlStrEqual(value, BAD_CAST "head")))
156
0
        ctxt->html = INSERT_IN_HEAD;
157
17.5k
    if ((ctxt->html < INSERT_IN_BODY) && (xmlStrEqual(value, BAD_CAST "body")))
158
2.38k
        ctxt->html = INSERT_IN_BODY;
159
17.5k
    if (ctxt->nameNr >= ctxt->nameMax) {
160
3.74k
        const xmlChar **tmp;
161
3.74k
        int newSize;
162
163
3.74k
        newSize = xmlGrowCapacity(ctxt->nameMax, sizeof(tmp[0]),
164
3.74k
                                  10, XML_MAX_ITEMS);
165
3.74k
        if (newSize < 0) {
166
0
            htmlErrMemory(ctxt);
167
0
            return (-1);
168
0
        }
169
3.74k
        tmp = xmlRealloc(ctxt->nameTab, newSize * sizeof(tmp[0]));
170
3.74k
        if (tmp == NULL) {
171
0
            htmlErrMemory(ctxt);
172
0
            return(-1);
173
0
        }
174
3.74k
        ctxt->nameTab = tmp;
175
3.74k
        ctxt->nameMax = newSize;
176
3.74k
    }
177
17.5k
    ctxt->nameTab[ctxt->nameNr] = value;
178
17.5k
    ctxt->name = value;
179
17.5k
    return (ctxt->nameNr++);
180
17.5k
}
181
/**
182
 * Pops the top element name from the name stack
183
 *
184
 * @param ctxt  an HTML parser context
185
 * @returns the name just removed
186
 */
187
static const xmlChar *
188
htmlnamePop(htmlParserCtxtPtr ctxt)
189
15.8k
{
190
15.8k
    const xmlChar *ret;
191
192
15.8k
    if (ctxt->nameNr <= 0)
193
0
        return (NULL);
194
15.8k
    ctxt->nameNr--;
195
15.8k
    if (ctxt->nameNr < 0)
196
0
        return (NULL);
197
15.8k
    if (ctxt->nameNr > 0)
198
13.5k
        ctxt->name = ctxt->nameTab[ctxt->nameNr - 1];
199
2.32k
    else
200
2.32k
        ctxt->name = NULL;
201
15.8k
    ret = ctxt->nameTab[ctxt->nameNr];
202
15.8k
    ctxt->nameTab[ctxt->nameNr] = NULL;
203
15.8k
    return (ret);
204
15.8k
}
205
206
/**
207
 * Pushes a new element name on top of the node info stack
208
 *
209
 * @param ctxt  an HTML parser context
210
 * @param value  the node info
211
 * @returns 0 in case of error, the index in the stack otherwise
212
 */
213
static int
214
htmlNodeInfoPush(htmlParserCtxtPtr ctxt, htmlParserNodeInfo *value)
215
0
{
216
0
    if (ctxt->nodeInfoNr >= ctxt->nodeInfoMax) {
217
0
        xmlParserNodeInfo *tmp;
218
0
        int newSize;
219
220
0
        newSize = xmlGrowCapacity(ctxt->nodeInfoMax, sizeof(tmp[0]),
221
0
                                  5, XML_MAX_ITEMS);
222
0
        if (newSize < 0) {
223
0
            htmlErrMemory(ctxt);
224
0
            return (0);
225
0
        }
226
0
        tmp = xmlRealloc(ctxt->nodeInfoTab, newSize * sizeof(tmp[0]));
227
0
        if (tmp == NULL) {
228
0
            htmlErrMemory(ctxt);
229
0
            return (0);
230
0
        }
231
0
        ctxt->nodeInfoTab = tmp;
232
0
        ctxt->nodeInfoMax = newSize;
233
0
    }
234
0
    ctxt->nodeInfoTab[ctxt->nodeInfoNr] = *value;
235
0
    ctxt->nodeInfo = &ctxt->nodeInfoTab[ctxt->nodeInfoNr];
236
0
    return (ctxt->nodeInfoNr++);
237
0
}
238
239
/**
240
 * Pops the top element name from the node info stack
241
 *
242
 * @param ctxt  an HTML parser context
243
 * @returns 0 in case of error, the pointer to NodeInfo otherwise
244
 */
245
static htmlParserNodeInfo *
246
htmlNodeInfoPop(htmlParserCtxtPtr ctxt)
247
0
{
248
0
    if (ctxt->nodeInfoNr <= 0)
249
0
        return (NULL);
250
0
    ctxt->nodeInfoNr--;
251
0
    if (ctxt->nodeInfoNr < 0)
252
0
        return (NULL);
253
0
    if (ctxt->nodeInfoNr > 0)
254
0
        ctxt->nodeInfo = &ctxt->nodeInfoTab[ctxt->nodeInfoNr - 1];
255
0
    else
256
0
        ctxt->nodeInfo = NULL;
257
0
    return &ctxt->nodeInfoTab[ctxt->nodeInfoNr];
258
0
}
259
260
/*
261
 * Macros for accessing the content. Those should be used only by the parser,
262
 * and not exported.
263
 *
264
 * Dirty macros, i.e. one need to make assumption on the context to use them
265
 *
266
 *   CUR_PTR return the current pointer to the xmlChar to be parsed.
267
 *   CUR     returns the current xmlChar value, i.e. a 8 bit value if compiled
268
 *           in ISO-Latin or UTF-8, and the current 16 bit value if compiled
269
 *           in UNICODE mode. This should be used internally by the parser
270
 *           only to compare to ASCII values otherwise it would break when
271
 *           running with UTF-8 encoding.
272
 *   NXT(n)  returns the n'th next xmlChar. Same as CUR is should be used only
273
 *           to compare on ASCII based substring.
274
 *   UPP(n)  returns the n'th next xmlChar converted to uppercase. Same as CUR
275
 *           it should be used only to compare on ASCII based substring.
276
 *   SKIP(n) Skip n xmlChar, and must also be used only to skip ASCII defined
277
 *           strings without newlines within the parser.
278
 *
279
 * Clean macros, not dependent of an ASCII context, expect UTF-8 encoding
280
 *
281
 *   COPY(to) copy one char to *to, increment CUR_PTR and to accordingly
282
 */
283
284
636
#define UPPER (toupper(*ctxt->input->cur))
285
286
116k
#define SKIP(val) ctxt->input->cur += (val),ctxt->input->col+=(val)
287
288
235k
#define NXT(val) ctxt->input->cur[(val)]
289
290
25.2k
#define UPP(val) (toupper(ctxt->input->cur[(val)]))
291
292
0
#define CUR_PTR ctxt->input->cur
293
#define BASE_PTR ctxt->input->base
294
295
#define SHRINK \
296
4.81M
    if ((!PARSER_PROGRESSIVE(ctxt)) && \
297
4.81M
        (ctxt->input->cur - ctxt->input->base > 2 * INPUT_CHUNK) && \
298
4.81M
  (ctxt->input->end - ctxt->input->cur < 2 * INPUT_CHUNK)) \
299
4.81M
  xmlParserShrink(ctxt);
300
301
#define GROW \
302
212k
    if ((!PARSER_PROGRESSIVE(ctxt)) && \
303
212k
        (ctxt->input->end - ctxt->input->cur < INPUT_CHUNK)) \
304
17.2k
  xmlParserGrow(ctxt);
305
306
186k
#define SKIP_BLANKS htmlSkipBlankChars(ctxt)
307
308
/* Imported from XML */
309
310
732k
#define CUR (*ctxt->input->cur)
311
312
/**
313
 * Prescan to find encoding.
314
 *
315
 * Try to find an encoding in the current data available in the input
316
 * buffer.
317
 *
318
 * TODO: Implement HTML5 prescan algorithm.
319
 *
320
 * @param ctxt  the HTML parser context
321
 * @returns  an encoding string or NULL if not found
322
 */
323
static xmlChar *
324
0
htmlFindEncoding(xmlParserCtxtPtr ctxt) {
325
0
    const xmlChar *start, *cur, *end;
326
0
    xmlChar *ret;
327
0
328
0
    if ((ctxt == NULL) || (ctxt->input == NULL) ||
329
0
        (ctxt->input->flags & XML_INPUT_HAS_ENCODING))
330
0
        return(NULL);
331
0
    if ((ctxt->input->cur == NULL) || (ctxt->input->end == NULL))
332
0
        return(NULL);
333
0
334
0
    start = ctxt->input->cur;
335
0
    end = ctxt->input->end;
336
0
    /* we also expect the input buffer to be zero terminated */
337
0
    if (*end != 0)
338
0
        return(NULL);
339
0
340
0
    cur = xmlStrcasestr(start, BAD_CAST "HTTP-EQUIV");
341
0
    if (cur == NULL)
342
0
        return(NULL);
343
0
    cur = xmlStrcasestr(cur, BAD_CAST  "CONTENT");
344
0
    if (cur == NULL)
345
0
        return(NULL);
346
0
    cur = xmlStrcasestr(cur, BAD_CAST  "CHARSET=");
347
0
    if (cur == NULL)
348
0
        return(NULL);
349
0
    cur += 8;
350
0
    start = cur;
351
0
    while ((IS_ALNUM(*cur)) ||
352
0
           (*cur == '-') || (*cur == '_') || (*cur == ':') || (*cur == '/'))
353
0
           cur++;
354
0
    if (cur == start)
355
0
        return(NULL);
356
0
    ret = xmlStrndup(start, cur - start);
357
0
    if (ret == NULL)
358
0
        htmlErrMemory(ctxt);
359
0
    return(ret);
360
0
}
361
362
static int
363
2.89M
htmlMaskMatch(htmlAsciiMask mask, unsigned c) {
364
2.89M
    if (c >= 64)
365
2.03M
        return(0);
366
857k
    return((mask[c/32] >> (c & 31)) & 1);
367
2.89M
}
368
369
static int
370
htmlValidateUtf8(xmlParserCtxtPtr ctxt, const xmlChar *str, size_t len,
371
25.0M
                 int partial) {
372
25.0M
    unsigned c = str[0];
373
25.0M
    int size;
374
375
25.0M
    if (c < 0xC2) {
376
2.01M
        goto invalid;
377
22.9M
    } else if (c < 0xE0) {
378
18.4M
        if (len < 2)
379
5
            goto incomplete;
380
18.4M
        if ((str[1] & 0xC0) != 0x80)
381
792k
            goto invalid;
382
17.7M
        size = 2;
383
17.7M
    } else if (c < 0xF0) {
384
3.72M
        unsigned v;
385
386
3.72M
        if (len < 3)
387
8
            goto incomplete;
388
389
3.72M
        v = str[1] << 8 | str[2]; /* hint to generate 16-bit load */
390
3.72M
        v |= c << 16;
391
392
3.72M
        if (((v & 0x00C0C0) != 0x008080) ||
393
3.06M
            ((v & 0x0F2000) == 0x000000) ||
394
3.06M
            ((v & 0x0F2000) == 0x0D2000))
395
653k
            goto invalid;
396
397
3.06M
        size = 3;
398
3.06M
    } else {
399
775k
        unsigned v;
400
401
775k
        if (len < 4)
402
40
            goto incomplete;
403
404
775k
        v = c << 24 | str[1] << 16 | str[2] << 8 | str[3];
405
406
775k
        if (((v & 0x00C0C0C0) != 0x00808080) ||
407
1.87k
            (v < 0xF0900000) || (v >= 0xF4900000))
408
773k
            goto invalid;
409
410
1.76k
        size = 4;
411
1.76k
    }
412
413
20.7M
    return(size);
414
415
53
incomplete:
416
53
    if (partial)
417
0
        return(0);
418
419
4.23M
invalid:
420
    /* Only report the first error */
421
4.23M
    if ((ctxt->input->flags & XML_INPUT_ENCODING_ERROR) == 0) {
422
160
        htmlParseErr(ctxt, XML_ERR_INVALID_ENCODING,
423
160
                     "Invalid bytes in character encoding\n", NULL, NULL);
424
160
        ctxt->input->flags |= XML_INPUT_ENCODING_ERROR;
425
160
    }
426
427
4.23M
    return(-1);
428
53
}
429
430
/**
431
 * skip all blanks character found at that point in the input streams.
432
 *
433
 * @param ctxt  the HTML parser context
434
 * @returns the number of space chars skipped
435
 */
436
437
static int
438
186k
htmlSkipBlankChars(xmlParserCtxtPtr ctxt) {
439
186k
    const xmlChar *cur = ctxt->input->cur;
440
186k
    size_t avail = ctxt->input->end - cur;
441
186k
    int res = 0;
442
186k
    int line = ctxt->input->line;
443
186k
    int col = ctxt->input->col;
444
445
257k
    while (!PARSER_STOPPED(ctxt)) {
446
257k
        if (avail == 0) {
447
459
            ctxt->input->cur = cur;
448
459
            GROW;
449
459
            cur = ctxt->input->cur;
450
459
            avail = ctxt->input->end - cur;
451
452
459
            if (avail == 0)
453
459
                break;
454
459
        }
455
456
257k
        if (*cur == '\n') {
457
34.6k
            line++;
458
34.6k
            col = 1;
459
222k
        } else if (IS_WS_HTML(*cur)) {
460
36.5k
            col++;
461
186k
        } else {
462
186k
            break;
463
186k
        }
464
465
71.2k
        cur += 1;
466
71.2k
        avail -= 1;
467
468
71.2k
  if (res < INT_MAX)
469
71.2k
      res++;
470
71.2k
    }
471
472
186k
    ctxt->input->cur = cur;
473
186k
    ctxt->input->line = line;
474
186k
    ctxt->input->col = col;
475
476
186k
    if (res > 8)
477
23
        GROW;
478
479
186k
    return(res);
480
186k
}
481
482
483
484
/************************************************************************
485
 *                  *
486
 *  The list of HTML elements and their properties    *
487
 *                  *
488
 ************************************************************************/
489
490
/*
491
 *  Start Tag: 1 means the start tag can be omitted
492
 *  End Tag:   1 means the end tag can be omitted
493
 *             2 means it's forbidden (empty elements)
494
 *             3 means the tag is stylistic and should be closed easily
495
 *  Depr:      this element is deprecated
496
 *  DTD:       1 means that this element is valid only in the Loose DTD
497
 *             2 means that this element is valid only in the Frameset DTD
498
 *
499
 * Name,Start Tag,End Tag,Save End,Empty,Deprecated,DTD,inline,Description
500
 */
501
502
static const htmlElemDesc
503
html40ElementTable[] = {
504
{ "a",    0, 0, 0, 0, 0, 0, 1, "anchor ",
505
  NULL, NULL, NULL, NULL, NULL,
506
  0
507
},
508
{ "abbr", 0, 0, 0, 0, 0, 0, 1, "abbreviated form",
509
  NULL, NULL, NULL, NULL, NULL,
510
  0
511
},
512
{ "acronym",  0, 0, 0, 0, 0, 0, 1, "",
513
  NULL, NULL, NULL, NULL, NULL,
514
  0
515
},
516
{ "address",  0, 0, 0, 0, 0, 0, 0, "information on author ",
517
  NULL, NULL, NULL, NULL, NULL,
518
  0
519
},
520
{ "applet", 0, 0, 0, 0, 1, 1, 2, "java applet ",
521
  NULL, NULL, NULL, NULL, NULL,
522
  0
523
},
524
{ "area", 0, 2, 2, 1, 0, 0, 0, "client-side image map area ",
525
  NULL, NULL, NULL, NULL, NULL,
526
  0
527
},
528
{ "b",    0, 3, 0, 0, 0, 0, 1, "bold text style",
529
  NULL, NULL, NULL, NULL, NULL,
530
  0
531
},
532
{ "base", 0, 2, 2, 1, 0, 0, 0, "document base uri ",
533
  NULL, NULL, NULL, NULL, NULL,
534
  0
535
},
536
{ "basefont", 0, 2, 2, 1, 1, 1, 1, "base font size " ,
537
  NULL, NULL, NULL, NULL, NULL,
538
  0
539
},
540
{ "bdo",  0, 0, 0, 0, 0, 0, 1, "i18n bidi over-ride ",
541
  NULL, NULL, NULL, NULL, NULL,
542
  0
543
},
544
{ "bgsound",  0, 0, 2, 1, 0, 0, 0, "",
545
  NULL, NULL, NULL, NULL, NULL,
546
  0
547
},
548
{ "big",  0, 3, 0, 0, 0, 0, 1, "large text style",
549
  NULL, NULL, NULL, NULL, NULL,
550
  0
551
},
552
{ "blockquote", 0, 0, 0, 0, 0, 0, 0, "long quotation ",
553
  NULL, NULL, NULL, NULL, NULL,
554
  0
555
},
556
{ "body", 1, 1, 0, 0, 0, 0, 0, "document body ",
557
  NULL, NULL, NULL, NULL, NULL,
558
  0
559
},
560
{ "br",   0, 2, 2, 1, 0, 0, 1, "forced line break ",
561
  NULL, NULL, NULL, NULL, NULL,
562
  0
563
},
564
{ "button", 0, 0, 0, 0, 0, 0, 2, "push button ",
565
  NULL, NULL, NULL, NULL, NULL,
566
  0
567
},
568
{ "caption",  0, 0, 0, 0, 0, 0, 0, "table caption ",
569
  NULL, NULL, NULL, NULL, NULL,
570
  0
571
},
572
{ "center", 0, 3, 0, 0, 1, 1, 0, "shorthand for div align=center ",
573
  NULL, NULL, NULL, NULL, NULL,
574
  0
575
},
576
{ "cite", 0, 0, 0, 0, 0, 0, 1, "citation",
577
  NULL, NULL, NULL, NULL, NULL,
578
  0
579
},
580
{ "code", 0, 0, 0, 0, 0, 0, 1, "computer code fragment",
581
  NULL, NULL, NULL, NULL, NULL,
582
  0
583
},
584
{ "col",  0, 2, 2, 1, 0, 0, 0, "table column ",
585
  NULL, NULL, NULL, NULL, NULL,
586
  0
587
},
588
{ "colgroup", 0, 1, 0, 0, 0, 0, 0, "table column group ",
589
  NULL, NULL, NULL, NULL, NULL,
590
  0
591
},
592
{ "dd",   0, 1, 0, 0, 0, 0, 0, "definition description ",
593
  NULL, NULL, NULL, NULL, NULL,
594
  0
595
},
596
{ "del",  0, 0, 0, 0, 0, 0, 2, "deleted text ",
597
  NULL, NULL, NULL, NULL, NULL,
598
  0
599
},
600
{ "dfn",  0, 0, 0, 0, 0, 0, 1, "instance definition",
601
  NULL, NULL, NULL, NULL, NULL,
602
  0
603
},
604
{ "dir",  0, 0, 0, 0, 1, 1, 0, "directory list",
605
  NULL, NULL, NULL, NULL, NULL,
606
  0
607
},
608
{ "div",  0, 0, 0, 0, 0, 0, 0, "generic language/style container",
609
  NULL, NULL, NULL, NULL, NULL,
610
  0
611
},
612
{ "dl",   0, 0, 0, 0, 0, 0, 0, "definition list ",
613
  NULL, NULL, NULL, NULL, NULL,
614
  0
615
},
616
{ "dt",   0, 1, 0, 0, 0, 0, 0, "definition term ",
617
  NULL, NULL, NULL, NULL, NULL,
618
  0
619
},
620
{ "em",   0, 3, 0, 0, 0, 0, 1, "emphasis",
621
  NULL, NULL, NULL, NULL, NULL,
622
  0
623
},
624
{ "embed",  0, 1, 2, 1, 1, 1, 1, "generic embedded object ",
625
  NULL, NULL, NULL, NULL, NULL,
626
  0
627
},
628
{ "fieldset", 0, 0, 0, 0, 0, 0, 0, "form control group ",
629
  NULL, NULL, NULL, NULL, NULL,
630
  0
631
},
632
{ "font", 0, 3, 0, 0, 1, 1, 1, "local change to font ",
633
  NULL, NULL, NULL, NULL, NULL,
634
  0
635
},
636
{ "form", 0, 0, 0, 0, 0, 0, 0, "interactive form ",
637
  NULL, NULL, NULL, NULL, NULL,
638
  0
639
},
640
{ "frame",  0, 2, 2, 1, 0, 2, 0, "subwindow " ,
641
  NULL, NULL, NULL, NULL, NULL,
642
  0
643
},
644
{ "frameset", 0, 0, 0, 0, 0, 2, 0, "window subdivision" ,
645
  NULL, NULL, NULL, NULL, NULL,
646
  0
647
},
648
{ "h1",   0, 0, 0, 0, 0, 0, 0, "heading ",
649
  NULL, NULL, NULL, NULL, NULL,
650
  0
651
},
652
{ "h2",   0, 0, 0, 0, 0, 0, 0, "heading ",
653
  NULL, NULL, NULL, NULL, NULL,
654
  0
655
},
656
{ "h3",   0, 0, 0, 0, 0, 0, 0, "heading ",
657
  NULL, NULL, NULL, NULL, NULL,
658
  0
659
},
660
{ "h4",   0, 0, 0, 0, 0, 0, 0, "heading ",
661
  NULL, NULL, NULL, NULL, NULL,
662
  0
663
},
664
{ "h5",   0, 0, 0, 0, 0, 0, 0, "heading ",
665
  NULL, NULL, NULL, NULL, NULL,
666
  0
667
},
668
{ "h6",   0, 0, 0, 0, 0, 0, 0, "heading ",
669
  NULL, NULL, NULL, NULL, NULL,
670
  0
671
},
672
{ "head", 1, 1, 0, 0, 0, 0, 0, "document head ",
673
  NULL, NULL, NULL, NULL, NULL,
674
  0
675
},
676
{ "hr",   0, 2, 2, 1, 0, 0, 0, "horizontal rule " ,
677
  NULL, NULL, NULL, NULL, NULL,
678
  0
679
},
680
{ "html", 1, 1, 0, 0, 0, 0, 0, "document root element ",
681
  NULL, NULL, NULL, NULL, NULL,
682
  0
683
},
684
{ "i",    0, 3, 0, 0, 0, 0, 1, "italic text style",
685
  NULL, NULL, NULL, NULL, NULL,
686
  0
687
},
688
{ "iframe", 0, 0, 0, 0, 0, 1, 2, "inline subwindow ",
689
  NULL, NULL, NULL, NULL, NULL,
690
  DATA_RAWTEXT
691
},
692
{ "img",  0, 2, 2, 1, 0, 0, 1, "embedded image ",
693
  NULL, NULL, NULL, NULL, NULL,
694
  0
695
},
696
{ "input",  0, 2, 2, 1, 0, 0, 1, "form control ",
697
  NULL, NULL, NULL, NULL, NULL,
698
  0
699
},
700
{ "ins",  0, 0, 0, 0, 0, 0, 2, "inserted text",
701
  NULL, NULL, NULL, NULL, NULL,
702
  0
703
},
704
{ "isindex",  0, 2, 2, 1, 1, 1, 0, "single line prompt ",
705
  NULL, NULL, NULL, NULL, NULL,
706
  0
707
},
708
{ "kbd",  0, 0, 0, 0, 0, 0, 1, "text to be entered by the user",
709
  NULL, NULL, NULL, NULL, NULL,
710
  0
711
},
712
{ "keygen", 0, 0, 2, 1, 0, 0, 0, "",
713
  NULL, NULL, NULL, NULL, NULL,
714
  0
715
},
716
{ "label",  0, 0, 0, 0, 0, 0, 1, "form field label text ",
717
  NULL, NULL, NULL, NULL, NULL,
718
  0
719
},
720
{ "legend", 0, 0, 0, 0, 0, 0, 0, "fieldset legend ",
721
  NULL, NULL, NULL, NULL, NULL,
722
  0
723
},
724
{ "li",   0, 1, 1, 0, 0, 0, 0, "list item ",
725
  NULL, NULL, NULL, NULL, NULL,
726
  0
727
},
728
{ "link", 0, 2, 2, 1, 0, 0, 0, "a media-independent link ",
729
  NULL, NULL, NULL, NULL, NULL,
730
  0
731
},
732
{ "map",  0, 0, 0, 0, 0, 0, 2, "client-side image map ",
733
  NULL, NULL, NULL, NULL, NULL,
734
  0
735
},
736
{ "menu", 0, 0, 0, 0, 1, 1, 0, "menu list ",
737
  NULL, NULL, NULL, NULL, NULL,
738
  0
739
},
740
{ "meta", 0, 2, 2, 1, 0, 0, 0, "generic metainformation ",
741
  NULL, NULL, NULL, NULL, NULL,
742
  0
743
},
744
{ "noembed",  0, 0, 0, 0, 0, 0, 0, "",
745
  NULL, NULL, NULL, NULL, NULL,
746
  DATA_RAWTEXT
747
},
748
{ "noframes", 0, 0, 0, 0, 0, 2, 0, "alternate content container for non frame-based rendering ",
749
  NULL, NULL, NULL, NULL, NULL,
750
  DATA_RAWTEXT
751
},
752
{ "noscript", 0, 0, 0, 0, 0, 0, 0, "alternate content container for non script-based rendering ",
753
  NULL, NULL, NULL, NULL, NULL,
754
  0
755
},
756
{ "object", 0, 0, 0, 0, 0, 0, 2, "generic embedded object ",
757
  NULL, NULL, NULL, NULL, NULL,
758
  0
759
},
760
{ "ol",   0, 0, 0, 0, 0, 0, 0, "ordered list ",
761
  NULL, NULL, NULL, NULL, NULL,
762
  0
763
},
764
{ "optgroup", 0, 0, 0, 0, 0, 0, 0, "option group ",
765
  NULL, NULL, NULL, NULL, NULL,
766
  0
767
},
768
{ "option", 0, 1, 0, 0, 0, 0, 0, "selectable choice " ,
769
  NULL, NULL, NULL, NULL, NULL,
770
  0
771
},
772
{ "p",    0, 1, 0, 0, 0, 0, 0, "paragraph ",
773
  NULL, NULL, NULL, NULL, NULL,
774
  0
775
},
776
{ "param",  0, 2, 2, 1, 0, 0, 0, "named property value ",
777
  NULL, NULL, NULL, NULL, NULL,
778
  0
779
},
780
{ "plaintext",  0, 0, 0, 0, 0, 0, 0, "",
781
  NULL, NULL, NULL, NULL, NULL,
782
  DATA_PLAINTEXT
783
},
784
{ "pre",  0, 0, 0, 0, 0, 0, 0, "preformatted text ",
785
  NULL, NULL, NULL, NULL, NULL,
786
  0
787
},
788
{ "q",    0, 0, 0, 0, 0, 0, 1, "short inline quotation ",
789
  NULL, NULL, NULL, NULL, NULL,
790
  0
791
},
792
{ "s",    0, 3, 0, 0, 1, 1, 1, "strike-through text style",
793
  NULL, NULL, NULL, NULL, NULL,
794
  0
795
},
796
{ "samp", 0, 0, 0, 0, 0, 0, 1, "sample program output, scripts, etc.",
797
  NULL, NULL, NULL, NULL, NULL,
798
  0
799
},
800
{ "script", 0, 0, 0, 0, 0, 0, 2, "script statements ",
801
  NULL, NULL, NULL, NULL, NULL,
802
  DATA_SCRIPT
803
},
804
{ "select", 0, 0, 0, 0, 0, 0, 1, "option selector ",
805
  NULL, NULL, NULL, NULL, NULL,
806
  0
807
},
808
{ "small",  0, 3, 0, 0, 0, 0, 1, "small text style",
809
  NULL, NULL, NULL, NULL, NULL,
810
  0
811
},
812
{ "source", 0, 0, 2, 1, 0, 0, 0, "",
813
  NULL, NULL, NULL, NULL, NULL,
814
  0
815
},
816
{ "span", 0, 0, 0, 0, 0, 0, 1, "generic language/style container ",
817
  NULL, NULL, NULL, NULL, NULL,
818
  0
819
},
820
{ "strike", 0, 3, 0, 0, 1, 1, 1, "strike-through text",
821
  NULL, NULL, NULL, NULL, NULL,
822
  0
823
},
824
{ "strong", 0, 3, 0, 0, 0, 0, 1, "strong emphasis",
825
  NULL, NULL, NULL, NULL, NULL,
826
  0
827
},
828
{ "style",  0, 0, 0, 0, 0, 0, 0, "style info ",
829
  NULL, NULL, NULL, NULL, NULL,
830
  DATA_RAWTEXT
831
},
832
{ "sub",  0, 3, 0, 0, 0, 0, 1, "subscript",
833
  NULL, NULL, NULL, NULL, NULL,
834
  0
835
},
836
{ "sup",  0, 3, 0, 0, 0, 0, 1, "superscript ",
837
  NULL, NULL, NULL, NULL, NULL,
838
  0
839
},
840
{ "table",  0, 0, 0, 0, 0, 0, 0, "",
841
  NULL, NULL, NULL, NULL, NULL,
842
  0
843
},
844
{ "tbody",  1, 0, 0, 0, 0, 0, 0, "table body ",
845
  NULL, NULL, NULL, NULL, NULL,
846
  0
847
},
848
{ "td",   0, 0, 0, 0, 0, 0, 0, "table data cell",
849
  NULL, NULL, NULL, NULL, NULL,
850
  0
851
},
852
{ "textarea", 0, 0, 0, 0, 0, 0, 1, "multi-line text field ",
853
  NULL, NULL, NULL, NULL, NULL,
854
  DATA_RCDATA
855
},
856
{ "tfoot",  0, 1, 0, 0, 0, 0, 0, "table footer ",
857
  NULL, NULL, NULL, NULL, NULL,
858
  0
859
},
860
{ "th",   0, 1, 0, 0, 0, 0, 0, "table header cell",
861
  NULL, NULL, NULL, NULL, NULL,
862
  0
863
},
864
{ "thead",  0, 1, 0, 0, 0, 0, 0, "table header ",
865
  NULL, NULL, NULL, NULL, NULL,
866
  0
867
},
868
{ "title",  0, 0, 0, 0, 0, 0, 0, "document title ",
869
  NULL, NULL, NULL, NULL, NULL,
870
  DATA_RCDATA
871
},
872
{ "tr",   0, 0, 0, 0, 0, 0, 0, "table row ",
873
  NULL, NULL, NULL, NULL, NULL,
874
  0
875
},
876
{ "track",  0, 0, 2, 1, 0, 0, 0, "",
877
  NULL, NULL, NULL, NULL, NULL,
878
  0
879
},
880
{ "tt",   0, 3, 0, 0, 0, 0, 1, "teletype or monospaced text style",
881
  NULL, NULL, NULL, NULL, NULL,
882
  0
883
},
884
{ "u",    0, 3, 0, 0, 1, 1, 1, "underlined text style",
885
  NULL, NULL, NULL, NULL, NULL,
886
  0
887
},
888
{ "ul",   0, 0, 0, 0, 0, 0, 0, "unordered list ",
889
  NULL, NULL, NULL, NULL, NULL,
890
  0
891
},
892
{ "var",  0, 0, 0, 0, 0, 0, 1, "instance of a variable or program argument",
893
  NULL, NULL, NULL, NULL, NULL,
894
  0
895
},
896
{ "wbr",  0, 0, 2, 1, 0, 0, 0, "",
897
  NULL, NULL, NULL, NULL, NULL,
898
  0
899
},
900
{ "xmp",  0, 0, 0, 0, 0, 0, 1, "",
901
  NULL, NULL, NULL, NULL, NULL,
902
  DATA_RAWTEXT
903
}
904
};
905
906
typedef struct {
907
    const char *oldTag;
908
    const char *newTag;
909
} htmlStartCloseEntry;
910
911
/*
912
 * start tags that imply the end of current element
913
 */
914
static const htmlStartCloseEntry htmlStartClose[] = {
915
    { "a", "a" },
916
    { "a", "fieldset" },
917
    { "a", "table" },
918
    { "a", "td" },
919
    { "a", "th" },
920
    { "address", "dd" },
921
    { "address", "dl" },
922
    { "address", "dt" },
923
    { "address", "form" },
924
    { "address", "li" },
925
    { "address", "ul" },
926
    { "b", "center" },
927
    { "b", "p" },
928
    { "b", "td" },
929
    { "b", "th" },
930
    { "big", "p" },
931
    { "caption", "col" },
932
    { "caption", "colgroup" },
933
    { "caption", "tbody" },
934
    { "caption", "tfoot" },
935
    { "caption", "thead" },
936
    { "caption", "tr" },
937
    { "col", "col" },
938
    { "col", "colgroup" },
939
    { "col", "tbody" },
940
    { "col", "tfoot" },
941
    { "col", "thead" },
942
    { "col", "tr" },
943
    { "colgroup", "colgroup" },
944
    { "colgroup", "tbody" },
945
    { "colgroup", "tfoot" },
946
    { "colgroup", "thead" },
947
    { "colgroup", "tr" },
948
    { "dd", "dt" },
949
    { "dir", "dd" },
950
    { "dir", "dl" },
951
    { "dir", "dt" },
952
    { "dir", "form" },
953
    { "dir", "ul" },
954
    { "dl", "form" },
955
    { "dl", "li" },
956
    { "dt", "dd" },
957
    { "dt", "dl" },
958
    { "font", "center" },
959
    { "font", "td" },
960
    { "font", "th" },
961
    { "form", "form" },
962
    { "h1", "fieldset" },
963
    { "h1", "form" },
964
    { "h1", "li" },
965
    { "h1", "p" },
966
    { "h1", "table" },
967
    { "h2", "fieldset" },
968
    { "h2", "form" },
969
    { "h2", "li" },
970
    { "h2", "p" },
971
    { "h2", "table" },
972
    { "h3", "fieldset" },
973
    { "h3", "form" },
974
    { "h3", "li" },
975
    { "h3", "p" },
976
    { "h3", "table" },
977
    { "h4", "fieldset" },
978
    { "h4", "form" },
979
    { "h4", "li" },
980
    { "h4", "p" },
981
    { "h4", "table" },
982
    { "h5", "fieldset" },
983
    { "h5", "form" },
984
    { "h5", "li" },
985
    { "h5", "p" },
986
    { "h5", "table" },
987
    { "h6", "fieldset" },
988
    { "h6", "form" },
989
    { "h6", "li" },
990
    { "h6", "p" },
991
    { "h6", "table" },
992
    { "head", "a" },
993
    { "head", "abbr" },
994
    { "head", "acronym" },
995
    { "head", "address" },
996
    { "head", "b" },
997
    { "head", "bdo" },
998
    { "head", "big" },
999
    { "head", "blockquote" },
1000
    { "head", "body" },
1001
    { "head", "br" },
1002
    { "head", "center" },
1003
    { "head", "cite" },
1004
    { "head", "code" },
1005
    { "head", "dd" },
1006
    { "head", "dfn" },
1007
    { "head", "dir" },
1008
    { "head", "div" },
1009
    { "head", "dl" },
1010
    { "head", "dt" },
1011
    { "head", "em" },
1012
    { "head", "fieldset" },
1013
    { "head", "font" },
1014
    { "head", "form" },
1015
    { "head", "frameset" },
1016
    { "head", "h1" },
1017
    { "head", "h2" },
1018
    { "head", "h3" },
1019
    { "head", "h4" },
1020
    { "head", "h5" },
1021
    { "head", "h6" },
1022
    { "head", "hr" },
1023
    { "head", "i" },
1024
    { "head", "iframe" },
1025
    { "head", "img" },
1026
    { "head", "kbd" },
1027
    { "head", "li" },
1028
    { "head", "listing" },
1029
    { "head", "map" },
1030
    { "head", "menu" },
1031
    { "head", "ol" },
1032
    { "head", "p" },
1033
    { "head", "pre" },
1034
    { "head", "q" },
1035
    { "head", "s" },
1036
    { "head", "samp" },
1037
    { "head", "small" },
1038
    { "head", "span" },
1039
    { "head", "strike" },
1040
    { "head", "strong" },
1041
    { "head", "sub" },
1042
    { "head", "sup" },
1043
    { "head", "table" },
1044
    { "head", "tt" },
1045
    { "head", "u" },
1046
    { "head", "ul" },
1047
    { "head", "var" },
1048
    { "head", "xmp" },
1049
    { "hr", "form" },
1050
    { "i", "center" },
1051
    { "i", "p" },
1052
    { "i", "td" },
1053
    { "i", "th" },
1054
    { "legend", "fieldset" },
1055
    { "li", "li" },
1056
    { "link", "body" },
1057
    { "link", "frameset" },
1058
    { "listing", "dd" },
1059
    { "listing", "dl" },
1060
    { "listing", "dt" },
1061
    { "listing", "fieldset" },
1062
    { "listing", "form" },
1063
    { "listing", "li" },
1064
    { "listing", "table" },
1065
    { "listing", "ul" },
1066
    { "menu", "dd" },
1067
    { "menu", "dl" },
1068
    { "menu", "dt" },
1069
    { "menu", "form" },
1070
    { "menu", "ul" },
1071
    { "ol", "form" },
1072
    { "option", "optgroup" },
1073
    { "option", "option" },
1074
    { "p", "address" },
1075
    { "p", "blockquote" },
1076
    { "p", "body" },
1077
    { "p", "caption" },
1078
    { "p", "center" },
1079
    { "p", "col" },
1080
    { "p", "colgroup" },
1081
    { "p", "dd" },
1082
    { "p", "dir" },
1083
    { "p", "div" },
1084
    { "p", "dl" },
1085
    { "p", "dt" },
1086
    { "p", "fieldset" },
1087
    { "p", "form" },
1088
    { "p", "frameset" },
1089
    { "p", "h1" },
1090
    { "p", "h2" },
1091
    { "p", "h3" },
1092
    { "p", "h4" },
1093
    { "p", "h5" },
1094
    { "p", "h6" },
1095
    { "p", "head" },
1096
    { "p", "hr" },
1097
    { "p", "li" },
1098
    { "p", "listing" },
1099
    { "p", "menu" },
1100
    { "p", "ol" },
1101
    { "p", "p" },
1102
    { "p", "pre" },
1103
    { "p", "table" },
1104
    { "p", "tbody" },
1105
    { "p", "td" },
1106
    { "p", "tfoot" },
1107
    { "p", "th" },
1108
    { "p", "title" },
1109
    { "p", "tr" },
1110
    { "p", "ul" },
1111
    { "p", "xmp" },
1112
    { "pre", "dd" },
1113
    { "pre", "dl" },
1114
    { "pre", "dt" },
1115
    { "pre", "fieldset" },
1116
    { "pre", "form" },
1117
    { "pre", "li" },
1118
    { "pre", "table" },
1119
    { "pre", "ul" },
1120
    { "s", "p" },
1121
    { "script", "noscript" },
1122
    { "small", "p" },
1123
    { "span", "td" },
1124
    { "span", "th" },
1125
    { "strike", "p" },
1126
    { "style", "body" },
1127
    { "style", "frameset" },
1128
    { "tbody", "tbody" },
1129
    { "tbody", "tfoot" },
1130
    { "td", "tbody" },
1131
    { "td", "td" },
1132
    { "td", "tfoot" },
1133
    { "td", "th" },
1134
    { "td", "tr" },
1135
    { "tfoot", "tbody" },
1136
    { "th", "tbody" },
1137
    { "th", "td" },
1138
    { "th", "tfoot" },
1139
    { "th", "th" },
1140
    { "th", "tr" },
1141
    { "thead", "tbody" },
1142
    { "thead", "tfoot" },
1143
    { "title", "body" },
1144
    { "title", "frameset" },
1145
    { "tr", "tbody" },
1146
    { "tr", "tfoot" },
1147
    { "tr", "tr" },
1148
    { "tt", "p" },
1149
    { "u", "p" },
1150
    { "u", "td" },
1151
    { "u", "th" },
1152
    { "ul", "address" },
1153
    { "ul", "form" },
1154
    { "ul", "menu" },
1155
    { "ul", "pre" },
1156
    { "xmp", "dd" },
1157
    { "xmp", "dl" },
1158
    { "xmp", "dt" },
1159
    { "xmp", "fieldset" },
1160
    { "xmp", "form" },
1161
    { "xmp", "li" },
1162
    { "xmp", "table" },
1163
    { "xmp", "ul" }
1164
};
1165
1166
/*
1167
 * The list of HTML attributes which are of content %Script;
1168
 * NOTE: when adding ones, check #htmlIsScriptAttribute since
1169
 *       it assumes the name starts with 'on'
1170
 */
1171
static const char *const htmlScriptAttributes[] = {
1172
    "onclick",
1173
    "ondblclick",
1174
    "onmousedown",
1175
    "onmouseup",
1176
    "onmouseover",
1177
    "onmousemove",
1178
    "onmouseout",
1179
    "onkeypress",
1180
    "onkeydown",
1181
    "onkeyup",
1182
    "onload",
1183
    "onunload",
1184
    "onfocus",
1185
    "onblur",
1186
    "onsubmit",
1187
    "onreset",
1188
    "onchange",
1189
    "onselect"
1190
};
1191
1192
/*
1193
 * This table is used by the htmlparser to know what to do with
1194
 * broken html pages. By assigning different priorities to different
1195
 * elements the parser can decide how to handle extra endtags.
1196
 * Endtags are only allowed to close elements with lower or equal
1197
 * priority.
1198
 */
1199
1200
typedef struct {
1201
    const char *name;
1202
    int priority;
1203
} elementPriority;
1204
1205
static const elementPriority htmlEndPriority[] = {
1206
    {"div",   150},
1207
    {"td",    160},
1208
    {"th",    160},
1209
    {"tr",    170},
1210
    {"thead", 180},
1211
    {"tbody", 180},
1212
    {"tfoot", 180},
1213
    {"table", 190},
1214
    {"head",  200},
1215
    {"body",  200},
1216
    {"html",  220},
1217
    {NULL,    100} /* Default priority */
1218
};
1219
1220
/************************************************************************
1221
 *                  *
1222
 *  functions to handle HTML specific data      *
1223
 *                  *
1224
 ************************************************************************/
1225
1226
static void
1227
15.8k
htmlParserFinishElementParsing(htmlParserCtxtPtr ctxt) {
1228
    /*
1229
     * Capture end position and add node
1230
     */
1231
15.8k
    if ( ctxt->node != NULL && ctxt->record_info ) {
1232
0
       ctxt->nodeInfo->end_pos = ctxt->input->consumed +
1233
0
                                (CUR_PTR - ctxt->input->base);
1234
0
       ctxt->nodeInfo->end_line = ctxt->input->line;
1235
0
       ctxt->nodeInfo->node = ctxt->node;
1236
0
       xmlParserAddNodeInfo(ctxt, ctxt->nodeInfo);
1237
0
       htmlNodeInfoPop(ctxt);
1238
0
    }
1239
15.8k
}
1240
1241
/**
1242
 * @deprecated This is a no-op.
1243
 */
1244
void
1245
0
htmlInitAutoClose(void) {
1246
0
}
1247
1248
static int
1249
95.8k
htmlCompareTags(const void *key, const void *member) {
1250
95.8k
    const xmlChar *tag = (const xmlChar *) key;
1251
95.8k
    const htmlElemDesc *desc = (const htmlElemDesc *) member;
1252
1253
95.8k
    return(xmlStrcasecmp(tag, BAD_CAST desc->name));
1254
95.8k
}
1255
1256
/**
1257
 * Lookup the HTML tag in the ElementTable
1258
 *
1259
 * @deprecated Only supports HTML 4.
1260
 *
1261
 * @param tag  The tag name in lowercase
1262
 * @returns the related htmlElemDesc or NULL if not found.
1263
 */
1264
const htmlElemDesc *
1265
13.8k
htmlTagLookup(const xmlChar *tag) {
1266
13.8k
    if (tag == NULL)
1267
0
        return(NULL);
1268
1269
13.8k
    return((const htmlElemDesc *) bsearch(tag, html40ElementTable,
1270
13.8k
                sizeof(html40ElementTable) / sizeof(htmlElemDesc),
1271
13.8k
                sizeof(htmlElemDesc), htmlCompareTags));
1272
13.8k
}
1273
1274
/**
1275
 * @param name  The name of the element to look up the priority for.
1276
 * @returns value: The "endtag" priority.
1277
 **/
1278
static int
1279
2.90k
htmlGetEndPriority (const xmlChar *name) {
1280
2.90k
    int i = 0;
1281
1282
34.8k
    while ((htmlEndPriority[i].name != NULL) &&
1283
31.9k
     (!xmlStrEqual((const xmlChar *)htmlEndPriority[i].name, name)))
1284
31.9k
  i++;
1285
1286
2.90k
    return(htmlEndPriority[i].priority);
1287
2.90k
}
1288
1289
1290
static int
1291
123k
htmlCompareStartClose(const void *vkey, const void *member) {
1292
123k
    const htmlStartCloseEntry *key = (const htmlStartCloseEntry *) vkey;
1293
123k
    const htmlStartCloseEntry *entry = (const htmlStartCloseEntry *) member;
1294
123k
    int ret;
1295
1296
123k
    ret = strcmp(key->oldTag, entry->oldTag);
1297
123k
    if (ret == 0)
1298
8.15k
        ret = strcmp(key->newTag, entry->newTag);
1299
1300
123k
    return(ret);
1301
123k
}
1302
1303
/**
1304
 * Checks whether the new tag is one of the registered valid tags for
1305
 * closing old.
1306
 *
1307
 * @param newtag  The new tag name
1308
 * @param oldtag  The old tag name
1309
 * @returns 0 if no, 1 if yes.
1310
 */
1311
static int
1312
htmlCheckAutoClose(const xmlChar * newtag, const xmlChar * oldtag)
1313
15.4k
{
1314
15.4k
    htmlStartCloseEntry key;
1315
15.4k
    void *res;
1316
1317
15.4k
    key.oldTag = (const char *) oldtag;
1318
15.4k
    key.newTag = (const char *) newtag;
1319
15.4k
    res = (void*)bsearch(&key, htmlStartClose,
1320
15.4k
            sizeof(htmlStartClose) / sizeof(htmlStartCloseEntry),
1321
15.4k
            sizeof(htmlStartCloseEntry), htmlCompareStartClose);
1322
15.4k
    return(res != NULL);
1323
15.4k
}
1324
1325
/**
1326
 * The HTML DTD allows an ending tag to implicitly close other tags.
1327
 *
1328
 * @param ctxt  an HTML parser context
1329
 * @param newtag  The new tag name
1330
 */
1331
static void
1332
htmlAutoCloseOnClose(htmlParserCtxtPtr ctxt, const xmlChar * newtag)
1333
2.74k
{
1334
2.74k
    const htmlElemDesc *info;
1335
2.74k
    int i, priority;
1336
1337
2.74k
    if (ctxt->options & HTML_PARSE_HTML5)
1338
0
        return;
1339
1340
2.74k
    priority = htmlGetEndPriority(newtag);
1341
1342
2.90k
    for (i = (ctxt->nameNr - 1); i >= 0; i--) {
1343
1344
2.90k
        if (xmlStrEqual(newtag, ctxt->nameTab[i]))
1345
2.74k
            break;
1346
        /*
1347
         * A misplaced endtag can only close elements with lower
1348
         * or equal priority, so if we find an element with higher
1349
         * priority before we find an element with
1350
         * matching name, we just ignore this endtag
1351
         */
1352
162
        if (htmlGetEndPriority(ctxt->nameTab[i]) > priority)
1353
0
            return;
1354
162
    }
1355
2.74k
    if (i < 0)
1356
0
        return;
1357
1358
2.90k
    while (!xmlStrEqual(newtag, ctxt->name)) {
1359
162
        info = htmlTagLookup(ctxt->name);
1360
162
        if ((info != NULL) && (info->endTag == 3)) {
1361
1
            htmlParseErr(ctxt, XML_ERR_TAG_NAME_MISMATCH,
1362
1
                   "Opening and ending tag mismatch: %s and %s\n",
1363
1
       newtag, ctxt->name);
1364
1
        }
1365
162
  htmlParserFinishElementParsing(ctxt);
1366
162
        if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
1367
162
            ctxt->sax->endElement(ctxt->userData, ctxt->name);
1368
162
  htmlnamePop(ctxt);
1369
162
    }
1370
2.74k
}
1371
1372
/**
1373
 * Close all remaining tags at the end of the stream
1374
 *
1375
 * @param ctxt  an HTML parser context
1376
 */
1377
static void
1378
htmlAutoCloseOnEnd(htmlParserCtxtPtr ctxt)
1379
2.41k
{
1380
2.41k
    int i;
1381
1382
2.41k
    if (ctxt->options & HTML_PARSE_HTML5)
1383
0
        return;
1384
1385
2.41k
    if (ctxt->nameNr == 0)
1386
84
        return;
1387
12.5k
    for (i = (ctxt->nameNr - 1); i >= 0; i--) {
1388
10.1k
  htmlParserFinishElementParsing(ctxt);
1389
10.1k
        if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
1390
10.1k
            ctxt->sax->endElement(ctxt->userData, ctxt->name);
1391
10.1k
  htmlnamePop(ctxt);
1392
10.1k
    }
1393
2.32k
}
1394
1395
/**
1396
 * The HTML DTD allows a tag to implicitly close other tags.
1397
 * The list is kept in htmlStartClose array. This function is
1398
 * called when a new tag has been detected and generates the
1399
 * appropriates closes if possible/needed.
1400
 * If newtag is NULL this mean we are at the end of the resource
1401
 * and we should check
1402
 *
1403
 * @param ctxt  an HTML parser context
1404
 * @param newtag  The new tag name or NULL
1405
 */
1406
static void
1407
htmlAutoClose(htmlParserCtxtPtr ctxt, const xmlChar * newtag)
1408
12.9k
{
1409
12.9k
    if (ctxt->options & HTML_PARSE_HTML5)
1410
0
        return;
1411
1412
12.9k
    if (newtag == NULL)
1413
0
        return;
1414
1415
15.4k
    while ((ctxt->name != NULL) &&
1416
15.4k
           (htmlCheckAutoClose(newtag, ctxt->name))) {
1417
2.57k
  htmlParserFinishElementParsing(ctxt);
1418
2.57k
        if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
1419
2.57k
            ctxt->sax->endElement(ctxt->userData, ctxt->name);
1420
2.57k
  htmlnamePop(ctxt);
1421
2.57k
    }
1422
12.9k
}
1423
1424
/**
1425
 * The HTML DTD allows a tag to implicitly close other tags.
1426
 * The list is kept in htmlStartClose array. This function checks
1427
 * if the element or one of it's children would autoclose the
1428
 * given tag.
1429
 *
1430
 * @deprecated Internal function, don't use.
1431
 *
1432
 * @param doc  the HTML document
1433
 * @param name  The tag name
1434
 * @param elem  the HTML element
1435
 * @returns 1 if autoclose, 0 otherwise
1436
 */
1437
int
1438
0
htmlAutoCloseTag(xmlDoc *doc, const xmlChar *name, xmlNode *elem) {
1439
0
    htmlNodePtr child;
1440
1441
0
    if (elem == NULL) return(1);
1442
0
    if (xmlStrEqual(name, elem->name)) return(0);
1443
0
    if (htmlCheckAutoClose(elem->name, name)) return(1);
1444
0
    child = elem->children;
1445
0
    while (child != NULL) {
1446
0
        if (htmlAutoCloseTag(doc, name, child)) return(1);
1447
0
  child = child->next;
1448
0
    }
1449
0
    return(0);
1450
0
}
1451
1452
/**
1453
 * The HTML DTD allows a tag to implicitly close other tags.
1454
 * The list is kept in htmlStartClose array. This function checks
1455
 * if a tag is autoclosed by one of it's child
1456
 *
1457
 * @deprecated Internal function, don't use.
1458
 *
1459
 * @param doc  the HTML document
1460
 * @param elem  the HTML element
1461
 * @returns 1 if autoclosed, 0 otherwise
1462
 */
1463
int
1464
0
htmlIsAutoClosed(xmlDoc *doc, xmlNode *elem) {
1465
0
    htmlNodePtr child;
1466
1467
0
    if (elem == NULL) return(1);
1468
0
    child = elem->children;
1469
0
    while (child != NULL) {
1470
0
  if (htmlAutoCloseTag(doc, elem->name, child)) return(1);
1471
0
  child = child->next;
1472
0
    }
1473
0
    return(0);
1474
0
}
1475
1476
/**
1477
 * The HTML DTD allows a tag to exists only implicitly
1478
 * called when a new tag has been detected and generates the
1479
 * appropriates implicit tags if missing
1480
 *
1481
 * @param ctxt  an HTML parser context
1482
 * @param newtag  The new tag name
1483
 */
1484
static void
1485
42.7k
htmlCheckImplied(htmlParserCtxtPtr ctxt, const xmlChar *newtag) {
1486
42.7k
    int i;
1487
1488
42.7k
    if (ctxt->options & (HTML_PARSE_NOIMPLIED | HTML_PARSE_HTML5))
1489
0
        return;
1490
42.7k
    if (!htmlOmittedDefaultValue)
1491
0
  return;
1492
42.7k
    if (xmlStrEqual(newtag, BAD_CAST"html"))
1493
0
  return;
1494
42.7k
    if (ctxt->nameNr <= 0) {
1495
2.38k
  htmlnamePush(ctxt, BAD_CAST"html");
1496
2.38k
  if ((ctxt->sax != NULL) && (ctxt->sax->startElement != NULL))
1497
2.38k
      ctxt->sax->startElement(ctxt->userData, BAD_CAST"html", NULL);
1498
2.38k
    }
1499
42.7k
    if ((xmlStrEqual(newtag, BAD_CAST"body")) || (xmlStrEqual(newtag, BAD_CAST"head")))
1500
4
        return;
1501
42.7k
    if ((ctxt->nameNr <= 1) &&
1502
2.38k
        ((xmlStrEqual(newtag, BAD_CAST"script")) ||
1503
2.38k
   (xmlStrEqual(newtag, BAD_CAST"style")) ||
1504
2.38k
   (xmlStrEqual(newtag, BAD_CAST"meta")) ||
1505
2.38k
   (xmlStrEqual(newtag, BAD_CAST"link")) ||
1506
2.38k
   (xmlStrEqual(newtag, BAD_CAST"title")) ||
1507
2.38k
   (xmlStrEqual(newtag, BAD_CAST"base")))) {
1508
0
        if (ctxt->html >= INSERT_IN_HEAD) {
1509
            /* we already saw or generated an <head> before */
1510
0
            return;
1511
0
        }
1512
        /*
1513
         * dropped OBJECT ... i you put it first BODY will be
1514
         * assumed !
1515
         */
1516
0
        htmlnamePush(ctxt, BAD_CAST"head");
1517
0
        if ((ctxt->sax != NULL) && (ctxt->sax->startElement != NULL))
1518
0
            ctxt->sax->startElement(ctxt->userData, BAD_CAST"head", NULL);
1519
42.7k
    } else if ((!xmlStrEqual(newtag, BAD_CAST"noframes")) &&
1520
42.7k
         (!xmlStrEqual(newtag, BAD_CAST"frame")) &&
1521
42.7k
         (!xmlStrEqual(newtag, BAD_CAST"frameset"))) {
1522
42.7k
        if (ctxt->html >= INSERT_IN_BODY) {
1523
            /* we already saw or generated a <body> before */
1524
40.3k
            return;
1525
40.3k
        }
1526
4.77k
  for (i = 0;i < ctxt->nameNr;i++) {
1527
2.38k
      if (xmlStrEqual(ctxt->nameTab[i], BAD_CAST"body")) {
1528
0
    return;
1529
0
      }
1530
2.38k
      if (xmlStrEqual(ctxt->nameTab[i], BAD_CAST"head")) {
1531
0
    return;
1532
0
      }
1533
2.38k
  }
1534
1535
2.38k
  htmlnamePush(ctxt, BAD_CAST"body");
1536
2.38k
  if ((ctxt->sax != NULL) && (ctxt->sax->startElement != NULL))
1537
2.38k
      ctxt->sax->startElement(ctxt->userData, BAD_CAST"body", NULL);
1538
2.38k
    }
1539
42.7k
}
1540
1541
/**
1542
 * Prepare for non-whitespace character data.
1543
 *
1544
 * @param ctxt  an HTML parser context
1545
 */
1546
1547
static void
1548
29.8k
htmlStartCharData(htmlParserCtxtPtr ctxt) {
1549
29.8k
    if (ctxt->options & (HTML_PARSE_NOIMPLIED | HTML_PARSE_HTML5))
1550
0
        return;
1551
29.8k
    if (!htmlOmittedDefaultValue)
1552
0
  return;
1553
1554
29.8k
    if (xmlStrEqual(ctxt->name, BAD_CAST "head"))
1555
0
        htmlAutoClose(ctxt, BAD_CAST "p");
1556
29.8k
    htmlCheckImplied(ctxt, BAD_CAST "p");
1557
29.8k
}
1558
1559
/**
1560
 * Check if an attribute is of content type Script
1561
 *
1562
 * @deprecated Only supports HTML 4.
1563
 *
1564
 * @param name  an attribute name
1565
 * @returns 1 is the attribute is a script 0 otherwise
1566
 */
1567
int
1568
0
htmlIsScriptAttribute(const xmlChar *name) {
1569
0
    unsigned int i;
1570
1571
0
    if (name == NULL)
1572
0
      return(0);
1573
    /*
1574
     * all script attributes start with 'on'
1575
     */
1576
0
    if ((name[0] != 'o') || (name[1] != 'n'))
1577
0
      return(0);
1578
0
    for (i = 0;
1579
0
   i < sizeof(htmlScriptAttributes)/sizeof(htmlScriptAttributes[0]);
1580
0
   i++) {
1581
0
  if (xmlStrEqual(name, (const xmlChar *) htmlScriptAttributes[i]))
1582
0
      return(1);
1583
0
    }
1584
0
    return(0);
1585
0
}
1586
1587
/************************************************************************
1588
 *                  *
1589
 *  The list of HTML predefined entities      *
1590
 *                  *
1591
 ************************************************************************/
1592
1593
1594
static const htmlEntityDesc  html40EntitiesTable[] = {
1595
/*
1596
 * the 4 absolute ones, plus apostrophe.
1597
 */
1598
{ 34, "quot", "quotation mark = APL quote, U+0022 ISOnum" },
1599
{ 38, "amp",  "ampersand, U+0026 ISOnum" },
1600
{ 39, "apos", "single quote" },
1601
{ 60, "lt", "less-than sign, U+003C ISOnum" },
1602
{ 62, "gt", "greater-than sign, U+003E ISOnum" },
1603
1604
/*
1605
 * A bunch still in the 128-255 range
1606
 * Replacing them depend really on the charset used.
1607
 */
1608
{ 160,  "nbsp", "no-break space = non-breaking space, U+00A0 ISOnum" },
1609
{ 161,  "iexcl","inverted exclamation mark, U+00A1 ISOnum" },
1610
{ 162,  "cent", "cent sign, U+00A2 ISOnum" },
1611
{ 163,  "pound","pound sign, U+00A3 ISOnum" },
1612
{ 164,  "curren","currency sign, U+00A4 ISOnum" },
1613
{ 165,  "yen",  "yen sign = yuan sign, U+00A5 ISOnum" },
1614
{ 166,  "brvbar","broken bar = broken vertical bar, U+00A6 ISOnum" },
1615
{ 167,  "sect", "section sign, U+00A7 ISOnum" },
1616
{ 168,  "uml",  "diaeresis = spacing diaeresis, U+00A8 ISOdia" },
1617
{ 169,  "copy", "copyright sign, U+00A9 ISOnum" },
1618
{ 170,  "ordf", "feminine ordinal indicator, U+00AA ISOnum" },
1619
{ 171,  "laquo","left-pointing double angle quotation mark = left pointing guillemet, U+00AB ISOnum" },
1620
{ 172,  "not",  "not sign, U+00AC ISOnum" },
1621
{ 173,  "shy",  "soft hyphen = discretionary hyphen, U+00AD ISOnum" },
1622
{ 174,  "reg",  "registered sign = registered trade mark sign, U+00AE ISOnum" },
1623
{ 175,  "macr", "macron = spacing macron = overline = APL overbar, U+00AF ISOdia" },
1624
{ 176,  "deg",  "degree sign, U+00B0 ISOnum" },
1625
{ 177,  "plusmn","plus-minus sign = plus-or-minus sign, U+00B1 ISOnum" },
1626
{ 178,  "sup2", "superscript two = superscript digit two = squared, U+00B2 ISOnum" },
1627
{ 179,  "sup3", "superscript three = superscript digit three = cubed, U+00B3 ISOnum" },
1628
{ 180,  "acute","acute accent = spacing acute, U+00B4 ISOdia" },
1629
{ 181,  "micro","micro sign, U+00B5 ISOnum" },
1630
{ 182,  "para", "pilcrow sign = paragraph sign, U+00B6 ISOnum" },
1631
{ 183,  "middot","middle dot = Georgian comma Greek middle dot, U+00B7 ISOnum" },
1632
{ 184,  "cedil","cedilla = spacing cedilla, U+00B8 ISOdia" },
1633
{ 185,  "sup1", "superscript one = superscript digit one, U+00B9 ISOnum" },
1634
{ 186,  "ordm", "masculine ordinal indicator, U+00BA ISOnum" },
1635
{ 187,  "raquo","right-pointing double angle quotation mark right pointing guillemet, U+00BB ISOnum" },
1636
{ 188,  "frac14","vulgar fraction one quarter = fraction one quarter, U+00BC ISOnum" },
1637
{ 189,  "frac12","vulgar fraction one half = fraction one half, U+00BD ISOnum" },
1638
{ 190,  "frac34","vulgar fraction three quarters = fraction three quarters, U+00BE ISOnum" },
1639
{ 191,  "iquest","inverted question mark = turned question mark, U+00BF ISOnum" },
1640
{ 192,  "Agrave","latin capital letter A with grave = latin capital letter A grave, U+00C0 ISOlat1" },
1641
{ 193,  "Aacute","latin capital letter A with acute, U+00C1 ISOlat1" },
1642
{ 194,  "Acirc","latin capital letter A with circumflex, U+00C2 ISOlat1" },
1643
{ 195,  "Atilde","latin capital letter A with tilde, U+00C3 ISOlat1" },
1644
{ 196,  "Auml", "latin capital letter A with diaeresis, U+00C4 ISOlat1" },
1645
{ 197,  "Aring","latin capital letter A with ring above = latin capital letter A ring, U+00C5 ISOlat1" },
1646
{ 198,  "AElig","latin capital letter AE = latin capital ligature AE, U+00C6 ISOlat1" },
1647
{ 199,  "Ccedil","latin capital letter C with cedilla, U+00C7 ISOlat1" },
1648
{ 200,  "Egrave","latin capital letter E with grave, U+00C8 ISOlat1" },
1649
{ 201,  "Eacute","latin capital letter E with acute, U+00C9 ISOlat1" },
1650
{ 202,  "Ecirc","latin capital letter E with circumflex, U+00CA ISOlat1" },
1651
{ 203,  "Euml", "latin capital letter E with diaeresis, U+00CB ISOlat1" },
1652
{ 204,  "Igrave","latin capital letter I with grave, U+00CC ISOlat1" },
1653
{ 205,  "Iacute","latin capital letter I with acute, U+00CD ISOlat1" },
1654
{ 206,  "Icirc","latin capital letter I with circumflex, U+00CE ISOlat1" },
1655
{ 207,  "Iuml", "latin capital letter I with diaeresis, U+00CF ISOlat1" },
1656
{ 208,  "ETH",  "latin capital letter ETH, U+00D0 ISOlat1" },
1657
{ 209,  "Ntilde","latin capital letter N with tilde, U+00D1 ISOlat1" },
1658
{ 210,  "Ograve","latin capital letter O with grave, U+00D2 ISOlat1" },
1659
{ 211,  "Oacute","latin capital letter O with acute, U+00D3 ISOlat1" },
1660
{ 212,  "Ocirc","latin capital letter O with circumflex, U+00D4 ISOlat1" },
1661
{ 213,  "Otilde","latin capital letter O with tilde, U+00D5 ISOlat1" },
1662
{ 214,  "Ouml", "latin capital letter O with diaeresis, U+00D6 ISOlat1" },
1663
{ 215,  "times","multiplication sign, U+00D7 ISOnum" },
1664
{ 216,  "Oslash","latin capital letter O with stroke latin capital letter O slash, U+00D8 ISOlat1" },
1665
{ 217,  "Ugrave","latin capital letter U with grave, U+00D9 ISOlat1" },
1666
{ 218,  "Uacute","latin capital letter U with acute, U+00DA ISOlat1" },
1667
{ 219,  "Ucirc","latin capital letter U with circumflex, U+00DB ISOlat1" },
1668
{ 220,  "Uuml", "latin capital letter U with diaeresis, U+00DC ISOlat1" },
1669
{ 221,  "Yacute","latin capital letter Y with acute, U+00DD ISOlat1" },
1670
{ 222,  "THORN","latin capital letter THORN, U+00DE ISOlat1" },
1671
{ 223,  "szlig","latin small letter sharp s = ess-zed, U+00DF ISOlat1" },
1672
{ 224,  "agrave","latin small letter a with grave = latin small letter a grave, U+00E0 ISOlat1" },
1673
{ 225,  "aacute","latin small letter a with acute, U+00E1 ISOlat1" },
1674
{ 226,  "acirc","latin small letter a with circumflex, U+00E2 ISOlat1" },
1675
{ 227,  "atilde","latin small letter a with tilde, U+00E3 ISOlat1" },
1676
{ 228,  "auml", "latin small letter a with diaeresis, U+00E4 ISOlat1" },
1677
{ 229,  "aring","latin small letter a with ring above = latin small letter a ring, U+00E5 ISOlat1" },
1678
{ 230,  "aelig","latin small letter ae = latin small ligature ae, U+00E6 ISOlat1" },
1679
{ 231,  "ccedil","latin small letter c with cedilla, U+00E7 ISOlat1" },
1680
{ 232,  "egrave","latin small letter e with grave, U+00E8 ISOlat1" },
1681
{ 233,  "eacute","latin small letter e with acute, U+00E9 ISOlat1" },
1682
{ 234,  "ecirc","latin small letter e with circumflex, U+00EA ISOlat1" },
1683
{ 235,  "euml", "latin small letter e with diaeresis, U+00EB ISOlat1" },
1684
{ 236,  "igrave","latin small letter i with grave, U+00EC ISOlat1" },
1685
{ 237,  "iacute","latin small letter i with acute, U+00ED ISOlat1" },
1686
{ 238,  "icirc","latin small letter i with circumflex, U+00EE ISOlat1" },
1687
{ 239,  "iuml", "latin small letter i with diaeresis, U+00EF ISOlat1" },
1688
{ 240,  "eth",  "latin small letter eth, U+00F0 ISOlat1" },
1689
{ 241,  "ntilde","latin small letter n with tilde, U+00F1 ISOlat1" },
1690
{ 242,  "ograve","latin small letter o with grave, U+00F2 ISOlat1" },
1691
{ 243,  "oacute","latin small letter o with acute, U+00F3 ISOlat1" },
1692
{ 244,  "ocirc","latin small letter o with circumflex, U+00F4 ISOlat1" },
1693
{ 245,  "otilde","latin small letter o with tilde, U+00F5 ISOlat1" },
1694
{ 246,  "ouml", "latin small letter o with diaeresis, U+00F6 ISOlat1" },
1695
{ 247,  "divide","division sign, U+00F7 ISOnum" },
1696
{ 248,  "oslash","latin small letter o with stroke, = latin small letter o slash, U+00F8 ISOlat1" },
1697
{ 249,  "ugrave","latin small letter u with grave, U+00F9 ISOlat1" },
1698
{ 250,  "uacute","latin small letter u with acute, U+00FA ISOlat1" },
1699
{ 251,  "ucirc","latin small letter u with circumflex, U+00FB ISOlat1" },
1700
{ 252,  "uuml", "latin small letter u with diaeresis, U+00FC ISOlat1" },
1701
{ 253,  "yacute","latin small letter y with acute, U+00FD ISOlat1" },
1702
{ 254,  "thorn","latin small letter thorn with, U+00FE ISOlat1" },
1703
{ 255,  "yuml", "latin small letter y with diaeresis, U+00FF ISOlat1" },
1704
1705
{ 338,  "OElig","latin capital ligature OE, U+0152 ISOlat2" },
1706
{ 339,  "oelig","latin small ligature oe, U+0153 ISOlat2" },
1707
{ 352,  "Scaron","latin capital letter S with caron, U+0160 ISOlat2" },
1708
{ 353,  "scaron","latin small letter s with caron, U+0161 ISOlat2" },
1709
{ 376,  "Yuml", "latin capital letter Y with diaeresis, U+0178 ISOlat2" },
1710
1711
/*
1712
 * Anything below should really be kept as entities references
1713
 */
1714
{ 402,  "fnof", "latin small f with hook = function = florin, U+0192 ISOtech" },
1715
1716
{ 710,  "circ", "modifier letter circumflex accent, U+02C6 ISOpub" },
1717
{ 732,  "tilde","small tilde, U+02DC ISOdia" },
1718
1719
{ 913,  "Alpha","greek capital letter alpha, U+0391" },
1720
{ 914,  "Beta", "greek capital letter beta, U+0392" },
1721
{ 915,  "Gamma","greek capital letter gamma, U+0393 ISOgrk3" },
1722
{ 916,  "Delta","greek capital letter delta, U+0394 ISOgrk3" },
1723
{ 917,  "Epsilon","greek capital letter epsilon, U+0395" },
1724
{ 918,  "Zeta", "greek capital letter zeta, U+0396" },
1725
{ 919,  "Eta",  "greek capital letter eta, U+0397" },
1726
{ 920,  "Theta","greek capital letter theta, U+0398 ISOgrk3" },
1727
{ 921,  "Iota", "greek capital letter iota, U+0399" },
1728
{ 922,  "Kappa","greek capital letter kappa, U+039A" },
1729
{ 923,  "Lambda", "greek capital letter lambda, U+039B ISOgrk3" },
1730
{ 924,  "Mu", "greek capital letter mu, U+039C" },
1731
{ 925,  "Nu", "greek capital letter nu, U+039D" },
1732
{ 926,  "Xi", "greek capital letter xi, U+039E ISOgrk3" },
1733
{ 927,  "Omicron","greek capital letter omicron, U+039F" },
1734
{ 928,  "Pi", "greek capital letter pi, U+03A0 ISOgrk3" },
1735
{ 929,  "Rho",  "greek capital letter rho, U+03A1" },
1736
{ 931,  "Sigma","greek capital letter sigma, U+03A3 ISOgrk3" },
1737
{ 932,  "Tau",  "greek capital letter tau, U+03A4" },
1738
{ 933,  "Upsilon","greek capital letter upsilon, U+03A5 ISOgrk3" },
1739
{ 934,  "Phi",  "greek capital letter phi, U+03A6 ISOgrk3" },
1740
{ 935,  "Chi",  "greek capital letter chi, U+03A7" },
1741
{ 936,  "Psi",  "greek capital letter psi, U+03A8 ISOgrk3" },
1742
{ 937,  "Omega","greek capital letter omega, U+03A9 ISOgrk3" },
1743
1744
{ 945,  "alpha","greek small letter alpha, U+03B1 ISOgrk3" },
1745
{ 946,  "beta", "greek small letter beta, U+03B2 ISOgrk3" },
1746
{ 947,  "gamma","greek small letter gamma, U+03B3 ISOgrk3" },
1747
{ 948,  "delta","greek small letter delta, U+03B4 ISOgrk3" },
1748
{ 949,  "epsilon","greek small letter epsilon, U+03B5 ISOgrk3" },
1749
{ 950,  "zeta", "greek small letter zeta, U+03B6 ISOgrk3" },
1750
{ 951,  "eta",  "greek small letter eta, U+03B7 ISOgrk3" },
1751
{ 952,  "theta","greek small letter theta, U+03B8 ISOgrk3" },
1752
{ 953,  "iota", "greek small letter iota, U+03B9 ISOgrk3" },
1753
{ 954,  "kappa","greek small letter kappa, U+03BA ISOgrk3" },
1754
{ 955,  "lambda","greek small letter lambda, U+03BB ISOgrk3" },
1755
{ 956,  "mu", "greek small letter mu, U+03BC ISOgrk3" },
1756
{ 957,  "nu", "greek small letter nu, U+03BD ISOgrk3" },
1757
{ 958,  "xi", "greek small letter xi, U+03BE ISOgrk3" },
1758
{ 959,  "omicron","greek small letter omicron, U+03BF NEW" },
1759
{ 960,  "pi", "greek small letter pi, U+03C0 ISOgrk3" },
1760
{ 961,  "rho",  "greek small letter rho, U+03C1 ISOgrk3" },
1761
{ 962,  "sigmaf","greek small letter final sigma, U+03C2 ISOgrk3" },
1762
{ 963,  "sigma","greek small letter sigma, U+03C3 ISOgrk3" },
1763
{ 964,  "tau",  "greek small letter tau, U+03C4 ISOgrk3" },
1764
{ 965,  "upsilon","greek small letter upsilon, U+03C5 ISOgrk3" },
1765
{ 966,  "phi",  "greek small letter phi, U+03C6 ISOgrk3" },
1766
{ 967,  "chi",  "greek small letter chi, U+03C7 ISOgrk3" },
1767
{ 968,  "psi",  "greek small letter psi, U+03C8 ISOgrk3" },
1768
{ 969,  "omega","greek small letter omega, U+03C9 ISOgrk3" },
1769
{ 977,  "thetasym","greek small letter theta symbol, U+03D1 NEW" },
1770
{ 978,  "upsih","greek upsilon with hook symbol, U+03D2 NEW" },
1771
{ 982,  "piv",  "greek pi symbol, U+03D6 ISOgrk3" },
1772
1773
{ 8194, "ensp", "en space, U+2002 ISOpub" },
1774
{ 8195, "emsp", "em space, U+2003 ISOpub" },
1775
{ 8201, "thinsp","thin space, U+2009 ISOpub" },
1776
{ 8204, "zwnj", "zero width non-joiner, U+200C NEW RFC 2070" },
1777
{ 8205, "zwj",  "zero width joiner, U+200D NEW RFC 2070" },
1778
{ 8206, "lrm",  "left-to-right mark, U+200E NEW RFC 2070" },
1779
{ 8207, "rlm",  "right-to-left mark, U+200F NEW RFC 2070" },
1780
{ 8211, "ndash","en dash, U+2013 ISOpub" },
1781
{ 8212, "mdash","em dash, U+2014 ISOpub" },
1782
{ 8216, "lsquo","left single quotation mark, U+2018 ISOnum" },
1783
{ 8217, "rsquo","right single quotation mark, U+2019 ISOnum" },
1784
{ 8218, "sbquo","single low-9 quotation mark, U+201A NEW" },
1785
{ 8220, "ldquo","left double quotation mark, U+201C ISOnum" },
1786
{ 8221, "rdquo","right double quotation mark, U+201D ISOnum" },
1787
{ 8222, "bdquo","double low-9 quotation mark, U+201E NEW" },
1788
{ 8224, "dagger","dagger, U+2020 ISOpub" },
1789
{ 8225, "Dagger","double dagger, U+2021 ISOpub" },
1790
1791
{ 8226, "bull", "bullet = black small circle, U+2022 ISOpub" },
1792
{ 8230, "hellip","horizontal ellipsis = three dot leader, U+2026 ISOpub" },
1793
1794
{ 8240, "permil","per mille sign, U+2030 ISOtech" },
1795
1796
{ 8242, "prime","prime = minutes = feet, U+2032 ISOtech" },
1797
{ 8243, "Prime","double prime = seconds = inches, U+2033 ISOtech" },
1798
1799
{ 8249, "lsaquo","single left-pointing angle quotation mark, U+2039 ISO proposed" },
1800
{ 8250, "rsaquo","single right-pointing angle quotation mark, U+203A ISO proposed" },
1801
1802
{ 8254, "oline","overline = spacing overscore, U+203E NEW" },
1803
{ 8260, "frasl","fraction slash, U+2044 NEW" },
1804
1805
{ 8364, "euro", "euro sign, U+20AC NEW" },
1806
1807
{ 8465, "image","blackletter capital I = imaginary part, U+2111 ISOamso" },
1808
{ 8472, "weierp","script capital P = power set = Weierstrass p, U+2118 ISOamso" },
1809
{ 8476, "real", "blackletter capital R = real part symbol, U+211C ISOamso" },
1810
{ 8482, "trade","trade mark sign, U+2122 ISOnum" },
1811
{ 8501, "alefsym","alef symbol = first transfinite cardinal, U+2135 NEW" },
1812
{ 8592, "larr", "leftwards arrow, U+2190 ISOnum" },
1813
{ 8593, "uarr", "upwards arrow, U+2191 ISOnum" },
1814
{ 8594, "rarr", "rightwards arrow, U+2192 ISOnum" },
1815
{ 8595, "darr", "downwards arrow, U+2193 ISOnum" },
1816
{ 8596, "harr", "left right arrow, U+2194 ISOamsa" },
1817
{ 8629, "crarr","downwards arrow with corner leftwards = carriage return, U+21B5 NEW" },
1818
{ 8656, "lArr", "leftwards double arrow, U+21D0 ISOtech" },
1819
{ 8657, "uArr", "upwards double arrow, U+21D1 ISOamsa" },
1820
{ 8658, "rArr", "rightwards double arrow, U+21D2 ISOtech" },
1821
{ 8659, "dArr", "downwards double arrow, U+21D3 ISOamsa" },
1822
{ 8660, "hArr", "left right double arrow, U+21D4 ISOamsa" },
1823
1824
{ 8704, "forall","for all, U+2200 ISOtech" },
1825
{ 8706, "part", "partial differential, U+2202 ISOtech" },
1826
{ 8707, "exist","there exists, U+2203 ISOtech" },
1827
{ 8709, "empty","empty set = null set = diameter, U+2205 ISOamso" },
1828
{ 8711, "nabla","nabla = backward difference, U+2207 ISOtech" },
1829
{ 8712, "isin", "element of, U+2208 ISOtech" },
1830
{ 8713, "notin","not an element of, U+2209 ISOtech" },
1831
{ 8715, "ni", "contains as member, U+220B ISOtech" },
1832
{ 8719, "prod", "n-ary product = product sign, U+220F ISOamsb" },
1833
{ 8721, "sum",  "n-ary summation, U+2211 ISOamsb" },
1834
{ 8722, "minus","minus sign, U+2212 ISOtech" },
1835
{ 8727, "lowast","asterisk operator, U+2217 ISOtech" },
1836
{ 8730, "radic","square root = radical sign, U+221A ISOtech" },
1837
{ 8733, "prop", "proportional to, U+221D ISOtech" },
1838
{ 8734, "infin","infinity, U+221E ISOtech" },
1839
{ 8736, "ang",  "angle, U+2220 ISOamso" },
1840
{ 8743, "and",  "logical and = wedge, U+2227 ISOtech" },
1841
{ 8744, "or", "logical or = vee, U+2228 ISOtech" },
1842
{ 8745, "cap",  "intersection = cap, U+2229 ISOtech" },
1843
{ 8746, "cup",  "union = cup, U+222A ISOtech" },
1844
{ 8747, "int",  "integral, U+222B ISOtech" },
1845
{ 8756, "there4","therefore, U+2234 ISOtech" },
1846
{ 8764, "sim",  "tilde operator = varies with = similar to, U+223C ISOtech" },
1847
{ 8773, "cong", "approximately equal to, U+2245 ISOtech" },
1848
{ 8776, "asymp","almost equal to = asymptotic to, U+2248 ISOamsr" },
1849
{ 8800, "ne", "not equal to, U+2260 ISOtech" },
1850
{ 8801, "equiv","identical to, U+2261 ISOtech" },
1851
{ 8804, "le", "less-than or equal to, U+2264 ISOtech" },
1852
{ 8805, "ge", "greater-than or equal to, U+2265 ISOtech" },
1853
{ 8834, "sub",  "subset of, U+2282 ISOtech" },
1854
{ 8835, "sup",  "superset of, U+2283 ISOtech" },
1855
{ 8836, "nsub", "not a subset of, U+2284 ISOamsn" },
1856
{ 8838, "sube", "subset of or equal to, U+2286 ISOtech" },
1857
{ 8839, "supe", "superset of or equal to, U+2287 ISOtech" },
1858
{ 8853, "oplus","circled plus = direct sum, U+2295 ISOamsb" },
1859
{ 8855, "otimes","circled times = vector product, U+2297 ISOamsb" },
1860
{ 8869, "perp", "up tack = orthogonal to = perpendicular, U+22A5 ISOtech" },
1861
{ 8901, "sdot", "dot operator, U+22C5 ISOamsb" },
1862
{ 8968, "lceil","left ceiling = apl upstile, U+2308 ISOamsc" },
1863
{ 8969, "rceil","right ceiling, U+2309 ISOamsc" },
1864
{ 8970, "lfloor","left floor = apl downstile, U+230A ISOamsc" },
1865
{ 8971, "rfloor","right floor, U+230B ISOamsc" },
1866
{ 9001, "lang", "left-pointing angle bracket = bra, U+2329 ISOtech" },
1867
{ 9002, "rang", "right-pointing angle bracket = ket, U+232A ISOtech" },
1868
{ 9674, "loz",  "lozenge, U+25CA ISOpub" },
1869
1870
{ 9824, "spades","black spade suit, U+2660 ISOpub" },
1871
{ 9827, "clubs","black club suit = shamrock, U+2663 ISOpub" },
1872
{ 9829, "hearts","black heart suit = valentine, U+2665 ISOpub" },
1873
{ 9830, "diams","black diamond suit, U+2666 ISOpub" },
1874
1875
};
1876
1877
/************************************************************************
1878
 *                  *
1879
 *    Commodity functions to handle entities      *
1880
 *                  *
1881
 ************************************************************************/
1882
1883
/**
1884
 * Lookup the given entity in EntitiesTable
1885
 *
1886
 * @deprecated Only supports HTML 4.
1887
 *
1888
 * TODO: the linear scan is really ugly, an hash table is really needed.
1889
 *
1890
 * @param name  the entity name
1891
 * @returns the associated htmlEntityDesc if found, NULL otherwise.
1892
 */
1893
const htmlEntityDesc *
1894
0
htmlEntityLookup(const xmlChar *name) {
1895
0
    unsigned int i;
1896
1897
0
    for (i = 0;i < (sizeof(html40EntitiesTable)/
1898
0
                    sizeof(html40EntitiesTable[0]));i++) {
1899
0
        if (xmlStrEqual(name, BAD_CAST html40EntitiesTable[i].name)) {
1900
0
            return((htmlEntityDescPtr) &html40EntitiesTable[i]);
1901
0
  }
1902
0
    }
1903
0
    return(NULL);
1904
0
}
1905
1906
static int
1907
0
htmlCompareEntityDesc(const void *vkey, const void *vdesc) {
1908
0
    const unsigned *key = vkey;
1909
0
    const htmlEntityDesc *desc = vdesc;
1910
1911
0
    return((int) *key - (int) desc->value);
1912
0
}
1913
1914
/**
1915
 * Lookup the given entity in EntitiesTable
1916
 *
1917
 * @deprecated Only supports HTML 4.
1918
 *
1919
 * TODO: the linear scan is really ugly, an hash table is really needed.
1920
 *
1921
 * @param value  the entity's unicode value
1922
 * @returns the associated htmlEntityDesc if found, NULL otherwise.
1923
 */
1924
const htmlEntityDesc *
1925
0
htmlEntityValueLookup(unsigned int value) {
1926
0
    const htmlEntityDesc *desc;
1927
0
    size_t nmemb;
1928
1929
0
    nmemb = sizeof(html40EntitiesTable) / sizeof(html40EntitiesTable[0]);
1930
0
    desc = bsearch(&value, html40EntitiesTable, nmemb, sizeof(htmlEntityDesc),
1931
0
                   htmlCompareEntityDesc);
1932
1933
0
    return(desc);
1934
0
}
1935
1936
/**
1937
 * Take a block of UTF-8 chars in and try to convert it to an ASCII
1938
 * plus HTML entities block of chars out.
1939
 *
1940
 * @deprecated Internal function, don't use.
1941
 *
1942
 * @param out  a pointer to an array of bytes to store the result
1943
 * @param outlen  the length of `out`
1944
 * @param in  a pointer to an array of UTF-8 chars
1945
 * @param inlen  the length of `in`
1946
 * @returns 0 if success, -2 if the transcoding fails, or -1 otherwise
1947
 * The value of `inlen` after return is the number of octets consumed
1948
 *     as the return value is positive, else unpredictable.
1949
 * The value of `outlen` after return is the number of octets consumed.
1950
 */
1951
int
1952
htmlUTF8ToHtml(unsigned char* out, int *outlen,
1953
0
               const unsigned char* in, int *inlen) {
1954
0
    const unsigned char* instart = in;
1955
0
    const unsigned char* inend;
1956
0
    unsigned char* outstart = out;
1957
0
    unsigned char* outend;
1958
0
    int ret = XML_ENC_ERR_SPACE;
1959
1960
0
    if ((out == NULL) || (outlen == NULL) || (inlen == NULL))
1961
0
        return(XML_ENC_ERR_INTERNAL);
1962
1963
0
    if (in == NULL) {
1964
        /*
1965
   * initialization nothing to do
1966
   */
1967
0
  *outlen = 0;
1968
0
  *inlen = 0;
1969
0
  return(XML_ENC_ERR_SUCCESS);
1970
0
    }
1971
1972
0
    inend = in + *inlen;
1973
0
    outend = out + *outlen;
1974
0
    while (in < inend) {
1975
0
        const htmlEntityDesc *ent;
1976
0
        const char *cp;
1977
0
        char nbuf[16];
1978
0
        unsigned c, d;
1979
0
        int seqlen, len, i;
1980
1981
0
  d = *in;
1982
1983
0
  if (d < 0x80) {
1984
0
            if (out >= outend)
1985
0
                goto done;
1986
0
            *out++ = d;
1987
0
            in += 1;
1988
0
            continue;
1989
0
        }
1990
1991
0
        if (d < 0xE0)      { c = d & 0x1F; seqlen = 2; }
1992
0
        else if (d < 0xF0) { c = d & 0x0F; seqlen = 3; }
1993
0
        else               { c = d & 0x07; seqlen = 4; }
1994
1995
0
  if (inend - in < seqlen)
1996
0
      break;
1997
1998
0
  for (i = 1; i < seqlen; i++) {
1999
0
      d = in[i];
2000
0
      c <<= 6;
2001
0
      c |= d & 0x3F;
2002
0
  }
2003
2004
        /*
2005
         * Try to lookup a predefined HTML entity for it
2006
         */
2007
0
        ent = htmlEntityValueLookup(c);
2008
2009
0
        if (ent == NULL) {
2010
0
          snprintf(nbuf, sizeof(nbuf), "#%u", c);
2011
0
          cp = nbuf;
2012
0
        } else {
2013
0
          cp = ent->name;
2014
0
        }
2015
2016
0
        len = strlen(cp);
2017
0
        if (outend - out < len + 2)
2018
0
            goto done;
2019
2020
0
        *out++ = '&';
2021
0
        memcpy(out, cp, len);
2022
0
        out += len;
2023
0
        *out++ = ';';
2024
2025
0
        in += seqlen;
2026
0
    }
2027
2028
0
    ret = out - outstart;
2029
2030
0
done:
2031
0
    *outlen = out - outstart;
2032
0
    *inlen = in - instart;
2033
0
    return(ret);
2034
0
}
2035
2036
/**
2037
 * Take a block of UTF-8 chars in and try to convert it to an ASCII
2038
 * plus HTML entities block of chars out.
2039
 *
2040
 * @deprecated Only supports HTML 4.
2041
 *
2042
 * @param out  a pointer to an array of bytes to store the result
2043
 * @param outlen  the length of `out`
2044
 * @param in  a pointer to an array of UTF-8 chars
2045
 * @param inlen  the length of `in`
2046
 * @param quoteChar  the quote character to escape (' or ") or zero.
2047
 * @returns 0 if success, -2 if the transcoding fails, or -1 otherwise
2048
 * The value of `inlen` after return is the number of octets consumed
2049
 *     as the return value is positive, else unpredictable.
2050
 * The value of `outlen` after return is the number of octets consumed.
2051
 */
2052
int
2053
htmlEncodeEntities(unsigned char* out, int *outlen,
2054
0
       const unsigned char* in, int *inlen, int quoteChar) {
2055
0
    const unsigned char* processed = in;
2056
0
    const unsigned char* outend;
2057
0
    const unsigned char* outstart = out;
2058
0
    const unsigned char* instart = in;
2059
0
    const unsigned char* inend;
2060
0
    unsigned int c, d;
2061
0
    int trailing;
2062
2063
0
    if ((out == NULL) || (outlen == NULL) || (inlen == NULL) || (in == NULL))
2064
0
        return(-1);
2065
0
    outend = out + (*outlen);
2066
0
    inend = in + (*inlen);
2067
0
    while (in < inend) {
2068
0
  d = *in++;
2069
0
  if      (d < 0x80)  { c= d; trailing= 0; }
2070
0
  else if (d < 0xC0) {
2071
      /* trailing byte in leading position */
2072
0
      *outlen = out - outstart;
2073
0
      *inlen = processed - instart;
2074
0
      return(-2);
2075
0
        } else if (d < 0xE0)  { c= d & 0x1F; trailing= 1; }
2076
0
        else if (d < 0xF0)  { c= d & 0x0F; trailing= 2; }
2077
0
        else if (d < 0xF8)  { c= d & 0x07; trailing= 3; }
2078
0
  else {
2079
      /* no chance for this in Ascii */
2080
0
      *outlen = out - outstart;
2081
0
      *inlen = processed - instart;
2082
0
      return(-2);
2083
0
  }
2084
2085
0
  if (inend - in < trailing)
2086
0
      break;
2087
2088
0
  while (trailing--) {
2089
0
      if (((d= *in++) & 0xC0) != 0x80) {
2090
0
    *outlen = out - outstart;
2091
0
    *inlen = processed - instart;
2092
0
    return(-2);
2093
0
      }
2094
0
      c <<= 6;
2095
0
      c |= d & 0x3F;
2096
0
  }
2097
2098
  /* assertion: c is a single UTF-4 value */
2099
0
  if ((c < 0x80) && (c != (unsigned int) quoteChar) &&
2100
0
      (c != '&') && (c != '<') && (c != '>')) {
2101
0
      if (out >= outend)
2102
0
    break;
2103
0
      *out++ = c;
2104
0
  } else {
2105
0
      const htmlEntityDesc * ent;
2106
0
      const char *cp;
2107
0
      char nbuf[16];
2108
0
      int len;
2109
2110
      /*
2111
       * Try to lookup a predefined HTML entity for it
2112
       */
2113
0
      ent = htmlEntityValueLookup(c);
2114
0
      if (ent == NULL) {
2115
0
    snprintf(nbuf, sizeof(nbuf), "#%u", c);
2116
0
    cp = nbuf;
2117
0
      }
2118
0
      else
2119
0
    cp = ent->name;
2120
0
      len = strlen(cp);
2121
0
      if (outend - out < len + 2)
2122
0
    break;
2123
0
      *out++ = '&';
2124
0
      memcpy(out, cp, len);
2125
0
      out += len;
2126
0
      *out++ = ';';
2127
0
  }
2128
0
  processed = in;
2129
0
    }
2130
0
    *outlen = out - outstart;
2131
0
    *inlen = processed - instart;
2132
0
    return(0);
2133
0
}
2134
2135
/************************************************************************
2136
 *                  *
2137
 *    Commodity functions, cleanup needed ?     *
2138
 *                  *
2139
 ************************************************************************/
2140
/*
2141
 * all tags allowing pc data from the html 4.01 loose dtd
2142
 * NOTE: it might be more appropriate to integrate this information
2143
 * into the html40ElementTable array but I don't want to risk any
2144
 * binary incompatibility
2145
 */
2146
static const char *const allowPCData[] = {
2147
    "a", "abbr", "acronym", "address", "applet", "b", "bdo", "big",
2148
    "blockquote", "body", "button", "caption", "center", "cite", "code",
2149
    "dd", "del", "dfn", "div", "dt", "em", "font", "form", "h1", "h2",
2150
    "h3", "h4", "h5", "h6", "i", "iframe", "ins", "kbd", "label", "legend",
2151
    "li", "noframes", "noscript", "object", "p", "pre", "q", "s", "samp",
2152
    "small", "span", "strike", "strong", "td", "th", "tt", "u", "var"
2153
};
2154
2155
/**
2156
 * Is this a sequence of blank chars that one can ignore ?
2157
 *
2158
 * @param ctxt  an HTML parser context
2159
 * @param str  a xmlChar *
2160
 * @param len  the size of `str`
2161
 * @returns 1 if ignorable 0 if whitespace, -1 otherwise.
2162
 */
2163
2164
1.15M
static int areBlanks(htmlParserCtxtPtr ctxt, const xmlChar *str, int len) {
2165
1.15M
    unsigned int i;
2166
1.15M
    int j;
2167
1.15M
    xmlNodePtr lastChild;
2168
1.15M
    xmlDtdPtr dtd;
2169
2170
1.24M
    for (j = 0;j < len;j++)
2171
1.16M
        if (!(IS_WS_HTML(str[j]))) return(-1);
2172
2173
77.8k
    if (CUR == 0) return(1);
2174
77.8k
    if (CUR != '<') return(0);
2175
53
    if (ctxt->name == NULL)
2176
0
  return(1);
2177
53
    if (xmlStrEqual(ctxt->name, BAD_CAST"html"))
2178
0
  return(1);
2179
53
    if (xmlStrEqual(ctxt->name, BAD_CAST"head"))
2180
0
  return(1);
2181
2182
    /* Only strip CDATA children of the body tag for strict HTML DTDs */
2183
53
    if (xmlStrEqual(ctxt->name, BAD_CAST "body") && ctxt->myDoc != NULL) {
2184
21
        dtd = xmlGetIntSubset(ctxt->myDoc);
2185
21
        if (dtd != NULL && dtd->ExternalID != NULL) {
2186
0
            if (!xmlStrcasecmp(dtd->ExternalID, BAD_CAST "-//W3C//DTD HTML 4.01//EN") ||
2187
0
                    !xmlStrcasecmp(dtd->ExternalID, BAD_CAST "-//W3C//DTD HTML 4//EN"))
2188
0
                return(1);
2189
0
        }
2190
21
    }
2191
2192
53
    if (ctxt->node == NULL) return(0);
2193
53
    lastChild = xmlGetLastChild(ctxt->node);
2194
56
    while ((lastChild) && (lastChild->type == XML_COMMENT_NODE))
2195
3
  lastChild = lastChild->prev;
2196
53
    if (lastChild == NULL) {
2197
30
        if ((ctxt->node->type != XML_ELEMENT_NODE) &&
2198
0
            (ctxt->node->content != NULL)) return(0);
2199
  /* keep ws in constructs like ...<b> </b>...
2200
     for all tags "b" allowing PCDATA */
2201
30
  for ( i = 0; i < sizeof(allowPCData)/sizeof(allowPCData[0]); i++ ) {
2202
30
      if ( xmlStrEqual(ctxt->name, BAD_CAST allowPCData[i]) ) {
2203
30
    return(0);
2204
30
      }
2205
30
  }
2206
30
    } else if (xmlNodeIsText(lastChild)) {
2207
22
        return(0);
2208
22
    } else {
2209
  /* keep ws in constructs like <p><b>xy</b> <i>z</i><p>
2210
     for all tags "p" allowing PCDATA */
2211
1
  for ( i = 0; i < sizeof(allowPCData)/sizeof(allowPCData[0]); i++ ) {
2212
1
      if ( xmlStrEqual(lastChild->name, BAD_CAST allowPCData[i]) ) {
2213
1
    return(0);
2214
1
      }
2215
1
  }
2216
1
    }
2217
0
    return(1);
2218
53
}
2219
2220
/**
2221
 * Creates a new HTML document without a DTD node if `URI` and `publicId`
2222
 * are NULL
2223
 *
2224
 * @param URI  system ID (URI) of the DTD (optional)
2225
 * @param publicId  public ID of the DTD (optional)
2226
 * @returns a new document, do not initialize the DTD if not provided
2227
 */
2228
xmlDoc *
2229
2.50k
htmlNewDocNoDtD(const xmlChar *URI, const xmlChar *publicId) {
2230
2.50k
    xmlDocPtr cur;
2231
2232
    /*
2233
     * Allocate a new document and fill the fields.
2234
     */
2235
2.50k
    cur = (xmlDocPtr) xmlMalloc(sizeof(xmlDoc));
2236
2.50k
    if (cur == NULL)
2237
3
  return(NULL);
2238
2.49k
    memset(cur, 0, sizeof(xmlDoc));
2239
2240
2.49k
    cur->type = XML_HTML_DOCUMENT_NODE;
2241
2.49k
    cur->version = NULL;
2242
2.49k
    cur->intSubset = NULL;
2243
2.49k
    cur->doc = cur;
2244
2.49k
    cur->name = NULL;
2245
2.49k
    cur->children = NULL;
2246
2.49k
    cur->extSubset = NULL;
2247
2.49k
    cur->oldNs = NULL;
2248
2.49k
    cur->encoding = NULL;
2249
2.49k
    cur->standalone = 1;
2250
2.49k
    cur->compression = 0;
2251
2.49k
    cur->ids = NULL;
2252
2.49k
    cur->refs = NULL;
2253
2.49k
    cur->_private = NULL;
2254
2.49k
    cur->charset = XML_CHAR_ENCODING_UTF8;
2255
2.49k
    cur->properties = XML_DOC_HTML | XML_DOC_USERBUILT;
2256
2.49k
    if ((publicId != NULL) ||
2257
2.49k
  (URI != NULL)) {
2258
0
        xmlDtdPtr intSubset;
2259
2260
0
  intSubset = xmlCreateIntSubset(cur, BAD_CAST "html", publicId, URI);
2261
0
        if (intSubset == NULL) {
2262
0
            xmlFree(cur);
2263
0
            return(NULL);
2264
0
        }
2265
0
    }
2266
2.49k
    if ((xmlRegisterCallbacks) && (xmlRegisterNodeDefaultValue))
2267
0
  xmlRegisterNodeDefaultValue((xmlNodePtr)cur);
2268
2.49k
    return(cur);
2269
2.49k
}
2270
2271
/**
2272
 * Creates a new HTML document
2273
 *
2274
 * @param URI  system ID (URI) of the DTD (optional)
2275
 * @param publicId  public ID of the DTD (optional)
2276
 * @returns a new document
2277
 */
2278
xmlDoc *
2279
0
htmlNewDoc(const xmlChar *URI, const xmlChar *publicId) {
2280
0
    if ((URI == NULL) && (publicId == NULL))
2281
0
  return(htmlNewDocNoDtD(
2282
0
        BAD_CAST "http://www.w3.org/TR/REC-html40/loose.dtd",
2283
0
        BAD_CAST "-//W3C//DTD HTML 4.0 Transitional//EN"));
2284
2285
0
    return(htmlNewDocNoDtD(URI, publicId));
2286
0
}
2287
2288
2289
/************************************************************************
2290
 *                  *
2291
 *      The parser itself       *
2292
 *  Relates to http://www.w3.org/TR/html40        *
2293
 *                  *
2294
 ************************************************************************/
2295
2296
/************************************************************************
2297
 *                  *
2298
 *      The parser itself       *
2299
 *                  *
2300
 ************************************************************************/
2301
2302
/**
2303
 * parse an HTML tag or attribute name, note that we convert it to lowercase
2304
 * since HTML names are not case-sensitive.
2305
 *
2306
 * @param ctxt  an HTML parser context
2307
 * @param attr  whether this is an attribute name
2308
 * @returns the Tag Name parsed or NULL
2309
 */
2310
2311
static xmlHashedString
2312
86.3k
htmlParseHTMLName(htmlParserCtxtPtr ctxt, int attr) {
2313
86.3k
    xmlHashedString ret;
2314
86.3k
    xmlChar buf[HTML_PARSER_BUFFER_SIZE];
2315
86.3k
    const xmlChar *in;
2316
86.3k
    size_t avail;
2317
86.3k
    int eof = PARSER_PROGRESSIVE(ctxt);
2318
86.3k
    int nbchar = 0;
2319
86.3k
    int stop = attr ? '=' : ' ';
2320
2321
86.3k
    in = ctxt->input->cur;
2322
86.3k
    avail = ctxt->input->end - in;
2323
2324
3.53M
    while (1) {
2325
3.53M
        int c, size;
2326
2327
3.53M
        if ((!eof) && (avail < 32)) {
2328
1.67k
            size_t oldAvail = avail;
2329
2330
1.67k
            ctxt->input->cur = in;
2331
2332
1.67k
            SHRINK;
2333
1.67k
            xmlParserGrow(ctxt);
2334
2335
1.67k
            in = ctxt->input->cur;
2336
1.67k
            avail = ctxt->input->end - in;
2337
2338
1.67k
            if (oldAvail == avail)
2339
1.57k
                eof = 1;
2340
1.67k
        }
2341
2342
3.53M
        if (avail == 0)
2343
198
            break;
2344
2345
3.53M
        c = *in;
2346
3.53M
        size = 1;
2347
2348
3.53M
        if ((nbchar != 0) &&
2349
3.44M
            ((c == '/') || (c == '>') || (c == stop) ||
2350
3.40M
             (IS_WS_HTML(c))))
2351
86.1k
            break;
2352
2353
3.44M
        if (c == 0) {
2354
11.8k
            if (nbchar + 3 <= HTML_PARSER_BUFFER_SIZE) {
2355
9.69k
                buf[nbchar++] = 0xEF;
2356
9.69k
                buf[nbchar++] = 0xBF;
2357
9.69k
                buf[nbchar++] = 0xBD;
2358
9.69k
            }
2359
3.43M
        } else if (c < 0x80) {
2360
1.47M
            if (nbchar < HTML_PARSER_BUFFER_SIZE) {
2361
1.16M
                if (IS_UPPER(c))
2362
229k
                    c += 0x20;
2363
1.16M
                buf[nbchar++] = c;
2364
1.16M
            }
2365
1.95M
        } else {
2366
1.95M
            size = htmlValidateUtf8(ctxt, in, avail, /* partial */ 0);
2367
2368
1.95M
            if (size > 0) {
2369
1.61M
                if (nbchar + size <= HTML_PARSER_BUFFER_SIZE) {
2370
1.20M
                    memcpy(buf + nbchar, in, size);
2371
1.20M
                    nbchar += size;
2372
1.20M
                }
2373
1.61M
            } else {
2374
337k
                size = 1;
2375
2376
337k
                if (nbchar + 3 <= HTML_PARSER_BUFFER_SIZE) {
2377
9.45k
                    buf[nbchar++] = 0xEF;
2378
9.45k
                    buf[nbchar++] = 0xBF;
2379
9.45k
                    buf[nbchar++] = 0xBD;
2380
9.45k
                }
2381
337k
            }
2382
1.95M
        }
2383
2384
3.44M
        in += size;
2385
3.44M
        avail -= size;
2386
3.44M
    }
2387
2388
86.3k
    ctxt->input->cur = in;
2389
2390
86.3k
    SHRINK;
2391
2392
86.3k
    ret = xmlDictLookupHashed(ctxt->dict, buf, nbchar);
2393
86.3k
    if (ret.name == NULL)
2394
6
        htmlErrMemory(ctxt);
2395
2396
86.3k
    return(ret);
2397
86.3k
}
2398
2399
static const short htmlC1Remap[32] = {
2400
    0x20AC, 0x0081, 0x201A, 0x0192, 0x201E, 0x2026, 0x2020, 0x2021,
2401
    0x02C6, 0x2030, 0x0160, 0x2039, 0x0152, 0x008D, 0x017D, 0x008F,
2402
    0x0090, 0x2018, 0x2019, 0x201C, 0x201D, 0x2022, 0x2013, 0x2014,
2403
    0x02DC, 0x2122, 0x0161, 0x203A, 0x0153, 0x009D, 0x017E, 0x0178
2404
};
2405
2406
static const xmlChar *
2407
153
htmlCodePointToUtf8(int c, xmlChar *out, int *osize) {
2408
153
    int i = 0;
2409
153
    int bits, hi;
2410
2411
153
    if ((c >= 0x80) && (c < 0xA0)) {
2412
5
        c = htmlC1Remap[c - 0x80];
2413
148
    } else if ((c <= 0) ||
2414
101
               ((c >= 0xD800) && (c < 0xE000)) ||
2415
101
               (c > 0x10FFFF)) {
2416
73
        c = 0xFFFD;
2417
73
    }
2418
2419
153
    if      (c <    0x80) { bits =  0; hi = 0x00; }
2420
106
    else if (c <   0x800) { bits =  6; hi = 0xC0; }
2421
99
    else if (c < 0x10000) { bits = 12; hi = 0xE0; }
2422
17
    else                  { bits = 18; hi = 0xF0; }
2423
2424
153
    out[i++] = (c >> bits) | hi;
2425
2426
375
    while (bits > 0) {
2427
222
        bits -= 6;
2428
222
        out[i++] = ((c >> bits) & 0x3F) | 0x80;
2429
222
    }
2430
2431
153
    *osize = i;
2432
153
    return(out);
2433
153
}
2434
2435
#include "codegen/html5ent.inc"
2436
2437
61
#define ENT_F_SEMICOLON 0x80u
2438
109
#define ENT_F_SUBTABLE  0x40u
2439
1.04M
#define ENT_F_ALL       0xC0u
2440
2441
static const xmlChar *
2442
htmlFindEntityPrefix(const xmlChar *string, size_t slen, int isAttr,
2443
182k
                     int *nlen, int *rlen) {
2444
182k
    const xmlChar *match = NULL;
2445
182k
    unsigned left, right;
2446
182k
    int first = string[0];
2447
182k
    size_t matchLen = 0;
2448
182k
    size_t soff = 1;
2449
2450
182k
    if (slen < 2)
2451
24
        return(NULL);
2452
182k
    if (!IS_ASCII_LETTER(first))
2453
28.3k
        return(NULL);
2454
2455
    /*
2456
     * Look up range by first character
2457
     */
2458
153k
    first &= 63;
2459
153k
    left = htmlEntAlpha[first*3] | htmlEntAlpha[first*3+1] << 8;
2460
153k
    right = left + htmlEntAlpha[first*3+2];
2461
2462
    /*
2463
     * Binary search
2464
     */
2465
1.19M
    while (left < right) {
2466
1.04M
        const xmlChar *bytes;
2467
1.04M
        unsigned mid;
2468
1.04M
        size_t len;
2469
1.04M
        int cmp;
2470
2471
1.04M
        mid = left + (right - left) / 2;
2472
1.04M
        bytes = htmlEntStrings + htmlEntValues[mid];
2473
1.04M
        len = bytes[0] & ~ENT_F_ALL;
2474
2475
1.04M
        cmp = string[soff] - bytes[1];
2476
2477
1.04M
        if (cmp == 0) {
2478
1.78k
            if (slen < len) {
2479
1
                cmp = strncmp((const char *) string + soff + 1,
2480
1
                              (const char *) bytes + 2,
2481
1
                              slen - 1);
2482
                /* Prefix can never match */
2483
1
                if (cmp == 0)
2484
0
                    break;
2485
1.78k
            } else {
2486
1.78k
                cmp = strncmp((const char *) string + soff + 1,
2487
1.78k
                              (const char *) bytes + 2,
2488
1.78k
                              len - 1);
2489
1.78k
            }
2490
1.78k
        }
2491
2492
1.04M
        if (cmp < 0) {
2493
1.02M
            right = mid;
2494
1.02M
        } else if (cmp > 0) {
2495
24.5k
            left = mid + 1;
2496
24.5k
        } else {
2497
109
            int term = soff + len < slen ? string[soff + len] : 0;
2498
109
            int isAlnum, isTerm;
2499
2500
109
            isAlnum = IS_ALNUM(term);
2501
109
            isTerm = ((term == ';') ||
2502
61
                      ((bytes[0] & ENT_F_SEMICOLON) &&
2503
33
                       ((!isAttr) ||
2504
3
                        ((!isAlnum) && (term != '=')))));
2505
2506
109
            if (isTerm) {
2507
81
                match = bytes + len + 1;
2508
81
                matchLen = soff + len;
2509
81
                if (term == ';')
2510
48
                    matchLen += 1;
2511
81
            }
2512
2513
109
            if (bytes[0] & ENT_F_SUBTABLE) {
2514
84
                if (isTerm)
2515
56
                    match += 2;
2516
2517
84
                if ((isAlnum) && (soff + len < slen)) {
2518
26
                    left = mid + bytes[len + 1];
2519
26
                    right = left + bytes[len + 2];
2520
26
                    soff += len;
2521
26
                    continue;
2522
26
                }
2523
84
            }
2524
2525
83
            break;
2526
109
        }
2527
1.04M
    }
2528
2529
153k
    if (match == NULL)
2530
153k
        return(NULL);
2531
2532
81
    *nlen = matchLen;
2533
81
    *rlen = match[0];
2534
81
    return(match + 1);
2535
153k
}
2536
2537
/**
2538
 * Parse data until terminator is reached.
2539
 *
2540
 * @param ctxt  an HTML parser context
2541
 * @param mask  mask of terminating characters
2542
 * @param comment  true if parsing a comment
2543
 * @param refs  true if references are allowed
2544
 * @param maxLength  maximum output length
2545
 * @returns the parsed string or NULL in case of errors.
2546
 */
2547
2548
static xmlChar *
2549
htmlParseData(htmlParserCtxtPtr ctxt, htmlAsciiMask mask,
2550
30.5k
              int comment, int refs, int maxLength) {
2551
30.5k
    xmlParserInputPtr input = ctxt->input;
2552
30.5k
    xmlChar *ret = NULL;
2553
30.5k
    xmlChar *buffer;
2554
30.5k
    xmlChar utf8Char[4];
2555
30.5k
    size_t buffer_size;
2556
30.5k
    size_t used;
2557
30.5k
    int eof = PARSER_PROGRESSIVE(ctxt);
2558
30.5k
    int line, col;
2559
30.5k
    int termSkip = -1;
2560
2561
30.5k
    used = 0;
2562
30.5k
    buffer_size = ctxt->spaceMax;
2563
30.5k
    buffer = (xmlChar *) ctxt->spaceTab;
2564
30.5k
    if (buffer == NULL) {
2565
1.28k
        buffer_size = 500;
2566
1.28k
        buffer = xmlMalloc(buffer_size + 1);
2567
1.28k
        if (buffer == NULL) {
2568
2
            htmlErrMemory(ctxt);
2569
2
            return(NULL);
2570
2
        }
2571
1.28k
    }
2572
2573
30.5k
    line = input->line;
2574
30.5k
    col = input->col;
2575
2576
215k
    while (!PARSER_STOPPED(ctxt)) {
2577
215k
        const xmlChar *chunk, *in, *repl;
2578
215k
        size_t avail, chunkSize, extraSize;
2579
215k
        int replSize;
2580
215k
        int skip = 0;
2581
215k
        int ncr = 0;
2582
215k
        int ncrSize = 0;
2583
215k
        int cp = 0;
2584
2585
215k
        chunk = input->cur;
2586
215k
        avail = input->end - chunk;
2587
215k
        in = chunk;
2588
2589
215k
        repl = BAD_CAST "";
2590
215k
        replSize = 0;
2591
2592
2.89M
        while (!PARSER_STOPPED(ctxt)) {
2593
2.89M
            size_t j;
2594
2.89M
            int cur, size;
2595
2596
2.89M
            if ((!eof) && (avail <= 64)) {
2597
1.54k
                size_t oldAvail = avail;
2598
1.54k
                size_t off = in - chunk;
2599
2600
1.54k
                input->cur = in;
2601
2602
1.54k
                xmlParserGrow(ctxt);
2603
2604
1.54k
                in = input->cur;
2605
1.54k
                chunk = in - off;
2606
1.54k
                input->cur = chunk;
2607
1.54k
                avail = input->end - in;
2608
2609
1.54k
                if (oldAvail == avail)
2610
724
                    eof = 1;
2611
1.54k
            }
2612
2613
2.89M
            if (avail == 0) {
2614
102
                termSkip = 0;
2615
102
                break;
2616
102
            }
2617
2618
2.89M
            cur = *in;
2619
2.89M
            size = 1;
2620
2.89M
            col += 1;
2621
2622
2.89M
            if (htmlMaskMatch(mask, cur)) {
2623
33.7k
                if (comment) {
2624
23.6k
                    if (avail < 2) {
2625
0
                        termSkip = 1;
2626
23.6k
                    } else if (in[1] == '-') {
2627
20.4k
                        if  (avail < 3) {
2628
0
                            termSkip = 2;
2629
20.4k
                        } else if (in[2] == '>') {
2630
20.2k
                            termSkip = 3;
2631
20.2k
                        } else if (in[2] == '!') {
2632
1
                            if (avail < 4)
2633
0
                                termSkip = 3;
2634
1
                            else if (in[3] == '>')
2635
0
                                termSkip = 4;
2636
1
                        }
2637
20.4k
                    }
2638
2639
23.6k
                    if (termSkip >= 0)
2640
20.2k
                        break;
2641
23.6k
                } else {
2642
10.1k
                    termSkip = 0;
2643
10.1k
                    break;
2644
10.1k
                }
2645
33.7k
            }
2646
2647
2.86M
            if (ncr) {
2648
53
                int lc = cur | 0x20;
2649
53
                int digit;
2650
2651
53
                if ((cur >= '0') && (cur <= '9')) {
2652
13
                    digit = cur - '0';
2653
40
                } else if ((ncr == 16) && (lc >= 'a') && (lc <= 'f')) {
2654
24
                    digit = (lc - 'a') + 10;
2655
24
                } else {
2656
16
                    if (cur == ';') {
2657
6
                        in += 1;
2658
6
                        size += 1;
2659
6
                        ncrSize += 1;
2660
6
                    }
2661
16
                    goto next_chunk;
2662
16
                }
2663
2664
37
                cp = cp * ncr + digit;
2665
37
                if (cp >= 0x110000)
2666
0
                    cp = 0x110000;
2667
2668
37
                ncrSize += 1;
2669
2670
37
                goto next_char;
2671
53
            }
2672
2673
2.86M
            switch (cur) {
2674
9.72k
            case '&':
2675
9.72k
                if (!refs)
2676
5.42k
                    break;
2677
2678
4.29k
                j = 1;
2679
2680
4.29k
                if ((j < avail) && (in[j] == '#')) {
2681
27
                    j += 1;
2682
27
                    if (j < avail) {
2683
27
                        if ((in[j] | 0x20) == 'x') {
2684
20
                            j += 1;
2685
20
                            if ((j < avail) && (IS_HEX_DIGIT(in[j]))) {
2686
15
                                ncr = 16;
2687
15
                                size = 3;
2688
15
                                ncrSize = 3;
2689
15
                                cp = 0;
2690
15
                            }
2691
20
                        } else if (IS_ASCII_DIGIT(in[j])) {
2692
1
                            ncr = 10;
2693
1
                            size = 2;
2694
1
                            ncrSize = 2;
2695
1
                            cp = 0;
2696
1
                        }
2697
27
                    }
2698
4.27k
                } else {
2699
4.27k
                    repl = htmlFindEntityPrefix(in + j,
2700
4.27k
                                                avail - j,
2701
4.27k
                                                /* isAttr */ 1,
2702
4.27k
                                                &skip, &replSize);
2703
4.27k
                    if (repl != NULL) {
2704
11
                        skip += 1;
2705
11
                        goto next_chunk;
2706
11
                    }
2707
2708
4.26k
                    skip = 0;
2709
4.26k
                }
2710
2711
4.28k
                break;
2712
2713
14.7k
            case '\0':
2714
14.7k
                skip = 1;
2715
14.7k
                repl = BAD_CAST "\xEF\xBF\xBD";
2716
14.7k
                replSize = 3;
2717
14.7k
                goto next_chunk;
2718
2719
6.82k
            case '\n':
2720
6.82k
                line += 1;
2721
6.82k
                col = 1;
2722
6.82k
                break;
2723
2724
12.6k
            case '\r':
2725
12.6k
                skip = 1;
2726
12.6k
                if (in[1] != 0x0A) {
2727
12.6k
                    repl = BAD_CAST "\x0A";
2728
12.6k
                    replSize = 1;
2729
12.6k
                }
2730
12.6k
                goto next_chunk;
2731
2732
2.81M
            default:
2733
2.81M
                if (cur < 0x80)
2734
1.43M
                    break;
2735
2736
1.37M
                if ((input->flags & XML_INPUT_HAS_ENCODING) == 0) {
2737
208
                    xmlChar * guess;
2738
2739
208
                    if (in > chunk)
2740
101
                        goto next_chunk;
2741
2742
107
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
2743
107
                    guess = NULL;
2744
#else
2745
                    guess = htmlFindEncoding(ctxt);
2746
#endif
2747
107
                    if (guess == NULL) {
2748
107
                        xmlSwitchEncoding(ctxt,
2749
107
                                XML_CHAR_ENCODING_WINDOWS_1252);
2750
107
                    } else {
2751
0
                        xmlSwitchEncodingName(ctxt, (const char *) guess);
2752
0
                        xmlFree(guess);
2753
0
                    }
2754
107
                    input->flags |= XML_INPUT_HAS_ENCODING;
2755
2756
107
                    eof = PARSER_PROGRESSIVE(ctxt);
2757
107
                    goto restart;
2758
208
                }
2759
2760
1.37M
                size = htmlValidateUtf8(ctxt, in, avail, /* partial */ 0);
2761
2762
1.37M
                if (size <= 0) {
2763
157k
                    skip = 1;
2764
157k
                    repl = BAD_CAST "\xEF\xBF\xBD";
2765
157k
                    replSize = 3;
2766
157k
                    goto next_chunk;
2767
157k
                }
2768
2769
1.22M
                break;
2770
2.86M
            }
2771
2772
2.67M
next_char:
2773
2.67M
            in += size;
2774
2.67M
            avail -= size;
2775
2.67M
        }
2776
2777
215k
next_chunk:
2778
215k
        if (ncrSize > 0) {
2779
16
            skip = ncrSize;
2780
16
            in -= ncrSize;
2781
2782
16
            repl = htmlCodePointToUtf8(cp, utf8Char, &replSize);
2783
16
        }
2784
2785
215k
        chunkSize = in - chunk;
2786
215k
        extraSize = chunkSize + replSize;
2787
2788
215k
        if (extraSize > maxLength - used) {
2789
0
            htmlParseErr(ctxt, XML_ERR_RESOURCE_LIMIT,
2790
0
                         "value too long\n", NULL, NULL);
2791
0
            goto error;
2792
0
        }
2793
2794
215k
        if (extraSize > buffer_size - used) {
2795
316
            size_t newSize = (used + extraSize) * 2;
2796
316
            xmlChar *tmp = xmlRealloc(buffer, newSize + 1);
2797
2798
316
            if (tmp == NULL) {
2799
0
                htmlErrMemory(ctxt);
2800
0
                goto error;
2801
0
            }
2802
316
            buffer = tmp;
2803
316
            buffer_size = newSize;
2804
316
        }
2805
2806
215k
        if (chunkSize > 0) {
2807
143k
            input->cur += chunkSize;
2808
143k
            memcpy(buffer + used, chunk, chunkSize);
2809
143k
            used += chunkSize;
2810
143k
        }
2811
2812
215k
        input->cur += skip;
2813
215k
        if (replSize > 0) {
2814
185k
            memcpy(buffer + used, repl, replSize);
2815
185k
            used += replSize;
2816
185k
        }
2817
2818
215k
        SHRINK;
2819
2820
215k
        if (termSkip >= 0)
2821
30.5k
            break;
2822
2823
185k
restart:
2824
185k
        ;
2825
185k
    }
2826
2827
30.5k
    if (termSkip > 0) {
2828
20.2k
        input->cur += termSkip;
2829
20.2k
        col += termSkip;
2830
20.2k
    }
2831
2832
30.5k
    input->line = line;
2833
30.5k
    input->col = col;
2834
2835
30.5k
    ret = xmlMalloc(used + 1);
2836
30.5k
    if (ret == NULL) {
2837
0
        htmlErrMemory(ctxt);
2838
30.5k
    } else {
2839
30.5k
        memcpy(ret, buffer, used);
2840
30.5k
        ret[used] = 0;
2841
30.5k
    }
2842
2843
30.5k
error:
2844
30.5k
    ctxt->spaceTab = (void *) buffer;
2845
30.5k
    ctxt->spaceMax = buffer_size;
2846
2847
30.5k
    return(ret);
2848
30.5k
}
2849
2850
/**
2851
 * @deprecated Internal function, don't use.
2852
 *
2853
 * @param ctxt  an HTML parser context
2854
 * @param str  location to store the entity name
2855
 * @returns NULL.
2856
 */
2857
const htmlEntityDesc *
2858
htmlParseEntityRef(htmlParserCtxt *ctxt ATTRIBUTE_UNUSED,
2859
0
                   const xmlChar **str ATTRIBUTE_UNUSED) {
2860
0
    return(NULL);
2861
0
}
2862
2863
/**
2864
 * parse a value for an attribute
2865
 * Note: the parser won't do substitution of entities here, this
2866
 * will be handled later in #xmlStringGetNodeList, unless it was
2867
 * asked for ctxt->replaceEntities != 0
2868
 *
2869
 * @param ctxt  an HTML parser context
2870
 * @returns the AttValue parsed or NULL.
2871
 */
2872
2873
static xmlChar *
2874
7.40k
htmlParseAttValue(htmlParserCtxtPtr ctxt) {
2875
7.40k
    xmlChar *ret = NULL;
2876
7.40k
    int maxLength = (ctxt->options & HTML_PARSE_HUGE) ?
2877
1.83k
                    XML_MAX_HUGE_LENGTH :
2878
7.40k
                    XML_MAX_TEXT_LENGTH;
2879
2880
7.40k
    if (CUR == '"') {
2881
17
        SKIP(1);
2882
17
  ret = htmlParseData(ctxt, MASK_DQ, 0, 1, maxLength);
2883
17
        if (CUR == '"')
2884
13
            SKIP(1);
2885
7.39k
    } else if (CUR == '\'') {
2886
51
        SKIP(1);
2887
51
  ret = htmlParseData(ctxt, MASK_SQ, 0, 1, maxLength);
2888
51
        if (CUR == '\'')
2889
51
            SKIP(1);
2890
7.34k
    } else {
2891
7.34k
  ret = htmlParseData(ctxt, MASK_WS_GT, 0, 1, maxLength);
2892
7.34k
    }
2893
7.40k
    return(ret);
2894
7.40k
}
2895
2896
static void
2897
htmlCharDataSAXCallback(htmlParserCtxtPtr ctxt, const xmlChar *buf,
2898
6.63M
                        int size, int mode) {
2899
6.63M
    if ((ctxt->sax == NULL) || (ctxt->disableSAX))
2900
1
        return;
2901
2902
6.63M
    if ((mode == 0) || (mode == DATA_RCDATA) ||
2903
6.63M
        (ctxt->sax->cdataBlock == NULL)) {
2904
6.63M
        if ((ctxt->name == NULL) ||
2905
6.63M
            (xmlStrEqual(ctxt->name, BAD_CAST "html")) ||
2906
6.63M
            (xmlStrEqual(ctxt->name, BAD_CAST "head"))) {
2907
2.25k
            int i;
2908
2909
            /*
2910
             * Add leading whitespace to html or head elements before
2911
             * calling htmlStartCharData.
2912
             */
2913
2.31k
            for (i = 0; i < size; i++)
2914
2.31k
                if (!IS_WS_HTML(buf[i]))
2915
2.25k
                    break;
2916
2917
2.25k
            if (i > 0) {
2918
61
                if (!ctxt->keepBlanks) {
2919
24
                    if (ctxt->sax->ignorableWhitespace != NULL)
2920
24
                        ctxt->sax->ignorableWhitespace(ctxt->userData, buf, i);
2921
37
                } else {
2922
37
                    if (ctxt->sax->characters != NULL)
2923
37
                        ctxt->sax->characters(ctxt->userData, buf, i);
2924
37
                }
2925
2926
61
                buf += i;
2927
61
                size -= i;
2928
61
            }
2929
2930
2.25k
            if (size <= 0)
2931
3
                return;
2932
2933
2.25k
            htmlStartCharData(ctxt);
2934
2935
2.25k
            if (PARSER_STOPPED(ctxt))
2936
7
                return;
2937
2.25k
        }
2938
2939
6.63M
        if ((mode == 0) &&
2940
6.63M
            (!ctxt->keepBlanks) &&
2941
1.15M
            (areBlanks(ctxt, buf, size) > 0)) {
2942
22
            if (ctxt->sax->ignorableWhitespace != NULL)
2943
22
                ctxt->sax->ignorableWhitespace(ctxt->userData, buf, size);
2944
6.63M
        } else {
2945
6.63M
            if (ctxt->sax->characters != NULL)
2946
6.63M
                ctxt->sax->characters(ctxt->userData, buf, size);
2947
6.63M
        }
2948
6.63M
    } else {
2949
        /*
2950
         * Insert as CDATA, which is the same as HTML_PRESERVE_NODE
2951
         */
2952
0
        ctxt->sax->cdataBlock(ctxt->userData, buf, size);
2953
0
    }
2954
6.63M
}
2955
2956
/**
2957
 * Parse character data and references.
2958
 *
2959
 * @param ctxt  an HTML parser context
2960
 * @param partial  true if the input buffer is incomplete
2961
 * @returns 1 if all data was parsed, 0 otherwise.
2962
 */
2963
2964
static int
2965
59.6k
htmlParseCharData(htmlParserCtxtPtr ctxt, int partial) {
2966
59.6k
    xmlParserInputPtr input = ctxt->input;
2967
59.6k
    xmlChar utf8Char[4];
2968
59.6k
    int complete = 0;
2969
59.6k
    int done = 0;
2970
59.6k
    int mode;
2971
59.6k
    int eof = PARSER_PROGRESSIVE(ctxt);
2972
59.6k
    int line, col;
2973
2974
59.6k
    mode = ctxt->endCheckState;
2975
2976
59.6k
    line = input->line;
2977
59.6k
    col = input->col;
2978
2979
4.14M
    while (!PARSER_STOPPED(ctxt)) {
2980
4.14M
        const xmlChar *chunk, *in, *repl;
2981
4.14M
        size_t avail;
2982
4.14M
        int replSize;
2983
4.14M
        int skip = 0;
2984
4.14M
        int ncr = 0;
2985
4.14M
        int ncrSize = 0;
2986
4.14M
        int cp = 0;
2987
2988
4.14M
        chunk = input->cur;
2989
4.14M
        avail = input->end - chunk;
2990
4.14M
        in = chunk;
2991
2992
4.14M
        repl = BAD_CAST "";
2993
4.14M
        replSize = 0;
2994
2995
22.5M
        while (!PARSER_STOPPED(ctxt)) {
2996
22.5M
            size_t j;
2997
22.5M
            int cur, size;
2998
2999
22.5M
            if (avail <= 64) {
3000
48.2k
                if (!eof) {
3001
14.2k
                    size_t oldAvail = avail;
3002
14.2k
                    size_t off = in - chunk;
3003
3004
14.2k
                    input->cur = in;
3005
3006
14.2k
                    xmlParserGrow(ctxt);
3007
3008
14.2k
                    in = input->cur;
3009
14.2k
                    chunk = in - off;
3010
14.2k
                    input->cur = chunk;
3011
14.2k
                    avail = input->end - in;
3012
3013
14.2k
                    if (oldAvail == avail)
3014
2.61k
                        eof = 1;
3015
14.2k
                }
3016
3017
48.2k
                if (avail == 0) {
3018
1.98k
                    if ((partial) && (ncr)) {
3019
0
                        in -= ncrSize;
3020
0
                        ncrSize = 0;
3021
0
                    }
3022
3023
1.98k
                    done = 1;
3024
1.98k
                    break;
3025
1.98k
                }
3026
48.2k
            }
3027
3028
            /* Accelerator */
3029
22.5M
            if (!ncr) {
3030
42.6M
                while (avail > 0) {
3031
42.6M
                    static const unsigned mask[8] = {
3032
42.6M
                        0x00002401, 0x10002040,
3033
42.6M
                        0x00000000, 0x00000000,
3034
42.6M
                        0xFFFFFFFF, 0xFFFFFFFF,
3035
42.6M
                        0xFFFFFFFF, 0xFFFFFFFF
3036
42.6M
                    };
3037
42.6M
                    cur = *in;
3038
42.6M
                    if ((1u << (cur & 0x1F)) & mask[cur >> 5])
3039
22.5M
                        break;
3040
20.1M
                    col += 1;
3041
20.1M
                    in += 1;
3042
20.1M
                    avail -= 1;
3043
20.1M
                }
3044
3045
22.5M
                if ((!eof) && (avail <= 64))
3046
4.27k
                    continue;
3047
22.4M
                if (avail == 0)
3048
496
                    continue;
3049
22.4M
            }
3050
3051
22.4M
            cur = *in;
3052
22.4M
            size = 1;
3053
22.4M
            col += 1;
3054
3055
22.4M
            if (ncr) {
3056
1.01k
                int lc = cur | 0x20;
3057
1.01k
                int digit;
3058
3059
1.01k
                if ((cur >= '0') && (cur <= '9')) {
3060
835
                    digit = cur - '0';
3061
835
                } else if ((ncr == 16) && (lc >= 'a') && (lc <= 'f')) {
3062
42
                    digit = (lc - 'a') + 10;
3063
137
                } else {
3064
137
                    if (cur == ';') {
3065
106
                        in += 1;
3066
106
                        size += 1;
3067
106
                        ncrSize += 1;
3068
106
                    }
3069
137
                    goto next_chunk;
3070
137
                }
3071
3072
877
                cp = cp * ncr + digit;
3073
877
                if (cp >= 0x110000)
3074
445
                    cp = 0x110000;
3075
3076
877
                ncrSize += 1;
3077
3078
877
                goto next_char;
3079
1.01k
            }
3080
3081
22.4M
            switch (cur) {
3082
57.5k
            case '<':
3083
57.5k
                if (mode == 0) {
3084
57.5k
                    done = 1;
3085
57.5k
                    complete = 1;
3086
57.5k
                    goto next_chunk;
3087
57.5k
                }
3088
0
                if (mode == DATA_PLAINTEXT)
3089
0
                    break;
3090
3091
0
                j = 1;
3092
0
                if (j < avail) {
3093
0
                    if ((mode == DATA_SCRIPT) && (in[j] == '!')) {
3094
                        /* Check for comment start */
3095
3096
0
                        j += 1;
3097
0
                        if ((j < avail) && (in[j] == '-')) {
3098
0
                            j += 1;
3099
0
                            if ((j < avail) && (in[j] == '-'))
3100
0
                                mode = DATA_SCRIPT_ESC1;
3101
0
                        }
3102
0
                    } else {
3103
0
                        int i = 0;
3104
0
                        int solidus = 0;
3105
3106
                        /* Check for tag */
3107
3108
0
                        if (in[j] == '/') {
3109
0
                            j += 1;
3110
0
                            solidus = 1;
3111
0
                        }
3112
3113
0
                        if ((solidus) || (mode == DATA_SCRIPT_ESC1)) {
3114
0
                            while ((j < avail) &&
3115
0
                                   (ctxt->name[i] != 0) &&
3116
0
                                   (ctxt->name[i] == (in[j] | 0x20))) {
3117
0
                                i += 1;
3118
0
                                j += 1;
3119
0
                            }
3120
3121
0
                            if ((ctxt->name[i] == 0) && (j < avail)) {
3122
0
                                int c = in[j];
3123
3124
0
                                if ((c == '>') || (c == '/') ||
3125
0
                                    (IS_WS_HTML(c))) {
3126
0
                                    if ((mode == DATA_SCRIPT_ESC1) &&
3127
0
                                        (!solidus)) {
3128
0
                                        mode = DATA_SCRIPT_ESC2;
3129
0
                                    } else if (mode == DATA_SCRIPT_ESC2) {
3130
0
                                        mode = DATA_SCRIPT_ESC1;
3131
0
                                    } else {
3132
0
                                        complete = 1;
3133
0
                                        done = 1;
3134
0
                                        goto next_chunk;
3135
0
                                    }
3136
0
                                }
3137
0
                            }
3138
0
                        }
3139
0
                    }
3140
0
                }
3141
3142
0
                if ((partial) && (j >= avail)) {
3143
0
                    done = 1;
3144
0
                    goto next_chunk;
3145
0
                }
3146
3147
0
                break;
3148
3149
189k
            case '-':
3150
189k
                if ((mode != DATA_SCRIPT_ESC1) && (mode != DATA_SCRIPT_ESC2))
3151
189k
                    break;
3152
3153
                /* Check for comment end */
3154
3155
0
                j = 1;
3156
0
                if ((j < avail) && (in[j] == '-')) {
3157
0
                    j += 1;
3158
0
                    if ((j < avail) && (in[j] == '>'))
3159
0
                        mode = DATA_SCRIPT;
3160
0
                }
3161
3162
0
                if ((partial) && (j >= avail)) {
3163
0
                    done = 1;
3164
0
                    goto next_chunk;
3165
0
                }
3166
3167
0
                break;
3168
3169
178k
            case '&':
3170
178k
                if ((mode != 0) && (mode != DATA_RCDATA))
3171
0
                    break;
3172
3173
178k
                j = 1;
3174
3175
178k
                if ((j < avail) && (in[j] == '#')) {
3176
238
                    j += 1;
3177
238
                    if (j < avail) {
3178
238
                        if ((in[j] | 0x20) == 'x') {
3179
108
                            j += 1;
3180
108
                            if ((j < avail) && (IS_HEX_DIGIT(in[j]))) {
3181
38
                                ncr = 16;
3182
38
                                size = 3;
3183
38
                                ncrSize = 3;
3184
38
                                cp = 0;
3185
38
                            }
3186
130
                        } else if (IS_ASCII_DIGIT(in[j])) {
3187
99
                            ncr = 10;
3188
99
                            size = 2;
3189
99
                            ncrSize = 2;
3190
99
                            cp = 0;
3191
99
                        }
3192
238
                    }
3193
177k
                } else {
3194
177k
                    if (partial) {
3195
0
                        int terminated = 0;
3196
0
                        size_t i;
3197
3198
                        /*
3199
                         * &CounterClockwiseContourIntegral; has 33 bytes.
3200
                         */
3201
0
                        for (i = 1; i < avail; i++) {
3202
0
                            if ((i >= 32) ||
3203
0
                                (!IS_ASCII_LETTER(in[i]) &&
3204
0
                                 ((i < 2) || !IS_ASCII_DIGIT(in[i])))) {
3205
0
                                terminated = 1;
3206
0
                                break;
3207
0
                            }
3208
0
                        }
3209
3210
0
                        if (!terminated) {
3211
0
                            done = 1;
3212
0
                            goto next_chunk;
3213
0
                        }
3214
0
                    }
3215
3216
177k
                    repl = htmlFindEntityPrefix(in + j,
3217
177k
                                                avail - j,
3218
177k
                                                /* isAttr */ 0,
3219
177k
                                                &skip, &replSize);
3220
177k
                    if (repl != NULL) {
3221
70
                        skip += 1;
3222
70
                        goto next_chunk;
3223
70
                    }
3224
3225
177k
                    skip = 0;
3226
177k
                }
3227
3228
178k
                if ((partial) && (j >= avail)) {
3229
0
                    done = 1;
3230
0
                    goto next_chunk;
3231
0
                }
3232
3233
178k
                break;
3234
3235
178k
            case '\0':
3236
42.3k
                skip = 1;
3237
3238
42.3k
                if (mode == 0) {
3239
                    /*
3240
                     * The HTML5 spec says that the tokenizer should
3241
                     * pass on U+0000 unmodified in normal data mode.
3242
                     * These characters should then be ignored in body
3243
                     * and other text, but should be replaced with
3244
                     * U+FFFD in foreign content.
3245
                     *
3246
                     * At least for now, we always strip U+0000 when
3247
                     * tokenizing.
3248
                     */
3249
42.3k
                    repl = BAD_CAST "";
3250
42.3k
                    replSize = 0;
3251
42.3k
                } else {
3252
0
                    repl = BAD_CAST "\xEF\xBF\xBD";
3253
0
                    replSize = 3;
3254
0
                }
3255
3256
42.3k
                goto next_chunk;
3257
3258
59.4k
            case '\n':
3259
59.4k
                line += 1;
3260
59.4k
                col = 1;
3261
59.4k
                break;
3262
3263
298k
            case '\r':
3264
298k
                if (partial && avail < 2) {
3265
0
                    done = 1;
3266
0
                    goto next_chunk;
3267
0
                }
3268
3269
298k
                skip = 1;
3270
298k
                if (in[1] != 0x0A) {
3271
298k
                    repl = BAD_CAST "\x0A";
3272
298k
                    replSize = 1;
3273
298k
                }
3274
298k
                goto next_chunk;
3275
3276
21.6M
            default:
3277
21.6M
                if (cur < 0x80)
3278
0
                    break;
3279
3280
21.6M
                if ((input->flags & XML_INPUT_HAS_ENCODING) == 0) {
3281
2.79k
                    xmlChar * guess;
3282
3283
2.79k
                    if (in > chunk)
3284
1.13k
                        goto next_chunk;
3285
3286
1.65k
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
3287
1.65k
                    guess = NULL;
3288
#else
3289
                    guess = htmlFindEncoding(ctxt);
3290
#endif
3291
1.65k
                    if (guess == NULL) {
3292
1.65k
                        xmlSwitchEncoding(ctxt,
3293
1.65k
                                XML_CHAR_ENCODING_WINDOWS_1252);
3294
1.65k
                    } else {
3295
0
                        xmlSwitchEncodingName(ctxt, (const char *) guess);
3296
0
                        xmlFree(guess);
3297
0
                    }
3298
1.65k
                    input->flags |= XML_INPUT_HAS_ENCODING;
3299
3300
1.65k
                    eof = PARSER_PROGRESSIVE(ctxt);
3301
1.65k
                    goto restart;
3302
2.79k
                }
3303
3304
21.6M
                size = htmlValidateUtf8(ctxt, in, avail, partial);
3305
3306
21.6M
                if ((partial) && (size == 0)) {
3307
0
                    done = 1;
3308
0
                    goto next_chunk;
3309
0
                }
3310
3311
21.6M
                if (size <= 0) {
3312
3.73M
                    skip = 1;
3313
3.73M
                    repl = BAD_CAST "\xEF\xBF\xBD";
3314
3.73M
                    replSize = 3;
3315
3.73M
                    goto next_chunk;
3316
3.73M
                }
3317
3318
17.9M
                break;
3319
22.4M
            }
3320
3321
18.3M
next_char:
3322
18.3M
            in += size;
3323
18.3M
            avail -= size;
3324
18.3M
        }
3325
3326
4.13M
next_chunk:
3327
4.13M
        if (ncrSize > 0) {
3328
137
            skip = ncrSize;
3329
137
            in -= ncrSize;
3330
3331
137
            repl = htmlCodePointToUtf8(cp, utf8Char, &replSize);
3332
137
        }
3333
3334
4.13M
        if (in > chunk) {
3335
2.60M
            input->cur += in - chunk;
3336
2.60M
            htmlCharDataSAXCallback(ctxt, chunk, in - chunk, mode);
3337
2.60M
        }
3338
3339
4.13M
        input->cur += skip;
3340
4.13M
        if (replSize > 0)
3341
4.03M
            htmlCharDataSAXCallback(ctxt, repl, replSize, mode);
3342
3343
4.13M
        SHRINK;
3344
3345
4.13M
        if (done)
3346
59.5k
            break;
3347
3348
4.08M
restart:
3349
4.08M
        ;
3350
4.08M
    }
3351
3352
59.6k
    input->line = line;
3353
59.6k
    input->col = col;
3354
3355
59.6k
    if (complete)
3356
57.5k
        ctxt->endCheckState = 0;
3357
2.04k
    else
3358
2.04k
        ctxt->endCheckState = mode;
3359
3360
59.6k
    return(complete);
3361
59.6k
}
3362
3363
/**
3364
 * Parse an HTML comment
3365
 *
3366
 * @param ctxt  an HTML parser context
3367
 * @param bogus  true if this is a bogus comment
3368
 */
3369
static void
3370
22.3k
htmlParseComment(htmlParserCtxtPtr ctxt, int bogus) {
3371
22.3k
    const xmlChar *comment = BAD_CAST "";
3372
22.3k
    xmlChar *buf = NULL;
3373
22.3k
    int maxLength = (ctxt->options & HTML_PARSE_HUGE) ?
3374
152
                    XML_MAX_HUGE_LENGTH :
3375
22.3k
                    XML_MAX_TEXT_LENGTH;
3376
3377
22.3k
    if (bogus) {
3378
2.00k
        buf = htmlParseData(ctxt, MASK_GT, 0, 0, maxLength);
3379
2.00k
        if (CUR == '>')
3380
1.92k
            SKIP(1);
3381
2.00k
        comment = buf;
3382
20.3k
    } else {
3383
20.3k
        if ((!PARSER_PROGRESSIVE(ctxt)) &&
3384
20.3k
            (ctxt->input->end - ctxt->input->cur < 2))
3385
4
            xmlParserGrow(ctxt);
3386
3387
20.3k
        if (CUR == '>') {
3388
3
            SKIP(1);
3389
20.3k
        } else if ((CUR == '-') && (NXT(1) == '>')) {
3390
0
            SKIP(2);
3391
20.3k
        } else {
3392
20.3k
            buf = htmlParseData(ctxt, MASK_DASH, 1, 0, maxLength);
3393
20.3k
            comment = buf;
3394
20.3k
        }
3395
20.3k
    }
3396
3397
22.3k
    if (comment == NULL)
3398
2
        return;
3399
3400
22.3k
    if ((ctxt->sax != NULL) && (ctxt->sax->comment != NULL) &&
3401
22.3k
        (!ctxt->disableSAX))
3402
22.3k
        ctxt->sax->comment(ctxt->userData, comment);
3403
3404
22.3k
    xmlFree(buf);
3405
22.3k
}
3406
3407
/**
3408
 * @deprecated Internal function, don't use.
3409
 *
3410
 * @param ctxt  an HTML parser context
3411
 * @returns 0
3412
 */
3413
int
3414
0
htmlParseCharRef(htmlParserCtxt *ctxt ATTRIBUTE_UNUSED) {
3415
0
    return(0);
3416
0
}
3417
3418
3419
/**
3420
 * Parse a DOCTYPE SYTSTEM or PUBLIC literal.
3421
 *
3422
 * @param ctxt  an HTML parser context
3423
 * @returns the literal or NULL in case of error.
3424
 */
3425
3426
static xmlChar *
3427
435
htmlParseDoctypeLiteral(htmlParserCtxtPtr ctxt) {
3428
435
    xmlChar *ret;
3429
435
    int maxLength = (ctxt->options & HTML_PARSE_HUGE) ?
3430
172
                    XML_MAX_TEXT_LENGTH :
3431
435
                    XML_MAX_NAME_LENGTH;
3432
3433
435
    if (CUR == '"') {
3434
5
        SKIP(1);
3435
5
        ret = htmlParseData(ctxt, MASK_DQ_GT, 0, 0, maxLength);
3436
5
        if (CUR == '"')
3437
0
            SKIP(1);
3438
430
    } else if (CUR == '\'') {
3439
416
        SKIP(1);
3440
416
        ret = htmlParseData(ctxt, MASK_SQ_GT, 0, 0, maxLength);
3441
416
        if (CUR == '\'')
3442
249
            SKIP(1);
3443
416
    } else {
3444
14
        return(NULL);
3445
14
    }
3446
3447
421
    return(ret);
3448
435
}
3449
3450
static void
3451
386
htmlSkipBogusDoctype(htmlParserCtxtPtr ctxt) {
3452
386
    const xmlChar *in;
3453
386
    size_t avail;
3454
386
    int eof = PARSER_PROGRESSIVE(ctxt);
3455
386
    int line, col;
3456
3457
386
    line = ctxt->input->line;
3458
386
    col = ctxt->input->col;
3459
3460
386
    in = ctxt->input->cur;
3461
386
    avail = ctxt->input->end - in;
3462
3463
248k
    while (!PARSER_STOPPED(ctxt)) {
3464
248k
        int cur;
3465
3466
248k
        if ((!eof) && (avail <= 64)) {
3467
194
            size_t oldAvail = avail;
3468
3469
194
            ctxt->input->cur = in;
3470
3471
194
            xmlParserGrow(ctxt);
3472
3473
194
            in = ctxt->input->cur;
3474
194
            avail = ctxt->input->end - in;
3475
3476
194
            if (oldAvail == avail)
3477
179
                eof = 1;
3478
194
        }
3479
3480
248k
        if (avail == 0)
3481
25
            break;
3482
3483
248k
        col += 1;
3484
3485
248k
        cur = *in;
3486
248k
        if (cur == '>') {
3487
361
            in += 1;
3488
361
            break;
3489
248k
        } else if (cur == 0x0A) {
3490
54
            line += 1;
3491
54
            col = 1;
3492
54
        }
3493
3494
248k
        in += 1;
3495
248k
        avail -= 1;
3496
3497
248k
        SHRINK;
3498
248k
    }
3499
3500
386
    ctxt->input->cur = in;
3501
386
    ctxt->input->line = line;
3502
386
    ctxt->input->col = col;
3503
386
}
3504
3505
/**
3506
 * Parse a DOCTYPE declaration.
3507
 *
3508
 * @param ctxt  an HTML parser context
3509
 */
3510
3511
static void
3512
386
htmlParseDocTypeDecl(htmlParserCtxtPtr ctxt) {
3513
386
    xmlChar *name = NULL;
3514
386
    xmlChar *publicId = NULL;
3515
386
    xmlChar *URI = NULL;
3516
386
    int maxLength = (ctxt->options & HTML_PARSE_HUGE) ?
3517
96
                    XML_MAX_TEXT_LENGTH :
3518
386
                    XML_MAX_NAME_LENGTH;
3519
3520
    /*
3521
     * We know that '<!DOCTYPE' has been detected.
3522
     */
3523
386
    SKIP(9);
3524
3525
386
    SKIP_BLANKS;
3526
3527
386
    if ((ctxt->input->cur < ctxt->input->end) && (CUR != '>')) {
3528
381
        name = htmlParseData(ctxt, MASK_WS_GT, 0, 0, maxLength);
3529
3530
381
        if ((ctxt->options & HTML_PARSE_HTML5) && (name != NULL)) {
3531
0
            xmlChar *cur;
3532
3533
0
            for (cur = name; *cur; cur++) {
3534
0
                if (IS_UPPER(*cur))
3535
0
                    *cur += 0x20;
3536
0
            }
3537
0
        }
3538
3539
381
        SKIP_BLANKS;
3540
381
    }
3541
3542
    /*
3543
     * Check for SystemID and publicId
3544
     */
3545
386
    if ((UPPER == 'P') && (UPP(1) == 'U') &&
3546
148
  (UPP(2) == 'B') && (UPP(3) == 'L') &&
3547
142
  (UPP(4) == 'I') && (UPP(5) == 'C')) {
3548
136
        SKIP(6);
3549
136
        SKIP_BLANKS;
3550
136
  publicId = htmlParseDoctypeLiteral(ctxt);
3551
136
  if (publicId == NULL)
3552
3
            goto bogus;
3553
133
        SKIP_BLANKS;
3554
133
  URI = htmlParseDoctypeLiteral(ctxt);
3555
250
    } else if ((UPPER == 'S') && (UPP(1) == 'Y') &&
3556
181
               (UPP(2) == 'S') && (UPP(3) == 'T') &&
3557
175
         (UPP(4) == 'E') && (UPP(5) == 'M')) {
3558
166
        SKIP(6);
3559
166
        SKIP_BLANKS;
3560
166
  URI = htmlParseDoctypeLiteral(ctxt);
3561
166
    }
3562
3563
386
bogus:
3564
386
    htmlSkipBogusDoctype(ctxt);
3565
3566
    /*
3567
     * Create or update the document accordingly to the DOCTYPE
3568
     */
3569
386
    if ((ctxt->sax != NULL) && (ctxt->sax->internalSubset != NULL) &&
3570
386
  (!ctxt->disableSAX))
3571
386
  ctxt->sax->internalSubset(ctxt->userData, name, publicId, URI);
3572
3573
386
    xmlFree(name);
3574
386
    xmlFree(URI);
3575
386
    xmlFree(publicId);
3576
386
}
3577
3578
/**
3579
 * parse an attribute
3580
 *
3581
 * [41] Attribute ::= Name Eq AttValue
3582
 *
3583
 * [25] Eq ::= S? '=' S?
3584
 *
3585
 * With namespace:
3586
 *
3587
 * [NS 11] Attribute ::= QName Eq AttValue
3588
 *
3589
 * Also the case QName == xmlns:??? is handled independently as a namespace
3590
 * definition.
3591
 *
3592
 * @param ctxt  an HTML parser context
3593
 * @param value  a xmlChar ** used to store the value of the attribute
3594
 * @returns the attribute name, and the value in *value.
3595
 */
3596
3597
static xmlHashedString
3598
68.4k
htmlParseAttribute(htmlParserCtxtPtr ctxt, xmlChar **value) {
3599
68.4k
    xmlHashedString hname;
3600
68.4k
    xmlChar *val = NULL;
3601
3602
68.4k
    *value = NULL;
3603
68.4k
    hname = htmlParseHTMLName(ctxt, 1);
3604
68.4k
    if (hname.name == NULL)
3605
0
        return(hname);
3606
3607
    /*
3608
     * read the value
3609
     */
3610
68.4k
    SKIP_BLANKS;
3611
68.4k
    if (CUR == '=') {
3612
7.40k
        SKIP(1);
3613
7.40k
  SKIP_BLANKS;
3614
7.40k
  val = htmlParseAttValue(ctxt);
3615
7.40k
    }
3616
3617
68.4k
    *value = val;
3618
68.4k
    return(hname);
3619
68.4k
}
3620
3621
static int
3622
htmlCharEncCheckAsciiCompatible(htmlParserCtxt *ctxt,
3623
0
                                const xmlChar *encoding) {
3624
0
    xmlCharEncodingHandler *handler;
3625
0
    xmlChar in[9] = "<a A=\"/>";
3626
0
    xmlChar out[9];
3627
0
    int inlen, outlen;
3628
0
    int res;
3629
3630
0
    res = xmlCreateCharEncodingHandler(
3631
0
            (const char *) encoding,
3632
0
            XML_ENC_INPUT | XML_ENC_HTML,
3633
0
            ctxt->convImpl, ctxt->convCtxt,
3634
0
            &handler);
3635
0
    if (res != XML_ERR_OK) {
3636
0
        xmlFatalErr(ctxt, res, (const char *) encoding);
3637
0
        return(-1);
3638
0
    }
3639
3640
    /* UTF-8 */
3641
0
    if (handler == NULL)
3642
0
        return(0);
3643
3644
0
    inlen = 8;
3645
0
    outlen = 8;
3646
0
    res = xmlEncInputChunk(handler, out, &outlen, in, &inlen, /* flush */ 1);
3647
3648
0
    xmlCharEncCloseFunc(handler);
3649
3650
0
    if ((res != XML_ENC_ERR_SUCCESS) ||
3651
0
        (inlen != 8) || (outlen != 8) ||
3652
0
        (memcmp(in, out, 8) != 0)) {
3653
0
        htmlParseErr(ctxt, XML_ERR_UNSUPPORTED_ENCODING,
3654
0
                     "Encoding %s isn't ASCII-compatible", encoding, NULL);
3655
0
        return(-1);
3656
0
    }
3657
3658
0
    return(0);
3659
0
}
3660
3661
/**
3662
 * Handle charset encoding in meta tag.
3663
 *
3664
 * @param ctxt  an HTML parser context
3665
 * @param atts  the attributes values
3666
 */
3667
static void
3668
0
htmlCheckMeta(htmlParserCtxtPtr ctxt, const xmlChar **atts) {
3669
0
    int i;
3670
0
    const xmlChar *att, *value;
3671
0
    int isContentType = 0;
3672
0
    const xmlChar *content = NULL;
3673
0
    xmlChar *encoding = NULL;
3674
3675
0
    if ((ctxt == NULL) || (atts == NULL))
3676
0
  return;
3677
3678
0
    i = 0;
3679
0
    att = atts[i++];
3680
0
    while (att != NULL) {
3681
0
  value = atts[i++];
3682
0
        if (value != NULL) {
3683
0
            if ((!xmlStrcasecmp(att, BAD_CAST "http-equiv")) &&
3684
0
                (!xmlStrcasecmp(value, BAD_CAST "Content-Type"))) {
3685
0
                isContentType = 1;
3686
0
            } else if (!xmlStrcasecmp(att, BAD_CAST "charset")) {
3687
0
                encoding = xmlStrdup(value);
3688
0
                if (encoding == NULL)
3689
0
                    htmlErrMemory(ctxt);
3690
0
                break;
3691
0
            } else if (!xmlStrcasecmp(att, BAD_CAST "content")) {
3692
0
                content = value;
3693
0
            }
3694
0
        }
3695
0
  att = atts[i++];
3696
0
    }
3697
3698
0
    if ((encoding == NULL) && (isContentType) && (content != NULL)) {
3699
0
        htmlMetaEncodingOffsets off;
3700
3701
0
        if (htmlParseContentType(content, &off)) {
3702
0
            encoding = xmlStrndup(content + off.start, off.end - off.start);
3703
0
            if (encoding == NULL)
3704
0
                htmlErrMemory(ctxt);
3705
0
        }
3706
0
    }
3707
3708
0
    if (encoding != NULL) {
3709
0
        if (htmlCharEncCheckAsciiCompatible(ctxt, encoding) < 0) {
3710
0
            xmlFree(encoding);
3711
0
            return;
3712
0
        }
3713
3714
0
        xmlSetDeclaredEncoding(ctxt, encoding);
3715
0
    }
3716
0
}
3717
3718
/**
3719
 * Inserts a new attribute into the hash table.
3720
 *
3721
 * @param ctxt  parser context
3722
 * @param size  size of the hash table
3723
 * @param name  attribute name
3724
 * @param hashValue  hash value of name
3725
 * @param aindex  attribute index (this is a multiple of 5)
3726
 * @returns INT_MAX if no existing attribute was found, the attribute
3727
 * index if an attribute was found, -1 if a memory allocation failed.
3728
 */
3729
static int
3730
htmlAttrHashInsert(xmlParserCtxtPtr ctxt, unsigned size, const xmlChar *name,
3731
55.6k
                   unsigned hashValue, int aindex) {
3732
55.6k
    xmlAttrHashBucket *table = ctxt->attrHash;
3733
55.6k
    xmlAttrHashBucket *bucket;
3734
55.6k
    unsigned hindex;
3735
3736
55.6k
    hindex = hashValue & (size - 1);
3737
55.6k
    bucket = &table[hindex];
3738
3739
69.5k
    while (bucket->index >= 0) {
3740
26.5k
        const xmlChar **atts = &ctxt->atts[bucket->index];
3741
3742
26.5k
        if (name == atts[0])
3743
12.5k
            return(bucket->index);
3744
3745
13.9k
        hindex++;
3746
13.9k
        bucket++;
3747
13.9k
        if (hindex >= size) {
3748
1.14k
            hindex = 0;
3749
1.14k
            bucket = table;
3750
1.14k
        }
3751
13.9k
    }
3752
3753
43.0k
    bucket->index = aindex;
3754
3755
43.0k
    return(INT_MAX);
3756
55.6k
}
3757
3758
/**
3759
 * parse a start of tag either for rule element or
3760
 * EmptyElement. In both case we don't parse the tag closing chars.
3761
 *
3762
 * [40] STag ::= '<' Name (S Attribute)* S? '>'
3763
 *
3764
 * [44] EmptyElemTag ::= '<' Name (S Attribute)* S? '/>'
3765
 *
3766
 * With namespace:
3767
 *
3768
 * [NS 8] STag ::= '<' QName (S Attribute)* S? '>'
3769
 *
3770
 * [NS 10] EmptyElement ::= '<' QName (S Attribute)* S? '/>'
3771
 *
3772
 * @param ctxt  an HTML parser context
3773
 * @returns 0 in case of success, -1 in case of error and 1 if discarded
3774
 */
3775
3776
static void
3777
12.9k
htmlParseStartTag(htmlParserCtxtPtr ctxt) {
3778
12.9k
    const xmlChar *name;
3779
12.9k
    const xmlChar *attname;
3780
12.9k
    xmlChar *attvalue;
3781
12.9k
    const xmlChar **atts;
3782
12.9k
    int nbatts = 0;
3783
12.9k
    int maxatts;
3784
12.9k
    int i;
3785
12.9k
    int discardtag = 0;
3786
3787
12.9k
    ctxt->endCheckState = 0;
3788
3789
12.9k
    SKIP(1);
3790
3791
12.9k
    atts = ctxt->atts;
3792
12.9k
    maxatts = ctxt->maxatts;
3793
3794
12.9k
    GROW;
3795
12.9k
    name = htmlParseHTMLName(ctxt, 0).name;
3796
12.9k
    if (name == NULL)
3797
2
        return;
3798
3799
12.9k
    if ((ctxt->options & HTML_PARSE_HTML5) == 0) {
3800
        /*
3801
         * Check for auto-closure of HTML elements.
3802
         */
3803
12.9k
        htmlAutoClose(ctxt, name);
3804
3805
        /*
3806
         * Check for implied HTML elements.
3807
         */
3808
12.9k
        htmlCheckImplied(ctxt, name);
3809
3810
        /*
3811
         * Avoid html at any level > 0, head at any level != 1
3812
         * or any attempt to recurse body
3813
         */
3814
12.9k
        if ((ctxt->nameNr > 0) && (xmlStrEqual(name, BAD_CAST"html"))) {
3815
0
            htmlParseErr(ctxt, XML_HTML_STRUCURE_ERROR,
3816
0
                         "htmlParseStartTag: misplaced <html> tag\n",
3817
0
                         name, NULL);
3818
0
            discardtag = 1;
3819
0
            ctxt->depth++;
3820
0
        }
3821
12.9k
        if ((ctxt->nameNr != 1) &&
3822
12.9k
            (xmlStrEqual(name, BAD_CAST"head"))) {
3823
0
            htmlParseErr(ctxt, XML_HTML_STRUCURE_ERROR,
3824
0
                         "htmlParseStartTag: misplaced <head> tag\n",
3825
0
                         name, NULL);
3826
0
            discardtag = 1;
3827
0
            ctxt->depth++;
3828
0
        }
3829
12.9k
        if (xmlStrEqual(name, BAD_CAST"body")) {
3830
4
            int indx;
3831
26
            for (indx = 0;indx < ctxt->nameNr;indx++) {
3832
22
                if (xmlStrEqual(ctxt->nameTab[indx], BAD_CAST"body")) {
3833
4
                    htmlParseErr(ctxt, XML_HTML_STRUCURE_ERROR,
3834
4
                                 "htmlParseStartTag: misplaced <body> tag\n",
3835
4
                                 name, NULL);
3836
4
                    discardtag = 1;
3837
4
                    ctxt->depth++;
3838
4
                }
3839
22
            }
3840
4
        }
3841
12.9k
    }
3842
3843
    /*
3844
     * Now parse the attributes, it ends up with the ending
3845
     *
3846
     * (S Attribute)* S?
3847
     */
3848
12.9k
    SKIP_BLANKS;
3849
100k
    while ((ctxt->input->cur < ctxt->input->end) &&
3850
100k
           (CUR != '>') &&
3851
87.5k
     ((CUR != '/') || (NXT(1) != '>')) &&
3852
87.3k
           (PARSER_STOPPED(ctxt) == 0)) {
3853
87.3k
        xmlHashedString hattname;
3854
3855
        /*  unexpected-solidus-in-tag */
3856
87.3k
        if (CUR == '/') {
3857
19.4k
            SKIP(1);
3858
19.4k
            SKIP_BLANKS;
3859
19.4k
            continue;
3860
19.4k
        }
3861
67.9k
  GROW;
3862
67.9k
  hattname = htmlParseAttribute(ctxt, &attvalue);
3863
67.9k
        attname = hattname.name;
3864
3865
67.9k
        if (attname != NULL) {
3866
      /*
3867
       * Add the pair to atts
3868
       */
3869
67.9k
      if (nbatts + 4 > maxatts) {
3870
2.60k
          const xmlChar **tmp;
3871
2.60k
                unsigned *utmp;
3872
2.60k
                int newSize;
3873
3874
2.60k
                newSize = xmlGrowCapacity(maxatts,
3875
2.60k
                                          sizeof(tmp[0]) * 2 + sizeof(utmp[0]),
3876
2.60k
                                          11, HTML_MAX_ATTRS);
3877
2.60k
    if (newSize < 0) {
3878
0
        htmlErrMemory(ctxt);
3879
0
        goto failed;
3880
0
    }
3881
2.60k
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
3882
2.60k
                if (newSize < 2)
3883
876
                    newSize = 2;
3884
2.60k
#endif
3885
2.60k
          tmp = xmlRealloc(atts, newSize * sizeof(tmp[0]) * 2);
3886
2.60k
    if (tmp == NULL) {
3887
2
        htmlErrMemory(ctxt);
3888
2
        goto failed;
3889
2
    }
3890
2.60k
                atts = tmp;
3891
2.60k
    ctxt->atts = tmp;
3892
3893
2.60k
          utmp = xmlRealloc(ctxt->attallocs, newSize * sizeof(utmp[0]));
3894
2.60k
    if (utmp == NULL) {
3895
0
        htmlErrMemory(ctxt);
3896
0
        goto failed;
3897
0
    }
3898
2.60k
                ctxt->attallocs = utmp;
3899
3900
2.60k
                maxatts = newSize * 2;
3901
2.60k
    ctxt->maxatts = maxatts;
3902
2.60k
      }
3903
3904
67.9k
            ctxt->attallocs[nbatts/2] = hattname.hashValue;
3905
67.9k
      atts[nbatts++] = attname;
3906
67.9k
      atts[nbatts++] = attvalue;
3907
3908
67.9k
            attvalue = NULL;
3909
67.9k
  }
3910
3911
67.9k
failed:
3912
67.9k
        if (attvalue != NULL)
3913
0
            xmlFree(attvalue);
3914
3915
67.9k
  SKIP_BLANKS;
3916
67.9k
    }
3917
3918
12.9k
    if (ctxt->input->cur >= ctxt->input->end) {
3919
172
        discardtag = 1;
3920
172
        goto done;
3921
172
    }
3922
3923
    /*
3924
     * Verify that attribute names are unique.
3925
     */
3926
12.7k
    if (nbatts > 2) {
3927
5.86k
        unsigned attrHashSize;
3928
5.86k
        int j, k;
3929
3930
5.86k
        attrHashSize = 4;
3931
17.2k
        while (attrHashSize / 2 < (unsigned) nbatts / 2)
3932
11.4k
            attrHashSize *= 2;
3933
3934
5.86k
        if (attrHashSize > ctxt->attrHashMax) {
3935
843
            xmlAttrHashBucket *tmp;
3936
3937
843
            tmp = xmlRealloc(ctxt->attrHash, attrHashSize * sizeof(tmp[0]));
3938
843
            if (tmp == NULL) {
3939
0
                htmlErrMemory(ctxt);
3940
0
                goto done;
3941
0
            }
3942
3943
843
            ctxt->attrHash = tmp;
3944
843
            ctxt->attrHashMax = attrHashSize;
3945
843
        }
3946
3947
5.86k
        memset(ctxt->attrHash, -1, attrHashSize * sizeof(ctxt->attrHash[0]));
3948
3949
61.4k
        for (i = 0, j = 0, k = 0; i < nbatts; i += 2, k++) {
3950
55.6k
            unsigned hashValue;
3951
55.6k
            int res;
3952
3953
55.6k
            attname = atts[i];
3954
55.6k
            hashValue = ctxt->attallocs[k] | 0x80000000;
3955
3956
55.6k
            res = htmlAttrHashInsert(ctxt, attrHashSize, attname,
3957
55.6k
                                    hashValue, j);
3958
55.6k
            if (res < 0)
3959
0
                continue;
3960
3961
55.6k
            if (res == INT_MAX) {
3962
43.0k
                atts[j] = atts[i];
3963
43.0k
                atts[j+1] = atts[i+1];
3964
43.0k
                j += 2;
3965
43.0k
            } else {
3966
12.5k
                xmlFree((xmlChar *) atts[i+1]);
3967
12.5k
            }
3968
55.6k
        }
3969
3970
5.86k
        nbatts = j;
3971
5.86k
    }
3972
3973
12.7k
    if (nbatts > 0) {
3974
6.56k
        atts[nbatts] = NULL;
3975
6.56k
        atts[nbatts + 1] = NULL;
3976
3977
    /*
3978
     * Apple's new libiconv is so broken that you routinely run into
3979
     * issues when fuzz testing (by accident with an uninstrumented
3980
     * libiconv). Here's a harmless (?) example:
3981
     *
3982
     * printf '>'             | iconv -f shift_jis -t utf-8 | hexdump -C
3983
     * printf '\xfc\x00\x00'  | iconv -f shift_jis -t utf-8 | hexdump -C
3984
     * printf '>\xfc\x00\x00' | iconv -f shift_jis -t utf-8 | hexdump -C
3985
     *
3986
     * The last command fails to detect the illegal sequence.
3987
     */
3988
6.56k
#if !defined(__APPLE__) || \
3989
6.56k
    !defined(FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION)
3990
        /*
3991
         * Handle specific association to the META tag
3992
         */
3993
6.56k
        if (((ctxt->input->flags & XML_INPUT_HAS_ENCODING) == 0) &&
3994
104
            (strcmp((char *) name, "meta") == 0)) {
3995
0
            htmlCheckMeta(ctxt, atts);
3996
0
        }
3997
6.56k
#endif
3998
6.56k
    }
3999
4000
    /*
4001
     * SAX: Start of Element !
4002
     */
4003
12.7k
    if (!discardtag) {
4004
12.7k
        if (ctxt->options & HTML_PARSE_HTML5) {
4005
0
            if (ctxt->nameNr > 0)
4006
0
                htmlnamePop(ctxt);
4007
0
        }
4008
4009
12.7k
  htmlnamePush(ctxt, name);
4010
12.7k
  if ((ctxt->sax != NULL) && (ctxt->sax->startElement != NULL)) {
4011
12.7k
      if (nbatts != 0)
4012
6.56k
    ctxt->sax->startElement(ctxt->userData, name, atts);
4013
6.16k
      else
4014
6.16k
    ctxt->sax->startElement(ctxt->userData, name, NULL);
4015
12.7k
  }
4016
12.7k
    }
4017
4018
12.9k
done:
4019
12.9k
    if (atts != NULL) {
4020
64.1k
        for (i = 1;i < nbatts;i += 2) {
4021
55.3k
      if (atts[i] != NULL)
4022
6.72k
    xmlFree((xmlChar *) atts[i]);
4023
55.3k
  }
4024
8.84k
    }
4025
12.9k
}
4026
4027
/**
4028
 * parse an end of tag
4029
 *
4030
 * [42] ETag ::= '</' Name S? '>'
4031
 *
4032
 * With namespace
4033
 *
4034
 * [NS 9] ETag ::= '</' QName S? '>'
4035
 *
4036
 * @param ctxt  an HTML parser context
4037
 * @returns 1 if the current level should be closed.
4038
 */
4039
4040
static void
4041
htmlParseEndTag(htmlParserCtxtPtr ctxt)
4042
4.96k
{
4043
4.96k
    const xmlChar *name;
4044
4.96k
    const xmlChar *oldname;
4045
4.96k
    int i;
4046
4047
4.96k
    ctxt->endCheckState = 0;
4048
4049
4.96k
    SKIP(2);
4050
4051
4.96k
    if (ctxt->input->cur >= ctxt->input->end) {
4052
1
        htmlStartCharData(ctxt);
4053
1
        if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
4054
1
            (ctxt->sax->characters != NULL))
4055
1
            ctxt->sax->characters(ctxt->userData,
4056
1
                                  BAD_CAST "</", 2);
4057
1
        return;
4058
1
    }
4059
4060
4.96k
    if (CUR == '>') {
4061
3
        SKIP(1);
4062
3
        return;
4063
3
    }
4064
4065
4.96k
    if (!IS_ASCII_LETTER(CUR)) {
4066
29
        htmlParseComment(ctxt, /* bogus */ 1);
4067
29
        return;
4068
29
    }
4069
4070
4.93k
    name = htmlParseHTMLName(ctxt, 0).name;
4071
4.93k
    if (name == NULL)
4072
4
        return;
4073
4074
    /*
4075
     * Parse and ignore attributes.
4076
     */
4077
4.92k
    SKIP_BLANKS;
4078
6.23k
    while ((ctxt->input->cur < ctxt->input->end) &&
4079
6.19k
           (CUR != '>') &&
4080
1.30k
     ((CUR != '/') || (NXT(1) != '>')) &&
4081
1.30k
           (ctxt->instate != XML_PARSER_EOF)) {
4082
1.30k
        xmlChar *attvalue = NULL;
4083
4084
        /*  unexpected-solidus-in-tag */
4085
1.30k
        if (CUR == '/') {
4086
716
            SKIP(1);
4087
716
            SKIP_BLANKS;
4088
716
            continue;
4089
716
        }
4090
586
  GROW;
4091
586
  htmlParseAttribute(ctxt, &attvalue);
4092
586
        if (attvalue != NULL)
4093
15
            xmlFree(attvalue);
4094
4095
586
  SKIP_BLANKS;
4096
586
    }
4097
4098
4.92k
    if (CUR == '>') {
4099
4.88k
        SKIP(1);
4100
4.88k
    } else if ((CUR == '/') && (NXT(1) == '>')) {
4101
3
        SKIP(2);
4102
37
    } else {
4103
37
        return;
4104
37
    }
4105
4106
4.89k
    if (ctxt->options & HTML_PARSE_HTML5) {
4107
0
        if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
4108
0
            ctxt->sax->endElement(ctxt->userData, name);
4109
0
        return;
4110
0
    }
4111
4112
    /*
4113
     * if we ignored misplaced tags in htmlParseStartTag don't pop them
4114
     * out now.
4115
     */
4116
4.89k
    if ((ctxt->depth > 0) &&
4117
0
        (xmlStrEqual(name, BAD_CAST "html") ||
4118
0
         xmlStrEqual(name, BAD_CAST "body") ||
4119
0
   xmlStrEqual(name, BAD_CAST "head"))) {
4120
0
  ctxt->depth--;
4121
0
  return;
4122
0
    }
4123
4124
    /*
4125
     * If the name read is not one of the element in the parsing stack
4126
     * then return, it's just an error.
4127
     */
4128
9.33k
    for (i = (ctxt->nameNr - 1); i >= 0; i--) {
4129
7.18k
        if (xmlStrEqual(name, ctxt->nameTab[i]))
4130
2.74k
            break;
4131
7.18k
    }
4132
4.89k
    if (i < 0) {
4133
2.14k
        htmlParseErr(ctxt, XML_ERR_TAG_NAME_MISMATCH,
4134
2.14k
               "Unexpected end tag : %s\n", name, NULL);
4135
2.14k
        return;
4136
2.14k
    }
4137
4138
4139
    /*
4140
     * Check for auto-closure of HTML elements.
4141
     */
4142
4143
2.74k
    htmlAutoCloseOnClose(ctxt, name);
4144
4145
    /*
4146
     * Well formedness constraints, opening and closing must match.
4147
     * With the exception that the autoclose may have popped stuff out
4148
     * of the stack.
4149
     */
4150
2.74k
    if ((ctxt->name != NULL) && (!xmlStrEqual(ctxt->name, name))) {
4151
0
        htmlParseErr(ctxt, XML_ERR_TAG_NAME_MISMATCH,
4152
0
                     "Opening and ending tag mismatch: %s and %s\n",
4153
0
                     name, ctxt->name);
4154
0
    }
4155
4156
    /*
4157
     * SAX: End of Tag
4158
     */
4159
2.74k
    oldname = ctxt->name;
4160
2.74k
    if ((oldname != NULL) && (xmlStrEqual(oldname, name))) {
4161
2.74k
  htmlParserFinishElementParsing(ctxt);
4162
2.74k
        if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
4163
2.74k
            ctxt->sax->endElement(ctxt->userData, name);
4164
2.74k
        htmlnamePop(ctxt);
4165
2.74k
    }
4166
2.74k
}
4167
4168
/**
4169
 * Parse a content: comment, sub-element, reference or text.
4170
 * New version for non recursive htmlParseElementInternal
4171
 *
4172
 * @param ctxt  an HTML parser context
4173
 */
4174
4175
static void
4176
2.49k
htmlParseContent(htmlParserCtxtPtr ctxt) {
4177
2.49k
    GROW;
4178
4179
129k
    while ((PARSER_STOPPED(ctxt) == 0) &&
4180
129k
           (ctxt->input->cur < ctxt->input->end)) {
4181
127k
        int mode;
4182
4183
127k
        mode = ctxt->endCheckState;
4184
4185
127k
        if ((mode == 0) && (CUR == '<')) {
4186
67.7k
            if (NXT(1) == '/') {
4187
4.96k
          htmlParseEndTag(ctxt);
4188
62.7k
            } else if (NXT(1) == '!') {
4189
                /*
4190
                 * Sometimes DOCTYPE arrives in the middle of the document
4191
                 */
4192
20.8k
                if ((UPP(2) == 'D') && (UPP(3) == 'O') &&
4193
177
                    (UPP(4) == 'C') && (UPP(5) == 'T') &&
4194
161
                    (UPP(6) == 'Y') && (UPP(7) == 'P') &&
4195
155
                    (UPP(8) == 'E')) {
4196
146
                    htmlParseDocTypeDecl(ctxt);
4197
20.6k
                } else if ((NXT(2) == '-') && (NXT(3) == '-')) {
4198
20.2k
                    SKIP(4);
4199
20.2k
                    htmlParseComment(ctxt, /* bogus */ 0);
4200
20.2k
                } else {
4201
397
                    SKIP(2);
4202
397
                    htmlParseComment(ctxt, /* bogus */ 1);
4203
397
                }
4204
41.9k
            } else if (NXT(1) == '?') {
4205
1.42k
                SKIP(1);
4206
1.42k
                htmlParseComment(ctxt, /* bogus */ 1);
4207
40.5k
            } else if (IS_ASCII_LETTER(NXT(1))) {
4208
12.9k
                htmlParseElementInternal(ctxt);
4209
27.5k
            } else {
4210
27.5k
                htmlStartCharData(ctxt);
4211
27.5k
                if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
4212
27.5k
                    (ctxt->sax->characters != NULL))
4213
27.5k
                    ctxt->sax->characters(ctxt->userData, BAD_CAST "<", 1);
4214
27.5k
                SKIP(1);
4215
27.5k
            }
4216
67.7k
        } else {
4217
59.6k
            htmlParseCharData(ctxt, /* partial */ 0);
4218
59.6k
        }
4219
4220
127k
        SHRINK;
4221
127k
        GROW;
4222
127k
    }
4223
4224
2.49k
    if (ctxt->input->cur >= ctxt->input->end)
4225
2.40k
        htmlAutoCloseOnEnd(ctxt);
4226
2.49k
}
4227
4228
/**
4229
 * Parse an HTML element, new version, non recursive
4230
 *
4231
 * @param ctxt  an HTML parser context
4232
 */
4233
static int
4234
12.9k
htmlParseElementInternal(htmlParserCtxtPtr ctxt) {
4235
12.9k
    const xmlChar *name;
4236
12.9k
    const htmlElemDesc * info;
4237
12.9k
    htmlParserNodeInfo node_info = { NULL, 0, 0, 0, 0 };
4238
4239
12.9k
    if ((ctxt == NULL) || (ctxt->input == NULL))
4240
0
  return(0);
4241
4242
    /* Capture start position */
4243
12.9k
    if (ctxt->record_info) {
4244
0
        node_info.begin_pos = ctxt->input->consumed +
4245
0
                          (CUR_PTR - ctxt->input->base);
4246
0
  node_info.begin_line = ctxt->input->line;
4247
0
    }
4248
4249
12.9k
    htmlParseStartTag(ctxt);
4250
12.9k
    name = ctxt->name;
4251
12.9k
    if (name == NULL)
4252
2
        return(0);
4253
4254
12.9k
    if (ctxt->record_info)
4255
0
        htmlNodeInfoPush(ctxt, &node_info);
4256
4257
    /*
4258
     * Check for an Empty Element labeled the XML/SGML way
4259
     */
4260
12.9k
    if ((CUR == '/') && (NXT(1) == '>')) {
4261
184
        SKIP(2);
4262
184
        htmlParserFinishElementParsing(ctxt);
4263
184
        if ((ctxt->options & HTML_PARSE_HTML5) == 0) {
4264
184
            if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
4265
184
                ctxt->sax->endElement(ctxt->userData, name);
4266
184
        }
4267
184
  htmlnamePop(ctxt);
4268
184
  return(0);
4269
184
    }
4270
4271
12.7k
    if (CUR != '>')
4272
176
        return(0);
4273
12.5k
    SKIP(1);
4274
4275
    /*
4276
     * Lookup the info for that element.
4277
     */
4278
12.5k
    info = htmlTagLookup(name);
4279
4280
    /*
4281
     * Check for an Empty Element from DTD definition
4282
     */
4283
12.5k
    if ((info != NULL) && (info->empty)) {
4284
0
        htmlParserFinishElementParsing(ctxt);
4285
0
        if ((ctxt->options & HTML_PARSE_HTML5) == 0) {
4286
0
            if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL))
4287
0
                ctxt->sax->endElement(ctxt->userData, name);
4288
0
        }
4289
0
  htmlnamePop(ctxt);
4290
0
  return(0);
4291
0
    }
4292
4293
12.5k
    if (info != NULL)
4294
5.37k
        ctxt->endCheckState = info->dataMode;
4295
4296
12.5k
    return(1);
4297
12.5k
}
4298
4299
/**
4300
 * This is kept for compatibility with previous code versions
4301
 *
4302
 * @deprecated Internal function, don't use.
4303
 *
4304
 * @param ctxt  an HTML parser context
4305
 */
4306
void
4307
0
htmlParseElement(htmlParserCtxt *ctxt) {
4308
0
    const xmlChar *oldptr;
4309
0
    int depth;
4310
4311
0
    if ((ctxt == NULL) || (ctxt->input == NULL))
4312
0
  return;
4313
4314
0
    if (htmlParseElementInternal(ctxt) == 0)
4315
0
        return;
4316
4317
    /*
4318
     * Parse the content of the element:
4319
     */
4320
0
    depth = ctxt->nameNr;
4321
0
    while (CUR != 0) {
4322
0
  oldptr = ctxt->input->cur;
4323
0
  htmlParseContent(ctxt);
4324
0
  if (oldptr==ctxt->input->cur) break;
4325
0
  if (ctxt->nameNr < depth) break;
4326
0
    }
4327
4328
0
    if (CUR == 0) {
4329
0
  htmlAutoCloseOnEnd(ctxt);
4330
0
    }
4331
0
}
4332
4333
/**
4334
 * @param ctxt  parser context
4335
 * @param input  parser input
4336
 * @returns a node list.
4337
 */
4338
xmlNode *
4339
0
htmlCtxtParseContentInternal(htmlParserCtxt *ctxt, xmlParserInput *input) {
4340
0
    xmlNodePtr root;
4341
0
    xmlNodePtr list = NULL;
4342
0
    xmlChar *rootName = BAD_CAST "#root";
4343
4344
0
    root = xmlNewDocNode(ctxt->myDoc, NULL, rootName, NULL);
4345
0
    if (root == NULL) {
4346
0
        htmlErrMemory(ctxt);
4347
0
        return(NULL);
4348
0
    }
4349
4350
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
4351
0
        xmlFreeNode(root);
4352
0
        return(NULL);
4353
0
    }
4354
4355
0
    htmlnamePush(ctxt, rootName);
4356
0
    nodePush(ctxt, root);
4357
4358
0
    htmlParseContent(ctxt);
4359
4360
    /*
4361
     * Only check for truncated multi-byte sequences
4362
     */
4363
0
    xmlParserCheckEOF(ctxt, XML_ERR_INTERNAL_ERROR);
4364
4365
    /* TODO: Use xmlCtxtIsCatastrophicError */
4366
0
    if (ctxt->errNo != XML_ERR_NO_MEMORY) {
4367
0
        xmlNodePtr cur;
4368
4369
        /*
4370
         * Unlink newly created node list.
4371
         */
4372
0
        list = root->children;
4373
0
        root->children = NULL;
4374
0
        root->last = NULL;
4375
0
        for (cur = list; cur != NULL; cur = cur->next)
4376
0
            cur->parent = NULL;
4377
0
    }
4378
4379
0
    nodePop(ctxt);
4380
0
    htmlnamePop(ctxt);
4381
4382
0
    xmlCtxtPopInput(ctxt);
4383
4384
0
    xmlFreeNode(root);
4385
0
    return(list);
4386
0
}
4387
4388
/**
4389
 * Parse an HTML document and invoke the SAX handlers. This is useful
4390
 * if you're only interested in custom SAX callbacks. If you want a
4391
 * document tree, use #htmlCtxtParseDocument.
4392
 *
4393
 * @param ctxt  an HTML parser context
4394
 * @returns 0, -1 in case of error.
4395
 */
4396
int
4397
2.49k
htmlParseDocument(htmlParserCtxt *ctxt) {
4398
2.49k
    if ((ctxt == NULL) || (ctxt->input == NULL))
4399
0
  return(-1);
4400
4401
2.49k
    if ((ctxt->sax) && (ctxt->sax->setDocumentLocator)) {
4402
2.49k
        ctxt->sax->setDocumentLocator(ctxt->userData,
4403
2.49k
                (xmlSAXLocator *) &xmlDefaultSAXLocator);
4404
2.49k
    }
4405
4406
2.49k
    xmlDetectEncoding(ctxt);
4407
4408
    /*
4409
     * TODO: Implement HTML5 prescan algorithm
4410
     */
4411
4412
    /*
4413
     * This is wrong but matches long-standing behavior. In most
4414
     * cases, a document starting with an XML declaration will
4415
     * specify UTF-8. The HTML5 prescan algorithm handles
4416
     * XML declarations in a better way.
4417
     */
4418
2.49k
    if (((ctxt->input->flags & XML_INPUT_HAS_ENCODING) == 0) &&
4419
2.49k
        (xmlStrncmp(ctxt->input->cur, BAD_CAST "<?xm", 4) == 0))
4420
106
        xmlSwitchEncoding(ctxt, XML_CHAR_ENCODING_UTF8);
4421
4422
    /*
4423
     * Wipe out everything which is before the first '<'
4424
     */
4425
2.49k
    SKIP_BLANKS;
4426
4427
2.49k
    if ((ctxt->sax) && (ctxt->sax->startDocument) && (!ctxt->disableSAX))
4428
2.49k
  ctxt->sax->startDocument(ctxt->userData);
4429
4430
    /*
4431
     * Parse possible comments, PIs or doctype declarations
4432
     * before any content.
4433
     */
4434
2.49k
    ctxt->instate = XML_PARSER_MISC;
4435
2.92k
    while (CUR == '<') {
4436
603
        if (NXT(1) == '!') {
4437
320
            if ((NXT(2) == '-') && (NXT(3) == '-')) {
4438
31
                SKIP(4);
4439
31
                htmlParseComment(ctxt, /* bogus */ 0);
4440
289
            } else if ((UPP(2) == 'D') && (UPP(3) == 'O') &&
4441
255
                       (UPP(4) == 'C') && (UPP(5) == 'T') &&
4442
249
                       (UPP(6) == 'Y') && (UPP(7) == 'P') &&
4443
243
                       (UPP(8) == 'E')) {
4444
240
                htmlParseDocTypeDecl(ctxt);
4445
240
                ctxt->instate = XML_PARSER_PROLOG;
4446
240
            } else {
4447
49
                SKIP(2);
4448
49
                htmlParseComment(ctxt, /* bogus */ 1);
4449
49
            }
4450
320
        } else if (NXT(1) == '?') {
4451
106
            SKIP(1);
4452
106
            htmlParseComment(ctxt, /* bogus */ 1);
4453
177
        } else {
4454
177
            break;
4455
177
        }
4456
426
  SKIP_BLANKS;
4457
426
        GROW;
4458
426
    }
4459
4460
    /*
4461
     * Time to start parsing the tree itself
4462
     */
4463
2.49k
    ctxt->instate = XML_PARSER_CONTENT;
4464
2.49k
    htmlParseContent(ctxt);
4465
4466
    /*
4467
     * Only check for truncated multi-byte sequences
4468
     */
4469
2.49k
    xmlParserCheckEOF(ctxt, XML_ERR_INTERNAL_ERROR);
4470
4471
    /*
4472
     * SAX: end of the document processing.
4473
     */
4474
2.49k
    if ((ctxt->sax) && (ctxt->sax->endDocument != NULL))
4475
2.49k
        ctxt->sax->endDocument(ctxt->userData);
4476
4477
2.49k
    if (! ctxt->wellFormed) return(-1);
4478
2.38k
    return(0);
4479
2.49k
}
4480
4481
4482
/************************************************************************
4483
 *                  *
4484
 *      Parser contexts handling      *
4485
 *                  *
4486
 ************************************************************************/
4487
4488
/**
4489
 * Initialize a parser context
4490
 *
4491
 * @param ctxt  an HTML parser context
4492
 * @param sax  SAX handler
4493
 * @param userData  user data
4494
 * @returns 0 in case of success and -1 in case of error
4495
 */
4496
static int
4497
htmlInitParserCtxt(htmlParserCtxtPtr ctxt, const htmlSAXHandler *sax,
4498
                   void *userData)
4499
2.50k
{
4500
2.50k
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
4501
2.50k
    size_t initialNodeTabSize = 1;
4502
#else
4503
    size_t initialNodeTabSize = 10;
4504
#endif
4505
4506
2.50k
    if (ctxt == NULL) return(-1);
4507
2.50k
    memset(ctxt, 0, sizeof(htmlParserCtxt));
4508
4509
2.50k
    ctxt->dict = xmlDictCreate();
4510
2.50k
    if (ctxt->dict == NULL)
4511
1
  return(-1);
4512
4513
2.50k
    if (ctxt->sax == NULL)
4514
2.50k
        ctxt->sax = (htmlSAXHandler *) xmlMalloc(sizeof(htmlSAXHandler));
4515
2.50k
    if (ctxt->sax == NULL)
4516
0
  return(-1);
4517
2.50k
    if (sax == NULL) {
4518
2.50k
        memset(ctxt->sax, 0, sizeof(htmlSAXHandler));
4519
2.50k
        xmlSAX2InitHtmlDefaultSAXHandler(ctxt->sax);
4520
2.50k
        ctxt->userData = ctxt;
4521
2.50k
    } else {
4522
0
        memcpy(ctxt->sax, sax, sizeof(htmlSAXHandler));
4523
0
        ctxt->userData = userData ? userData : ctxt;
4524
0
    }
4525
4526
    /* Allocate the Input stack */
4527
2.50k
    ctxt->inputTab = (htmlParserInputPtr *)
4528
2.50k
                      xmlMalloc(sizeof(htmlParserInputPtr));
4529
2.50k
    if (ctxt->inputTab == NULL)
4530
0
  return(-1);
4531
2.50k
    ctxt->inputNr = 0;
4532
2.50k
    ctxt->inputMax = 1;
4533
2.50k
    ctxt->input = NULL;
4534
2.50k
    ctxt->version = NULL;
4535
2.50k
    ctxt->encoding = NULL;
4536
2.50k
    ctxt->standalone = -1;
4537
2.50k
    ctxt->instate = XML_PARSER_START;
4538
4539
    /* Allocate the Node stack */
4540
2.50k
    ctxt->nodeTab = xmlMalloc(initialNodeTabSize * sizeof(htmlNodePtr));
4541
2.50k
    if (ctxt->nodeTab == NULL)
4542
0
  return(-1);
4543
2.50k
    ctxt->nodeNr = 0;
4544
2.50k
    ctxt->nodeMax = initialNodeTabSize;
4545
2.50k
    ctxt->node = NULL;
4546
4547
    /* Allocate the Name stack */
4548
2.50k
    ctxt->nameTab = xmlMalloc(initialNodeTabSize * sizeof(xmlChar *));
4549
2.50k
    if (ctxt->nameTab == NULL)
4550
1
  return(-1);
4551
2.50k
    ctxt->nameNr = 0;
4552
2.50k
    ctxt->nameMax = initialNodeTabSize;
4553
2.50k
    ctxt->name = NULL;
4554
4555
2.50k
    ctxt->nodeInfoTab = NULL;
4556
2.50k
    ctxt->nodeInfoNr  = 0;
4557
2.50k
    ctxt->nodeInfoMax = 0;
4558
4559
2.50k
    ctxt->myDoc = NULL;
4560
2.50k
    ctxt->wellFormed = 1;
4561
2.50k
    ctxt->replaceEntities = 0;
4562
2.50k
    ctxt->keepBlanks = xmlKeepBlanksDefaultValue;
4563
2.50k
    ctxt->html = INSERT_INITIAL;
4564
2.50k
    ctxt->vctxt.flags = XML_VCTXT_USE_PCTXT;
4565
2.50k
    ctxt->vctxt.userData = ctxt;
4566
2.50k
    ctxt->vctxt.error = xmlParserValidityError;
4567
2.50k
    ctxt->vctxt.warning = xmlParserValidityWarning;
4568
2.50k
    ctxt->record_info = 0;
4569
2.50k
    ctxt->validate = 0;
4570
2.50k
    ctxt->checkIndex = 0;
4571
2.50k
    ctxt->catalogs = NULL;
4572
2.50k
    xmlInitNodeInfoSeq(&ctxt->node_seq);
4573
2.50k
    return(0);
4574
2.50k
}
4575
4576
/**
4577
 * Free all the memory used by a parser context. However the parsed
4578
 * document in `ctxt->myDoc` is not freed.
4579
 *
4580
 * @param ctxt  an HTML parser context
4581
 */
4582
void
4583
htmlFreeParserCtxt(htmlParserCtxt *ctxt)
4584
4
{
4585
4
    xmlFreeParserCtxt(ctxt);
4586
4
}
4587
4588
/**
4589
 * Allocate and initialize a new HTML parser context.
4590
 *
4591
 * This can be used to parse HTML documents into DOM trees with
4592
 * functions like #xmlCtxtReadFile or #xmlCtxtReadMemory.
4593
 *
4594
 * See #htmlCtxtUseOptions for parser options.
4595
 *
4596
 * See #xmlCtxtSetErrorHandler for advanced error handling.
4597
 *
4598
 * See #htmlNewSAXParserCtxt for custom SAX parsers.
4599
 *
4600
 * @returns the htmlParserCtxt or NULL in case of allocation error
4601
 */
4602
htmlParserCtxt *
4603
htmlNewParserCtxt(void)
4604
2.50k
{
4605
2.50k
    return(htmlNewSAXParserCtxt(NULL, NULL));
4606
2.50k
}
4607
4608
/**
4609
 * Allocate and initialize a new HTML SAX parser context. If `userData`
4610
 * is NULL, the parser context will be passed as user data.
4611
 *
4612
 * @since 2.11.0
4613
 *
4614
 * If you want support older versions, it's best to invoke
4615
 * #htmlNewParserCtxt and set `ctxt->sax` with struct assignment.
4616
 *
4617
 * Also see #htmlNewParserCtxt.
4618
 *
4619
 * @param sax  SAX handler
4620
 * @param userData  user data
4621
 * @returns the htmlParserCtxt or NULL in case of allocation error
4622
 */
4623
htmlParserCtxt *
4624
htmlNewSAXParserCtxt(const htmlSAXHandler *sax, void *userData)
4625
2.51k
{
4626
2.51k
    xmlParserCtxtPtr ctxt;
4627
4628
2.51k
    xmlInitParser();
4629
4630
2.51k
    ctxt = (xmlParserCtxtPtr) xmlMalloc(sizeof(xmlParserCtxt));
4631
2.51k
    if (ctxt == NULL)
4632
4
  return(NULL);
4633
2.50k
    memset(ctxt, 0, sizeof(xmlParserCtxt));
4634
2.50k
    if (htmlInitParserCtxt(ctxt, sax, userData) < 0) {
4635
2
        htmlFreeParserCtxt(ctxt);
4636
2
  return(NULL);
4637
2
    }
4638
2.50k
    return(ctxt);
4639
2.50k
}
4640
4641
static htmlParserCtxtPtr
4642
htmlCreateMemoryParserCtxtInternal(const char *url,
4643
                                   const char *buffer, size_t size,
4644
0
                                   const char *encoding) {
4645
0
    xmlParserCtxtPtr ctxt;
4646
0
    xmlParserInputPtr input;
4647
4648
0
    if (buffer == NULL)
4649
0
  return(NULL);
4650
4651
0
    ctxt = htmlNewParserCtxt();
4652
0
    if (ctxt == NULL)
4653
0
  return(NULL);
4654
4655
0
    input = xmlCtxtNewInputFromMemory(ctxt, url, buffer, size, encoding, 0);
4656
0
    if (input == NULL) {
4657
0
  xmlFreeParserCtxt(ctxt);
4658
0
        return(NULL);
4659
0
    }
4660
4661
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
4662
0
        xmlFreeInputStream(input);
4663
0
        xmlFreeParserCtxt(ctxt);
4664
0
        return(NULL);
4665
0
    }
4666
4667
0
    return(ctxt);
4668
0
}
4669
4670
/**
4671
 * Create a parser context for an HTML in-memory document. The input
4672
 * buffer must not contain any terminating null bytes.
4673
 *
4674
 * @deprecated Use #htmlNewParserCtxt and #htmlCtxtReadMemory.
4675
 *
4676
 * @param buffer  a pointer to a char array
4677
 * @param size  the size of the array
4678
 * @returns the new parser context or NULL
4679
 */
4680
htmlParserCtxt *
4681
0
htmlCreateMemoryParserCtxt(const char *buffer, int size) {
4682
0
    if (size <= 0)
4683
0
  return(NULL);
4684
4685
0
    return(htmlCreateMemoryParserCtxtInternal(NULL, buffer, size, NULL));
4686
0
}
4687
4688
/**
4689
 * Create a parser context for a null-terminated string.
4690
 *
4691
 * @param str  a pointer to an array of xmlChar
4692
 * @param url  URL of the document (optional)
4693
 * @param encoding  encoding (optional)
4694
 * @returns the new parser context or NULL if a memory allocation failed.
4695
 */
4696
static htmlParserCtxtPtr
4697
htmlCreateDocParserCtxt(const xmlChar *str, const char *url,
4698
0
                        const char *encoding) {
4699
0
    xmlParserCtxtPtr ctxt;
4700
0
    xmlParserInputPtr input;
4701
4702
0
    if (str == NULL)
4703
0
  return(NULL);
4704
4705
0
    ctxt = htmlNewParserCtxt();
4706
0
    if (ctxt == NULL)
4707
0
  return(NULL);
4708
4709
0
    input = xmlCtxtNewInputFromString(ctxt, url, (const char *) str,
4710
0
                                      encoding, 0);
4711
0
    if (input == NULL) {
4712
0
  xmlFreeParserCtxt(ctxt);
4713
0
  return(NULL);
4714
0
    }
4715
4716
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
4717
0
        xmlFreeInputStream(input);
4718
0
        xmlFreeParserCtxt(ctxt);
4719
0
        return(NULL);
4720
0
    }
4721
4722
0
    return(ctxt);
4723
0
}
4724
4725
#ifdef LIBXML_PUSH_ENABLED
4726
/************************************************************************
4727
 *                  *
4728
 *  Progressive parsing interfaces        *
4729
 *                  *
4730
 ************************************************************************/
4731
4732
typedef enum {
4733
    LSTATE_TAG_NAME = 0,
4734
    LSTATE_BEFORE_ATTR_NAME,
4735
    LSTATE_ATTR_NAME,
4736
    LSTATE_AFTER_ATTR_NAME,
4737
    LSTATE_BEFORE_ATTR_VALUE,
4738
    LSTATE_ATTR_VALUE_DQUOTED,
4739
    LSTATE_ATTR_VALUE_SQUOTED,
4740
    LSTATE_ATTR_VALUE_UNQUOTED
4741
} xmlLookupStates;
4742
4743
/**
4744
 * Check whether there's enough data in the input buffer to finish parsing
4745
 * a tag. This has to take quotes into account.
4746
 *
4747
 * @param ctxt  an HTML parser context
4748
 */
4749
static int
4750
0
htmlParseLookupGt(xmlParserCtxtPtr ctxt) {
4751
0
    const xmlChar *cur;
4752
0
    const xmlChar *end = ctxt->input->end;
4753
0
    int state = ctxt->endCheckState;
4754
0
    size_t index;
4755
4756
0
    if (ctxt->checkIndex == 0)
4757
0
        cur = ctxt->input->cur + 2; /* Skip '<a' or '</' */
4758
0
    else
4759
0
        cur = ctxt->input->cur + ctxt->checkIndex;
4760
4761
0
    while (cur < end) {
4762
0
        int c = *cur++;
4763
4764
0
        if (state != LSTATE_ATTR_VALUE_SQUOTED &&
4765
0
            state != LSTATE_ATTR_VALUE_DQUOTED) {
4766
0
            if (c == '/' &&
4767
0
                state != LSTATE_BEFORE_ATTR_VALUE &&
4768
0
                state != LSTATE_ATTR_VALUE_UNQUOTED) {
4769
0
                state = LSTATE_BEFORE_ATTR_NAME;
4770
0
                continue;
4771
0
            } else if (c == '>') {
4772
0
                ctxt->checkIndex = 0;
4773
0
                ctxt->endCheckState = 0;
4774
0
                return(0);
4775
0
            }
4776
0
        }
4777
4778
0
        switch (state) {
4779
0
            case LSTATE_TAG_NAME:
4780
0
                if (IS_WS_HTML(c))
4781
0
                    state = LSTATE_BEFORE_ATTR_NAME;
4782
0
                break;
4783
4784
0
            case LSTATE_BEFORE_ATTR_NAME:
4785
0
                if (!IS_WS_HTML(c))
4786
0
                    state = LSTATE_ATTR_NAME;
4787
0
                break;
4788
4789
0
            case LSTATE_ATTR_NAME:
4790
0
                if (c == '=')
4791
0
                    state = LSTATE_BEFORE_ATTR_VALUE;
4792
0
                else if (IS_WS_HTML(c))
4793
0
                    state = LSTATE_AFTER_ATTR_NAME;
4794
0
                break;
4795
4796
0
            case LSTATE_AFTER_ATTR_NAME:
4797
0
                if (c == '=')
4798
0
                    state = LSTATE_BEFORE_ATTR_VALUE;
4799
0
                else if (!IS_WS_HTML(c))
4800
0
                    state = LSTATE_ATTR_NAME;
4801
0
                break;
4802
4803
0
            case LSTATE_BEFORE_ATTR_VALUE:
4804
0
                if (c == '"')
4805
0
                    state = LSTATE_ATTR_VALUE_DQUOTED;
4806
0
                else if (c == '\'')
4807
0
                    state = LSTATE_ATTR_VALUE_SQUOTED;
4808
0
                else if (!IS_WS_HTML(c))
4809
0
                    state = LSTATE_ATTR_VALUE_UNQUOTED;
4810
0
                break;
4811
4812
0
            case LSTATE_ATTR_VALUE_DQUOTED:
4813
0
                if (c == '"')
4814
0
                    state = LSTATE_BEFORE_ATTR_NAME;
4815
0
                break;
4816
4817
0
            case LSTATE_ATTR_VALUE_SQUOTED:
4818
0
                if (c == '\'')
4819
0
                    state = LSTATE_BEFORE_ATTR_NAME;
4820
0
                break;
4821
4822
0
            case LSTATE_ATTR_VALUE_UNQUOTED:
4823
0
                if (IS_WS_HTML(c))
4824
0
                    state = LSTATE_BEFORE_ATTR_NAME;
4825
0
                break;
4826
0
        }
4827
0
    }
4828
4829
0
    index = cur - ctxt->input->cur;
4830
0
    if (index > LONG_MAX) {
4831
0
        ctxt->checkIndex = 0;
4832
0
        ctxt->endCheckState = 0;
4833
0
        return(0);
4834
0
    }
4835
0
    ctxt->checkIndex = index;
4836
0
    ctxt->endCheckState = state;
4837
0
    return(-1);
4838
0
}
4839
4840
/**
4841
 * Check whether the input buffer contains a string.
4842
 *
4843
 * @param ctxt  an XML parser context
4844
 * @param startDelta  delta to apply at the start
4845
 * @param str  string
4846
 * @param strLen  length of string
4847
 * @param extraLen  extra length
4848
 */
4849
static int
4850
htmlParseLookupString(xmlParserCtxtPtr ctxt, size_t startDelta,
4851
0
                      const char *str, size_t strLen, size_t extraLen) {
4852
0
    const xmlChar *end = ctxt->input->end;
4853
0
    const xmlChar *cur, *term;
4854
0
    size_t index, rescan;
4855
0
    int ret;
4856
4857
0
    if (ctxt->checkIndex == 0) {
4858
0
        cur = ctxt->input->cur + startDelta;
4859
0
    } else {
4860
0
        cur = ctxt->input->cur + ctxt->checkIndex;
4861
0
    }
4862
4863
0
    term = BAD_CAST strstr((const char *) cur, str);
4864
0
    if ((term != NULL) &&
4865
0
        ((size_t) (ctxt->input->end - term) >= extraLen + 1)) {
4866
0
        ctxt->checkIndex = 0;
4867
4868
0
        if (term - ctxt->input->cur > INT_MAX / 2)
4869
0
            ret = INT_MAX / 2;
4870
0
        else
4871
0
            ret = term - ctxt->input->cur;
4872
4873
0
        return(ret);
4874
0
    }
4875
4876
    /* Rescan (strLen + extraLen - 1) characters. */
4877
0
    rescan = strLen + extraLen - 1;
4878
0
    if ((size_t) (end - cur) <= rescan)
4879
0
        end = cur;
4880
0
    else
4881
0
        end -= rescan;
4882
0
    index = end - ctxt->input->cur;
4883
0
    if (index > INT_MAX / 2) {
4884
0
        ctxt->checkIndex = 0;
4885
0
        ret = INT_MAX / 2;
4886
0
    } else {
4887
0
        ctxt->checkIndex = index;
4888
0
        ret = -1;
4889
0
    }
4890
4891
0
    return(ret);
4892
0
}
4893
4894
/**
4895
 * Try to find a comment end tag in the input stream
4896
 * The search includes "-->" as well as WHATWG-recommended
4897
 * incorrectly-closed tags.
4898
 *
4899
 * @param ctxt  an HTML parser context
4900
 * @returns the index to the current parsing point if the full
4901
 * sequence is available, -1 otherwise.
4902
 */
4903
static int
4904
htmlParseLookupCommentEnd(htmlParserCtxtPtr ctxt)
4905
0
{
4906
0
    int mark = 0;
4907
0
    int offset;
4908
4909
0
    while (1) {
4910
0
  mark = htmlParseLookupString(ctxt, 2, "--", 2, 0);
4911
0
  if (mark < 0)
4912
0
            break;
4913
        /*
4914
         * <!-->    is a complete comment, but
4915
         * <!--!>   is not
4916
         * <!---!>  is not
4917
         * <!----!> is
4918
         */
4919
0
        if ((NXT(mark+2) == '>') ||
4920
0
      ((mark >= 4) && (NXT(mark+2) == '!') && (NXT(mark+3) == '>'))) {
4921
0
            ctxt->checkIndex = 0;
4922
0
      break;
4923
0
  }
4924
0
        offset = (NXT(mark+2) == '!') ? 3 : 2;
4925
0
        if (mark + offset >= ctxt->input->end - ctxt->input->cur) {
4926
0
      ctxt->checkIndex = mark;
4927
0
            return(-1);
4928
0
        }
4929
0
  ctxt->checkIndex = mark + 1;
4930
0
    }
4931
0
    return mark;
4932
0
}
4933
4934
4935
/**
4936
 * Try to progress on parsing
4937
 *
4938
 * @param ctxt  an HTML parser context
4939
 * @param terminate  last chunk indicator
4940
 * @returns zero if no parsing was possible
4941
 */
4942
static void
4943
3
htmlParseTryOrFinish(htmlParserCtxtPtr ctxt, int terminate) {
4944
6
    while (PARSER_STOPPED(ctxt) == 0) {
4945
6
        htmlParserInputPtr in;
4946
6
        size_t avail;
4947
4948
6
  in = ctxt->input;
4949
6
  if (in == NULL) break;
4950
6
  avail = in->end - in->cur;
4951
4952
6
        switch (ctxt->instate) {
4953
0
            case XML_PARSER_EOF:
4954
          /*
4955
     * Document parsing is done !
4956
     */
4957
0
          return;
4958
4959
3
            case XML_PARSER_START:
4960
                /*
4961
                 * Very first chars read from the document flow.
4962
                 */
4963
3
                if ((!terminate) && (avail < 4))
4964
0
                    return;
4965
4966
3
                xmlDetectEncoding(ctxt);
4967
4968
                /*
4969
                 * TODO: Implement HTML5 prescan algorithm
4970
                 */
4971
4972
                /*
4973
                 * This is wrong but matches long-standing behavior. In most
4974
                 * cases, a document starting with an XML declaration will
4975
                 * specify UTF-8. The HTML5 prescan algorithm handles
4976
                 * XML declarations in a better way.
4977
                 */
4978
3
                if (((ctxt->input->flags & XML_INPUT_HAS_ENCODING) == 0) &&
4979
3
                    (xmlStrncmp(ctxt->input->cur, BAD_CAST "<?xm", 4) == 0)) {
4980
0
                    xmlSwitchEncoding(ctxt, XML_CHAR_ENCODING_UTF8);
4981
0
                }
4982
4983
                /* fall through */
4984
4985
3
            case XML_PARSER_XML_DECL:
4986
3
                if ((ctxt->sax) && (ctxt->sax->setDocumentLocator)) {
4987
3
                    ctxt->sax->setDocumentLocator(ctxt->userData,
4988
3
                            (xmlSAXLocator *) &xmlDefaultSAXLocator);
4989
3
                }
4990
3
    if ((ctxt->sax) && (ctxt->sax->startDocument) &&
4991
3
              (!ctxt->disableSAX))
4992
3
        ctxt->sax->startDocument(ctxt->userData);
4993
4994
                /* Allow callback to modify state for tests */
4995
3
                if ((ctxt->instate == XML_PARSER_START) ||
4996
0
                    (ctxt->instate == XML_PARSER_XML_DECL))
4997
3
                    ctxt->instate = XML_PARSER_MISC;
4998
3
    break;
4999
5000
0
            case XML_PARSER_START_TAG:
5001
0
    if ((!terminate) &&
5002
0
        (htmlParseLookupGt(ctxt) < 0))
5003
0
        return;
5004
5005
0
                htmlParseElementInternal(ctxt);
5006
5007
0
    ctxt->instate = XML_PARSER_CONTENT;
5008
0
                break;
5009
5010
3
            case XML_PARSER_MISC: /* initial */
5011
3
            case XML_PARSER_PROLOG: /* before html */
5012
3
            case XML_PARSER_CONTENT: {
5013
3
                int mode;
5014
5015
3
                if ((ctxt->instate == XML_PARSER_MISC) ||
5016
3
                    (ctxt->instate == XML_PARSER_PROLOG)) {
5017
3
                    SKIP_BLANKS;
5018
3
                    avail = in->end - in->cur;
5019
3
                }
5020
5021
3
    if (avail < 1)
5022
3
        return;
5023
                /*
5024
                 * Note that endCheckState is also used by
5025
                 * xmlParseLookupGt.
5026
                 */
5027
0
                mode = ctxt->endCheckState;
5028
5029
0
                if (mode != 0) {
5030
0
                    if (htmlParseCharData(ctxt, !terminate) == 0)
5031
0
                        return;
5032
0
    } else if (in->cur[0] == '<') {
5033
0
                    int next;
5034
5035
0
                    if (avail < 2) {
5036
0
                        if (!terminate)
5037
0
                            return;
5038
0
                        next = ' ';
5039
0
                    } else {
5040
0
                        next = in->cur[1];
5041
0
                    }
5042
5043
0
                    if (next == '!') {
5044
0
                        if ((!terminate) && (avail < 4))
5045
0
                            return;
5046
0
                        if ((in->cur[2] == '-') && (in->cur[3] == '-')) {
5047
0
                            if ((!terminate) &&
5048
0
                                (htmlParseLookupCommentEnd(ctxt) < 0))
5049
0
                                return;
5050
0
                            SKIP(4);
5051
0
                            htmlParseComment(ctxt, /* bogus */ 0);
5052
                            /* don't change state */
5053
0
                            break;
5054
0
                        }
5055
5056
0
                        if ((!terminate) && (avail < 9))
5057
0
                            return;
5058
0
                        if ((UPP(2) == 'D') && (UPP(3) == 'O') &&
5059
0
                            (UPP(4) == 'C') && (UPP(5) == 'T') &&
5060
0
                            (UPP(6) == 'Y') && (UPP(7) == 'P') &&
5061
0
                            (UPP(8) == 'E')) {
5062
0
                            if ((!terminate) &&
5063
0
                                (htmlParseLookupString(ctxt, 9, ">", 1,
5064
0
                                                       0) < 0))
5065
0
                                return;
5066
0
                            htmlParseDocTypeDecl(ctxt);
5067
0
                            if (ctxt->instate == XML_PARSER_MISC)
5068
0
                                ctxt->instate = XML_PARSER_PROLOG;
5069
0
                        } else {
5070
0
                            if ((!terminate) &&
5071
0
                                (htmlParseLookupString(ctxt, 2, ">", 1, 0) < 0))
5072
0
                                return;
5073
0
                            SKIP(2);
5074
0
                            htmlParseComment(ctxt, /* bogus */ 1);
5075
0
                        }
5076
0
                    } else if (next == '?') {
5077
0
                        if ((!terminate) &&
5078
0
                            (htmlParseLookupString(ctxt, 2, ">", 1, 0) < 0))
5079
0
                            return;
5080
0
                        SKIP(1);
5081
0
                        htmlParseComment(ctxt, /* bogus */ 1);
5082
                        /* don't change state */
5083
0
                    } else if (next == '/') {
5084
0
                        ctxt->instate = XML_PARSER_END_TAG;
5085
0
                        ctxt->checkIndex = 0;
5086
0
                    } else if (IS_ASCII_LETTER(next)) {
5087
0
                        ctxt->instate = XML_PARSER_START_TAG;
5088
0
                        ctxt->checkIndex = 0;
5089
0
                    } else {
5090
0
                        ctxt->instate = XML_PARSER_CONTENT;
5091
0
                        htmlStartCharData(ctxt);
5092
0
                        if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
5093
0
                            (ctxt->sax->characters != NULL))
5094
0
                            ctxt->sax->characters(ctxt->userData,
5095
0
                                                  BAD_CAST "<", 1);
5096
0
                        SKIP(1);
5097
0
                    }
5098
0
                } else {
5099
0
                    ctxt->instate = XML_PARSER_CONTENT;
5100
                    /*
5101
                     * We follow the logic of the XML push parser
5102
                     */
5103
0
        if (avail < HTML_PARSER_BIG_BUFFER_SIZE) {
5104
0
                        if ((!terminate) &&
5105
0
                            (htmlParseLookupString(ctxt, 0, "<", 1, 0) < 0))
5106
0
                            return;
5107
0
                    }
5108
0
                    ctxt->checkIndex = 0;
5109
0
                    if (htmlParseCharData(ctxt, !terminate) == 0)
5110
0
                        return;
5111
0
    }
5112
5113
0
    break;
5114
0
      }
5115
5116
0
            case XML_PARSER_END_TAG:
5117
0
    if ((!terminate) &&
5118
0
        (htmlParseLookupGt(ctxt) < 0))
5119
0
        return;
5120
0
    htmlParseEndTag(ctxt);
5121
0
    ctxt->instate = XML_PARSER_CONTENT;
5122
0
    ctxt->checkIndex = 0;
5123
0
          break;
5124
5125
0
      default:
5126
0
    htmlParseErr(ctxt, XML_ERR_INTERNAL_ERROR,
5127
0
           "HPP: internal error\n", NULL, NULL);
5128
0
    ctxt->instate = XML_PARSER_EOF;
5129
0
    break;
5130
6
  }
5131
6
    }
5132
3
}
5133
5134
/**
5135
 * Parse a chunk of memory in push parser mode.
5136
 *
5137
 * Assumes that the parser context was initialized with
5138
 * #htmlCreatePushParserCtxt.
5139
 *
5140
 * The last chunk, which will often be empty, must be marked with
5141
 * the `terminate` flag. With the default SAX callbacks, the resulting
5142
 * document will be available in `ctxt->myDoc`. This pointer will not
5143
 * be freed by the library.
5144
 *
5145
 * If the document isn't well-formed, `ctxt->myDoc` is set to NULL.
5146
 *
5147
 * Since 2.14.0, #xmlCtxtGetDocument can be used to retrieve the
5148
 * result document.
5149
 *
5150
 * @param ctxt  an HTML parser context
5151
 * @param chunk  chunk of memory
5152
 * @param size  size of chunk in bytes
5153
 * @param terminate  last chunk indicator
5154
 * @returns an xmlParserErrors code (0 on success).
5155
 */
5156
int
5157
htmlParseChunk(htmlParserCtxt *ctxt, const char *chunk, int size,
5158
3
              int terminate) {
5159
3
    if ((ctxt == NULL) ||
5160
3
        (ctxt->input == NULL) || (ctxt->input->buf == NULL) ||
5161
3
        (size < 0) ||
5162
3
        ((size > 0) && (chunk == NULL)))
5163
0
  return(XML_ERR_ARGUMENT);
5164
3
    if (PARSER_STOPPED(ctxt) != 0)
5165
0
        return(ctxt->errNo);
5166
5167
3
    if (size > 0)  {
5168
0
  size_t pos = ctxt->input->cur - ctxt->input->base;
5169
0
  int res;
5170
5171
0
  res = xmlParserInputBufferPush(ctxt->input->buf, size, chunk);
5172
0
        xmlBufUpdateInput(ctxt->input->buf->buffer, ctxt->input, pos);
5173
0
  if (res < 0) {
5174
0
            htmlParseErr(ctxt, ctxt->input->buf->error,
5175
0
                         "xmlParserInputBufferPush failed", NULL, NULL);
5176
0
      return (ctxt->errNo);
5177
0
  }
5178
0
    }
5179
5180
3
    htmlParseTryOrFinish(ctxt, terminate);
5181
5182
3
    if ((terminate) && (ctxt->instate != XML_PARSER_EOF)) {
5183
3
        htmlAutoCloseOnEnd(ctxt);
5184
5185
        /*
5186
         * Only check for truncated multi-byte sequences
5187
         */
5188
3
        xmlParserCheckEOF(ctxt, XML_ERR_INTERNAL_ERROR);
5189
5190
3
        if ((ctxt->sax) && (ctxt->sax->endDocument != NULL))
5191
3
            ctxt->sax->endDocument(ctxt->userData);
5192
5193
3
  ctxt->instate = XML_PARSER_EOF;
5194
3
    }
5195
5196
3
    return((xmlParserErrors) ctxt->errNo);
5197
3
}
5198
5199
/************************************************************************
5200
 *                  *
5201
 *      User entry points       *
5202
 *                  *
5203
 ************************************************************************/
5204
5205
/**
5206
 * Create a parser context for using the HTML parser in push mode.
5207
 *
5208
 * @param sax  a SAX handler (optional)
5209
 * @param user_data  The user data returned on SAX callbacks (optional)
5210
 * @param chunk  a pointer to an array of chars (optional)
5211
 * @param size  number of chars in the array
5212
 * @param filename  only used for error reporting (optional)
5213
 * @param enc  encoding (deprecated, pass XML_CHAR_ENCODING_NONE)
5214
 * @returns the new parser context or NULL if a memory allocation
5215
 * failed.
5216
 */
5217
htmlParserCtxt *
5218
htmlCreatePushParserCtxt(htmlSAXHandler *sax, void *user_data,
5219
                         const char *chunk, int size, const char *filename,
5220
9
       xmlCharEncoding enc) {
5221
9
    htmlParserCtxtPtr ctxt;
5222
9
    htmlParserInputPtr input;
5223
9
    const char *encoding;
5224
5225
9
    ctxt = htmlNewSAXParserCtxt(sax, user_data);
5226
9
    if (ctxt == NULL)
5227
2
  return(NULL);
5228
5229
7
    encoding = xmlGetCharEncodingName(enc);
5230
7
    input = xmlNewPushInput(filename, chunk, size);
5231
7
    if (input == NULL) {
5232
2
  htmlFreeParserCtxt(ctxt);
5233
2
  return(NULL);
5234
2
    }
5235
5236
5
    if (xmlCtxtPushInput(ctxt, input) < 0) {
5237
0
        xmlFreeInputStream(input);
5238
0
        xmlFreeParserCtxt(ctxt);
5239
0
        return(NULL);
5240
0
    }
5241
5242
5
    if (encoding != NULL)
5243
0
        xmlSwitchEncodingName(ctxt, encoding);
5244
5245
5
    return(ctxt);
5246
5
}
5247
#endif /* LIBXML_PUSH_ENABLED */
5248
5249
/**
5250
 * Parse an HTML in-memory document. If sax is not NULL, use the SAX callbacks
5251
 * to handle parse events. If sax is NULL, fallback to the default DOM
5252
 * behavior and return a tree.
5253
 *
5254
 * @deprecated Use #htmlNewSAXParserCtxt and #htmlCtxtReadDoc.
5255
 *
5256
 * @param cur  a pointer to an array of xmlChar
5257
 * @param encoding  a free form C string describing the HTML document encoding, or NULL
5258
 * @param sax  the SAX handler block
5259
 * @param userData  if using SAX, this pointer will be provided on callbacks.
5260
 * @returns the resulting document tree unless SAX is NULL or the document is
5261
 *     not well formed.
5262
 */
5263
5264
xmlDoc *
5265
htmlSAXParseDoc(const xmlChar *cur, const char *encoding,
5266
0
                htmlSAXHandler *sax, void *userData) {
5267
0
    htmlDocPtr ret;
5268
0
    htmlParserCtxtPtr ctxt;
5269
5270
0
    if (cur == NULL)
5271
0
        return(NULL);
5272
5273
0
    ctxt = htmlCreateDocParserCtxt(cur, NULL, encoding);
5274
0
    if (ctxt == NULL)
5275
0
        return(NULL);
5276
5277
0
    if (sax != NULL) {
5278
0
        *ctxt->sax = *sax;
5279
0
        ctxt->userData = userData;
5280
0
    }
5281
5282
0
    htmlParseDocument(ctxt);
5283
0
    ret = ctxt->myDoc;
5284
0
    htmlFreeParserCtxt(ctxt);
5285
5286
0
    return(ret);
5287
0
}
5288
5289
/**
5290
 * Parse an HTML in-memory document and build a tree.
5291
 *
5292
 * @deprecated Use #htmlReadDoc.
5293
 *
5294
 * This function uses deprecated global parser options.
5295
 *
5296
 * @param cur  a pointer to an array of xmlChar
5297
 * @param encoding  the encoding (optional)
5298
 * @returns the resulting document tree
5299
 */
5300
5301
xmlDoc *
5302
0
htmlParseDoc(const xmlChar *cur, const char *encoding) {
5303
0
    return(htmlSAXParseDoc(cur, encoding, NULL, NULL));
5304
0
}
5305
5306
5307
/**
5308
 * Create a parser context to read from a file.
5309
 *
5310
 * @deprecated Use #htmlNewParserCtxt and #htmlCtxtReadFile.
5311
 *
5312
 * A non-NULL encoding overrides encoding declarations in the document.
5313
 *
5314
 * Automatic support for ZLIB/Compress compressed document is provided
5315
 * by default if found at compile-time.
5316
 *
5317
 * @param filename  the filename
5318
 * @param encoding  optional encoding
5319
 * @returns the new parser context or NULL if a memory allocation failed.
5320
 */
5321
htmlParserCtxt *
5322
htmlCreateFileParserCtxt(const char *filename, const char *encoding)
5323
0
{
5324
0
    htmlParserCtxtPtr ctxt;
5325
0
    htmlParserInputPtr input;
5326
5327
0
    if (filename == NULL)
5328
0
        return(NULL);
5329
5330
0
    ctxt = htmlNewParserCtxt();
5331
0
    if (ctxt == NULL) {
5332
0
  return(NULL);
5333
0
    }
5334
5335
0
    input = xmlCtxtNewInputFromUrl(ctxt, filename, NULL, encoding, 0);
5336
0
    if (input == NULL) {
5337
0
  xmlFreeParserCtxt(ctxt);
5338
0
  return(NULL);
5339
0
    }
5340
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
5341
0
        xmlFreeInputStream(input);
5342
0
        xmlFreeParserCtxt(ctxt);
5343
0
        return(NULL);
5344
0
    }
5345
5346
0
    return(ctxt);
5347
0
}
5348
5349
/**
5350
 * parse an HTML file and build a tree. Automatic support for ZLIB/Compress
5351
 * compressed document is provided by default if found at compile-time.
5352
 * It use the given SAX function block to handle the parsing callback.
5353
 * If sax is NULL, fallback to the default DOM tree building routines.
5354
 *
5355
 * @deprecated Use #htmlNewSAXParserCtxt and #htmlCtxtReadFile.
5356
 *
5357
 * @param filename  the filename
5358
 * @param encoding  encoding (optional)
5359
 * @param sax  the SAX handler block
5360
 * @param userData  if using SAX, this pointer will be provided on callbacks.
5361
 * @returns the resulting document tree unless SAX is NULL or the document is
5362
 *     not well formed.
5363
 */
5364
5365
xmlDoc *
5366
htmlSAXParseFile(const char *filename, const char *encoding, htmlSAXHandler *sax,
5367
0
                 void *userData) {
5368
0
    htmlDocPtr ret;
5369
0
    htmlParserCtxtPtr ctxt;
5370
0
    htmlSAXHandlerPtr oldsax = NULL;
5371
5372
0
    ctxt = htmlCreateFileParserCtxt(filename, encoding);
5373
0
    if (ctxt == NULL) return(NULL);
5374
0
    if (sax != NULL) {
5375
0
  oldsax = ctxt->sax;
5376
0
        ctxt->sax = sax;
5377
0
        ctxt->userData = userData;
5378
0
    }
5379
5380
0
    htmlParseDocument(ctxt);
5381
5382
0
    ret = ctxt->myDoc;
5383
0
    if (sax != NULL) {
5384
0
        ctxt->sax = oldsax;
5385
0
        ctxt->userData = NULL;
5386
0
    }
5387
0
    htmlFreeParserCtxt(ctxt);
5388
5389
0
    return(ret);
5390
0
}
5391
5392
/**
5393
 * Parse an HTML file and build a tree.
5394
 *
5395
 * @param filename  the filename
5396
 * @param encoding  encoding (optional)
5397
 * @returns the resulting document tree
5398
 */
5399
5400
xmlDoc *
5401
0
htmlParseFile(const char *filename, const char *encoding) {
5402
0
    return(htmlSAXParseFile(filename, encoding, NULL, NULL));
5403
0
}
5404
5405
/**
5406
 * Set and return the previous value for handling HTML omitted tags.
5407
 *
5408
 * @deprecated Use HTML_PARSE_NOIMPLIED
5409
 *
5410
 * @param val  int 0 or 1
5411
 * @returns the last value for 0 for no handling, 1 for auto insertion.
5412
 */
5413
5414
int
5415
0
htmlHandleOmittedElem(int val) {
5416
0
    int old = htmlOmittedDefaultValue;
5417
5418
0
    htmlOmittedDefaultValue = val;
5419
0
    return(old);
5420
0
}
5421
5422
/**
5423
 * @deprecated Don't use.
5424
 *
5425
 * @param parent  HTML parent element
5426
 * @param elt  HTML element
5427
 * @returns 1
5428
 */
5429
int
5430
htmlElementAllowedHere(const htmlElemDesc* parent ATTRIBUTE_UNUSED,
5431
0
                       const xmlChar* elt ATTRIBUTE_UNUSED) {
5432
0
    return(1);
5433
0
}
5434
5435
/**
5436
 * @deprecated Don't use.
5437
 *
5438
 * @param parent  HTML parent element
5439
 * @param elt  HTML element
5440
 * @returns HTML_VALID
5441
 */
5442
htmlStatus
5443
htmlElementStatusHere(const htmlElemDesc* parent ATTRIBUTE_UNUSED,
5444
0
                      const htmlElemDesc* elt ATTRIBUTE_UNUSED) {
5445
0
    return(HTML_VALID);
5446
0
}
5447
5448
/**
5449
 * @deprecated Don't use.
5450
 *
5451
 * @param elt  HTML element
5452
 * @param attr  HTML attribute
5453
 * @param legacy  whether to allow deprecated attributes
5454
 * @returns HTML_VALID
5455
 */
5456
htmlStatus
5457
htmlAttrAllowed(const htmlElemDesc* elt ATTRIBUTE_UNUSED,
5458
                const xmlChar* attr ATTRIBUTE_UNUSED,
5459
0
                int legacy ATTRIBUTE_UNUSED) {
5460
0
    return(HTML_VALID);
5461
0
}
5462
5463
/**
5464
 * @deprecated Don't use.
5465
 *
5466
 * @param node  an xmlNode in a tree
5467
 * @param legacy  whether to allow deprecated elements (YES is faster here
5468
 *  for Element nodes)
5469
 * @returns HTML_VALID
5470
 */
5471
htmlStatus
5472
htmlNodeStatus(xmlNode *node ATTRIBUTE_UNUSED,
5473
0
               int legacy ATTRIBUTE_UNUSED) {
5474
0
    return(HTML_VALID);
5475
0
}
5476
5477
/************************************************************************
5478
 *                  *
5479
 *  New set (2.6.0) of simpler and more flexible APIs   *
5480
 *                  *
5481
 ************************************************************************/
5482
5483
/**
5484
 * Reset a parser context
5485
 *
5486
 * Same as #xmlCtxtReset.
5487
 *
5488
 * @param ctxt  an HTML parser context
5489
 */
5490
void
5491
htmlCtxtReset(htmlParserCtxt *ctxt)
5492
3.12k
{
5493
3.12k
    xmlCtxtReset(ctxt);
5494
3.12k
}
5495
5496
static int
5497
htmlCtxtSetOptionsInternal(xmlParserCtxtPtr ctxt, int options, int keepMask)
5498
5.63k
{
5499
5.63k
    int allMask;
5500
5501
5.63k
    if (ctxt == NULL)
5502
8
        return(-1);
5503
5504
5.62k
    allMask = HTML_PARSE_RECOVER |
5505
5.62k
              HTML_PARSE_HTML5 |
5506
5.62k
              HTML_PARSE_NODEFDTD |
5507
5.62k
              HTML_PARSE_NOERROR |
5508
5.62k
              HTML_PARSE_NOWARNING |
5509
5.62k
              HTML_PARSE_PEDANTIC |
5510
5.62k
              HTML_PARSE_NOBLANKS |
5511
5.62k
              HTML_PARSE_NONET |
5512
5.62k
              HTML_PARSE_NOIMPLIED |
5513
5.62k
              HTML_PARSE_COMPACT |
5514
5.62k
              HTML_PARSE_HUGE |
5515
5.62k
              HTML_PARSE_IGNORE_ENC |
5516
5.62k
              HTML_PARSE_BIG_LINES;
5517
5518
5.62k
    ctxt->options = (ctxt->options & keepMask) | (options & allMask);
5519
5520
    /*
5521
     * For some options, struct members are historically the source
5522
     * of truth. See xmlCtxtSetOptionsInternal.
5523
     */
5524
5.62k
    ctxt->keepBlanks = (options & HTML_PARSE_NOBLANKS) ? 0 : 1;
5525
5526
    /*
5527
     * Recover from character encoding errors
5528
     */
5529
5.62k
    ctxt->recovery = 1;
5530
5531
    /*
5532
     * Changing SAX callbacks is a bad idea. This should be fixed.
5533
     */
5534
5.62k
    if (options & HTML_PARSE_NOBLANKS) {
5535
2.44k
        ctxt->sax->ignorableWhitespace = xmlSAX2IgnorableWhitespace;
5536
2.44k
    }
5537
5.62k
    if (options & HTML_PARSE_HUGE) {
5538
1.33k
        if (ctxt->dict != NULL)
5539
1.33k
            xmlDictSetLimit(ctxt->dict, 0);
5540
1.33k
    }
5541
5542
    /*
5543
     * It would be useful to allow this feature.
5544
     */
5545
5.62k
    ctxt->dictNames = 0;
5546
5547
    /*
5548
     * Allow XML_PARSE_NOENT which many users set on the HTML parser.
5549
     */
5550
5.62k
    return(options & ~allMask & ~XML_PARSE_NOENT);
5551
5.63k
}
5552
5553
/**
5554
 * Applies the options to the parser context. Unset options are
5555
 * cleared.
5556
 *
5557
 * @since 2.14.0
5558
 *
5559
 * With older versions, you can use #htmlCtxtUseOptions.
5560
 *
5561
 * @param ctxt  an HTML parser context
5562
 * @param options  a bitmask of htmlParserOption values
5563
 * @returns 0 in case of success, the set of unknown or unimplemented options
5564
 *         in case of error.
5565
 */
5566
int
5567
htmlCtxtSetOptions(htmlParserCtxt *ctxt, int options)
5568
0
{
5569
0
    return(htmlCtxtSetOptionsInternal(ctxt, options, 0));
5570
0
}
5571
5572
/**
5573
 * Applies the options to the parser context. The following options
5574
 * are never cleared and can only be enabled:
5575
 *
5576
 * @deprecated Use #htmlCtxtSetOptions.
5577
 *
5578
 * - HTML_PARSE_NODEFDTD
5579
 * - HTML_PARSE_NOERROR
5580
 * - HTML_PARSE_NOWARNING
5581
 * - HTML_PARSE_NOIMPLIED
5582
 * - HTML_PARSE_COMPACT
5583
 * - HTML_PARSE_HUGE
5584
 * - HTML_PARSE_IGNORE_ENC
5585
 * - HTML_PARSE_BIG_LINES
5586
 *
5587
 * @param ctxt  an HTML parser context
5588
 * @param options  a combination of htmlParserOption values
5589
 * @returns 0 in case of success, the set of unknown or unimplemented options
5590
 *         in case of error.
5591
 */
5592
int
5593
htmlCtxtUseOptions(htmlParserCtxt *ctxt, int options)
5594
5.63k
{
5595
5.63k
    int keepMask;
5596
5597
    /*
5598
     * For historic reasons, some options can only be enabled.
5599
     */
5600
5.63k
    keepMask = HTML_PARSE_NODEFDTD |
5601
5.63k
               HTML_PARSE_NOERROR |
5602
5.63k
               HTML_PARSE_NOWARNING |
5603
5.63k
               HTML_PARSE_NOIMPLIED |
5604
5.63k
               HTML_PARSE_COMPACT |
5605
5.63k
               HTML_PARSE_HUGE |
5606
5.63k
               HTML_PARSE_IGNORE_ENC |
5607
5.63k
               HTML_PARSE_BIG_LINES;
5608
5609
5.63k
    return(htmlCtxtSetOptionsInternal(ctxt, options, keepMask));
5610
5.63k
}
5611
5612
/**
5613
 * Parse an HTML document and return the resulting document tree.
5614
 *
5615
 * @since 2.13.0
5616
 *
5617
 * @param ctxt  an HTML parser context
5618
 * @param input  parser input
5619
 * @returns the resulting document tree or NULL
5620
 */
5621
xmlDoc *
5622
htmlCtxtParseDocument(htmlParserCtxt *ctxt, xmlParserInput *input)
5623
2.49k
{
5624
2.49k
    htmlDocPtr ret;
5625
5626
2.49k
    if ((ctxt == NULL) || (input == NULL)) {
5627
0
        xmlFatalErr(ctxt, XML_ERR_ARGUMENT, NULL);
5628
0
        xmlFreeInputStream(input);
5629
0
        return(NULL);
5630
0
    }
5631
5632
    /* assert(ctxt->inputNr == 0); */
5633
2.49k
    while (ctxt->inputNr > 0)
5634
0
        xmlFreeInputStream(xmlCtxtPopInput(ctxt));
5635
5636
2.49k
    if (xmlCtxtPushInput(ctxt, input) < 0) {
5637
0
        xmlFreeInputStream(input);
5638
0
        return(NULL);
5639
0
    }
5640
5641
2.49k
    ctxt->html = INSERT_INITIAL;
5642
2.49k
    htmlParseDocument(ctxt);
5643
5644
2.49k
    ret = xmlCtxtGetDocument(ctxt);
5645
5646
    /* assert(ctxt->inputNr == 1); */
5647
4.99k
    while (ctxt->inputNr > 0)
5648
2.49k
        xmlFreeInputStream(xmlCtxtPopInput(ctxt));
5649
5650
2.49k
    return(ret);
5651
2.49k
}
5652
5653
/**
5654
 * Convenience function to parse an HTML document from a zero-terminated
5655
 * string.
5656
 *
5657
 * See #htmlCtxtReadDoc for details.
5658
 *
5659
 * @param str  a pointer to a zero terminated string
5660
 * @param url  only used for error reporting (optoinal)
5661
 * @param encoding  the document encoding (optional)
5662
 * @param options  a combination of htmlParserOption values
5663
 * @returns the resulting document tree.
5664
 */
5665
xmlDoc *
5666
htmlReadDoc(const xmlChar *str, const char *url, const char *encoding,
5667
            int options)
5668
0
{
5669
0
    htmlParserCtxtPtr ctxt;
5670
0
    xmlParserInputPtr input;
5671
0
    htmlDocPtr doc = NULL;
5672
5673
0
    ctxt = htmlNewParserCtxt();
5674
0
    if (ctxt == NULL)
5675
0
        return(NULL);
5676
5677
0
    htmlCtxtUseOptions(ctxt, options);
5678
5679
0
    input = xmlCtxtNewInputFromString(ctxt, url, (const char *) str, encoding,
5680
0
                                      XML_INPUT_BUF_STATIC);
5681
5682
0
    if (input != NULL)
5683
0
        doc = htmlCtxtParseDocument(ctxt, input);
5684
5685
0
    htmlFreeParserCtxt(ctxt);
5686
0
    return(doc);
5687
0
}
5688
5689
/**
5690
 * Convenience function to parse an HTML file from the filesystem,
5691
 * the network or a global user-defined resource loader.
5692
 *
5693
 * See #htmlCtxtReadFile for details.
5694
 *
5695
 * @param filename  a file or URL
5696
 * @param encoding  the document encoding (optional)
5697
 * @param options  a combination of htmlParserOption values
5698
 * @returns the resulting document tree.
5699
 */
5700
xmlDoc *
5701
htmlReadFile(const char *filename, const char *encoding, int options)
5702
0
{
5703
0
    htmlParserCtxtPtr ctxt;
5704
0
    xmlParserInputPtr input;
5705
0
    htmlDocPtr doc = NULL;
5706
5707
0
    ctxt = htmlNewParserCtxt();
5708
0
    if (ctxt == NULL)
5709
0
        return(NULL);
5710
5711
0
    htmlCtxtUseOptions(ctxt, options);
5712
5713
0
    input = xmlCtxtNewInputFromUrl(ctxt, filename, NULL, encoding, 0);
5714
5715
0
    if (input != NULL)
5716
0
        doc = htmlCtxtParseDocument(ctxt, input);
5717
5718
0
    htmlFreeParserCtxt(ctxt);
5719
0
    return(doc);
5720
0
}
5721
5722
/**
5723
 * Convenience function to parse an HTML document from memory.
5724
 * The input buffer must not contain any terminating null bytes.
5725
 *
5726
 * See #htmlCtxtReadMemory for details.
5727
 *
5728
 * @param buffer  a pointer to a char array
5729
 * @param size  the size of the array
5730
 * @param url  only used for error reporting (optional)
5731
 * @param encoding  the document encoding, or NULL
5732
 * @param options  a combination of htmlParserOption values
5733
 * @returns the resulting document tree
5734
 */
5735
xmlDoc *
5736
htmlReadMemory(const char *buffer, int size, const char *url,
5737
               const char *encoding, int options)
5738
0
{
5739
0
    htmlParserCtxtPtr ctxt;
5740
0
    xmlParserInputPtr input;
5741
0
    htmlDocPtr doc = NULL;
5742
5743
0
    if (size < 0)
5744
0
  return(NULL);
5745
5746
0
    ctxt = htmlNewParserCtxt();
5747
0
    if (ctxt == NULL)
5748
0
        return(NULL);
5749
5750
0
    htmlCtxtUseOptions(ctxt, options);
5751
5752
0
    input = xmlCtxtNewInputFromMemory(ctxt, url, buffer, size, encoding,
5753
0
                                      XML_INPUT_BUF_STATIC);
5754
5755
0
    if (input != NULL)
5756
0
        doc = htmlCtxtParseDocument(ctxt, input);
5757
5758
0
    htmlFreeParserCtxt(ctxt);
5759
0
    return(doc);
5760
0
}
5761
5762
/**
5763
 * Convenience function to parse an HTML document from a
5764
 * file descriptor.
5765
 *
5766
 * NOTE that the file descriptor will not be closed when the
5767
 * context is freed or reset.
5768
 *
5769
 * See #htmlCtxtReadFd for details.
5770
 *
5771
 * @param fd  an open file descriptor
5772
 * @param url  only used for error reporting (optional)
5773
 * @param encoding  the document encoding, or NULL
5774
 * @param options  a combination of htmlParserOption values
5775
 * @returns the resulting document tree
5776
 */
5777
xmlDoc *
5778
htmlReadFd(int fd, const char *url, const char *encoding, int options)
5779
0
{
5780
0
    htmlParserCtxtPtr ctxt;
5781
0
    xmlParserInputPtr input;
5782
0
    htmlDocPtr doc = NULL;
5783
5784
0
    ctxt = htmlNewParserCtxt();
5785
0
    if (ctxt == NULL)
5786
0
        return(NULL);
5787
5788
0
    htmlCtxtUseOptions(ctxt, options);
5789
5790
0
    input = xmlCtxtNewInputFromFd(ctxt, url, fd, encoding, 0);
5791
5792
0
    if (input != NULL)
5793
0
        doc = htmlCtxtParseDocument(ctxt, input);
5794
5795
0
    htmlFreeParserCtxt(ctxt);
5796
0
    return(doc);
5797
0
}
5798
5799
/**
5800
 * Convenience function to parse an HTML document from I/O functions
5801
 * and context.
5802
 *
5803
 * See #htmlCtxtReadIO for details.
5804
 *
5805
 * @param ioread  an I/O read function
5806
 * @param ioclose  an I/O close function (optional)
5807
 * @param ioctx  an I/O handler
5808
 * @param url  only used for error reporting (optional)
5809
 * @param encoding  the document encoding (optional)
5810
 * @param options  a combination of htmlParserOption values
5811
 * @returns the resulting document tree
5812
 */
5813
xmlDoc *
5814
htmlReadIO(xmlInputReadCallback ioread, xmlInputCloseCallback ioclose,
5815
          void *ioctx, const char *url, const char *encoding, int options)
5816
0
{
5817
0
    htmlParserCtxtPtr ctxt;
5818
0
    xmlParserInputPtr input;
5819
0
    htmlDocPtr doc = NULL;
5820
5821
0
    ctxt = htmlNewParserCtxt();
5822
0
    if (ctxt == NULL)
5823
0
        return (NULL);
5824
5825
0
    htmlCtxtUseOptions(ctxt, options);
5826
5827
0
    input = xmlCtxtNewInputFromIO(ctxt, url, ioread, ioclose, ioctx,
5828
0
                                  encoding, 0);
5829
5830
0
    if (input != NULL)
5831
0
        doc = htmlCtxtParseDocument(ctxt, input);
5832
5833
0
    htmlFreeParserCtxt(ctxt);
5834
0
    return(doc);
5835
0
}
5836
5837
/**
5838
 * Parse an HTML in-memory document and build a tree.
5839
 *
5840
 * See #htmlCtxtUseOptions for details.
5841
 *
5842
 * @param ctxt  an HTML parser context
5843
 * @param str  a pointer to a zero terminated string
5844
 * @param URL  only used for error reporting (optional)
5845
 * @param encoding  the document encoding (optional)
5846
 * @param options  a combination of htmlParserOption values
5847
 * @returns the resulting document tree
5848
 */
5849
xmlDoc *
5850
htmlCtxtReadDoc(xmlParserCtxt *ctxt, const xmlChar *str,
5851
                const char *URL, const char *encoding, int options)
5852
0
{
5853
0
    xmlParserInputPtr input;
5854
5855
0
    if (ctxt == NULL)
5856
0
        return (NULL);
5857
5858
0
    htmlCtxtReset(ctxt);
5859
0
    htmlCtxtUseOptions(ctxt, options);
5860
5861
0
    input = xmlCtxtNewInputFromString(ctxt, URL, (const char *) str,
5862
0
                                      encoding, 0);
5863
0
    if (input == NULL)
5864
0
        return(NULL);
5865
5866
0
    return(htmlCtxtParseDocument(ctxt, input));
5867
0
}
5868
5869
/**
5870
 * Parse an HTML file from the filesystem, the network or a
5871
 * user-defined resource loader.
5872
 *
5873
 * See #htmlCtxtUseOptions for details.
5874
 *
5875
 * @param ctxt  an HTML parser context
5876
 * @param filename  a file or URL
5877
 * @param encoding  the document encoding (optional)
5878
 * @param options  a combination of htmlParserOption values
5879
 * @returns the resulting document tree
5880
 */
5881
xmlDoc *
5882
htmlCtxtReadFile(xmlParserCtxt *ctxt, const char *filename,
5883
                const char *encoding, int options)
5884
3.12k
{
5885
3.12k
    xmlParserInputPtr input;
5886
5887
3.12k
    if (ctxt == NULL)
5888
0
        return (NULL);
5889
5890
3.12k
    htmlCtxtReset(ctxt);
5891
3.12k
    htmlCtxtUseOptions(ctxt, options);
5892
5893
3.12k
    input = xmlCtxtNewInputFromUrl(ctxt, filename, NULL, encoding, 0);
5894
3.12k
    if (input == NULL)
5895
622
        return(NULL);
5896
5897
2.49k
    return(htmlCtxtParseDocument(ctxt, input));
5898
3.12k
}
5899
5900
/**
5901
 * Parse an HTML in-memory document and build a tree. The input buffer must
5902
 * not contain any terminating null bytes.
5903
 *
5904
 * See #htmlCtxtUseOptions for details.
5905
 *
5906
 * @param ctxt  an HTML parser context
5907
 * @param buffer  a pointer to a char array
5908
 * @param size  the size of the array
5909
 * @param URL  only used for error reporting (optional)
5910
 * @param encoding  the document encoding (optinal)
5911
 * @param options  a combination of htmlParserOption values
5912
 * @returns the resulting document tree
5913
 */
5914
xmlDoc *
5915
htmlCtxtReadMemory(xmlParserCtxt *ctxt, const char *buffer, int size,
5916
                  const char *URL, const char *encoding, int options)
5917
0
{
5918
0
    xmlParserInputPtr input;
5919
5920
0
    if ((ctxt == NULL) || (size < 0))
5921
0
        return (NULL);
5922
5923
0
    htmlCtxtReset(ctxt);
5924
0
    htmlCtxtUseOptions(ctxt, options);
5925
5926
0
    input = xmlCtxtNewInputFromMemory(ctxt, URL, buffer, size, encoding,
5927
0
                                      XML_INPUT_BUF_STATIC);
5928
0
    if (input == NULL)
5929
0
        return(NULL);
5930
5931
0
    return(htmlCtxtParseDocument(ctxt, input));
5932
0
}
5933
5934
/**
5935
 * Parse an HTML from a file descriptor and build a tree.
5936
 *
5937
 * See #htmlCtxtUseOptions for details.
5938
 *
5939
 * NOTE that the file descriptor will not be closed when the
5940
 * context is freed or reset.
5941
 *
5942
 * @param ctxt  an HTML parser context
5943
 * @param fd  an open file descriptor
5944
 * @param URL  only used for error reporting (optional)
5945
 * @param encoding  the document encoding (optinal)
5946
 * @param options  a combination of htmlParserOption values
5947
 * @returns the resulting document tree
5948
 */
5949
xmlDoc *
5950
htmlCtxtReadFd(xmlParserCtxt *ctxt, int fd,
5951
              const char *URL, const char *encoding, int options)
5952
0
{
5953
0
    xmlParserInputPtr input;
5954
5955
0
    if (ctxt == NULL)
5956
0
        return(NULL);
5957
5958
0
    htmlCtxtReset(ctxt);
5959
0
    htmlCtxtUseOptions(ctxt, options);
5960
5961
0
    input = xmlCtxtNewInputFromFd(ctxt, URL, fd, encoding, 0);
5962
0
    if (input == NULL)
5963
0
        return(NULL);
5964
5965
0
    return(htmlCtxtParseDocument(ctxt, input));
5966
0
}
5967
5968
/**
5969
 * Parse an HTML document from I/O functions and source and build a tree.
5970
 *
5971
 * See #htmlCtxtUseOptions for details.
5972
 *
5973
 * @param ctxt  an HTML parser context
5974
 * @param ioread  an I/O read function
5975
 * @param ioclose  an I/O close function
5976
 * @param ioctx  an I/O handler
5977
 * @param URL  the base URL to use for the document
5978
 * @param encoding  the document encoding, or NULL
5979
 * @param options  a combination of htmlParserOption values
5980
 * @returns the resulting document tree
5981
 */
5982
xmlDoc *
5983
htmlCtxtReadIO(xmlParserCtxt *ctxt, xmlInputReadCallback ioread,
5984
              xmlInputCloseCallback ioclose, void *ioctx,
5985
        const char *URL,
5986
              const char *encoding, int options)
5987
0
{
5988
0
    xmlParserInputPtr input;
5989
5990
0
    if (ctxt == NULL)
5991
0
        return (NULL);
5992
5993
0
    htmlCtxtReset(ctxt);
5994
0
    htmlCtxtUseOptions(ctxt, options);
5995
5996
0
    input = xmlCtxtNewInputFromIO(ctxt, URL, ioread, ioclose, ioctx,
5997
0
                                  encoding, 0);
5998
0
    if (input == NULL)
5999
0
        return(NULL);
6000
6001
0
    return(htmlCtxtParseDocument(ctxt, input));
6002
0
}
6003
6004
#endif /* LIBXML_HTML_ENABLED */