Coverage Report

Created: 2026-07-16 06:49

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libxml2/parser.c
Line
Count
Source
1
/*
2
 * parser.c : an XML 1.0 parser, namespaces and validity support are mostly
3
 *            implemented on top of the SAX interfaces
4
 *
5
 * References:
6
 *   The XML specification:
7
 *     http://www.w3.org/TR/REC-xml
8
 *   Original 1.0 version:
9
 *     http://www.w3.org/TR/1998/REC-xml-19980210
10
 *   XML second edition working draft
11
 *     http://www.w3.org/TR/2000/WD-xml-2e-20000814
12
 *
13
 * Okay this is a big file, the parser core is around 7000 lines, then it
14
 * is followed by the progressive parser top routines, then the various
15
 * high level APIs to call the parser and a few miscellaneous functions.
16
 * A number of helper functions and deprecated ones have been moved to
17
 * parserInternals.c to reduce this file size.
18
 * As much as possible the functions are associated with their relative
19
 * production in the XML specification. A few productions defining the
20
 * different ranges of character are actually implanted either in
21
 * parserInternals.h or parserInternals.c
22
 * The DOM tree build is realized from the default SAX callbacks in
23
 * the module SAX2.c.
24
 * The routines doing the validation checks are in valid.c and called either
25
 * from the SAX callbacks or as standalone functions using a preparsed
26
 * document.
27
 *
28
 * See Copyright for the status of this software.
29
 *
30
 * Author: Daniel Veillard
31
 */
32
33
/* To avoid EBCDIC trouble when parsing on zOS */
34
#if defined(__MVS__)
35
#pragma convert("ISO8859-1")
36
#endif
37
38
#define IN_LIBXML
39
#include "libxml.h"
40
41
#if defined(_WIN32)
42
#define XML_DIR_SEP '\\'
43
#else
44
#define XML_DIR_SEP '/'
45
#endif
46
47
#include <stdlib.h>
48
#include <limits.h>
49
#include <string.h>
50
#include <stdarg.h>
51
#include <stddef.h>
52
#include <ctype.h>
53
#include <stdlib.h>
54
#include <libxml/parser.h>
55
#include <libxml/xmlmemory.h>
56
#include <libxml/tree.h>
57
#include <libxml/parserInternals.h>
58
#include <libxml/valid.h>
59
#include <libxml/entities.h>
60
#include <libxml/xmlerror.h>
61
#include <libxml/encoding.h>
62
#include <libxml/xmlIO.h>
63
#include <libxml/uri.h>
64
#include <libxml/SAX2.h>
65
#include <libxml/HTMLparser.h>
66
#ifdef LIBXML_CATALOG_ENABLED
67
#include <libxml/catalog.h>
68
#endif
69
70
#include "private/buf.h"
71
#include "private/dict.h"
72
#include "private/entities.h"
73
#include "private/error.h"
74
#include "private/html.h"
75
#include "private/io.h"
76
#include "private/memory.h"
77
#include "private/parser.h"
78
#include "private/tree.h"
79
80
0
#define NS_INDEX_EMPTY  INT_MAX
81
0
#define NS_INDEX_XML    (INT_MAX - 1)
82
0
#define URI_HASH_EMPTY  0xD943A04E
83
0
#define URI_HASH_XML    0xF0451F02
84
85
#ifndef STDIN_FILENO
86
0
  #define STDIN_FILENO 0
87
#endif
88
89
#ifndef SIZE_MAX
90
  #define SIZE_MAX ((size_t) -1)
91
#endif
92
93
0
#define XML_MAX_ATTRS 100000000 /* 100 million */
94
95
0
#define XML_SPECIAL_EXTERNAL    (1 << 20)
96
0
#define XML_SPECIAL_TYPE_MASK   (XML_SPECIAL_EXTERNAL - 1)
97
98
0
#define XML_ATTVAL_ALLOC        (1 << 0)
99
0
#define XML_ATTVAL_NORM_CHANGE  (1 << 1)
100
101
struct _xmlStartTag {
102
    const xmlChar *prefix;
103
    const xmlChar *URI;
104
    int line;
105
    int nsNr;
106
};
107
108
typedef struct {
109
    void *saxData;
110
    unsigned prefixHashValue;
111
    unsigned uriHashValue;
112
    unsigned elementId;
113
    int oldIndex;
114
} xmlParserNsExtra;
115
116
typedef struct {
117
    unsigned hashValue;
118
    int index;
119
} xmlParserNsBucket;
120
121
struct _xmlParserNsData {
122
    xmlParserNsExtra *extra;
123
124
    unsigned hashSize;
125
    unsigned hashElems;
126
    xmlParserNsBucket *hash;
127
128
    unsigned elementId;
129
    int defaultNsIndex;
130
    int minNsIndex;
131
};
132
133
static int
134
xmlParseElementStart(xmlParserCtxtPtr ctxt);
135
136
static void
137
xmlParseElementEnd(xmlParserCtxtPtr ctxt);
138
139
static xmlEntityPtr
140
xmlLookupGeneralEntity(xmlParserCtxtPtr ctxt, const xmlChar *name, int inAttr);
141
142
static const xmlChar *
143
xmlParseEntityRefInternal(xmlParserCtxtPtr ctxt);
144
145
/************************************************************************
146
 *                  *
147
 *  Arbitrary limits set in the parser. See XML_PARSE_HUGE    *
148
 *                  *
149
 ************************************************************************/
150
151
#define XML_PARSER_BIG_ENTITY 1000
152
#define XML_PARSER_LOT_ENTITY 5000
153
154
/*
155
 * Constants for protection against abusive entity expansion
156
 * ("billion laughs").
157
 */
158
159
/*
160
 * A certain amount of entity expansion which is always allowed.
161
 */
162
0
#define XML_PARSER_ALLOWED_EXPANSION 1000000
163
164
/*
165
 * Fixed cost for each entity reference. This crudely models processing time
166
 * as well to protect, for example, against exponential expansion of empty
167
 * or very short entities.
168
 */
169
0
#define XML_ENT_FIXED_COST 20
170
171
0
#define XML_PARSER_BIG_BUFFER_SIZE 300
172
0
#define XML_PARSER_BUFFER_SIZE 100
173
0
#define SAX_COMPAT_MODE BAD_CAST "SAX compatibility mode document"
174
175
/**
176
 * XML_PARSER_CHUNK_SIZE
177
 *
178
 * When calling GROW that's the minimal amount of data
179
 * the parser expected to have received. It is not a hard
180
 * limit but an optimization when reading strings like Names
181
 * It is not strictly needed as long as inputs available characters
182
 * are followed by 0, which should be provided by the I/O level
183
 */
184
#define XML_PARSER_CHUNK_SIZE 100
185
186
/**
187
 * Constant string describing the version of the library used at
188
 * run-time.
189
 */
190
const char *const
191
xmlParserVersion = LIBXML_VERSION_STRING LIBXML_VERSION_EXTRA;
192
193
/*
194
 * List of XML prefixed PI allowed by W3C specs
195
 */
196
197
static const char* const xmlW3CPIs[] = {
198
    "xml-stylesheet",
199
    "xml-model",
200
    NULL
201
};
202
203
204
/* DEPR void xmlParserHandleReference(xmlParserCtxtPtr ctxt); */
205
static xmlEntityPtr xmlParseStringPEReference(xmlParserCtxtPtr ctxt,
206
                                              const xmlChar **str);
207
208
static void
209
xmlCtxtParseEntity(xmlParserCtxtPtr ctxt, xmlEntityPtr ent);
210
211
static int
212
xmlLoadEntityContent(xmlParserCtxtPtr ctxt, xmlEntityPtr entity);
213
214
static void
215
xmlParsePERefInternal(xmlParserCtxt *ctxt, int markupDecl);
216
217
/************************************************************************
218
 *                  *
219
 *    Some factorized error routines        *
220
 *                  *
221
 ************************************************************************/
222
223
static void
224
0
xmlErrMemory(xmlParserCtxtPtr ctxt) {
225
0
    xmlCtxtErrMemory(ctxt);
226
0
}
227
228
/**
229
 * Handle a redefinition of attribute error
230
 *
231
 * @param ctxt  an XML parser context
232
 * @param prefix  the attribute prefix
233
 * @param localname  the attribute localname
234
 */
235
static void
236
xmlErrAttributeDup(xmlParserCtxtPtr ctxt, const xmlChar * prefix,
237
                   const xmlChar * localname)
238
0
{
239
0
    if (prefix == NULL)
240
0
        xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, XML_ERR_ATTRIBUTE_REDEFINED,
241
0
                   XML_ERR_FATAL, localname, NULL, NULL, 0,
242
0
                   "Attribute %s redefined\n", localname);
243
0
    else
244
0
        xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, XML_ERR_ATTRIBUTE_REDEFINED,
245
0
                   XML_ERR_FATAL, prefix, localname, NULL, 0,
246
0
                   "Attribute %s:%s redefined\n", prefix, localname);
247
0
}
248
249
/**
250
 * Handle a fatal parser error, i.e. violating Well-Formedness constraints
251
 *
252
 * @param ctxt  an XML parser context
253
 * @param error  the error number
254
 * @param msg  the error message
255
 */
256
static void LIBXML_ATTR_FORMAT(3,0)
257
xmlFatalErrMsg(xmlParserCtxtPtr ctxt, xmlParserErrors error,
258
               const char *msg)
259
0
{
260
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, error, XML_ERR_FATAL,
261
0
               NULL, NULL, NULL, 0, "%s", msg);
262
0
}
263
264
/**
265
 * Handle a warning.
266
 *
267
 * @param ctxt  an XML parser context
268
 * @param error  the error number
269
 * @param msg  the error message
270
 * @param str1  extra data
271
 * @param str2  extra data
272
 */
273
void LIBXML_ATTR_FORMAT(3,0)
274
xmlWarningMsg(xmlParserCtxtPtr ctxt, xmlParserErrors error,
275
              const char *msg, const xmlChar *str1, const xmlChar *str2)
276
0
{
277
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, error, XML_ERR_WARNING,
278
0
               str1, str2, NULL, 0, msg, str1, str2);
279
0
}
280
281
#ifdef LIBXML_VALID_ENABLED
282
/**
283
 * Handle a validity error.
284
 *
285
 * @param ctxt  an XML parser context
286
 * @param error  the error number
287
 * @param msg  the error message
288
 * @param str1  extra data
289
 * @param str2  extra data
290
 */
291
static void LIBXML_ATTR_FORMAT(3,0)
292
xmlValidityError(xmlParserCtxtPtr ctxt, xmlParserErrors error,
293
              const char *msg, const xmlChar *str1, const xmlChar *str2)
294
0
{
295
0
    ctxt->valid = 0;
296
297
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_DTD, error, XML_ERR_ERROR,
298
0
               str1, str2, NULL, 0, msg, str1, str2);
299
0
}
300
#endif
301
302
/**
303
 * Handle a fatal parser error, i.e. violating Well-Formedness constraints
304
 *
305
 * @param ctxt  an XML parser context
306
 * @param error  the error number
307
 * @param msg  the error message
308
 * @param val  an integer value
309
 */
310
static void LIBXML_ATTR_FORMAT(3,0)
311
xmlFatalErrMsgInt(xmlParserCtxtPtr ctxt, xmlParserErrors error,
312
                  const char *msg, int val)
313
0
{
314
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, error, XML_ERR_FATAL,
315
0
               NULL, NULL, NULL, val, msg, val);
316
0
}
317
318
/**
319
 * Handle a fatal parser error, i.e. violating Well-Formedness constraints
320
 *
321
 * @param ctxt  an XML parser context
322
 * @param error  the error number
323
 * @param msg  the error message
324
 * @param str1  an string info
325
 * @param val  an integer value
326
 * @param str2  an string info
327
 */
328
static void LIBXML_ATTR_FORMAT(3,0)
329
xmlFatalErrMsgStrIntStr(xmlParserCtxtPtr ctxt, xmlParserErrors error,
330
                  const char *msg, const xmlChar *str1, int val,
331
      const xmlChar *str2)
332
0
{
333
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, error, XML_ERR_FATAL,
334
0
               str1, str2, NULL, val, msg, str1, val, str2);
335
0
}
336
337
/**
338
 * Handle a fatal parser error, i.e. violating Well-Formedness constraints
339
 *
340
 * @param ctxt  an XML parser context
341
 * @param error  the error number
342
 * @param msg  the error message
343
 * @param val  a string value
344
 */
345
static void LIBXML_ATTR_FORMAT(3,0)
346
xmlFatalErrMsgStr(xmlParserCtxtPtr ctxt, xmlParserErrors error,
347
                  const char *msg, const xmlChar * val)
348
0
{
349
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, error, XML_ERR_FATAL,
350
0
               val, NULL, NULL, 0, msg, val);
351
0
}
352
353
/**
354
 * Handle a non fatal parser error
355
 *
356
 * @param ctxt  an XML parser context
357
 * @param error  the error number
358
 * @param msg  the error message
359
 * @param val  a string value
360
 */
361
static void LIBXML_ATTR_FORMAT(3,0)
362
xmlErrMsgStr(xmlParserCtxtPtr ctxt, xmlParserErrors error,
363
                  const char *msg, const xmlChar * val)
364
0
{
365
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_PARSER, error, XML_ERR_ERROR,
366
0
               val, NULL, NULL, 0, msg, val);
367
0
}
368
369
/**
370
 * Handle a fatal parser error, i.e. violating Well-Formedness constraints
371
 *
372
 * @param ctxt  an XML parser context
373
 * @param error  the error number
374
 * @param msg  the message
375
 * @param info1  extra information string
376
 * @param info2  extra information string
377
 * @param info3  extra information string
378
 */
379
static void LIBXML_ATTR_FORMAT(3,0)
380
xmlNsErr(xmlParserCtxtPtr ctxt, xmlParserErrors error,
381
         const char *msg,
382
         const xmlChar * info1, const xmlChar * info2,
383
         const xmlChar * info3)
384
0
{
385
0
    ctxt->nsWellFormed = 0;
386
387
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_NAMESPACE, error, XML_ERR_ERROR,
388
0
               info1, info2, info3, 0, msg, info1, info2, info3);
389
0
}
390
391
/**
392
 * Handle a namespace warning error
393
 *
394
 * @param ctxt  an XML parser context
395
 * @param error  the error number
396
 * @param msg  the message
397
 * @param info1  extra information string
398
 * @param info2  extra information string
399
 * @param info3  extra information string
400
 */
401
static void LIBXML_ATTR_FORMAT(3,0)
402
xmlNsWarn(xmlParserCtxtPtr ctxt, xmlParserErrors error,
403
         const char *msg,
404
         const xmlChar * info1, const xmlChar * info2,
405
         const xmlChar * info3)
406
0
{
407
0
    xmlCtxtErr(ctxt, NULL, XML_FROM_NAMESPACE, error, XML_ERR_WARNING,
408
0
               info1, info2, info3, 0, msg, info1, info2, info3);
409
0
}
410
411
/**
412
 * Check for non-linear entity expansion behaviour.
413
 *
414
 * In some cases like xmlExpandEntityInAttValue, this function is called
415
 * for each, possibly nested entity and its unexpanded content length.
416
 *
417
 * In other cases like #xmlParseReference, it's only called for each
418
 * top-level entity with its unexpanded content length plus the sum of
419
 * the unexpanded content lengths (plus fixed cost) of all nested
420
 * entities.
421
 *
422
 * Summing the unexpanded lengths also adds the length of the reference.
423
 * This is by design. Taking the length of the entity name into account
424
 * discourages attacks that try to waste CPU time with abusively long
425
 * entity names. See test/recurse/lol6.xml for example. Each call also
426
 * adds some fixed cost XML_ENT_FIXED_COST to discourage attacks with
427
 * short entities.
428
 *
429
 * @param ctxt  parser context
430
 * @param extra  sum of unexpanded entity sizes
431
 * @returns 1 on error, 0 on success.
432
 */
433
static int
434
xmlParserEntityCheck(xmlParserCtxtPtr ctxt, unsigned long extra)
435
0
{
436
0
    unsigned long consumed;
437
0
    unsigned long *expandedSize;
438
0
    xmlParserInputPtr input = ctxt->input;
439
0
    xmlEntityPtr entity = input->entity;
440
441
0
    if ((entity) && (entity->flags & XML_ENT_CHECKED))
442
0
        return(0);
443
444
    /*
445
     * Compute total consumed bytes so far, including input streams of
446
     * external entities.
447
     */
448
0
    consumed = input->consumed;
449
0
    xmlSaturatedAddSizeT(&consumed, input->cur - input->base);
450
0
    xmlSaturatedAdd(&consumed, ctxt->sizeentities);
451
452
0
    if (entity)
453
0
        expandedSize = &entity->expandedSize;
454
0
    else
455
0
        expandedSize = &ctxt->sizeentcopy;
456
457
    /*
458
     * Add extra cost and some fixed cost.
459
     */
460
0
    xmlSaturatedAdd(expandedSize, extra);
461
0
    xmlSaturatedAdd(expandedSize, XML_ENT_FIXED_COST);
462
463
    /*
464
     * It's important to always use saturation arithmetic when tracking
465
     * entity sizes to make the size checks reliable. If "sizeentcopy"
466
     * overflows, we have to abort.
467
     */
468
0
    if ((ctxt->maxAmpl > 0) &&
469
0
        (*expandedSize > XML_PARSER_ALLOWED_EXPANSION) &&
470
0
        ((*expandedSize >= ULONG_MAX) ||
471
0
         (*expandedSize / ctxt->maxAmpl > consumed))) {
472
0
        xmlFatalErrMsg(ctxt, XML_ERR_RESOURCE_LIMIT,
473
0
                       "Maximum entity amplification factor exceeded, see "
474
0
                       "xmlCtxtSetMaxAmplification.\n");
475
0
        return(1);
476
0
    }
477
478
0
    return(0);
479
0
}
480
481
/************************************************************************
482
 *                  *
483
 *    Library wide options          *
484
 *                  *
485
 ************************************************************************/
486
487
/**
488
 * Examines if the library has been compiled with a given feature.
489
 *
490
 * @param feature  the feature to be examined
491
 * @returns zero (0) if the feature does not exist or an unknown
492
 * feature is requested, non-zero otherwise.
493
 */
494
int
495
xmlHasFeature(xmlFeature feature)
496
0
{
497
0
    switch (feature) {
498
0
  case XML_WITH_THREAD:
499
0
#ifdef LIBXML_THREAD_ENABLED
500
0
      return(1);
501
#else
502
      return(0);
503
#endif
504
0
        case XML_WITH_TREE:
505
0
            return(1);
506
0
        case XML_WITH_OUTPUT:
507
0
#ifdef LIBXML_OUTPUT_ENABLED
508
0
            return(1);
509
#else
510
            return(0);
511
#endif
512
0
        case XML_WITH_PUSH:
513
0
#ifdef LIBXML_PUSH_ENABLED
514
0
            return(1);
515
#else
516
            return(0);
517
#endif
518
0
        case XML_WITH_READER:
519
0
#ifdef LIBXML_READER_ENABLED
520
0
            return(1);
521
#else
522
            return(0);
523
#endif
524
0
        case XML_WITH_PATTERN:
525
0
#ifdef LIBXML_PATTERN_ENABLED
526
0
            return(1);
527
#else
528
            return(0);
529
#endif
530
0
        case XML_WITH_WRITER:
531
0
#ifdef LIBXML_WRITER_ENABLED
532
0
            return(1);
533
#else
534
            return(0);
535
#endif
536
0
        case XML_WITH_SAX1:
537
0
#ifdef LIBXML_SAX1_ENABLED
538
0
            return(1);
539
#else
540
            return(0);
541
#endif
542
0
        case XML_WITH_HTTP:
543
0
            return(0);
544
0
        case XML_WITH_VALID:
545
0
#ifdef LIBXML_VALID_ENABLED
546
0
            return(1);
547
#else
548
            return(0);
549
#endif
550
0
        case XML_WITH_HTML:
551
0
#ifdef LIBXML_HTML_ENABLED
552
0
            return(1);
553
#else
554
            return(0);
555
#endif
556
0
        case XML_WITH_LEGACY:
557
0
            return(0);
558
0
        case XML_WITH_C14N:
559
0
#ifdef LIBXML_C14N_ENABLED
560
0
            return(1);
561
#else
562
            return(0);
563
#endif
564
0
        case XML_WITH_CATALOG:
565
0
#ifdef LIBXML_CATALOG_ENABLED
566
0
            return(1);
567
#else
568
            return(0);
569
#endif
570
0
        case XML_WITH_XPATH:
571
0
#ifdef LIBXML_XPATH_ENABLED
572
0
            return(1);
573
#else
574
            return(0);
575
#endif
576
0
        case XML_WITH_XPTR:
577
0
#ifdef LIBXML_XPTR_ENABLED
578
0
            return(1);
579
#else
580
            return(0);
581
#endif
582
0
        case XML_WITH_XINCLUDE:
583
0
#ifdef LIBXML_XINCLUDE_ENABLED
584
0
            return(1);
585
#else
586
            return(0);
587
#endif
588
0
        case XML_WITH_ICONV:
589
0
#ifdef LIBXML_ICONV_ENABLED
590
0
            return(1);
591
#else
592
            return(0);
593
#endif
594
0
        case XML_WITH_ISO8859X:
595
0
#ifdef LIBXML_ISO8859X_ENABLED
596
0
            return(1);
597
#else
598
            return(0);
599
#endif
600
0
        case XML_WITH_UNICODE:
601
0
            return(0);
602
0
        case XML_WITH_REGEXP:
603
0
#ifdef LIBXML_REGEXP_ENABLED
604
0
            return(1);
605
#else
606
            return(0);
607
#endif
608
0
        case XML_WITH_AUTOMATA:
609
0
#ifdef LIBXML_REGEXP_ENABLED
610
0
            return(1);
611
#else
612
            return(0);
613
#endif
614
0
        case XML_WITH_EXPR:
615
0
            return(0);
616
0
        case XML_WITH_RELAXNG:
617
0
#ifdef LIBXML_RELAXNG_ENABLED
618
0
            return(1);
619
#else
620
            return(0);
621
#endif
622
0
        case XML_WITH_SCHEMAS:
623
0
#ifdef LIBXML_SCHEMAS_ENABLED
624
0
            return(1);
625
#else
626
            return(0);
627
#endif
628
0
        case XML_WITH_SCHEMATRON:
629
#ifdef LIBXML_SCHEMATRON_ENABLED
630
            return(1);
631
#else
632
0
            return(0);
633
0
#endif
634
0
        case XML_WITH_MODULES:
635
0
#ifdef LIBXML_MODULES_ENABLED
636
0
            return(1);
637
#else
638
            return(0);
639
#endif
640
0
        case XML_WITH_DEBUG:
641
0
#ifdef LIBXML_DEBUG_ENABLED
642
0
            return(1);
643
#else
644
            return(0);
645
#endif
646
0
        case XML_WITH_DEBUG_MEM:
647
0
            return(0);
648
0
        case XML_WITH_ZLIB:
649
#ifdef LIBXML_ZLIB_ENABLED
650
            return(1);
651
#else
652
0
            return(0);
653
0
#endif
654
0
        case XML_WITH_LZMA:
655
0
            return(0);
656
0
        case XML_WITH_ICU:
657
#ifdef LIBXML_ICU_ENABLED
658
            return(1);
659
#else
660
0
            return(0);
661
0
#endif
662
0
        default:
663
0
      break;
664
0
     }
665
0
     return(0);
666
0
}
667
668
/************************************************************************
669
 *                  *
670
 *      Simple string buffer        *
671
 *                  *
672
 ************************************************************************/
673
674
typedef struct {
675
    xmlChar *mem;
676
    unsigned size;
677
    unsigned cap; /* size < cap */
678
    unsigned max; /* size <= max */
679
    xmlParserErrors code;
680
} xmlSBuf;
681
682
static void
683
0
xmlSBufInit(xmlSBuf *buf, unsigned max) {
684
0
    buf->mem = NULL;
685
0
    buf->size = 0;
686
0
    buf->cap = 0;
687
0
    buf->max = max;
688
0
    buf->code = XML_ERR_OK;
689
0
}
690
691
static int
692
0
xmlSBufGrow(xmlSBuf *buf, unsigned len) {
693
0
    xmlChar *mem;
694
0
    unsigned cap;
695
696
0
    if (len >= UINT_MAX / 2 - buf->size) {
697
0
        if (buf->code == XML_ERR_OK)
698
0
            buf->code = XML_ERR_RESOURCE_LIMIT;
699
0
        return(-1);
700
0
    }
701
702
0
    cap = (buf->size + len) * 2;
703
0
    if (cap < 240)
704
0
        cap = 240;
705
706
0
    mem = xmlRealloc(buf->mem, cap);
707
0
    if (mem == NULL) {
708
0
        buf->code = XML_ERR_NO_MEMORY;
709
0
        return(-1);
710
0
    }
711
712
0
    buf->mem = mem;
713
0
    buf->cap = cap;
714
715
0
    return(0);
716
0
}
717
718
static void
719
0
xmlSBufAddString(xmlSBuf *buf, const xmlChar *str, unsigned len) {
720
0
    if (buf->max - buf->size < len) {
721
0
        if (buf->code == XML_ERR_OK)
722
0
            buf->code = XML_ERR_RESOURCE_LIMIT;
723
0
        return;
724
0
    }
725
726
0
    if (buf->cap - buf->size <= len) {
727
0
        if (xmlSBufGrow(buf, len) < 0)
728
0
            return;
729
0
    }
730
731
0
    if (len > 0)
732
0
        memcpy(buf->mem + buf->size, str, len);
733
0
    buf->size += len;
734
0
}
735
736
static void
737
0
xmlSBufAddCString(xmlSBuf *buf, const char *str, unsigned len) {
738
0
    xmlSBufAddString(buf, (const xmlChar *) str, len);
739
0
}
740
741
static void
742
0
xmlSBufAddChar(xmlSBuf *buf, int c) {
743
0
    xmlChar *end;
744
745
0
    if (buf->max - buf->size < 4) {
746
0
        if (buf->code == XML_ERR_OK)
747
0
            buf->code = XML_ERR_RESOURCE_LIMIT;
748
0
        return;
749
0
    }
750
751
0
    if (buf->cap - buf->size <= 4) {
752
0
        if (xmlSBufGrow(buf, 4) < 0)
753
0
            return;
754
0
    }
755
756
0
    end = buf->mem + buf->size;
757
758
0
    if (c < 0x80) {
759
0
        *end = (xmlChar) c;
760
0
        buf->size += 1;
761
0
    } else {
762
0
        buf->size += xmlCopyCharMultiByte(end, c);
763
0
    }
764
0
}
765
766
static void
767
0
xmlSBufAddReplChar(xmlSBuf *buf) {
768
0
    xmlSBufAddCString(buf, "\xEF\xBF\xBD", 3);
769
0
}
770
771
static void
772
0
xmlSBufReportError(xmlSBuf *buf, xmlParserCtxtPtr ctxt, const char *errMsg) {
773
0
    if (buf->code == XML_ERR_NO_MEMORY)
774
0
        xmlCtxtErrMemory(ctxt);
775
0
    else
776
0
        xmlFatalErr(ctxt, buf->code, errMsg);
777
0
}
778
779
static xmlChar *
780
xmlSBufFinish(xmlSBuf *buf, int *sizeOut, xmlParserCtxtPtr ctxt,
781
0
              const char *errMsg) {
782
0
    if (buf->mem == NULL) {
783
0
        buf->mem = xmlMalloc(1);
784
0
        if (buf->mem == NULL) {
785
0
            buf->code = XML_ERR_NO_MEMORY;
786
0
        } else {
787
0
            buf->mem[0] = 0;
788
0
        }
789
0
    } else {
790
0
        buf->mem[buf->size] = 0;
791
0
    }
792
793
0
    if (buf->code == XML_ERR_OK) {
794
0
        if (sizeOut != NULL)
795
0
            *sizeOut = buf->size;
796
0
        return(buf->mem);
797
0
    }
798
799
0
    xmlSBufReportError(buf, ctxt, errMsg);
800
801
0
    xmlFree(buf->mem);
802
803
0
    if (sizeOut != NULL)
804
0
        *sizeOut = 0;
805
0
    return(NULL);
806
0
}
807
808
static void
809
0
xmlSBufCleanup(xmlSBuf *buf, xmlParserCtxtPtr ctxt, const char *errMsg) {
810
0
    if (buf->code != XML_ERR_OK)
811
0
        xmlSBufReportError(buf, ctxt, errMsg);
812
813
0
    xmlFree(buf->mem);
814
0
}
815
816
static int
817
xmlUTF8MultibyteLen(xmlParserCtxtPtr ctxt, const xmlChar *str,
818
0
                    const char *errMsg) {
819
0
    int c = str[0];
820
0
    int c1 = str[1];
821
822
0
    if ((c1 & 0xC0) != 0x80)
823
0
        goto encoding_error;
824
825
0
    if (c < 0xE0) {
826
        /* 2-byte sequence */
827
0
        if (c < 0xC2)
828
0
            goto encoding_error;
829
830
0
        return(2);
831
0
    } else {
832
0
        int c2 = str[2];
833
834
0
        if ((c2 & 0xC0) != 0x80)
835
0
            goto encoding_error;
836
837
0
        if (c < 0xF0) {
838
            /* 3-byte sequence */
839
0
            if (c == 0xE0) {
840
                /* overlong */
841
0
                if (c1 < 0xA0)
842
0
                    goto encoding_error;
843
0
            } else if (c == 0xED) {
844
                /* surrogate */
845
0
                if (c1 >= 0xA0)
846
0
                    goto encoding_error;
847
0
            } else if (c == 0xEF) {
848
                /* U+FFFE and U+FFFF are invalid Chars */
849
0
                if ((c1 == 0xBF) && (c2 >= 0xBE))
850
0
                    xmlFatalErrMsg(ctxt, XML_ERR_INVALID_CHAR, errMsg);
851
0
            }
852
853
0
            return(3);
854
0
        } else {
855
            /* 4-byte sequence */
856
0
            if ((str[3] & 0xC0) != 0x80)
857
0
                goto encoding_error;
858
0
            if (c == 0xF0) {
859
                /* overlong */
860
0
                if (c1 < 0x90)
861
0
                    goto encoding_error;
862
0
            } else if (c >= 0xF4) {
863
                /* greater than 0x10FFFF */
864
0
                if ((c > 0xF4) || (c1 >= 0x90))
865
0
                    goto encoding_error;
866
0
            }
867
868
0
            return(4);
869
0
        }
870
0
    }
871
872
0
encoding_error:
873
    /* Only report the first error */
874
0
    if ((ctxt->input->flags & XML_INPUT_ENCODING_ERROR) == 0) {
875
0
        xmlCtxtErrIO(ctxt, XML_ERR_INVALID_ENCODING, NULL);
876
0
        ctxt->input->flags |= XML_INPUT_ENCODING_ERROR;
877
0
    }
878
879
0
    return(0);
880
0
}
881
882
/************************************************************************
883
 *                  *
884
 *    SAX2 defaulted attributes handling      *
885
 *                  *
886
 ************************************************************************/
887
888
/**
889
 * Final initialization of the parser context before starting to parse.
890
 *
891
 * This accounts for users modifying struct members of parser context
892
 * directly.
893
 *
894
 * @param ctxt  an XML parser context
895
 */
896
static void
897
0
xmlCtxtInitializeLate(xmlParserCtxtPtr ctxt) {
898
0
    xmlSAXHandlerPtr sax;
899
900
    /* Avoid unused variable warning if features are disabled. */
901
0
    (void) sax;
902
903
    /*
904
     * Changing the SAX struct directly is still widespread practice
905
     * in internal and external code.
906
     */
907
0
    if (ctxt == NULL) return;
908
0
    sax = ctxt->sax;
909
0
#ifdef LIBXML_SAX1_ENABLED
910
    /*
911
     * Only enable SAX2 if there SAX2 element handlers, except when there
912
     * are no element handlers at all.
913
     */
914
0
    if (((ctxt->options & XML_PARSE_SAX1) == 0) &&
915
0
        (sax) &&
916
0
        (sax->initialized == XML_SAX2_MAGIC) &&
917
0
        ((sax->startElementNs != NULL) ||
918
0
         (sax->endElementNs != NULL) ||
919
0
         ((sax->startElement == NULL) && (sax->endElement == NULL))))
920
0
        ctxt->sax2 = 1;
921
#else
922
    ctxt->sax2 = 1;
923
#endif /* LIBXML_SAX1_ENABLED */
924
925
    /*
926
     * Some users replace the dictionary directly in the context struct.
927
     * We really need an API function to do that cleanly.
928
     */
929
0
    ctxt->str_xml = xmlDictLookup(ctxt->dict, BAD_CAST "xml", 3);
930
0
    ctxt->str_xmlns = xmlDictLookup(ctxt->dict, BAD_CAST "xmlns", 5);
931
0
    ctxt->str_xml_ns = xmlDictLookup(ctxt->dict, XML_XML_NAMESPACE, 36);
932
0
    if ((ctxt->str_xml==NULL) || (ctxt->str_xmlns==NULL) ||
933
0
    (ctxt->str_xml_ns == NULL)) {
934
0
        xmlErrMemory(ctxt);
935
0
    }
936
937
0
    xmlDictSetLimit(ctxt->dict,
938
0
                    (ctxt->options & XML_PARSE_HUGE) ?
939
0
                        0 :
940
0
                        XML_MAX_DICTIONARY_LIMIT);
941
942
0
#ifdef LIBXML_VALID_ENABLED
943
0
    if (ctxt->validate)
944
0
        ctxt->vctxt.flags |= XML_VCTXT_VALIDATE;
945
0
    else
946
0
        ctxt->vctxt.flags &= ~XML_VCTXT_VALIDATE;
947
0
#endif /* LIBXML_VALID_ENABLED */
948
0
}
949
950
typedef struct {
951
    xmlHashedString prefix;
952
    xmlHashedString name;
953
    xmlHashedString value;
954
    const xmlChar *valueEnd;
955
    int external;
956
    int expandedSize;
957
} xmlDefAttr;
958
959
typedef struct _xmlDefAttrs xmlDefAttrs;
960
typedef xmlDefAttrs *xmlDefAttrsPtr;
961
struct _xmlDefAttrs {
962
    int nbAttrs;  /* number of defaulted attributes on that element */
963
    int maxAttrs;       /* the size of the array */
964
#if __STDC_VERSION__ >= 199901L
965
    /* Using a C99 flexible array member avoids UBSan errors. */
966
    xmlDefAttr attrs[] ATTRIBUTE_COUNTED_BY(maxAttrs);
967
#else
968
    xmlDefAttr attrs[1];
969
#endif
970
};
971
972
/**
973
 * Normalize the space in non CDATA attribute values:
974
 * If the attribute type is not CDATA, then the XML processor MUST further
975
 * process the normalized attribute value by discarding any leading and
976
 * trailing space (\#x20) characters, and by replacing sequences of space
977
 * (\#x20) characters by a single space (\#x20) character.
978
 * Note that the size of dst need to be at least src, and if one doesn't need
979
 * to preserve dst (and it doesn't come from a dictionary or read-only) then
980
 * passing src as dst is just fine.
981
 *
982
 * @param src  the source string
983
 * @param dst  the target string
984
 * @returns a pointer to the normalized value (dst) or NULL if no conversion
985
 *         is needed.
986
 */
987
xmlChar *
988
xmlAttrNormalizeSpace(const xmlChar *src, xmlChar *dst)
989
0
{
990
0
    if ((src == NULL) || (dst == NULL))
991
0
        return(NULL);
992
993
0
    while (*src == 0x20) src++;
994
0
    while (*src != 0) {
995
0
  if (*src == 0x20) {
996
0
      while (*src == 0x20) src++;
997
0
      if (*src != 0)
998
0
    *dst++ = 0x20;
999
0
  } else {
1000
0
      *dst++ = *src++;
1001
0
  }
1002
0
    }
1003
0
    *dst = 0;
1004
0
    if (dst == src)
1005
0
       return(NULL);
1006
0
    return(dst);
1007
0
}
1008
1009
/**
1010
 * TODO: This function should also remove leading and trailing
1011
 *       whitespaces, and also group whitespaces together, but right
1012
 *       now the parse doesn't do that with XML_PARSE_NOENT, so this
1013
 *       replacement is consistent with the parser normalization
1014
 *
1015
 *       Maybe merge with xmlAttrNormalizeSpace, or do something
1016
 *       similar
1017
 *
1018
 * Normalize attritube entity values
1019
 * Replaces any space character with 0x20
1020
 * https://www.w3.org/TR/REC-xml/#AVNormalize
1021
 */
1022
xmlChar *
1023
xmlAttrNormalize(xmlChar *src)
1024
0
{
1025
0
    xmlChar *out = NULL;
1026
0
    xmlChar *dst = NULL;
1027
1028
0
    if (src == NULL)
1029
0
        return(NULL);
1030
1031
0
    out = src;
1032
0
    dst = out;
1033
0
    while (*src != 0) {
1034
0
        if (*src < 0x20) {
1035
0
            src++;
1036
0
            *dst++ = 0x20;
1037
0
        } else {
1038
0
            *dst++ = *src++;
1039
0
        }
1040
0
    }
1041
0
    *dst = 0;
1042
0
    return(out);
1043
0
}
1044
1045
/**
1046
 * Add a defaulted attribute for an element
1047
 *
1048
 * @param ctxt  an XML parser context
1049
 * @param fullname  the element fullname
1050
 * @param fullattr  the attribute fullname
1051
 * @param value  the attribute value
1052
 */
1053
static void
1054
xmlAddDefAttrs(xmlParserCtxtPtr ctxt,
1055
               const xmlChar *fullname,
1056
               const xmlChar *fullattr,
1057
0
               const xmlChar *value) {
1058
0
    xmlDefAttrsPtr defaults;
1059
0
    xmlDefAttr *attr;
1060
0
    int len, expandedSize;
1061
0
    xmlHashedString name;
1062
0
    xmlHashedString prefix;
1063
0
    xmlHashedString hvalue;
1064
0
    const xmlChar *localname;
1065
1066
    /*
1067
     * Allows to detect attribute redefinitions
1068
     */
1069
0
    if (ctxt->attsSpecial != NULL) {
1070
0
        if (xmlHashLookup2(ctxt->attsSpecial, fullname, fullattr) != NULL)
1071
0
      return;
1072
0
    }
1073
1074
0
    if (ctxt->attsDefault == NULL) {
1075
0
        ctxt->attsDefault = xmlHashCreateDict(10, ctxt->dict);
1076
0
  if (ctxt->attsDefault == NULL)
1077
0
      goto mem_error;
1078
0
    }
1079
1080
    /*
1081
     * split the element name into prefix:localname , the string found
1082
     * are within the DTD and then not associated to namespace names.
1083
     */
1084
0
    localname = xmlSplitQName3(fullname, &len);
1085
0
    if (localname == NULL) {
1086
0
        name = xmlDictLookupHashed(ctxt->dict, fullname, -1);
1087
0
  prefix.name = NULL;
1088
0
    } else {
1089
0
        name = xmlDictLookupHashed(ctxt->dict, localname, -1);
1090
0
  prefix = xmlDictLookupHashed(ctxt->dict, fullname, len);
1091
0
        if (prefix.name == NULL)
1092
0
            goto mem_error;
1093
0
    }
1094
0
    if (name.name == NULL)
1095
0
        goto mem_error;
1096
1097
    /*
1098
     * make sure there is some storage
1099
     */
1100
0
    defaults = xmlHashLookup2(ctxt->attsDefault, name.name, prefix.name);
1101
0
    if ((defaults == NULL) ||
1102
0
        (defaults->nbAttrs >= defaults->maxAttrs)) {
1103
0
        xmlDefAttrsPtr temp;
1104
0
        int newSize;
1105
1106
0
        if (defaults == NULL) {
1107
0
            newSize = 4;
1108
0
        } else {
1109
0
            if ((defaults->maxAttrs >= XML_MAX_ATTRS) ||
1110
0
                ((size_t) defaults->maxAttrs >
1111
0
                     SIZE_MAX / 2 / sizeof(temp[0]) - sizeof(*defaults)))
1112
0
                goto mem_error;
1113
1114
0
            if (defaults->maxAttrs > XML_MAX_ATTRS / 2)
1115
0
                newSize = XML_MAX_ATTRS;
1116
0
            else
1117
0
                newSize = defaults->maxAttrs * 2;
1118
0
        }
1119
0
        temp = xmlRealloc(defaults,
1120
0
                          sizeof(*defaults) + newSize * sizeof(xmlDefAttr));
1121
0
  if (temp == NULL)
1122
0
      goto mem_error;
1123
0
        if (defaults == NULL)
1124
0
            temp->nbAttrs = 0;
1125
0
  temp->maxAttrs = newSize;
1126
0
        defaults = temp;
1127
0
  if (xmlHashUpdateEntry2(ctxt->attsDefault, name.name, prefix.name,
1128
0
                          defaults, NULL) < 0) {
1129
0
      xmlFree(defaults);
1130
0
      goto mem_error;
1131
0
  }
1132
0
    }
1133
1134
    /*
1135
     * Split the attribute name into prefix:localname , the string found
1136
     * are within the DTD and hen not associated to namespace names.
1137
     */
1138
0
    localname = xmlSplitQName3(fullattr, &len);
1139
0
    if (localname == NULL) {
1140
0
        name = xmlDictLookupHashed(ctxt->dict, fullattr, -1);
1141
0
  prefix.name = NULL;
1142
0
    } else {
1143
0
        name = xmlDictLookupHashed(ctxt->dict, localname, -1);
1144
0
  prefix = xmlDictLookupHashed(ctxt->dict, fullattr, len);
1145
0
        if (prefix.name == NULL)
1146
0
            goto mem_error;
1147
0
    }
1148
0
    if (name.name == NULL)
1149
0
        goto mem_error;
1150
1151
    /* intern the string and precompute the end */
1152
0
    len = strlen((const char *) value);
1153
0
    hvalue = xmlDictLookupHashed(ctxt->dict, value, len);
1154
0
    if (hvalue.name == NULL)
1155
0
        goto mem_error;
1156
1157
0
    expandedSize = strlen((const char *) name.name);
1158
0
    if (prefix.name != NULL)
1159
0
        expandedSize += strlen((const char *) prefix.name);
1160
0
    expandedSize += len;
1161
1162
0
    attr = &defaults->attrs[defaults->nbAttrs++];
1163
0
    attr->name = name;
1164
0
    attr->prefix = prefix;
1165
0
    attr->value = hvalue;
1166
0
    attr->valueEnd = hvalue.name + len;
1167
0
    attr->external = PARSER_EXTERNAL(ctxt);
1168
0
    attr->expandedSize = expandedSize;
1169
1170
0
    return;
1171
1172
0
mem_error:
1173
0
    xmlErrMemory(ctxt);
1174
0
}
1175
1176
/**
1177
 * Register this attribute type
1178
 *
1179
 * @param ctxt  an XML parser context
1180
 * @param fullname  the element fullname
1181
 * @param fullattr  the attribute fullname
1182
 * @param type  the attribute type
1183
 */
1184
static void
1185
xmlAddSpecialAttr(xmlParserCtxtPtr ctxt,
1186
      const xmlChar *fullname,
1187
      const xmlChar *fullattr,
1188
      int type)
1189
0
{
1190
0
    if (ctxt->attsSpecial == NULL) {
1191
0
        ctxt->attsSpecial = xmlHashCreateDict(10, ctxt->dict);
1192
0
  if (ctxt->attsSpecial == NULL)
1193
0
      goto mem_error;
1194
0
    }
1195
1196
0
    if (PARSER_EXTERNAL(ctxt))
1197
0
        type |= XML_SPECIAL_EXTERNAL;
1198
1199
0
    if (xmlHashAdd2(ctxt->attsSpecial, fullname, fullattr,
1200
0
                    XML_INT_TO_PTR(type)) < 0)
1201
0
        goto mem_error;
1202
0
    return;
1203
1204
0
mem_error:
1205
0
    xmlErrMemory(ctxt);
1206
0
}
1207
1208
/**
1209
 * Removes CDATA attributes from the special attribute table
1210
 */
1211
static void
1212
xmlCleanSpecialAttrCallback(void *payload, void *data,
1213
                            const xmlChar *fullname, const xmlChar *fullattr,
1214
0
                            const xmlChar *unused ATTRIBUTE_UNUSED) {
1215
0
    xmlParserCtxtPtr ctxt = (xmlParserCtxtPtr) data;
1216
1217
0
    if (XML_PTR_TO_INT(payload) == XML_ATTRIBUTE_CDATA) {
1218
0
        xmlHashRemoveEntry2(ctxt->attsSpecial, fullname, fullattr, NULL);
1219
0
    }
1220
0
}
1221
1222
/**
1223
 * Trim the list of attributes defined to remove all those of type
1224
 * CDATA as they are not special. This call should be done when finishing
1225
 * to parse the DTD and before starting to parse the document root.
1226
 *
1227
 * @param ctxt  an XML parser context
1228
 */
1229
static void
1230
xmlCleanSpecialAttr(xmlParserCtxtPtr ctxt)
1231
0
{
1232
0
    if (ctxt->attsSpecial == NULL)
1233
0
        return;
1234
1235
0
    xmlHashScanFull(ctxt->attsSpecial, xmlCleanSpecialAttrCallback, ctxt);
1236
1237
0
    if (xmlHashSize(ctxt->attsSpecial) == 0) {
1238
0
        xmlHashFree(ctxt->attsSpecial, NULL);
1239
0
        ctxt->attsSpecial = NULL;
1240
0
    }
1241
0
}
1242
1243
/**
1244
 * Checks that the value conforms to the LanguageID production:
1245
 *
1246
 * @deprecated Internal function, do not use.
1247
 *
1248
 * NOTE: this is somewhat deprecated, those productions were removed from
1249
 * the XML Second edition.
1250
 *
1251
 *     [33] LanguageID ::= Langcode ('-' Subcode)*
1252
 *     [34] Langcode ::= ISO639Code |  IanaCode |  UserCode
1253
 *     [35] ISO639Code ::= ([a-z] | [A-Z]) ([a-z] | [A-Z])
1254
 *     [36] IanaCode ::= ('i' | 'I') '-' ([a-z] | [A-Z])+
1255
 *     [37] UserCode ::= ('x' | 'X') '-' ([a-z] | [A-Z])+
1256
 *     [38] Subcode ::= ([a-z] | [A-Z])+
1257
 *
1258
 * The current REC reference the successors of RFC 1766, currently 5646
1259
 *
1260
 * http://www.rfc-editor.org/rfc/rfc5646.txt
1261
 *
1262
 *     langtag       = language
1263
 *                     ["-" script]
1264
 *                     ["-" region]
1265
 *                     *("-" variant)
1266
 *                     *("-" extension)
1267
 *                     ["-" privateuse]
1268
 *     language      = 2*3ALPHA            ; shortest ISO 639 code
1269
 *                     ["-" extlang]       ; sometimes followed by
1270
 *                                         ; extended language subtags
1271
 *                   / 4ALPHA              ; or reserved for future use
1272
 *                   / 5*8ALPHA            ; or registered language subtag
1273
 *
1274
 *     extlang       = 3ALPHA              ; selected ISO 639 codes
1275
 *                     *2("-" 3ALPHA)      ; permanently reserved
1276
 *
1277
 *     script        = 4ALPHA              ; ISO 15924 code
1278
 *
1279
 *     region        = 2ALPHA              ; ISO 3166-1 code
1280
 *                   / 3DIGIT              ; UN M.49 code
1281
 *
1282
 *     variant       = 5*8alphanum         ; registered variants
1283
 *                   / (DIGIT 3alphanum)
1284
 *
1285
 *     extension     = singleton 1*("-" (2*8alphanum))
1286
 *
1287
 *                                         ; Single alphanumerics
1288
 *                                         ; "x" reserved for private use
1289
 *     singleton     = DIGIT               ; 0 - 9
1290
 *                   / %x41-57             ; A - W
1291
 *                   / %x59-5A             ; Y - Z
1292
 *                   / %x61-77             ; a - w
1293
 *                   / %x79-7A             ; y - z
1294
 *
1295
 * it sounds right to still allow Irregular i-xxx IANA and user codes too
1296
 * The parser below doesn't try to cope with extension or privateuse
1297
 * that could be added but that's not interoperable anyway
1298
 *
1299
 * @param lang  pointer to the string value
1300
 * @returns 1 if correct 0 otherwise
1301
 **/
1302
int
1303
xmlCheckLanguageID(const xmlChar * lang)
1304
0
{
1305
0
    const xmlChar *cur = lang, *nxt;
1306
1307
0
    if (cur == NULL)
1308
0
        return (0);
1309
0
    if (((cur[0] == 'i') && (cur[1] == '-')) ||
1310
0
        ((cur[0] == 'I') && (cur[1] == '-')) ||
1311
0
        ((cur[0] == 'x') && (cur[1] == '-')) ||
1312
0
        ((cur[0] == 'X') && (cur[1] == '-'))) {
1313
        /*
1314
         * Still allow IANA code and user code which were coming
1315
         * from the previous version of the XML-1.0 specification
1316
         * it's deprecated but we should not fail
1317
         */
1318
0
        cur += 2;
1319
0
        while (((cur[0] >= 'A') && (cur[0] <= 'Z')) ||
1320
0
               ((cur[0] >= 'a') && (cur[0] <= 'z')))
1321
0
            cur++;
1322
0
        return(cur[0] == 0);
1323
0
    }
1324
0
    nxt = cur;
1325
0
    while (((nxt[0] >= 'A') && (nxt[0] <= 'Z')) ||
1326
0
           ((nxt[0] >= 'a') && (nxt[0] <= 'z')))
1327
0
           nxt++;
1328
0
    if (nxt - cur >= 4) {
1329
        /*
1330
         * Reserved
1331
         */
1332
0
        if ((nxt - cur > 8) || (nxt[0] != 0))
1333
0
            return(0);
1334
0
        return(1);
1335
0
    }
1336
0
    if (nxt - cur < 2)
1337
0
        return(0);
1338
    /* we got an ISO 639 code */
1339
0
    if (nxt[0] == 0)
1340
0
        return(1);
1341
0
    if (nxt[0] != '-')
1342
0
        return(0);
1343
1344
0
    nxt++;
1345
0
    cur = nxt;
1346
    /* now we can have extlang or script or region or variant */
1347
0
    if ((nxt[0] >= '0') && (nxt[0] <= '9'))
1348
0
        goto region_m49;
1349
1350
0
    while (((nxt[0] >= 'A') && (nxt[0] <= 'Z')) ||
1351
0
           ((nxt[0] >= 'a') && (nxt[0] <= 'z')))
1352
0
           nxt++;
1353
0
    if (nxt - cur == 4)
1354
0
        goto script;
1355
0
    if (nxt - cur == 2)
1356
0
        goto region;
1357
0
    if ((nxt - cur >= 5) && (nxt - cur <= 8))
1358
0
        goto variant;
1359
0
    if (nxt - cur != 3)
1360
0
        return(0);
1361
    /* we parsed an extlang */
1362
0
    if (nxt[0] == 0)
1363
0
        return(1);
1364
0
    if (nxt[0] != '-')
1365
0
        return(0);
1366
1367
0
    nxt++;
1368
0
    cur = nxt;
1369
    /* now we can have script or region or variant */
1370
0
    if ((nxt[0] >= '0') && (nxt[0] <= '9'))
1371
0
        goto region_m49;
1372
1373
0
    while (((nxt[0] >= 'A') && (nxt[0] <= 'Z')) ||
1374
0
           ((nxt[0] >= 'a') && (nxt[0] <= 'z')))
1375
0
           nxt++;
1376
0
    if (nxt - cur == 2)
1377
0
        goto region;
1378
0
    if ((nxt - cur >= 5) && (nxt - cur <= 8))
1379
0
        goto variant;
1380
0
    if (nxt - cur != 4)
1381
0
        return(0);
1382
    /* we parsed a script */
1383
0
script:
1384
0
    if (nxt[0] == 0)
1385
0
        return(1);
1386
0
    if (nxt[0] != '-')
1387
0
        return(0);
1388
1389
0
    nxt++;
1390
0
    cur = nxt;
1391
    /* now we can have region or variant */
1392
0
    if ((nxt[0] >= '0') && (nxt[0] <= '9'))
1393
0
        goto region_m49;
1394
1395
0
    while (((nxt[0] >= 'A') && (nxt[0] <= 'Z')) ||
1396
0
           ((nxt[0] >= 'a') && (nxt[0] <= 'z')))
1397
0
           nxt++;
1398
1399
0
    if ((nxt - cur >= 5) && (nxt - cur <= 8))
1400
0
        goto variant;
1401
0
    if (nxt - cur != 2)
1402
0
        return(0);
1403
    /* we parsed a region */
1404
0
region:
1405
0
    if (nxt[0] == 0)
1406
0
        return(1);
1407
0
    if (nxt[0] != '-')
1408
0
        return(0);
1409
1410
0
    nxt++;
1411
0
    cur = nxt;
1412
    /* now we can just have a variant */
1413
0
    while (((nxt[0] >= 'A') && (nxt[0] <= 'Z')) ||
1414
0
           ((nxt[0] >= 'a') && (nxt[0] <= 'z')))
1415
0
           nxt++;
1416
1417
0
    if ((nxt - cur < 5) || (nxt - cur > 8))
1418
0
        return(0);
1419
1420
    /* we parsed a variant */
1421
0
variant:
1422
0
    if (nxt[0] == 0)
1423
0
        return(1);
1424
0
    if (nxt[0] != '-')
1425
0
        return(0);
1426
    /* extensions and private use subtags not checked */
1427
0
    return (1);
1428
1429
0
region_m49:
1430
0
    if (((nxt[1] >= '0') && (nxt[1] <= '9')) &&
1431
0
        ((nxt[2] >= '0') && (nxt[2] <= '9'))) {
1432
0
        nxt += 3;
1433
0
        goto region;
1434
0
    }
1435
0
    return(0);
1436
0
}
1437
1438
/************************************************************************
1439
 *                  *
1440
 *    Parser stacks related functions and macros    *
1441
 *                  *
1442
 ************************************************************************/
1443
1444
static xmlChar *
1445
xmlParseStringEntityRef(xmlParserCtxtPtr ctxt, const xmlChar **str);
1446
1447
/**
1448
 * Create a new namespace database.
1449
 *
1450
 * @returns the new obejct.
1451
 */
1452
xmlParserNsData *
1453
0
xmlParserNsCreate(void) {
1454
0
    xmlParserNsData *nsdb = xmlMalloc(sizeof(*nsdb));
1455
1456
0
    if (nsdb == NULL)
1457
0
        return(NULL);
1458
0
    memset(nsdb, 0, sizeof(*nsdb));
1459
0
    nsdb->defaultNsIndex = INT_MAX;
1460
1461
0
    return(nsdb);
1462
0
}
1463
1464
/**
1465
 * Free a namespace database.
1466
 *
1467
 * @param nsdb  namespace database
1468
 */
1469
void
1470
0
xmlParserNsFree(xmlParserNsData *nsdb) {
1471
0
    if (nsdb == NULL)
1472
0
        return;
1473
1474
0
    xmlFree(nsdb->extra);
1475
0
    xmlFree(nsdb->hash);
1476
0
    xmlFree(nsdb);
1477
0
}
1478
1479
/**
1480
 * Reset a namespace database.
1481
 *
1482
 * @param nsdb  namespace database
1483
 */
1484
static void
1485
0
xmlParserNsReset(xmlParserNsData *nsdb) {
1486
0
    if (nsdb == NULL)
1487
0
        return;
1488
1489
0
    nsdb->hashElems = 0;
1490
0
    nsdb->elementId = 0;
1491
0
    nsdb->defaultNsIndex = INT_MAX;
1492
1493
0
    if (nsdb->hash)
1494
0
        memset(nsdb->hash, 0, nsdb->hashSize * sizeof(nsdb->hash[0]));
1495
0
}
1496
1497
/**
1498
 * Signal that a new element has started.
1499
 *
1500
 * @param nsdb  namespace database
1501
 * @returns 0 on success, -1 if the element counter overflowed.
1502
 */
1503
static int
1504
0
xmlParserNsStartElement(xmlParserNsData *nsdb) {
1505
0
    if (nsdb->elementId == UINT_MAX)
1506
0
        return(-1);
1507
0
    nsdb->elementId++;
1508
1509
0
    return(0);
1510
0
}
1511
1512
/**
1513
 * Lookup namespace with given prefix. If `bucketPtr` is non-NULL, it will
1514
 * be set to the matching bucket, or the first empty bucket if no match
1515
 * was found.
1516
 *
1517
 * @param ctxt  parser context
1518
 * @param prefix  namespace prefix
1519
 * @param bucketPtr  optional bucket (return value)
1520
 * @returns the namespace index on success, INT_MAX if no namespace was
1521
 * found.
1522
 */
1523
static int
1524
xmlParserNsLookup(xmlParserCtxtPtr ctxt, const xmlHashedString *prefix,
1525
0
                  xmlParserNsBucket **bucketPtr) {
1526
0
    xmlParserNsBucket *bucket, *tombstone;
1527
0
    unsigned index, hashValue;
1528
1529
0
    if (prefix->name == NULL)
1530
0
        return(ctxt->nsdb->defaultNsIndex);
1531
1532
0
    if (ctxt->nsdb->hashSize == 0)
1533
0
        return(INT_MAX);
1534
1535
0
    hashValue = prefix->hashValue;
1536
0
    index = hashValue & (ctxt->nsdb->hashSize - 1);
1537
0
    bucket = &ctxt->nsdb->hash[index];
1538
0
    tombstone = NULL;
1539
1540
0
    while (bucket->hashValue) {
1541
0
        if (bucket->index == INT_MAX) {
1542
0
            if (tombstone == NULL)
1543
0
                tombstone = bucket;
1544
0
        } else if (bucket->hashValue == hashValue) {
1545
0
            if (ctxt->nsTab[bucket->index * 2] == prefix->name) {
1546
0
                if (bucketPtr != NULL)
1547
0
                    *bucketPtr = bucket;
1548
0
                return(bucket->index);
1549
0
            }
1550
0
        }
1551
1552
0
        index++;
1553
0
        bucket++;
1554
0
        if (index == ctxt->nsdb->hashSize) {
1555
0
            index = 0;
1556
0
            bucket = ctxt->nsdb->hash;
1557
0
        }
1558
0
    }
1559
1560
0
    if (bucketPtr != NULL)
1561
0
        *bucketPtr = tombstone ? tombstone : bucket;
1562
0
    return(INT_MAX);
1563
0
}
1564
1565
/**
1566
 * Lookup namespace URI with given prefix.
1567
 *
1568
 * @param ctxt  parser context
1569
 * @param prefix  namespace prefix
1570
 * @returns the namespace URI on success, NULL if no namespace was found.
1571
 */
1572
static const xmlChar *
1573
0
xmlParserNsLookupUri(xmlParserCtxtPtr ctxt, const xmlHashedString *prefix) {
1574
0
    const xmlChar *ret;
1575
0
    int nsIndex;
1576
1577
0
    if (prefix->name == ctxt->str_xml)
1578
0
        return(ctxt->str_xml_ns);
1579
1580
    /*
1581
     * minNsIndex is used when building an entity tree. We must
1582
     * ignore namespaces declared outside the entity.
1583
     */
1584
0
    nsIndex = xmlParserNsLookup(ctxt, prefix, NULL);
1585
0
    if ((nsIndex == INT_MAX) || (nsIndex < ctxt->nsdb->minNsIndex))
1586
0
        return(NULL);
1587
1588
0
    ret = ctxt->nsTab[nsIndex * 2 + 1];
1589
0
    if (ret[0] == 0)
1590
0
        ret = NULL;
1591
0
    return(ret);
1592
0
}
1593
1594
/**
1595
 * Lookup extra data for the given prefix. This returns data stored
1596
 * with xmlParserNsUdpateSax.
1597
 *
1598
 * @param ctxt  parser context
1599
 * @param prefix  namespace prefix
1600
 * @returns the data on success, NULL if no namespace was found.
1601
 */
1602
void *
1603
0
xmlParserNsLookupSax(xmlParserCtxt *ctxt, const xmlChar *prefix) {
1604
0
    xmlHashedString hprefix;
1605
0
    int nsIndex;
1606
1607
0
    if (prefix == ctxt->str_xml)
1608
0
        return(NULL);
1609
1610
0
    hprefix.name = prefix;
1611
0
    if (prefix != NULL)
1612
0
        hprefix.hashValue = xmlDictComputeHash(ctxt->dict, prefix);
1613
0
    else
1614
0
        hprefix.hashValue = 0;
1615
0
    nsIndex = xmlParserNsLookup(ctxt, &hprefix, NULL);
1616
0
    if ((nsIndex == INT_MAX) || (nsIndex < ctxt->nsdb->minNsIndex))
1617
0
        return(NULL);
1618
1619
0
    return(ctxt->nsdb->extra[nsIndex].saxData);
1620
0
}
1621
1622
/**
1623
 * Sets or updates extra data for the given prefix. This value will be
1624
 * returned by xmlParserNsLookupSax as long as the namespace with the
1625
 * given prefix is in scope.
1626
 *
1627
 * @param ctxt  parser context
1628
 * @param prefix  namespace prefix
1629
 * @param saxData  extra data for SAX handler
1630
 * @returns the data on success, NULL if no namespace was found.
1631
 */
1632
int
1633
xmlParserNsUpdateSax(xmlParserCtxt *ctxt, const xmlChar *prefix,
1634
0
                     void *saxData) {
1635
0
    xmlHashedString hprefix;
1636
0
    int nsIndex;
1637
1638
0
    if (prefix == ctxt->str_xml)
1639
0
        return(-1);
1640
1641
0
    hprefix.name = prefix;
1642
0
    if (prefix != NULL)
1643
0
        hprefix.hashValue = xmlDictComputeHash(ctxt->dict, prefix);
1644
0
    else
1645
0
        hprefix.hashValue = 0;
1646
0
    nsIndex = xmlParserNsLookup(ctxt, &hprefix, NULL);
1647
0
    if ((nsIndex == INT_MAX) || (nsIndex < ctxt->nsdb->minNsIndex))
1648
0
        return(-1);
1649
1650
0
    ctxt->nsdb->extra[nsIndex].saxData = saxData;
1651
0
    return(0);
1652
0
}
1653
1654
/**
1655
 * Grows the namespace tables.
1656
 *
1657
 * @param ctxt  parser context
1658
 * @returns 0 on success, -1 if a memory allocation failed.
1659
 */
1660
static int
1661
0
xmlParserNsGrow(xmlParserCtxtPtr ctxt) {
1662
0
    const xmlChar **table;
1663
0
    xmlParserNsExtra *extra;
1664
0
    int newSize;
1665
1666
0
    newSize = xmlGrowCapacity(ctxt->nsMax,
1667
0
                              sizeof(table[0]) + sizeof(extra[0]),
1668
0
                              16, XML_MAX_ITEMS);
1669
0
    if (newSize < 0)
1670
0
        goto error;
1671
1672
0
    table = xmlRealloc(ctxt->nsTab, 2 * newSize * sizeof(table[0]));
1673
0
    if (table == NULL)
1674
0
        goto error;
1675
0
    ctxt->nsTab = table;
1676
1677
0
    extra = xmlRealloc(ctxt->nsdb->extra, newSize * sizeof(extra[0]));
1678
0
    if (extra == NULL)
1679
0
        goto error;
1680
0
    ctxt->nsdb->extra = extra;
1681
1682
0
    ctxt->nsMax = newSize;
1683
0
    return(0);
1684
1685
0
error:
1686
0
    xmlErrMemory(ctxt);
1687
0
    return(-1);
1688
0
}
1689
1690
/**
1691
 * Push a new namespace on the table.
1692
 *
1693
 * @param ctxt  parser context
1694
 * @param prefix  prefix with hash value
1695
 * @param uri  uri with hash value
1696
 * @param saxData  extra data for SAX handler
1697
 * @param defAttr  whether the namespace comes from a default attribute
1698
 * @returns 1 if the namespace was pushed, 0 if the namespace was ignored,
1699
 * -1 if a memory allocation failed.
1700
 */
1701
static int
1702
xmlParserNsPush(xmlParserCtxtPtr ctxt, const xmlHashedString *prefix,
1703
0
                const xmlHashedString *uri, void *saxData, int defAttr) {
1704
0
    xmlParserNsBucket *bucket = NULL;
1705
0
    xmlParserNsExtra *extra;
1706
0
    const xmlChar **ns;
1707
0
    unsigned hashValue, nsIndex, oldIndex;
1708
1709
0
    if ((prefix != NULL) && (prefix->name == ctxt->str_xml))
1710
0
        return(0);
1711
1712
0
    if ((ctxt->nsNr >= ctxt->nsMax) && (xmlParserNsGrow(ctxt) < 0)) {
1713
0
        xmlErrMemory(ctxt);
1714
0
        return(-1);
1715
0
    }
1716
1717
    /*
1718
     * Default namespace and 'xml' namespace
1719
     */
1720
0
    if ((prefix == NULL) || (prefix->name == NULL)) {
1721
0
        oldIndex = ctxt->nsdb->defaultNsIndex;
1722
1723
0
        if (oldIndex != INT_MAX) {
1724
0
            extra = &ctxt->nsdb->extra[oldIndex];
1725
1726
0
            if (extra->elementId == ctxt->nsdb->elementId) {
1727
0
                if (defAttr == 0)
1728
0
                    xmlErrAttributeDup(ctxt, NULL, BAD_CAST "xmlns");
1729
0
                return(0);
1730
0
            }
1731
1732
0
            if ((ctxt->options & XML_PARSE_NSCLEAN) &&
1733
0
                (uri->name == ctxt->nsTab[oldIndex * 2 + 1]))
1734
0
                return(0);
1735
0
        }
1736
1737
0
        ctxt->nsdb->defaultNsIndex = ctxt->nsNr;
1738
0
        goto populate_entry;
1739
0
    }
1740
1741
    /*
1742
     * Hash table lookup
1743
     */
1744
0
    oldIndex = xmlParserNsLookup(ctxt, prefix, &bucket);
1745
0
    if (oldIndex != INT_MAX) {
1746
0
        extra = &ctxt->nsdb->extra[oldIndex];
1747
1748
        /*
1749
         * Check for duplicate definitions on the same element.
1750
         */
1751
0
        if (extra->elementId == ctxt->nsdb->elementId) {
1752
0
            if (defAttr == 0)
1753
0
                xmlErrAttributeDup(ctxt, BAD_CAST "xmlns", prefix->name);
1754
0
            return(0);
1755
0
        }
1756
1757
0
        if ((ctxt->options & XML_PARSE_NSCLEAN) &&
1758
0
            (uri->name == ctxt->nsTab[bucket->index * 2 + 1]))
1759
0
            return(0);
1760
1761
0
        bucket->index = ctxt->nsNr;
1762
0
        goto populate_entry;
1763
0
    }
1764
1765
    /*
1766
     * Insert new bucket
1767
     */
1768
1769
0
    hashValue = prefix->hashValue;
1770
1771
    /*
1772
     * Grow hash table, 50% fill factor
1773
     */
1774
0
    if (ctxt->nsdb->hashElems + 1 > ctxt->nsdb->hashSize / 2) {
1775
0
        xmlParserNsBucket *newHash;
1776
0
        unsigned newSize, i, index;
1777
1778
0
        if (ctxt->nsdb->hashSize > UINT_MAX / 2) {
1779
0
            xmlErrMemory(ctxt);
1780
0
            return(-1);
1781
0
        }
1782
0
        newSize = ctxt->nsdb->hashSize ? ctxt->nsdb->hashSize * 2 : 16;
1783
0
        newHash = xmlMalloc(newSize * sizeof(newHash[0]));
1784
0
        if (newHash == NULL) {
1785
0
            xmlErrMemory(ctxt);
1786
0
            return(-1);
1787
0
        }
1788
0
        memset(newHash, 0, newSize * sizeof(newHash[0]));
1789
1790
0
        for (i = 0; i < ctxt->nsdb->hashSize; i++) {
1791
0
            unsigned hv = ctxt->nsdb->hash[i].hashValue;
1792
0
            unsigned newIndex;
1793
1794
0
            if ((hv == 0) || (ctxt->nsdb->hash[i].index == INT_MAX))
1795
0
                continue;
1796
0
            newIndex = hv & (newSize - 1);
1797
1798
0
            while (newHash[newIndex].hashValue != 0) {
1799
0
                newIndex++;
1800
0
                if (newIndex == newSize)
1801
0
                    newIndex = 0;
1802
0
            }
1803
1804
0
            newHash[newIndex] = ctxt->nsdb->hash[i];
1805
0
        }
1806
1807
0
        xmlFree(ctxt->nsdb->hash);
1808
0
        ctxt->nsdb->hash = newHash;
1809
0
        ctxt->nsdb->hashSize = newSize;
1810
1811
        /*
1812
         * Relookup
1813
         */
1814
0
        index = hashValue & (newSize - 1);
1815
1816
0
        while (newHash[index].hashValue != 0) {
1817
0
            index++;
1818
0
            if (index == newSize)
1819
0
                index = 0;
1820
0
        }
1821
1822
0
        bucket = &newHash[index];
1823
0
    }
1824
1825
0
    bucket->hashValue = hashValue;
1826
0
    bucket->index = ctxt->nsNr;
1827
0
    ctxt->nsdb->hashElems++;
1828
0
    oldIndex = INT_MAX;
1829
1830
0
populate_entry:
1831
0
    nsIndex = ctxt->nsNr;
1832
1833
0
    ns = &ctxt->nsTab[nsIndex * 2];
1834
0
    ns[0] = prefix ? prefix->name : NULL;
1835
0
    ns[1] = uri->name;
1836
1837
0
    extra = &ctxt->nsdb->extra[nsIndex];
1838
0
    extra->saxData = saxData;
1839
0
    extra->prefixHashValue = prefix ? prefix->hashValue : 0;
1840
0
    extra->uriHashValue = uri->hashValue;
1841
0
    extra->elementId = ctxt->nsdb->elementId;
1842
0
    extra->oldIndex = oldIndex;
1843
1844
0
    ctxt->nsNr++;
1845
1846
0
    return(1);
1847
0
}
1848
1849
/**
1850
 * Pops the top `nr` namespaces and restores the hash table.
1851
 *
1852
 * @param ctxt  an XML parser context
1853
 * @param nr  the number to pop
1854
 * @returns the number of namespaces popped.
1855
 */
1856
static int
1857
xmlParserNsPop(xmlParserCtxtPtr ctxt, int nr)
1858
0
{
1859
0
    int i;
1860
1861
    /* assert(nr <= ctxt->nsNr); */
1862
1863
0
    for (i = ctxt->nsNr - 1; i >= ctxt->nsNr - nr; i--) {
1864
0
        const xmlChar *prefix = ctxt->nsTab[i * 2];
1865
0
        xmlParserNsExtra *extra = &ctxt->nsdb->extra[i];
1866
1867
0
        if (prefix == NULL) {
1868
0
            ctxt->nsdb->defaultNsIndex = extra->oldIndex;
1869
0
        } else {
1870
0
            xmlHashedString hprefix;
1871
0
            xmlParserNsBucket *bucket = NULL;
1872
1873
0
            hprefix.name = prefix;
1874
0
            hprefix.hashValue = extra->prefixHashValue;
1875
0
            xmlParserNsLookup(ctxt, &hprefix, &bucket);
1876
            /* assert(bucket && bucket->hashValue); */
1877
0
            bucket->index = extra->oldIndex;
1878
0
        }
1879
0
    }
1880
1881
0
    ctxt->nsNr -= nr;
1882
0
    return(nr);
1883
0
}
1884
1885
static int
1886
0
xmlCtxtGrowAttrs(xmlParserCtxtPtr ctxt) {
1887
0
    const xmlChar **atts;
1888
0
    unsigned *attallocs;
1889
0
    int newSize;
1890
1891
0
    newSize = xmlGrowCapacity(ctxt->maxatts / 5,
1892
0
                              sizeof(atts[0]) * 5 + sizeof(attallocs[0]),
1893
0
                              10, XML_MAX_ATTRS);
1894
0
    if (newSize < 0) {
1895
0
        xmlFatalErr(ctxt, XML_ERR_RESOURCE_LIMIT,
1896
0
                    "Maximum number of attributes exceeded");
1897
0
        return(-1);
1898
0
    }
1899
1900
0
    atts = xmlRealloc(ctxt->atts, newSize * sizeof(atts[0]) * 5);
1901
0
    if (atts == NULL)
1902
0
        goto mem_error;
1903
0
    ctxt->atts = atts;
1904
1905
0
    attallocs = xmlRealloc(ctxt->attallocs,
1906
0
                           newSize * sizeof(attallocs[0]));
1907
0
    if (attallocs == NULL)
1908
0
        goto mem_error;
1909
0
    ctxt->attallocs = attallocs;
1910
1911
0
    ctxt->maxatts = newSize * 5;
1912
1913
0
    return(0);
1914
1915
0
mem_error:
1916
0
    xmlErrMemory(ctxt);
1917
0
    return(-1);
1918
0
}
1919
1920
/**
1921
 * Pushes a new parser input on top of the input stack
1922
 *
1923
 * @param ctxt  an XML parser context
1924
 * @param value  the parser input
1925
 * @returns -1 in case of error, the index in the stack otherwise
1926
 */
1927
int
1928
xmlCtxtPushInput(xmlParserCtxt *ctxt, xmlParserInput *value)
1929
0
{
1930
0
    char *directory = NULL;
1931
0
    int maxDepth;
1932
1933
0
    if ((ctxt == NULL) || (value == NULL))
1934
0
        return(-1);
1935
1936
0
    maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 40 : 20;
1937
1938
0
    if (ctxt->inputNr >= ctxt->inputMax) {
1939
0
        xmlParserInputPtr *tmp;
1940
0
        int newSize;
1941
1942
0
        newSize = xmlGrowCapacity(ctxt->inputMax, sizeof(tmp[0]),
1943
0
                                  5, maxDepth);
1944
0
        if (newSize < 0) {
1945
0
            xmlFatalErrMsg(ctxt, XML_ERR_RESOURCE_LIMIT,
1946
0
                           "Maximum entity nesting depth exceeded");
1947
0
            return(-1);
1948
0
        }
1949
0
        tmp = xmlRealloc(ctxt->inputTab, newSize * sizeof(tmp[0]));
1950
0
        if (tmp == NULL) {
1951
0
            xmlErrMemory(ctxt);
1952
0
            return(-1);
1953
0
        }
1954
0
        ctxt->inputTab = tmp;
1955
0
        ctxt->inputMax = newSize;
1956
0
    }
1957
1958
0
    if ((ctxt->inputNr == 0) && (value->filename != NULL)) {
1959
0
        directory = xmlParserGetDirectory(value->filename);
1960
0
        if (directory == NULL) {
1961
0
            xmlErrMemory(ctxt);
1962
0
            return(-1);
1963
0
        }
1964
0
    }
1965
1966
0
    if (ctxt->input_id >= INT_MAX) {
1967
0
        xmlFatalErrMsg(ctxt, XML_ERR_RESOURCE_LIMIT, "Input ID overflow\n");
1968
0
        return(-1);
1969
0
    }
1970
1971
0
    ctxt->inputTab[ctxt->inputNr] = value;
1972
0
    ctxt->input = value;
1973
1974
0
    if (ctxt->inputNr == 0) {
1975
0
        xmlFree(ctxt->directory);
1976
0
        ctxt->directory = directory;
1977
0
    }
1978
1979
    /*
1980
     * The input ID is unused internally, but there are entity
1981
     * loaders in downstream code that detect the main document
1982
     * by checking for "input_id == 1".
1983
     */
1984
0
    value->id = ctxt->input_id++;
1985
1986
0
    return(ctxt->inputNr++);
1987
0
}
1988
1989
/**
1990
 * Pops the top parser input from the input stack
1991
 *
1992
 * @param ctxt  an XML parser context
1993
 * @returns the input just removed
1994
 */
1995
xmlParserInput *
1996
xmlCtxtPopInput(xmlParserCtxt *ctxt)
1997
0
{
1998
0
    xmlParserInputPtr ret;
1999
2000
0
    if (ctxt == NULL)
2001
0
        return(NULL);
2002
0
    if (ctxt->inputNr <= 0)
2003
0
        return (NULL);
2004
0
    ctxt->inputNr--;
2005
0
    if (ctxt->inputNr > 0)
2006
0
        ctxt->input = ctxt->inputTab[ctxt->inputNr - 1];
2007
0
    else
2008
0
        ctxt->input = NULL;
2009
0
    ret = ctxt->inputTab[ctxt->inputNr];
2010
0
    ctxt->inputTab[ctxt->inputNr] = NULL;
2011
0
    return (ret);
2012
0
}
2013
2014
/**
2015
 * Pushes a new element node on top of the node stack
2016
 *
2017
 * @deprecated Internal function, do not use.
2018
 *
2019
 * @param ctxt  an XML parser context
2020
 * @param value  the element node
2021
 * @returns -1 in case of error, the index in the stack otherwise
2022
 */
2023
int
2024
nodePush(xmlParserCtxt *ctxt, xmlNode *value)
2025
0
{
2026
0
    if (ctxt == NULL)
2027
0
        return(0);
2028
2029
0
    if (ctxt->nodeNr >= ctxt->nodeMax) {
2030
0
        int maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 2048 : 256;
2031
0
        xmlNodePtr *tmp;
2032
0
        int newSize;
2033
2034
0
        newSize = xmlGrowCapacity(ctxt->nodeMax, sizeof(tmp[0]),
2035
0
                                  10, maxDepth);
2036
0
        if (newSize < 0) {
2037
0
            xmlFatalErrMsgInt(ctxt, XML_ERR_RESOURCE_LIMIT,
2038
0
                    "Excessive depth in document: %d,"
2039
0
                    " use XML_PARSE_HUGE option\n",
2040
0
                    ctxt->nodeNr);
2041
0
            return(-1);
2042
0
        }
2043
2044
0
  tmp = xmlRealloc(ctxt->nodeTab, newSize * sizeof(tmp[0]));
2045
0
        if (tmp == NULL) {
2046
0
            xmlErrMemory(ctxt);
2047
0
            return (-1);
2048
0
        }
2049
0
        ctxt->nodeTab = tmp;
2050
0
  ctxt->nodeMax = newSize;
2051
0
    }
2052
2053
0
    ctxt->nodeTab[ctxt->nodeNr] = value;
2054
0
    ctxt->node = value;
2055
0
    return (ctxt->nodeNr++);
2056
0
}
2057
2058
/**
2059
 * Pops the top element node from the node stack
2060
 *
2061
 * @deprecated Internal function, do not use.
2062
 *
2063
 * @param ctxt  an XML parser context
2064
 * @returns the node just removed
2065
 */
2066
xmlNode *
2067
nodePop(xmlParserCtxt *ctxt)
2068
0
{
2069
0
    xmlNodePtr ret;
2070
2071
0
    if (ctxt == NULL) return(NULL);
2072
0
    if (ctxt->nodeNr <= 0)
2073
0
        return (NULL);
2074
0
    ctxt->nodeNr--;
2075
0
    if (ctxt->nodeNr > 0)
2076
0
        ctxt->node = ctxt->nodeTab[ctxt->nodeNr - 1];
2077
0
    else
2078
0
        ctxt->node = NULL;
2079
0
    ret = ctxt->nodeTab[ctxt->nodeNr];
2080
0
    ctxt->nodeTab[ctxt->nodeNr] = NULL;
2081
0
    return (ret);
2082
0
}
2083
2084
/**
2085
 * Pushes a new element name/prefix/URL on top of the name stack
2086
 *
2087
 * @param ctxt  an XML parser context
2088
 * @param value  the element name
2089
 * @param prefix  the element prefix
2090
 * @param URI  the element namespace name
2091
 * @param line  the current line number for error messages
2092
 * @param nsNr  the number of namespaces pushed on the namespace table
2093
 * @returns -1 in case of error, the index in the stack otherwise
2094
 */
2095
static int
2096
nameNsPush(xmlParserCtxtPtr ctxt, const xmlChar * value,
2097
           const xmlChar *prefix, const xmlChar *URI, int line, int nsNr)
2098
0
{
2099
0
    xmlStartTag *tag;
2100
2101
0
    if (ctxt->nameNr >= ctxt->nameMax) {
2102
0
        const xmlChar **tmp;
2103
0
        xmlStartTag *tmp2;
2104
0
        int newSize;
2105
2106
0
        newSize = xmlGrowCapacity(ctxt->nameMax,
2107
0
                                  sizeof(tmp[0]) + sizeof(tmp2[0]),
2108
0
                                  10, XML_MAX_ITEMS);
2109
0
        if (newSize < 0)
2110
0
            goto mem_error;
2111
2112
0
        tmp = xmlRealloc(ctxt->nameTab, newSize * sizeof(tmp[0]));
2113
0
        if (tmp == NULL)
2114
0
      goto mem_error;
2115
0
  ctxt->nameTab = tmp;
2116
2117
0
        tmp2 = xmlRealloc(ctxt->pushTab, newSize * sizeof(tmp2[0]));
2118
0
        if (tmp2 == NULL)
2119
0
      goto mem_error;
2120
0
  ctxt->pushTab = tmp2;
2121
2122
0
        ctxt->nameMax = newSize;
2123
0
    } else if (ctxt->pushTab == NULL) {
2124
0
        ctxt->pushTab = xmlMalloc(ctxt->nameMax * sizeof(ctxt->pushTab[0]));
2125
0
        if (ctxt->pushTab == NULL)
2126
0
            goto mem_error;
2127
0
    }
2128
0
    ctxt->nameTab[ctxt->nameNr] = value;
2129
0
    ctxt->name = value;
2130
0
    tag = &ctxt->pushTab[ctxt->nameNr];
2131
0
    tag->prefix = prefix;
2132
0
    tag->URI = URI;
2133
0
    tag->line = line;
2134
0
    tag->nsNr = nsNr;
2135
0
    return (ctxt->nameNr++);
2136
0
mem_error:
2137
0
    xmlErrMemory(ctxt);
2138
0
    return (-1);
2139
0
}
2140
#ifdef LIBXML_PUSH_ENABLED
2141
/**
2142
 * Pops the top element/prefix/URI name from the name stack
2143
 *
2144
 * @param ctxt  an XML parser context
2145
 * @returns the name just removed
2146
 */
2147
static const xmlChar *
2148
nameNsPop(xmlParserCtxtPtr ctxt)
2149
0
{
2150
0
    const xmlChar *ret;
2151
2152
0
    if (ctxt->nameNr <= 0)
2153
0
        return (NULL);
2154
0
    ctxt->nameNr--;
2155
0
    if (ctxt->nameNr > 0)
2156
0
        ctxt->name = ctxt->nameTab[ctxt->nameNr - 1];
2157
0
    else
2158
0
        ctxt->name = NULL;
2159
0
    ret = ctxt->nameTab[ctxt->nameNr];
2160
0
    ctxt->nameTab[ctxt->nameNr] = NULL;
2161
0
    return (ret);
2162
0
}
2163
#endif /* LIBXML_PUSH_ENABLED */
2164
2165
/**
2166
 * Pops the top element name from the name stack
2167
 *
2168
 * @deprecated Internal function, do not use.
2169
 *
2170
 * @param ctxt  an XML parser context
2171
 * @returns the name just removed
2172
 */
2173
static const xmlChar *
2174
namePop(xmlParserCtxtPtr ctxt)
2175
0
{
2176
0
    const xmlChar *ret;
2177
2178
0
    if ((ctxt == NULL) || (ctxt->nameNr <= 0))
2179
0
        return (NULL);
2180
0
    ctxt->nameNr--;
2181
0
    if (ctxt->nameNr > 0)
2182
0
        ctxt->name = ctxt->nameTab[ctxt->nameNr - 1];
2183
0
    else
2184
0
        ctxt->name = NULL;
2185
0
    ret = ctxt->nameTab[ctxt->nameNr];
2186
0
    ctxt->nameTab[ctxt->nameNr] = NULL;
2187
0
    return (ret);
2188
0
}
2189
2190
0
static int spacePush(xmlParserCtxtPtr ctxt, int val) {
2191
0
    if (ctxt->spaceNr >= ctxt->spaceMax) {
2192
0
        int *tmp;
2193
0
        int newSize;
2194
2195
0
        newSize = xmlGrowCapacity(ctxt->spaceMax, sizeof(tmp[0]),
2196
0
                                  10, XML_MAX_ITEMS);
2197
0
        if (newSize < 0) {
2198
0
      xmlErrMemory(ctxt);
2199
0
      return(-1);
2200
0
        }
2201
2202
0
        tmp = xmlRealloc(ctxt->spaceTab, newSize * sizeof(tmp[0]));
2203
0
        if (tmp == NULL) {
2204
0
      xmlErrMemory(ctxt);
2205
0
      return(-1);
2206
0
  }
2207
0
  ctxt->spaceTab = tmp;
2208
2209
0
        ctxt->spaceMax = newSize;
2210
0
    }
2211
0
    ctxt->spaceTab[ctxt->spaceNr] = val;
2212
0
    ctxt->space = &ctxt->spaceTab[ctxt->spaceNr];
2213
0
    return(ctxt->spaceNr++);
2214
0
}
2215
2216
0
static int spacePop(xmlParserCtxtPtr ctxt) {
2217
0
    int ret;
2218
0
    if (ctxt->spaceNr <= 0) return(0);
2219
0
    ctxt->spaceNr--;
2220
0
    if (ctxt->spaceNr > 0)
2221
0
  ctxt->space = &ctxt->spaceTab[ctxt->spaceNr - 1];
2222
0
    else
2223
0
        ctxt->space = &ctxt->spaceTab[0];
2224
0
    ret = ctxt->spaceTab[ctxt->spaceNr];
2225
0
    ctxt->spaceTab[ctxt->spaceNr] = -1;
2226
0
    return(ret);
2227
0
}
2228
2229
/*
2230
 * Macros for accessing the content. Those should be used only by the parser,
2231
 * and not exported.
2232
 *
2233
 * Dirty macros, i.e. one often need to make assumption on the context to
2234
 * use them
2235
 *
2236
 *   CUR_PTR return the current pointer to the xmlChar to be parsed.
2237
 *           To be used with extreme caution since operations consuming
2238
 *           characters may move the input buffer to a different location !
2239
 *   CUR     returns the current xmlChar value, i.e. a 8 bit value if compiled
2240
 *           This should be used internally by the parser
2241
 *           only to compare to ASCII values otherwise it would break when
2242
 *           running with UTF-8 encoding.
2243
 *   RAW     same as CUR but in the input buffer, bypass any token
2244
 *           extraction that may have been done
2245
 *   NXT(n)  returns the n'th next xmlChar. Same as CUR is should be used only
2246
 *           to compare on ASCII based substring.
2247
 *   SKIP(n) Skip n xmlChar, and must also be used only to skip ASCII defined
2248
 *           strings without newlines within the parser.
2249
 *   NEXT1(l) Skip 1 xmlChar, and must also be used only to skip 1 non-newline ASCII
2250
 *           defined char within the parser.
2251
 * Clean macros, not dependent of an ASCII context, expect UTF-8 encoding
2252
 *
2253
 *   NEXT    Skip to the next character, this does the proper decoding
2254
 *           in UTF-8 mode. It also pop-up unfinished entities on the fly.
2255
 *   NEXTL(l) Skip the current unicode character of l xmlChars long.
2256
 *   COPY_BUF  copy the current unicode char to the target buffer, increment
2257
 *            the index
2258
 *   GROW, SHRINK  handling of input buffers
2259
 */
2260
2261
0
#define RAW (*ctxt->input->cur)
2262
0
#define CUR (*ctxt->input->cur)
2263
0
#define NXT(val) ctxt->input->cur[(val)]
2264
0
#define CUR_PTR ctxt->input->cur
2265
0
#define BASE_PTR ctxt->input->base
2266
2267
#define CMP4( s, c1, c2, c3, c4 ) \
2268
0
  ( ((unsigned char *) s)[ 0 ] == c1 && ((unsigned char *) s)[ 1 ] == c2 && \
2269
0
    ((unsigned char *) s)[ 2 ] == c3 && ((unsigned char *) s)[ 3 ] == c4 )
2270
#define CMP5( s, c1, c2, c3, c4, c5 ) \
2271
0
  ( CMP4( s, c1, c2, c3, c4 ) && ((unsigned char *) s)[ 4 ] == c5 )
2272
#define CMP6( s, c1, c2, c3, c4, c5, c6 ) \
2273
0
  ( CMP5( s, c1, c2, c3, c4, c5 ) && ((unsigned char *) s)[ 5 ] == c6 )
2274
#define CMP7( s, c1, c2, c3, c4, c5, c6, c7 ) \
2275
0
  ( CMP6( s, c1, c2, c3, c4, c5, c6 ) && ((unsigned char *) s)[ 6 ] == c7 )
2276
#define CMP8( s, c1, c2, c3, c4, c5, c6, c7, c8 ) \
2277
0
  ( CMP7( s, c1, c2, c3, c4, c5, c6, c7 ) && ((unsigned char *) s)[ 7 ] == c8 )
2278
#define CMP9( s, c1, c2, c3, c4, c5, c6, c7, c8, c9 ) \
2279
0
  ( CMP8( s, c1, c2, c3, c4, c5, c6, c7, c8 ) && \
2280
0
    ((unsigned char *) s)[ 8 ] == c9 )
2281
#define CMP10( s, c1, c2, c3, c4, c5, c6, c7, c8, c9, c10 ) \
2282
0
  ( CMP9( s, c1, c2, c3, c4, c5, c6, c7, c8, c9 ) && \
2283
0
    ((unsigned char *) s)[ 9 ] == c10 )
2284
2285
0
#define SKIP(val) do {             \
2286
0
    ctxt->input->cur += (val),ctxt->input->col+=(val);      \
2287
0
    if (*ctxt->input->cur == 0)           \
2288
0
        xmlParserGrow(ctxt);           \
2289
0
  } while (0)
2290
2291
#define SKIPL(val) do {             \
2292
    int skipl;                \
2293
    for(skipl=0; skipl<val; skipl++) {          \
2294
  if (*(ctxt->input->cur) == '\n') {        \
2295
  ctxt->input->line++; ctxt->input->col = 1;      \
2296
  } else ctxt->input->col++;          \
2297
  ctxt->input->cur++;           \
2298
    }                 \
2299
    if (*ctxt->input->cur == 0)           \
2300
        xmlParserGrow(ctxt);            \
2301
  } while (0)
2302
2303
#define SHRINK \
2304
0
    if (!PARSER_PROGRESSIVE(ctxt)) \
2305
0
  xmlParserShrink(ctxt);
2306
2307
#define GROW \
2308
0
    if ((!PARSER_PROGRESSIVE(ctxt)) && \
2309
0
        (ctxt->input->end - ctxt->input->cur < INPUT_CHUNK)) \
2310
0
  xmlParserGrow(ctxt);
2311
2312
0
#define SKIP_BLANKS xmlSkipBlankChars(ctxt)
2313
2314
0
#define SKIP_BLANKS_PE xmlSkipBlankCharsPE(ctxt)
2315
2316
0
#define NEXT xmlNextChar(ctxt)
2317
2318
0
#define NEXT1 {               \
2319
0
  ctxt->input->col++;           \
2320
0
  ctxt->input->cur++;           \
2321
0
  if (*ctxt->input->cur == 0)         \
2322
0
      xmlParserGrow(ctxt);           \
2323
0
    }
2324
2325
0
#define NEXTL(l) do {             \
2326
0
    if (*(ctxt->input->cur) == '\n') {         \
2327
0
  ctxt->input->line++; ctxt->input->col = 1;      \
2328
0
    } else ctxt->input->col++;           \
2329
0
    ctxt->input->cur += l;        \
2330
0
  } while (0)
2331
2332
#define COPY_BUF(b, i, v)           \
2333
0
    if (v < 0x80) b[i++] = v;           \
2334
0
    else i += xmlCopyCharMultiByte(&b[i],v)
2335
2336
static int
2337
0
xmlCurrentCharRecover(xmlParserCtxtPtr ctxt, int *len) {
2338
0
    int c = xmlCurrentChar(ctxt, len);
2339
2340
0
    if (c == XML_INVALID_CHAR)
2341
0
        c = 0xFFFD; /* replacement character */
2342
2343
0
    return(c);
2344
0
}
2345
2346
/**
2347
 * Skip whitespace in the input stream.
2348
 *
2349
 * @deprecated Internal function, do not use.
2350
 *
2351
 * @param ctxt  the XML parser context
2352
 * @returns the number of space chars skipped
2353
 */
2354
int
2355
0
xmlSkipBlankChars(xmlParserCtxt *ctxt) {
2356
0
    const xmlChar *cur;
2357
0
    int res = 0;
2358
2359
0
    cur = ctxt->input->cur;
2360
0
    while (IS_BLANK_CH(*cur)) {
2361
0
        if (*cur == '\n') {
2362
0
            ctxt->input->line++; ctxt->input->col = 1;
2363
0
        } else {
2364
0
            ctxt->input->col++;
2365
0
        }
2366
0
        cur++;
2367
0
        if (res < INT_MAX)
2368
0
            res++;
2369
0
        if (*cur == 0) {
2370
0
            ctxt->input->cur = cur;
2371
0
            xmlParserGrow(ctxt);
2372
0
            cur = ctxt->input->cur;
2373
0
        }
2374
0
    }
2375
0
    ctxt->input->cur = cur;
2376
2377
0
    if (res > 4)
2378
0
        GROW;
2379
2380
0
    return(res);
2381
0
}
2382
2383
static void
2384
0
xmlPopPE(xmlParserCtxtPtr ctxt) {
2385
0
    unsigned long consumed;
2386
0
    xmlEntityPtr ent;
2387
2388
0
    ent = ctxt->input->entity;
2389
2390
0
    ent->flags &= ~XML_ENT_EXPANDING;
2391
2392
0
    if ((ent->flags & XML_ENT_CHECKED) == 0) {
2393
0
        int result;
2394
2395
        /*
2396
         * Read the rest of the stream in case of errors. We want
2397
         * to account for the whole entity size.
2398
         */
2399
0
        do {
2400
0
            ctxt->input->cur = ctxt->input->end;
2401
0
            xmlParserShrink(ctxt);
2402
0
            result = xmlParserGrow(ctxt);
2403
0
        } while (result > 0);
2404
2405
0
        consumed = ctxt->input->consumed;
2406
0
        xmlSaturatedAddSizeT(&consumed,
2407
0
                             ctxt->input->end - ctxt->input->base);
2408
2409
0
        xmlSaturatedAdd(&ent->expandedSize, consumed);
2410
2411
        /*
2412
         * Add to sizeentities when parsing an external entity
2413
         * for the first time.
2414
         */
2415
0
        if (ent->etype == XML_EXTERNAL_PARAMETER_ENTITY) {
2416
0
            xmlSaturatedAdd(&ctxt->sizeentities, consumed);
2417
0
        }
2418
2419
0
        ent->flags |= XML_ENT_CHECKED;
2420
0
    }
2421
2422
0
    xmlFreeInputStream(xmlCtxtPopInput(ctxt));
2423
2424
0
    xmlParserEntityCheck(ctxt, ent->expandedSize);
2425
2426
0
    GROW;
2427
0
}
2428
2429
/**
2430
 * Skip whitespace in the input stream, also handling parameter
2431
 * entities.
2432
 *
2433
 * @param ctxt  the XML parser context
2434
 * @returns the number of space chars skipped
2435
 */
2436
static int
2437
0
xmlSkipBlankCharsPE(xmlParserCtxtPtr ctxt) {
2438
0
    int res = 0;
2439
0
    int inParam;
2440
0
    int expandParam;
2441
2442
0
    inParam = PARSER_IN_PE(ctxt);
2443
0
    expandParam = PARSER_EXTERNAL(ctxt);
2444
2445
0
    if (!inParam && !expandParam)
2446
0
        return(xmlSkipBlankChars(ctxt));
2447
2448
    /*
2449
     * It's Okay to use CUR/NEXT here since all the blanks are on
2450
     * the ASCII range.
2451
     */
2452
0
    while (PARSER_STOPPED(ctxt) == 0) {
2453
0
        if (IS_BLANK_CH(CUR)) { /* CHECKED tstblanks.xml */
2454
0
            NEXT;
2455
0
        } else if (CUR == '%') {
2456
0
            if ((expandParam == 0) ||
2457
0
                (IS_BLANK_CH(NXT(1))) || (NXT(1) == 0))
2458
0
                break;
2459
2460
            /*
2461
             * Expand parameter entity. We continue to consume
2462
             * whitespace at the start of the entity and possible
2463
             * even consume the whole entity and pop it. We might
2464
             * even pop multiple PEs in this loop.
2465
             */
2466
0
            xmlParsePERefInternal(ctxt, 0);
2467
2468
0
            inParam = PARSER_IN_PE(ctxt);
2469
0
            expandParam = PARSER_EXTERNAL(ctxt);
2470
0
        } else if (CUR == 0) {
2471
0
            if (inParam == 0)
2472
0
                break;
2473
2474
            /*
2475
             * Don't pop parameter entities that start a markup
2476
             * declaration to detect Well-formedness constraint:
2477
             * PE Between Declarations.
2478
             */
2479
0
            if (ctxt->input->flags & XML_INPUT_MARKUP_DECL)
2480
0
                break;
2481
2482
0
            xmlPopPE(ctxt);
2483
2484
0
            inParam = PARSER_IN_PE(ctxt);
2485
0
            expandParam = PARSER_EXTERNAL(ctxt);
2486
0
        } else {
2487
0
            break;
2488
0
        }
2489
2490
        /*
2491
         * Also increase the counter when entering or exiting a PERef.
2492
         * The spec says: "When a parameter-entity reference is recognized
2493
         * in the DTD and included, its replacement text MUST be enlarged
2494
         * by the attachment of one leading and one following space (#x20)
2495
         * character."
2496
         */
2497
0
        if (res < INT_MAX)
2498
0
            res++;
2499
0
    }
2500
2501
0
    return(res);
2502
0
}
2503
2504
/************************************************************************
2505
 *                  *
2506
 *    Commodity functions to handle entities      *
2507
 *                  *
2508
 ************************************************************************/
2509
2510
/**
2511
 * @deprecated Internal function, don't use.
2512
 *
2513
 * @param ctxt  an XML parser context
2514
 * @returns the current xmlChar in the parser context
2515
 */
2516
xmlChar
2517
0
xmlPopInput(xmlParserCtxt *ctxt) {
2518
0
    xmlParserInputPtr input;
2519
2520
0
    if ((ctxt == NULL) || (ctxt->inputNr <= 1)) return(0);
2521
0
    input = xmlCtxtPopInput(ctxt);
2522
0
    xmlFreeInputStream(input);
2523
0
    if (*ctxt->input->cur == 0)
2524
0
        xmlParserGrow(ctxt);
2525
0
    return(CUR);
2526
0
}
2527
2528
/**
2529
 * Push an input stream onto the stack.
2530
 *
2531
 * @deprecated Internal function, don't use.
2532
 *
2533
 * @param ctxt  an XML parser context
2534
 * @param input  an XML parser input fragment (entity, XML fragment ...).
2535
 * @returns -1 in case of error or the index in the input stack
2536
 */
2537
int
2538
0
xmlPushInput(xmlParserCtxt *ctxt, xmlParserInput *input) {
2539
0
    int ret;
2540
2541
0
    if ((ctxt == NULL) || (input == NULL))
2542
0
        return(-1);
2543
2544
0
    ret = xmlCtxtPushInput(ctxt, input);
2545
0
    if (ret >= 0)
2546
0
        GROW;
2547
0
    return(ret);
2548
0
}
2549
2550
/**
2551
 * Parse a numeric character reference. Always consumes '&'.
2552
 *
2553
 * @deprecated Internal function, don't use.
2554
 *
2555
 *     [66] CharRef ::= '&#' [0-9]+ ';' |
2556
 *                      '&#x' [0-9a-fA-F]+ ';'
2557
 *
2558
 * [ WFC: Legal Character ]
2559
 * Characters referred to using character references must match the
2560
 * production for Char.
2561
 *
2562
 * @param ctxt  an XML parser context
2563
 * @returns the value parsed (as an int), 0 in case of error
2564
 */
2565
int
2566
0
xmlParseCharRef(xmlParserCtxt *ctxt) {
2567
0
    int val = 0;
2568
0
    int count = 0;
2569
2570
    /*
2571
     * Using RAW/CUR/NEXT is okay since we are working on ASCII range here
2572
     */
2573
0
    if ((RAW == '&') && (NXT(1) == '#') &&
2574
0
        (NXT(2) == 'x')) {
2575
0
  SKIP(3);
2576
0
  GROW;
2577
0
  while ((RAW != ';') && (PARSER_STOPPED(ctxt) == 0)) {
2578
0
      if (count++ > 20) {
2579
0
    count = 0;
2580
0
    GROW;
2581
0
      }
2582
0
      if ((RAW >= '0') && (RAW <= '9'))
2583
0
          val = val * 16 + (CUR - '0');
2584
0
      else if ((RAW >= 'a') && (RAW <= 'f') && (count < 20))
2585
0
          val = val * 16 + (CUR - 'a') + 10;
2586
0
      else if ((RAW >= 'A') && (RAW <= 'F') && (count < 20))
2587
0
          val = val * 16 + (CUR - 'A') + 10;
2588
0
      else {
2589
0
    xmlFatalErr(ctxt, XML_ERR_INVALID_HEX_CHARREF, NULL);
2590
0
    val = 0;
2591
0
    break;
2592
0
      }
2593
0
      if (val > 0x110000)
2594
0
          val = 0x110000;
2595
2596
0
      NEXT;
2597
0
      count++;
2598
0
  }
2599
0
  if (RAW == ';') {
2600
      /* on purpose to avoid reentrancy problems with NEXT and SKIP */
2601
0
      ctxt->input->col++;
2602
0
      ctxt->input->cur++;
2603
0
  }
2604
0
    } else if  ((RAW == '&') && (NXT(1) == '#')) {
2605
0
  SKIP(2);
2606
0
  GROW;
2607
0
  while (RAW != ';') { /* loop blocked by count */
2608
0
      if (count++ > 20) {
2609
0
    count = 0;
2610
0
    GROW;
2611
0
      }
2612
0
      if ((RAW >= '0') && (RAW <= '9'))
2613
0
          val = val * 10 + (CUR - '0');
2614
0
      else {
2615
0
    xmlFatalErr(ctxt, XML_ERR_INVALID_DEC_CHARREF, NULL);
2616
0
    val = 0;
2617
0
    break;
2618
0
      }
2619
0
      if (val > 0x110000)
2620
0
          val = 0x110000;
2621
2622
0
      NEXT;
2623
0
      count++;
2624
0
  }
2625
0
  if (RAW == ';') {
2626
      /* on purpose to avoid reentrancy problems with NEXT and SKIP */
2627
0
      ctxt->input->col++;
2628
0
      ctxt->input->cur++;
2629
0
  }
2630
0
    } else {
2631
0
        if (RAW == '&')
2632
0
            SKIP(1);
2633
0
        xmlFatalErr(ctxt, XML_ERR_INVALID_CHARREF, NULL);
2634
0
    }
2635
2636
    /*
2637
     * [ WFC: Legal Character ]
2638
     * Characters referred to using character references must match the
2639
     * production for Char.
2640
     */
2641
0
    if (val >= 0x110000) {
2642
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
2643
0
                "xmlParseCharRef: character reference out of bounds\n",
2644
0
          val);
2645
0
        val = 0xFFFD;
2646
0
    } else if (!IS_CHAR(val)) {
2647
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
2648
0
                          "xmlParseCharRef: invalid xmlChar value %d\n",
2649
0
                    val);
2650
0
    }
2651
0
    return(val);
2652
0
}
2653
2654
/**
2655
 * Parse Reference declarations, variant parsing from a string rather
2656
 * than an an input flow.
2657
 *
2658
 *     [66] CharRef ::= '&#' [0-9]+ ';' |
2659
 *                      '&#x' [0-9a-fA-F]+ ';'
2660
 *
2661
 * [ WFC: Legal Character ]
2662
 * Characters referred to using character references must match the
2663
 * production for Char.
2664
 *
2665
 * @param ctxt  an XML parser context
2666
 * @param str  a pointer to an index in the string
2667
 * @returns the value parsed (as an int), 0 in case of error, str will be
2668
 *         updated to the current value of the index
2669
 */
2670
static int
2671
0
xmlParseStringCharRef(xmlParserCtxtPtr ctxt, const xmlChar **str) {
2672
0
    const xmlChar *ptr;
2673
0
    xmlChar cur;
2674
0
    int val = 0;
2675
2676
0
    if ((str == NULL) || (*str == NULL)) return(0);
2677
0
    ptr = *str;
2678
0
    cur = *ptr;
2679
0
    if ((cur == '&') && (ptr[1] == '#') && (ptr[2] == 'x')) {
2680
0
  ptr += 3;
2681
0
  cur = *ptr;
2682
0
  while (cur != ';') { /* Non input consuming loop */
2683
0
      if ((cur >= '0') && (cur <= '9'))
2684
0
          val = val * 16 + (cur - '0');
2685
0
      else if ((cur >= 'a') && (cur <= 'f'))
2686
0
          val = val * 16 + (cur - 'a') + 10;
2687
0
      else if ((cur >= 'A') && (cur <= 'F'))
2688
0
          val = val * 16 + (cur - 'A') + 10;
2689
0
      else {
2690
0
    xmlFatalErr(ctxt, XML_ERR_INVALID_HEX_CHARREF, NULL);
2691
0
    val = 0;
2692
0
    break;
2693
0
      }
2694
0
      if (val > 0x110000)
2695
0
          val = 0x110000;
2696
2697
0
      ptr++;
2698
0
      cur = *ptr;
2699
0
  }
2700
0
  if (cur == ';')
2701
0
      ptr++;
2702
0
    } else if  ((cur == '&') && (ptr[1] == '#')){
2703
0
  ptr += 2;
2704
0
  cur = *ptr;
2705
0
  while (cur != ';') { /* Non input consuming loops */
2706
0
      if ((cur >= '0') && (cur <= '9'))
2707
0
          val = val * 10 + (cur - '0');
2708
0
      else {
2709
0
    xmlFatalErr(ctxt, XML_ERR_INVALID_DEC_CHARREF, NULL);
2710
0
    val = 0;
2711
0
    break;
2712
0
      }
2713
0
      if (val > 0x110000)
2714
0
          val = 0x110000;
2715
2716
0
      ptr++;
2717
0
      cur = *ptr;
2718
0
  }
2719
0
  if (cur == ';')
2720
0
      ptr++;
2721
0
    } else {
2722
0
  xmlFatalErr(ctxt, XML_ERR_INVALID_CHARREF, NULL);
2723
0
  return(0);
2724
0
    }
2725
0
    *str = ptr;
2726
2727
    /*
2728
     * [ WFC: Legal Character ]
2729
     * Characters referred to using character references must match the
2730
     * production for Char.
2731
     */
2732
0
    if (val >= 0x110000) {
2733
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
2734
0
                "xmlParseStringCharRef: character reference out of bounds\n",
2735
0
                val);
2736
0
    } else if (IS_CHAR(val)) {
2737
0
        return(val);
2738
0
    } else {
2739
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
2740
0
        "xmlParseStringCharRef: invalid xmlChar value %d\n",
2741
0
        val);
2742
0
    }
2743
0
    return(0);
2744
0
}
2745
2746
/**
2747
 *     [69] PEReference ::= '%' Name ';'
2748
 *
2749
 * @deprecated Internal function, do not use.
2750
 *
2751
 * [ WFC: No Recursion ]
2752
 * A parsed entity must not contain a recursive
2753
 * reference to itself, either directly or indirectly.
2754
 *
2755
 * [ WFC: Entity Declared ]
2756
 * In a document without any DTD, a document with only an internal DTD
2757
 * subset which contains no parameter entity references, or a document
2758
 * with "standalone='yes'", ...  ... The declaration of a parameter
2759
 * entity must precede any reference to it...
2760
 *
2761
 * [ VC: Entity Declared ]
2762
 * In a document with an external subset or external parameter entities
2763
 * with "standalone='no'", ...  ... The declaration of a parameter entity
2764
 * must precede any reference to it...
2765
 *
2766
 * [ WFC: In DTD ]
2767
 * Parameter-entity references may only appear in the DTD.
2768
 * NOTE: misleading but this is handled.
2769
 *
2770
 * A PEReference may have been detected in the current input stream
2771
 * the handling is done accordingly to
2772
 *      http://www.w3.org/TR/REC-xml#entproc
2773
 * i.e.
2774
 *   - Included in literal in entity values
2775
 *   - Included as Parameter Entity reference within DTDs
2776
 * @param ctxt  the parser context
2777
 */
2778
void
2779
0
xmlParserHandlePEReference(xmlParserCtxt *ctxt) {
2780
0
    xmlParsePERefInternal(ctxt, 0);
2781
0
}
2782
2783
/**
2784
 * @deprecated Internal function, don't use.
2785
 *
2786
 * @param ctxt  the parser context
2787
 * @param str  the input string
2788
 * @param len  the string length
2789
 * @param what  combination of XML_SUBSTITUTE_REF and XML_SUBSTITUTE_PEREF
2790
 * @param end  an end marker xmlChar, 0 if none
2791
 * @param end2  an end marker xmlChar, 0 if none
2792
 * @param end3  an end marker xmlChar, 0 if none
2793
 * @returns A newly allocated string with the substitution done. The caller
2794
 *      must deallocate it !
2795
 */
2796
xmlChar *
2797
xmlStringLenDecodeEntities(xmlParserCtxt *ctxt, const xmlChar *str, int len,
2798
                           int what ATTRIBUTE_UNUSED,
2799
0
                           xmlChar end, xmlChar end2, xmlChar end3) {
2800
0
    if ((ctxt == NULL) || (str == NULL) || (len < 0))
2801
0
        return(NULL);
2802
2803
0
    if ((str[len] != 0) ||
2804
0
        (end != 0) || (end2 != 0) || (end3 != 0))
2805
0
        return(NULL);
2806
2807
0
    return(xmlExpandEntitiesInAttValue(ctxt, str, 0));
2808
0
}
2809
2810
/**
2811
 * @deprecated Internal function, don't use.
2812
 *
2813
 * @param ctxt  the parser context
2814
 * @param str  the input string
2815
 * @param what  combination of XML_SUBSTITUTE_REF and XML_SUBSTITUTE_PEREF
2816
 * @param end  an end marker xmlChar, 0 if none
2817
 * @param end2  an end marker xmlChar, 0 if none
2818
 * @param end3  an end marker xmlChar, 0 if none
2819
 * @returns A newly allocated string with the substitution done. The caller
2820
 *      must deallocate it !
2821
 */
2822
xmlChar *
2823
xmlStringDecodeEntities(xmlParserCtxt *ctxt, const xmlChar *str,
2824
                        int what ATTRIBUTE_UNUSED,
2825
0
            xmlChar end, xmlChar  end2, xmlChar end3) {
2826
0
    if ((ctxt == NULL) || (str == NULL))
2827
0
        return(NULL);
2828
2829
0
    if ((end != 0) || (end2 != 0) || (end3 != 0))
2830
0
        return(NULL);
2831
2832
0
    return(xmlExpandEntitiesInAttValue(ctxt, str, 0));
2833
0
}
2834
2835
/************************************************************************
2836
 *                  *
2837
 *    Commodity functions, cleanup needed ?     *
2838
 *                  *
2839
 ************************************************************************/
2840
2841
/**
2842
 * Is this a sequence of blank chars that one can ignore ?
2843
 *
2844
 * @param ctxt  an XML parser context
2845
 * @param str  a xmlChar *
2846
 * @param len  the size of `str`
2847
 * @param blank_chars  we know the chars are blanks
2848
 * @returns 1 if ignorable 0 otherwise.
2849
 */
2850
2851
static int areBlanks(xmlParserCtxtPtr ctxt, const xmlChar *str, int len,
2852
0
                     int blank_chars) {
2853
0
    int i;
2854
0
    xmlNodePtr lastChild;
2855
2856
    /*
2857
     * Check for xml:space value.
2858
     */
2859
0
    if ((ctxt->space == NULL) || (*(ctxt->space) == 1) ||
2860
0
        (*(ctxt->space) == -2))
2861
0
  return(0);
2862
2863
    /*
2864
     * Check that the string is made of blanks
2865
     */
2866
0
    if (blank_chars == 0) {
2867
0
  for (i = 0;i < len;i++)
2868
0
      if (!(IS_BLANK_CH(str[i]))) return(0);
2869
0
    }
2870
2871
    /*
2872
     * Look if the element is mixed content in the DTD if available
2873
     */
2874
0
    if (ctxt->node == NULL) return(0);
2875
0
    if (ctxt->myDoc != NULL) {
2876
0
        xmlElementPtr elemDecl = NULL;
2877
0
        xmlDocPtr doc = ctxt->myDoc;
2878
0
        const xmlChar *prefix = NULL;
2879
2880
0
        if (ctxt->node->ns)
2881
0
            prefix = ctxt->node->ns->prefix;
2882
0
        if (doc->intSubset != NULL)
2883
0
            elemDecl = xmlHashLookup2(doc->intSubset->elements, ctxt->node->name,
2884
0
                                      prefix);
2885
0
        if ((elemDecl == NULL) && (doc->extSubset != NULL))
2886
0
            elemDecl = xmlHashLookup2(doc->extSubset->elements, ctxt->node->name,
2887
0
                                      prefix);
2888
0
        if (elemDecl != NULL) {
2889
0
            if (elemDecl->etype == XML_ELEMENT_TYPE_ELEMENT)
2890
0
                return(1);
2891
0
            if ((elemDecl->etype == XML_ELEMENT_TYPE_ANY) ||
2892
0
                (elemDecl->etype == XML_ELEMENT_TYPE_MIXED))
2893
0
                return(0);
2894
0
        }
2895
0
    }
2896
2897
    /*
2898
     * Otherwise, heuristic :-\
2899
     *
2900
     * When push parsing, we could be at the end of a chunk.
2901
     * This makes the look-ahead and consequently the NOBLANKS
2902
     * option unreliable.
2903
     */
2904
0
    if ((RAW != '<') && (RAW != 0xD)) return(0);
2905
0
    if ((ctxt->node->children == NULL) &&
2906
0
  (RAW == '<') && (NXT(1) == '/')) return(0);
2907
2908
0
    lastChild = xmlGetLastChild(ctxt->node);
2909
0
    if (lastChild == NULL) {
2910
0
        if ((ctxt->node->type != XML_ELEMENT_NODE) &&
2911
0
            (ctxt->node->content != NULL)) return(0);
2912
0
    } else if (xmlNodeIsText(lastChild))
2913
0
        return(0);
2914
0
    else if ((ctxt->node->children != NULL) &&
2915
0
             (xmlNodeIsText(ctxt->node->children)))
2916
0
        return(0);
2917
0
    return(1);
2918
0
}
2919
2920
/************************************************************************
2921
 *                  *
2922
 *    Extra stuff for namespace support     *
2923
 *  Relates to http://www.w3.org/TR/WD-xml-names      *
2924
 *                  *
2925
 ************************************************************************/
2926
2927
/**
2928
 * Parse an UTF8 encoded XML qualified name string
2929
 *
2930
 * @deprecated Don't use.
2931
 *
2932
 * @param ctxt  an XML parser context
2933
 * @param name  an XML parser context
2934
 * @param prefixOut  a xmlChar **
2935
 * @returns the local part, and prefix is updated
2936
 *   to get the Prefix if any.
2937
 */
2938
2939
xmlChar *
2940
0
xmlSplitQName(xmlParserCtxt *ctxt, const xmlChar *name, xmlChar **prefixOut) {
2941
0
    xmlChar *ret;
2942
0
    const xmlChar *localname;
2943
2944
0
    localname = xmlSplitQName4(name, prefixOut);
2945
0
    if (localname == NULL) {
2946
0
        xmlCtxtErrMemory(ctxt);
2947
0
        return(NULL);
2948
0
    }
2949
2950
0
    ret = xmlStrdup(localname);
2951
0
    if (ret == NULL) {
2952
0
        xmlCtxtErrMemory(ctxt);
2953
0
        xmlFree(*prefixOut);
2954
0
    }
2955
2956
0
    return(ret);
2957
0
}
2958
2959
/************************************************************************
2960
 *                  *
2961
 *      The parser itself       *
2962
 *  Relates to http://www.w3.org/TR/REC-xml       *
2963
 *                  *
2964
 ************************************************************************/
2965
2966
/************************************************************************
2967
 *                  *
2968
 *  Routines to parse Name, NCName and NmToken      *
2969
 *                  *
2970
 ************************************************************************/
2971
2972
/*
2973
 * The two following functions are related to the change of accepted
2974
 * characters for Name and NmToken in the Revision 5 of XML-1.0
2975
 * They correspond to the modified production [4] and the new production [4a]
2976
 * changes in that revision. Also note that the macros used for the
2977
 * productions Letter, Digit, CombiningChar and Extender are not needed
2978
 * anymore.
2979
 * We still keep compatibility to pre-revision5 parsing semantic if the
2980
 * new XML_PARSE_OLD10 option is given to the parser.
2981
 */
2982
2983
static int
2984
0
xmlIsNameStartCharNew(int c) {
2985
    /*
2986
     * Use the new checks of production [4] [4a] amd [5] of the
2987
     * Update 5 of XML-1.0
2988
     */
2989
0
    if ((c != ' ') && (c != '>') && (c != '/') && /* accelerators */
2990
0
        (((c >= 'a') && (c <= 'z')) ||
2991
0
         ((c >= 'A') && (c <= 'Z')) ||
2992
0
         (c == '_') || (c == ':') ||
2993
0
         ((c >= 0xC0) && (c <= 0xD6)) ||
2994
0
         ((c >= 0xD8) && (c <= 0xF6)) ||
2995
0
         ((c >= 0xF8) && (c <= 0x2FF)) ||
2996
0
         ((c >= 0x370) && (c <= 0x37D)) ||
2997
0
         ((c >= 0x37F) && (c <= 0x1FFF)) ||
2998
0
         ((c >= 0x200C) && (c <= 0x200D)) ||
2999
0
         ((c >= 0x2070) && (c <= 0x218F)) ||
3000
0
         ((c >= 0x2C00) && (c <= 0x2FEF)) ||
3001
0
         ((c >= 0x3001) && (c <= 0xD7FF)) ||
3002
0
         ((c >= 0xF900) && (c <= 0xFDCF)) ||
3003
0
         ((c >= 0xFDF0) && (c <= 0xFFFD)) ||
3004
0
         ((c >= 0x10000) && (c <= 0xEFFFF))))
3005
0
        return(1);
3006
0
    return(0);
3007
0
}
3008
3009
static int
3010
0
xmlIsNameCharNew(int c) {
3011
    /*
3012
     * Use the new checks of production [4] [4a] amd [5] of the
3013
     * Update 5 of XML-1.0
3014
     */
3015
0
    if ((c != ' ') && (c != '>') && (c != '/') && /* accelerators */
3016
0
        (((c >= 'a') && (c <= 'z')) ||
3017
0
         ((c >= 'A') && (c <= 'Z')) ||
3018
0
         ((c >= '0') && (c <= '9')) || /* !start */
3019
0
         (c == '_') || (c == ':') ||
3020
0
         (c == '-') || (c == '.') || (c == 0xB7) || /* !start */
3021
0
         ((c >= 0xC0) && (c <= 0xD6)) ||
3022
0
         ((c >= 0xD8) && (c <= 0xF6)) ||
3023
0
         ((c >= 0xF8) && (c <= 0x2FF)) ||
3024
0
         ((c >= 0x300) && (c <= 0x36F)) || /* !start */
3025
0
         ((c >= 0x370) && (c <= 0x37D)) ||
3026
0
         ((c >= 0x37F) && (c <= 0x1FFF)) ||
3027
0
         ((c >= 0x200C) && (c <= 0x200D)) ||
3028
0
         ((c >= 0x203F) && (c <= 0x2040)) || /* !start */
3029
0
         ((c >= 0x2070) && (c <= 0x218F)) ||
3030
0
         ((c >= 0x2C00) && (c <= 0x2FEF)) ||
3031
0
         ((c >= 0x3001) && (c <= 0xD7FF)) ||
3032
0
         ((c >= 0xF900) && (c <= 0xFDCF)) ||
3033
0
         ((c >= 0xFDF0) && (c <= 0xFFFD)) ||
3034
0
         ((c >= 0x10000) && (c <= 0xEFFFF))))
3035
0
         return(1);
3036
0
    return(0);
3037
0
}
3038
3039
static int
3040
0
xmlIsNameStartCharOld(int c) {
3041
0
    if ((c != ' ') && (c != '>') && (c != '/') && /* accelerators */
3042
0
        ((IS_LETTER(c) || (c == '_') || (c == ':'))))
3043
0
        return(1);
3044
0
    return(0);
3045
0
}
3046
3047
static int
3048
0
xmlIsNameCharOld(int c) {
3049
0
    if ((c != ' ') && (c != '>') && (c != '/') && /* accelerators */
3050
0
        ((IS_LETTER(c)) || (IS_DIGIT(c)) ||
3051
0
         (c == '.') || (c == '-') ||
3052
0
         (c == '_') || (c == ':') ||
3053
0
         (IS_COMBINING(c)) ||
3054
0
         (IS_EXTENDER(c))))
3055
0
        return(1);
3056
0
    return(0);
3057
0
}
3058
3059
static int
3060
0
xmlIsNameStartChar(int c, int old10) {
3061
0
    if (!old10)
3062
0
        return(xmlIsNameStartCharNew(c));
3063
0
    else
3064
0
        return(xmlIsNameStartCharOld(c));
3065
0
}
3066
3067
static int
3068
0
xmlIsNameChar(int c, int old10) {
3069
0
    if (!old10)
3070
0
        return(xmlIsNameCharNew(c));
3071
0
    else
3072
0
        return(xmlIsNameCharOld(c));
3073
0
}
3074
3075
/*
3076
 * Scan an XML Name, NCName or Nmtoken.
3077
 *
3078
 * Returns a pointer to the end of the name on success. If the
3079
 * name is invalid, returns `ptr`. If the name is longer than
3080
 * `maxSize` bytes, returns NULL.
3081
 *
3082
 * @param ptr  pointer to the start of the name
3083
 * @param maxSize  maximum size in bytes
3084
 * @param flags  XML_SCAN_* flags
3085
 * @returns a pointer to the end of the name or NULL
3086
 */
3087
const xmlChar *
3088
0
xmlScanName(const xmlChar *ptr, size_t maxSize, int flags) {
3089
0
    int stop = flags & XML_SCAN_NC ? ':' : 0;
3090
0
    int old10 = flags & XML_SCAN_OLD10 ? 1 : 0;
3091
3092
0
    while (1) {
3093
0
        int c, len;
3094
3095
0
        c = *ptr;
3096
0
        if (c < 0x80) {
3097
0
            if (c == stop)
3098
0
                break;
3099
0
            len = 1;
3100
0
        } else {
3101
0
            len = 4;
3102
0
            c = xmlGetUTF8Char(ptr, &len);
3103
0
            if (c < 0)
3104
0
                break;
3105
0
        }
3106
3107
0
        if (flags & XML_SCAN_NMTOKEN ?
3108
0
                !xmlIsNameChar(c, old10) :
3109
0
                !xmlIsNameStartChar(c, old10))
3110
0
            break;
3111
3112
0
        if ((size_t) len > maxSize)
3113
0
            return(NULL);
3114
0
        ptr += len;
3115
0
        maxSize -= len;
3116
0
        flags |= XML_SCAN_NMTOKEN;
3117
0
    }
3118
3119
0
    return(ptr);
3120
0
}
3121
3122
static const xmlChar *
3123
0
xmlParseNameComplex(xmlParserCtxtPtr ctxt) {
3124
0
    const xmlChar *ret;
3125
0
    int len = 0, l;
3126
0
    int c;
3127
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3128
0
                    XML_MAX_TEXT_LENGTH :
3129
0
                    XML_MAX_NAME_LENGTH;
3130
0
    int old10 = (ctxt->options & XML_PARSE_OLD10) ? 1 : 0;
3131
3132
    /*
3133
     * Handler for more complex cases
3134
     */
3135
0
    c = xmlCurrentChar(ctxt, &l);
3136
0
    if (!xmlIsNameStartChar(c, old10))
3137
0
        return(NULL);
3138
0
    len += l;
3139
0
    NEXTL(l);
3140
0
    c = xmlCurrentChar(ctxt, &l);
3141
0
    while (xmlIsNameChar(c, old10)) {
3142
0
        if (len <= INT_MAX - l)
3143
0
            len += l;
3144
0
        NEXTL(l);
3145
0
        c = xmlCurrentChar(ctxt, &l);
3146
0
    }
3147
0
    if (len > maxLength) {
3148
0
        xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "Name");
3149
0
        return(NULL);
3150
0
    }
3151
0
    if (ctxt->input->cur - ctxt->input->base < len) {
3152
        /*
3153
         * There were a couple of bugs where PERefs lead to to a change
3154
         * of the buffer. Check the buffer size to avoid passing an invalid
3155
         * pointer to xmlDictLookup.
3156
         */
3157
0
        xmlFatalErr(ctxt, XML_ERR_INTERNAL_ERROR,
3158
0
                    "unexpected change of input buffer");
3159
0
        return (NULL);
3160
0
    }
3161
0
    if ((*ctxt->input->cur == '\n') && (ctxt->input->cur[-1] == '\r'))
3162
0
        ret = xmlDictLookup(ctxt->dict, ctxt->input->cur - (len + 1), len);
3163
0
    else
3164
0
        ret = xmlDictLookup(ctxt->dict, ctxt->input->cur - len, len);
3165
0
    if (ret == NULL)
3166
0
        xmlErrMemory(ctxt);
3167
0
    return(ret);
3168
0
}
3169
3170
/**
3171
 * Parse an XML name.
3172
 *
3173
 * @deprecated Internal function, don't use.
3174
 *
3175
 *     [4] NameChar ::= Letter | Digit | '.' | '-' | '_' | ':' |
3176
 *                      CombiningChar | Extender
3177
 *
3178
 *     [5] Name ::= (Letter | '_' | ':') (NameChar)*
3179
 *
3180
 *     [6] Names ::= Name (#x20 Name)*
3181
 *
3182
 * @param ctxt  an XML parser context
3183
 * @returns the Name parsed or NULL
3184
 */
3185
3186
const xmlChar *
3187
0
xmlParseName(xmlParserCtxt *ctxt) {
3188
0
    const xmlChar *in;
3189
0
    const xmlChar *ret;
3190
0
    size_t count = 0;
3191
0
    size_t maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3192
0
                       XML_MAX_TEXT_LENGTH :
3193
0
                       XML_MAX_NAME_LENGTH;
3194
3195
0
    GROW;
3196
3197
    /*
3198
     * Accelerator for simple ASCII names
3199
     */
3200
0
    in = ctxt->input->cur;
3201
0
    if (((*in >= 0x61) && (*in <= 0x7A)) ||
3202
0
  ((*in >= 0x41) && (*in <= 0x5A)) ||
3203
0
  (*in == '_') || (*in == ':')) {
3204
0
  in++;
3205
0
  while (((*in >= 0x61) && (*in <= 0x7A)) ||
3206
0
         ((*in >= 0x41) && (*in <= 0x5A)) ||
3207
0
         ((*in >= 0x30) && (*in <= 0x39)) ||
3208
0
         (*in == '_') || (*in == '-') ||
3209
0
         (*in == ':') || (*in == '.'))
3210
0
      in++;
3211
0
  if ((*in > 0) && (*in < 0x80)) {
3212
0
      count = in - ctxt->input->cur;
3213
0
            if (count > maxLength) {
3214
0
                xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "Name");
3215
0
                return(NULL);
3216
0
            }
3217
0
      ret = xmlDictLookup(ctxt->dict, ctxt->input->cur, count);
3218
0
      ctxt->input->cur = in;
3219
0
      ctxt->input->col += count;
3220
0
      if (ret == NULL)
3221
0
          xmlErrMemory(ctxt);
3222
0
      return(ret);
3223
0
  }
3224
0
    }
3225
    /* accelerator for special cases */
3226
0
    return(xmlParseNameComplex(ctxt));
3227
0
}
3228
3229
static xmlHashedString
3230
0
xmlParseNCNameComplex(xmlParserCtxtPtr ctxt) {
3231
0
    xmlHashedString ret;
3232
0
    int len = 0, l;
3233
0
    int c;
3234
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3235
0
                    XML_MAX_TEXT_LENGTH :
3236
0
                    XML_MAX_NAME_LENGTH;
3237
0
    int old10 = (ctxt->options & XML_PARSE_OLD10) ? 1 : 0;
3238
0
    size_t startPosition = 0;
3239
3240
0
    ret.name = NULL;
3241
0
    ret.hashValue = 0;
3242
3243
    /*
3244
     * Handler for more complex cases
3245
     */
3246
0
    startPosition = CUR_PTR - BASE_PTR;
3247
0
    c = xmlCurrentChar(ctxt, &l);
3248
0
    if ((c == ' ') || (c == '>') || (c == '/') || /* accelerators */
3249
0
  (!xmlIsNameStartChar(c, old10) || (c == ':'))) {
3250
0
  return(ret);
3251
0
    }
3252
3253
0
    while ((c != ' ') && (c != '>') && (c != '/') && /* test bigname.xml */
3254
0
     (xmlIsNameChar(c, old10) && (c != ':'))) {
3255
0
        if (len <= INT_MAX - l)
3256
0
      len += l;
3257
0
  NEXTL(l);
3258
0
  c = xmlCurrentChar(ctxt, &l);
3259
0
    }
3260
0
    if (len > maxLength) {
3261
0
        xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "NCName");
3262
0
        return(ret);
3263
0
    }
3264
0
    ret = xmlDictLookupHashed(ctxt->dict, (BASE_PTR + startPosition), len);
3265
0
    if (ret.name == NULL)
3266
0
        xmlErrMemory(ctxt);
3267
0
    return(ret);
3268
0
}
3269
3270
/**
3271
 * Parse an XML name.
3272
 *
3273
 *     [4NS] NCNameChar ::= Letter | Digit | '.' | '-' | '_' |
3274
 *                          CombiningChar | Extender
3275
 *
3276
 *     [5NS] NCName ::= (Letter | '_') (NCNameChar)*
3277
 *
3278
 * @param ctxt  an XML parser context
3279
 * @returns the Name parsed or NULL
3280
 */
3281
3282
static xmlHashedString
3283
0
xmlParseNCName(xmlParserCtxtPtr ctxt) {
3284
0
    const xmlChar *in, *e;
3285
0
    xmlHashedString ret;
3286
0
    size_t count = 0;
3287
0
    size_t maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3288
0
                       XML_MAX_TEXT_LENGTH :
3289
0
                       XML_MAX_NAME_LENGTH;
3290
3291
0
    ret.name = NULL;
3292
3293
    /*
3294
     * Accelerator for simple ASCII names
3295
     */
3296
0
    in = ctxt->input->cur;
3297
0
    e = ctxt->input->end;
3298
0
    if ((((*in >= 0x61) && (*in <= 0x7A)) ||
3299
0
   ((*in >= 0x41) && (*in <= 0x5A)) ||
3300
0
   (*in == '_')) && (in < e)) {
3301
0
  in++;
3302
0
  while ((((*in >= 0x61) && (*in <= 0x7A)) ||
3303
0
          ((*in >= 0x41) && (*in <= 0x5A)) ||
3304
0
          ((*in >= 0x30) && (*in <= 0x39)) ||
3305
0
          (*in == '_') || (*in == '-') ||
3306
0
          (*in == '.')) && (in < e))
3307
0
      in++;
3308
0
  if (in >= e)
3309
0
      goto complex;
3310
0
  if ((*in > 0) && (*in < 0x80)) {
3311
0
      count = in - ctxt->input->cur;
3312
0
            if (count > maxLength) {
3313
0
                xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "NCName");
3314
0
                return(ret);
3315
0
            }
3316
0
      ret = xmlDictLookupHashed(ctxt->dict, ctxt->input->cur, count);
3317
0
      ctxt->input->cur = in;
3318
0
      ctxt->input->col += count;
3319
0
      if (ret.name == NULL) {
3320
0
          xmlErrMemory(ctxt);
3321
0
      }
3322
0
      return(ret);
3323
0
  }
3324
0
    }
3325
0
complex:
3326
0
    return(xmlParseNCNameComplex(ctxt));
3327
0
}
3328
3329
/**
3330
 * Parse an XML name and compares for match
3331
 * (specialized for endtag parsing)
3332
 *
3333
 * @param ctxt  an XML parser context
3334
 * @param other  the name to compare with
3335
 * @returns NULL for an illegal name, (xmlChar*) 1 for success
3336
 * and the name for mismatch
3337
 */
3338
3339
static const xmlChar *
3340
0
xmlParseNameAndCompare(xmlParserCtxtPtr ctxt, xmlChar const *other) {
3341
0
    register const xmlChar *cmp = other;
3342
0
    register const xmlChar *in;
3343
0
    const xmlChar *ret;
3344
3345
0
    GROW;
3346
3347
0
    in = ctxt->input->cur;
3348
0
    while (*in != 0 && *in == *cmp) {
3349
0
  ++in;
3350
0
  ++cmp;
3351
0
    }
3352
0
    if (*cmp == 0 && (*in == '>' || IS_BLANK_CH (*in))) {
3353
  /* success */
3354
0
  ctxt->input->col += in - ctxt->input->cur;
3355
0
  ctxt->input->cur = in;
3356
0
  return (const xmlChar*) 1;
3357
0
    }
3358
    /* failure (or end of input buffer), check with full function */
3359
0
    ret = xmlParseName (ctxt);
3360
    /* strings coming from the dictionary direct compare possible */
3361
0
    if (ret == other) {
3362
0
  return (const xmlChar*) 1;
3363
0
    }
3364
0
    return ret;
3365
0
}
3366
3367
/**
3368
 * Parse an XML name.
3369
 *
3370
 * @param ctxt  an XML parser context
3371
 * @param str  a pointer to the string pointer (IN/OUT)
3372
 * @returns the Name parsed or NULL. The `str` pointer
3373
 * is updated to the current location in the string.
3374
 */
3375
3376
static xmlChar *
3377
0
xmlParseStringName(xmlParserCtxtPtr ctxt, const xmlChar** str) {
3378
0
    xmlChar *ret;
3379
0
    const xmlChar *cur = *str;
3380
0
    int flags = 0;
3381
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3382
0
                    XML_MAX_TEXT_LENGTH :
3383
0
                    XML_MAX_NAME_LENGTH;
3384
3385
0
    if (ctxt->options & XML_PARSE_OLD10)
3386
0
        flags |= XML_SCAN_OLD10;
3387
3388
0
    cur = xmlScanName(*str, maxLength, flags);
3389
0
    if (cur == NULL) {
3390
0
        xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "NCName");
3391
0
        return(NULL);
3392
0
    }
3393
0
    if (cur == *str)
3394
0
        return(NULL);
3395
3396
0
    ret = xmlStrndup(*str, cur - *str);
3397
0
    if (ret == NULL)
3398
0
        xmlErrMemory(ctxt);
3399
0
    *str = cur;
3400
0
    return(ret);
3401
0
}
3402
3403
/**
3404
 * Parse an XML Nmtoken.
3405
 *
3406
 * @deprecated Internal function, don't use.
3407
 *
3408
 *     [7] Nmtoken ::= (NameChar)+
3409
 *
3410
 *     [8] Nmtokens ::= Nmtoken (#x20 Nmtoken)*
3411
 *
3412
 * @param ctxt  an XML parser context
3413
 * @returns the Nmtoken parsed or NULL
3414
 */
3415
3416
xmlChar *
3417
0
xmlParseNmtoken(xmlParserCtxt *ctxt) {
3418
0
    xmlChar buf[XML_MAX_NAMELEN + 5];
3419
0
    xmlChar *ret;
3420
0
    int len = 0, l;
3421
0
    int c;
3422
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3423
0
                    XML_MAX_TEXT_LENGTH :
3424
0
                    XML_MAX_NAME_LENGTH;
3425
0
    int old10 = (ctxt->options & XML_PARSE_OLD10) ? 1 : 0;
3426
3427
0
    c = xmlCurrentChar(ctxt, &l);
3428
3429
0
    while (xmlIsNameChar(c, old10)) {
3430
0
  COPY_BUF(buf, len, c);
3431
0
  NEXTL(l);
3432
0
  c = xmlCurrentChar(ctxt, &l);
3433
0
  if (len >= XML_MAX_NAMELEN) {
3434
      /*
3435
       * Okay someone managed to make a huge token, so he's ready to pay
3436
       * for the processing speed.
3437
       */
3438
0
      xmlChar *buffer;
3439
0
      int max = len * 2;
3440
3441
0
      buffer = xmlMalloc(max);
3442
0
      if (buffer == NULL) {
3443
0
          xmlErrMemory(ctxt);
3444
0
    return(NULL);
3445
0
      }
3446
0
      memcpy(buffer, buf, len);
3447
0
      while (xmlIsNameChar(c, old10)) {
3448
0
    if (len + 10 > max) {
3449
0
        xmlChar *tmp;
3450
0
                    int newSize;
3451
3452
0
                    newSize = xmlGrowCapacity(max, 1, 1, maxLength);
3453
0
                    if (newSize < 0) {
3454
0
                        xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "NmToken");
3455
0
                        xmlFree(buffer);
3456
0
                        return(NULL);
3457
0
                    }
3458
0
        tmp = xmlRealloc(buffer, newSize);
3459
0
        if (tmp == NULL) {
3460
0
      xmlErrMemory(ctxt);
3461
0
      xmlFree(buffer);
3462
0
      return(NULL);
3463
0
        }
3464
0
        buffer = tmp;
3465
0
                    max = newSize;
3466
0
    }
3467
0
    COPY_BUF(buffer, len, c);
3468
0
    NEXTL(l);
3469
0
    c = xmlCurrentChar(ctxt, &l);
3470
0
      }
3471
0
      buffer[len] = 0;
3472
0
      return(buffer);
3473
0
  }
3474
0
    }
3475
0
    if (len == 0)
3476
0
        return(NULL);
3477
0
    if (len > maxLength) {
3478
0
        xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "NmToken");
3479
0
        return(NULL);
3480
0
    }
3481
0
    ret = xmlStrndup(buf, len);
3482
0
    if (ret == NULL)
3483
0
        xmlErrMemory(ctxt);
3484
0
    return(ret);
3485
0
}
3486
3487
/**
3488
 * Validate an entity value and expand parameter entities.
3489
 *
3490
 * @param ctxt  parser context
3491
 * @param buf  string buffer
3492
 * @param str  entity value
3493
 * @param length  size of entity value
3494
 * @param depth  nesting depth
3495
 */
3496
static void
3497
xmlExpandPEsInEntityValue(xmlParserCtxtPtr ctxt, xmlSBuf *buf,
3498
0
                          const xmlChar *str, int length, int depth) {
3499
0
    int maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 40 : 20;
3500
0
    const xmlChar *end, *chunk;
3501
0
    int c, l;
3502
3503
0
    if (str == NULL)
3504
0
        return;
3505
3506
0
    depth += 1;
3507
0
    if (depth > maxDepth) {
3508
0
  xmlFatalErrMsg(ctxt, XML_ERR_RESOURCE_LIMIT,
3509
0
                       "Maximum entity nesting depth exceeded");
3510
0
  return;
3511
0
    }
3512
3513
0
    end = str + length;
3514
0
    chunk = str;
3515
3516
0
    while ((str < end) && (!PARSER_STOPPED(ctxt))) {
3517
0
        c = *str;
3518
3519
0
        if (c >= 0x80) {
3520
0
            l = xmlUTF8MultibyteLen(ctxt, str,
3521
0
                    "invalid character in entity value\n");
3522
0
            if (l == 0) {
3523
0
                if (chunk < str)
3524
0
                    xmlSBufAddString(buf, chunk, str - chunk);
3525
0
                xmlSBufAddReplChar(buf);
3526
0
                str += 1;
3527
0
                chunk = str;
3528
0
            } else {
3529
0
                str += l;
3530
0
            }
3531
0
        } else if (c == '&') {
3532
0
            if (str[1] == '#') {
3533
0
                if (chunk < str)
3534
0
                    xmlSBufAddString(buf, chunk, str - chunk);
3535
3536
0
                c = xmlParseStringCharRef(ctxt, &str);
3537
0
                if (c == 0)
3538
0
                    return;
3539
3540
0
                xmlSBufAddChar(buf, c);
3541
3542
0
                chunk = str;
3543
0
            } else {
3544
0
                xmlChar *name;
3545
3546
                /*
3547
                 * General entity references are checked for
3548
                 * syntactic validity.
3549
                 */
3550
0
                str++;
3551
0
                name = xmlParseStringName(ctxt, &str);
3552
3553
0
                if ((name == NULL) || (*str++ != ';')) {
3554
0
                    xmlFatalErrMsg(ctxt, XML_ERR_ENTITY_CHAR_ERROR,
3555
0
                            "EntityValue: '&' forbidden except for entities "
3556
0
                            "references\n");
3557
0
                    xmlFree(name);
3558
0
                    return;
3559
0
                }
3560
3561
0
                xmlFree(name);
3562
0
            }
3563
0
        } else if (c == '%') {
3564
0
            xmlEntityPtr ent;
3565
3566
0
            if (chunk < str)
3567
0
                xmlSBufAddString(buf, chunk, str - chunk);
3568
3569
0
            ent = xmlParseStringPEReference(ctxt, &str);
3570
0
            if (ent == NULL)
3571
0
                return;
3572
3573
0
            if (!PARSER_EXTERNAL(ctxt)) {
3574
0
                xmlFatalErr(ctxt, XML_ERR_ENTITY_PE_INTERNAL, NULL);
3575
0
                return;
3576
0
            }
3577
3578
0
            if (ent->content == NULL) {
3579
                /*
3580
                 * Note: external parsed entities will not be loaded,
3581
                 * it is not required for a non-validating parser to
3582
                 * complete external PEReferences coming from the
3583
                 * internal subset
3584
                 */
3585
0
                if (((ctxt->options & XML_PARSE_NO_XXE) == 0) &&
3586
0
                    ((ctxt->replaceEntities) ||
3587
0
                     (ctxt->validate))) {
3588
0
                    xmlLoadEntityContent(ctxt, ent);
3589
0
                } else {
3590
0
                    xmlWarningMsg(ctxt, XML_ERR_ENTITY_PROCESSING,
3591
0
                                  "not validating will not read content for "
3592
0
                                  "PE entity %s\n", ent->name, NULL);
3593
0
                }
3594
0
            }
3595
3596
            /*
3597
             * TODO: Skip if ent->content is still NULL.
3598
             */
3599
3600
0
            if (xmlParserEntityCheck(ctxt, ent->length))
3601
0
                return;
3602
3603
0
            if (ent->flags & XML_ENT_EXPANDING) {
3604
0
                xmlFatalErr(ctxt, XML_ERR_ENTITY_LOOP, NULL);
3605
0
                return;
3606
0
            }
3607
3608
0
            ent->flags |= XML_ENT_EXPANDING;
3609
0
            xmlExpandPEsInEntityValue(ctxt, buf, ent->content, ent->length,
3610
0
                                      depth);
3611
0
            ent->flags &= ~XML_ENT_EXPANDING;
3612
3613
0
            chunk = str;
3614
0
        } else {
3615
            /* Normal ASCII char */
3616
0
            if (!IS_BYTE_CHAR(c)) {
3617
0
                xmlFatalErrMsg(ctxt, XML_ERR_INVALID_CHAR,
3618
0
                        "invalid character in entity value\n");
3619
0
                if (chunk < str)
3620
0
                    xmlSBufAddString(buf, chunk, str - chunk);
3621
0
                xmlSBufAddReplChar(buf);
3622
0
                str += 1;
3623
0
                chunk = str;
3624
0
            } else {
3625
0
                str += 1;
3626
0
            }
3627
0
        }
3628
0
    }
3629
3630
0
    if (chunk < str)
3631
0
        xmlSBufAddString(buf, chunk, str - chunk);
3632
0
}
3633
3634
/**
3635
 * Parse a value for ENTITY declarations
3636
 *
3637
 * @deprecated Internal function, don't use.
3638
 *
3639
 *     [9] EntityValue ::= '"' ([^%&"] | PEReference | Reference)* '"' |
3640
 *                         "'" ([^%&'] | PEReference | Reference)* "'"
3641
 *
3642
 * @param ctxt  an XML parser context
3643
 * @param orig  if non-NULL store a copy of the original entity value
3644
 * @returns the EntityValue parsed with reference substituted or NULL
3645
 */
3646
xmlChar *
3647
0
xmlParseEntityValue(xmlParserCtxt *ctxt, xmlChar **orig) {
3648
0
    unsigned maxLength = (ctxt->options & XML_PARSE_HUGE) ?
3649
0
                         XML_MAX_HUGE_LENGTH :
3650
0
                         XML_MAX_TEXT_LENGTH;
3651
0
    xmlSBuf buf;
3652
0
    const xmlChar *start;
3653
0
    int quote, length;
3654
3655
0
    xmlSBufInit(&buf, maxLength);
3656
3657
0
    GROW;
3658
3659
0
    quote = CUR;
3660
0
    if ((quote != '"') && (quote != '\'')) {
3661
0
  xmlFatalErr(ctxt, XML_ERR_ATTRIBUTE_NOT_STARTED, NULL);
3662
0
  return(NULL);
3663
0
    }
3664
0
    CUR_PTR++;
3665
3666
0
    length = 0;
3667
3668
    /*
3669
     * Copy raw content of the entity into a buffer
3670
     */
3671
0
    while (1) {
3672
0
        int c;
3673
3674
0
        if (PARSER_STOPPED(ctxt))
3675
0
            goto error;
3676
3677
0
        if (CUR_PTR >= ctxt->input->end) {
3678
0
            xmlFatalErrMsg(ctxt, XML_ERR_ENTITY_NOT_FINISHED, NULL);
3679
0
            goto error;
3680
0
        }
3681
3682
0
        c = CUR;
3683
3684
0
        if (c == 0) {
3685
0
            xmlFatalErrMsg(ctxt, XML_ERR_INVALID_CHAR,
3686
0
                    "invalid character in entity value\n");
3687
0
            goto error;
3688
0
        }
3689
0
        if (c == quote)
3690
0
            break;
3691
0
        NEXTL(1);
3692
0
        length += 1;
3693
3694
        /*
3695
         * TODO: Check growth threshold
3696
         */
3697
0
        if (ctxt->input->end - CUR_PTR < 10)
3698
0
            GROW;
3699
0
    }
3700
3701
0
    start = CUR_PTR - length;
3702
3703
0
    if (orig != NULL) {
3704
0
        *orig = xmlStrndup(start, length);
3705
0
        if (*orig == NULL)
3706
0
            xmlErrMemory(ctxt);
3707
0
    }
3708
3709
0
    xmlExpandPEsInEntityValue(ctxt, &buf, start, length, ctxt->inputNr);
3710
3711
0
    NEXTL(1);
3712
3713
0
    return(xmlSBufFinish(&buf, NULL, ctxt, "entity length too long"));
3714
3715
0
error:
3716
0
    xmlSBufCleanup(&buf, ctxt, "entity length too long");
3717
0
    return(NULL);
3718
0
}
3719
3720
/**
3721
 * Check an entity reference in an attribute value for validity
3722
 * without expanding it.
3723
 *
3724
 * @param ctxt  parser context
3725
 * @param pent  entity
3726
 * @param depth  nesting depth
3727
 */
3728
static void
3729
0
xmlCheckEntityInAttValue(xmlParserCtxtPtr ctxt, xmlEntityPtr pent, int depth) {
3730
0
    int maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 40 : 20;
3731
0
    const xmlChar *str;
3732
0
    unsigned long expandedSize = pent->length;
3733
0
    int c, flags;
3734
3735
0
    depth += 1;
3736
0
    if (depth > maxDepth) {
3737
0
  xmlFatalErrMsg(ctxt, XML_ERR_RESOURCE_LIMIT,
3738
0
                       "Maximum entity nesting depth exceeded");
3739
0
  return;
3740
0
    }
3741
3742
0
    if (pent->flags & XML_ENT_EXPANDING) {
3743
0
        xmlFatalErr(ctxt, XML_ERR_ENTITY_LOOP, NULL);
3744
0
        return;
3745
0
    }
3746
3747
    /*
3748
     * If we're parsing a default attribute value in DTD content,
3749
     * the entity might reference other entities which weren't
3750
     * defined yet, so the check isn't reliable.
3751
     */
3752
0
    if (ctxt->inSubset == 0)
3753
0
        flags = XML_ENT_CHECKED | XML_ENT_VALIDATED;
3754
0
    else
3755
0
        flags = XML_ENT_VALIDATED;
3756
3757
0
    str = pent->content;
3758
0
    if (str == NULL)
3759
0
        goto done;
3760
3761
    /*
3762
     * Note that entity values are already validated. We only check
3763
     * for illegal less-than signs and compute the expanded size
3764
     * of the entity. No special handling for multi-byte characters
3765
     * is needed.
3766
     */
3767
0
    while (!PARSER_STOPPED(ctxt)) {
3768
0
        c = *str;
3769
3770
0
  if (c != '&') {
3771
0
            if (c == 0)
3772
0
                break;
3773
3774
0
            if (c == '<')
3775
0
                xmlFatalErrMsgStr(ctxt, XML_ERR_LT_IN_ATTRIBUTE,
3776
0
                        "'<' in entity '%s' is not allowed in attributes "
3777
0
                        "values\n", pent->name);
3778
3779
0
            str += 1;
3780
0
        } else if (str[1] == '#') {
3781
0
            int val;
3782
3783
0
      val = xmlParseStringCharRef(ctxt, &str);
3784
0
      if (val == 0) {
3785
0
                pent->content[0] = 0;
3786
0
                break;
3787
0
            }
3788
0
  } else {
3789
0
            xmlChar *name;
3790
0
            xmlEntityPtr ent;
3791
3792
0
      name = xmlParseStringEntityRef(ctxt, &str);
3793
0
      if (name == NULL) {
3794
0
                pent->content[0] = 0;
3795
0
                break;
3796
0
            }
3797
3798
0
            ent = xmlLookupGeneralEntity(ctxt, name, /* inAttr */ 1);
3799
0
            xmlFree(name);
3800
3801
0
            if ((ent != NULL) &&
3802
0
                (ent->etype != XML_INTERNAL_PREDEFINED_ENTITY)) {
3803
0
                if ((ent->flags & flags) != flags) {
3804
0
                    pent->flags |= XML_ENT_EXPANDING;
3805
0
                    xmlCheckEntityInAttValue(ctxt, ent, depth);
3806
0
                    pent->flags &= ~XML_ENT_EXPANDING;
3807
0
                }
3808
3809
0
                xmlSaturatedAdd(&expandedSize, ent->expandedSize);
3810
0
                xmlSaturatedAdd(&expandedSize, XML_ENT_FIXED_COST);
3811
0
            }
3812
0
        }
3813
0
    }
3814
3815
0
done:
3816
0
    if (ctxt->inSubset == 0)
3817
0
        pent->expandedSize = expandedSize;
3818
3819
0
    pent->flags |= flags;
3820
0
}
3821
3822
/**
3823
 * Expand general entity references in an entity or attribute value.
3824
 * Perform attribute value normalization.
3825
 *
3826
 * @param ctxt  parser context
3827
 * @param buf  string buffer
3828
 * @param str  entity or attribute value
3829
 * @param pent  entity for entity value, NULL for attribute values
3830
 * @param normalize  whether to collapse whitespace
3831
 * @param inSpace  whitespace state
3832
 * @param depth  nesting depth
3833
 * @param check  whether to check for amplification
3834
 * @returns  whether there was a normalization change
3835
 */
3836
static int
3837
xmlExpandEntityInAttValue(xmlParserCtxtPtr ctxt, xmlSBuf *buf,
3838
                          const xmlChar *str, xmlEntityPtr pent, int normalize,
3839
0
                          int *inSpace, int depth, int check) {
3840
0
    int maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 40 : 20;
3841
0
    int c, chunkSize;
3842
0
    int normChange = 0;
3843
3844
0
    if (str == NULL)
3845
0
        return(0);
3846
3847
0
    depth += 1;
3848
0
    if (depth > maxDepth) {
3849
0
  xmlFatalErrMsg(ctxt, XML_ERR_RESOURCE_LIMIT,
3850
0
                       "Maximum entity nesting depth exceeded");
3851
0
  return(0);
3852
0
    }
3853
3854
0
    if (pent != NULL) {
3855
0
        if (pent->flags & XML_ENT_EXPANDING) {
3856
0
            xmlFatalErr(ctxt, XML_ERR_ENTITY_LOOP, NULL);
3857
0
            return(0);
3858
0
        }
3859
3860
0
        if (check) {
3861
0
            if (xmlParserEntityCheck(ctxt, pent->length))
3862
0
                return(0);
3863
0
        }
3864
0
    }
3865
3866
0
    chunkSize = 0;
3867
3868
    /*
3869
     * Note that entity values are already validated. No special
3870
     * handling for multi-byte characters is needed.
3871
     */
3872
0
    while (!PARSER_STOPPED(ctxt)) {
3873
0
        c = *str;
3874
3875
0
  if (c != '&') {
3876
0
            if (c == 0)
3877
0
                break;
3878
3879
            /*
3880
             * If this function is called without an entity, it is used to
3881
             * expand entities in an attribute content where less-than was
3882
             * already unscaped and is allowed.
3883
             */
3884
0
            if ((pent != NULL) && (c == '<')) {
3885
0
                xmlFatalErrMsgStr(ctxt, XML_ERR_LT_IN_ATTRIBUTE,
3886
0
                        "'<' in entity '%s' is not allowed in attributes "
3887
0
                        "values\n", pent->name);
3888
0
                break;
3889
0
            }
3890
3891
0
            if (c <= 0x20) {
3892
0
                if ((normalize) && (*inSpace)) {
3893
                    /* Skip char */
3894
0
                    if (chunkSize > 0) {
3895
0
                        xmlSBufAddString(buf, str - chunkSize, chunkSize);
3896
0
                        chunkSize = 0;
3897
0
                    }
3898
0
                    normChange = 1;
3899
0
                } else if (c < 0x20) {
3900
0
                    if (chunkSize > 0) {
3901
0
                        xmlSBufAddString(buf, str - chunkSize, chunkSize);
3902
0
                        chunkSize = 0;
3903
0
                    }
3904
3905
0
                    xmlSBufAddCString(buf, " ", 1);
3906
0
                } else {
3907
0
                    chunkSize += 1;
3908
0
                }
3909
3910
0
                *inSpace = 1;
3911
0
            } else {
3912
0
                chunkSize += 1;
3913
0
                *inSpace = 0;
3914
0
            }
3915
3916
0
            str += 1;
3917
0
        } else if (str[1] == '#') {
3918
0
            int val;
3919
3920
0
            if (chunkSize > 0) {
3921
0
                xmlSBufAddString(buf, str - chunkSize, chunkSize);
3922
0
                chunkSize = 0;
3923
0
            }
3924
3925
0
      val = xmlParseStringCharRef(ctxt, &str);
3926
0
      if (val == 0) {
3927
0
                if (pent != NULL)
3928
0
                    pent->content[0] = 0;
3929
0
                break;
3930
0
            }
3931
3932
0
            if (val == ' ') {
3933
0
                if ((normalize) && (*inSpace))
3934
0
                    normChange = 1;
3935
0
                else
3936
0
                    xmlSBufAddCString(buf, " ", 1);
3937
0
                *inSpace = 1;
3938
0
            } else {
3939
0
                xmlSBufAddChar(buf, val);
3940
0
                *inSpace = 0;
3941
0
            }
3942
0
  } else {
3943
0
            xmlChar *name;
3944
0
            xmlEntityPtr ent;
3945
3946
0
            if (chunkSize > 0) {
3947
0
                xmlSBufAddString(buf, str - chunkSize, chunkSize);
3948
0
                chunkSize = 0;
3949
0
            }
3950
3951
0
      name = xmlParseStringEntityRef(ctxt, &str);
3952
0
            if (name == NULL) {
3953
0
                if (pent != NULL)
3954
0
                    pent->content[0] = 0;
3955
0
                break;
3956
0
            }
3957
3958
0
            ent = xmlLookupGeneralEntity(ctxt, name, /* inAttr */ 1);
3959
0
            xmlFree(name);
3960
3961
0
      if ((ent != NULL) &&
3962
0
    (ent->etype == XML_INTERNAL_PREDEFINED_ENTITY)) {
3963
0
    if (ent->content == NULL) {
3964
0
        xmlFatalErrMsg(ctxt, XML_ERR_INTERNAL_ERROR,
3965
0
          "predefined entity has no content\n");
3966
0
                    break;
3967
0
                }
3968
3969
0
                xmlSBufAddString(buf, ent->content, ent->length);
3970
3971
0
                *inSpace = 0;
3972
0
      } else if ((ent != NULL) && (ent->content != NULL)) {
3973
0
                if (pent != NULL)
3974
0
                    pent->flags |= XML_ENT_EXPANDING;
3975
0
    normChange |= xmlExpandEntityInAttValue(ctxt, buf,
3976
0
                        ent->content, ent, normalize, inSpace, depth, check);
3977
0
                if (pent != NULL)
3978
0
                    pent->flags &= ~XML_ENT_EXPANDING;
3979
0
      }
3980
0
        }
3981
0
    }
3982
3983
0
    if (chunkSize > 0)
3984
0
        xmlSBufAddString(buf, str - chunkSize, chunkSize);
3985
3986
0
    return(normChange);
3987
0
}
3988
3989
/**
3990
 * Expand general entity references in an entity or attribute value.
3991
 * Perform attribute value normalization.
3992
 *
3993
 * @param ctxt  parser context
3994
 * @param str  entity or attribute value
3995
 * @param normalize  whether to collapse whitespace
3996
 * @returns the expanded attribtue value.
3997
 */
3998
xmlChar *
3999
xmlExpandEntitiesInAttValue(xmlParserCtxt *ctxt, const xmlChar *str,
4000
0
                            int normalize) {
4001
0
    unsigned maxLength = (ctxt->options & XML_PARSE_HUGE) ?
4002
0
                         XML_MAX_HUGE_LENGTH :
4003
0
                         XML_MAX_TEXT_LENGTH;
4004
0
    xmlSBuf buf;
4005
0
    int inSpace = 1;
4006
4007
0
    xmlSBufInit(&buf, maxLength);
4008
4009
0
    xmlExpandEntityInAttValue(ctxt, &buf, str, NULL, normalize, &inSpace,
4010
0
                              ctxt->inputNr, /* check */ 0);
4011
4012
0
    if ((normalize) && (inSpace) && (buf.size > 0))
4013
0
        buf.size--;
4014
4015
0
    return(xmlSBufFinish(&buf, NULL, ctxt, "AttValue length too long"));
4016
0
}
4017
4018
/**
4019
 * Parse a value for an attribute.
4020
 *
4021
 * NOTE: if no normalization is needed, the routine will return pointers
4022
 * directly from the data buffer.
4023
 *
4024
 * 3.3.3 Attribute-Value Normalization:
4025
 *
4026
 * Before the value of an attribute is passed to the application or
4027
 * checked for validity, the XML processor must normalize it as follows:
4028
 *
4029
 * - a character reference is processed by appending the referenced
4030
 *   character to the attribute value
4031
 * - an entity reference is processed by recursively processing the
4032
 *   replacement text of the entity
4033
 * - a whitespace character (\#x20, \#xD, \#xA, \#x9) is processed by
4034
 *   appending \#x20 to the normalized value, except that only a single
4035
 *   \#x20 is appended for a "#xD#xA" sequence that is part of an external
4036
 *   parsed entity or the literal entity value of an internal parsed entity
4037
 * - other characters are processed by appending them to the normalized value
4038
 *
4039
 * If the declared value is not CDATA, then the XML processor must further
4040
 * process the normalized attribute value by discarding any leading and
4041
 * trailing space (\#x20) characters, and by replacing sequences of space
4042
 * (\#x20) characters by a single space (\#x20) character.
4043
 * All attributes for which no declaration has been read should be treated
4044
 * by a non-validating parser as if declared CDATA.
4045
 *
4046
 * @param ctxt  an XML parser context
4047
 * @param attlen  attribute len result
4048
 * @param outFlags  resulting XML_ATTVAL_* flags
4049
 * @param special  value from attsSpecial
4050
 * @param isNamespace  whether this is a namespace declaration
4051
 * @returns the AttValue parsed or NULL. The value has to be freed by the
4052
 *     caller if it was copied, this can be detected by val[*len] == 0.
4053
 */
4054
static xmlChar *
4055
xmlParseAttValueInternal(xmlParserCtxtPtr ctxt, int *attlen, int *outFlags,
4056
0
                         int special, int isNamespace) {
4057
0
    unsigned maxLength = (ctxt->options & XML_PARSE_HUGE) ?
4058
0
                         XML_MAX_HUGE_LENGTH :
4059
0
                         XML_MAX_TEXT_LENGTH;
4060
0
    xmlSBuf buf;
4061
0
    xmlChar *ret;
4062
0
    int c, l, quote, entFlags, chunkSize;
4063
0
    int inSpace = 1;
4064
0
    int replaceEntities;
4065
0
    int normalize = (special & XML_SPECIAL_TYPE_MASK) > XML_ATTRIBUTE_CDATA;
4066
0
    int attvalFlags = 0;
4067
4068
    /* Always expand namespace URIs */
4069
0
    replaceEntities = (ctxt->replaceEntities) || (isNamespace);
4070
4071
0
    xmlSBufInit(&buf, maxLength);
4072
4073
0
    GROW;
4074
4075
0
    quote = CUR;
4076
0
    if ((quote != '"') && (quote != '\'')) {
4077
0
  xmlFatalErr(ctxt, XML_ERR_ATTRIBUTE_NOT_STARTED, NULL);
4078
0
  return(NULL);
4079
0
    }
4080
0
    NEXTL(1);
4081
4082
0
    if (ctxt->inSubset == 0)
4083
0
        entFlags = XML_ENT_CHECKED | XML_ENT_VALIDATED;
4084
0
    else
4085
0
        entFlags = XML_ENT_VALIDATED;
4086
4087
0
    inSpace = 1;
4088
0
    chunkSize = 0;
4089
4090
0
    while (1) {
4091
0
        if (PARSER_STOPPED(ctxt))
4092
0
            goto error;
4093
4094
0
        if (CUR_PTR >= ctxt->input->end) {
4095
0
            xmlFatalErrMsg(ctxt, XML_ERR_ATTRIBUTE_NOT_FINISHED,
4096
0
                           "AttValue: ' expected\n");
4097
0
            goto error;
4098
0
        }
4099
4100
        /*
4101
         * TODO: Check growth threshold
4102
         */
4103
0
        if (ctxt->input->end - CUR_PTR < 10)
4104
0
            GROW;
4105
4106
0
        c = CUR;
4107
4108
0
        if (c >= 0x80) {
4109
0
            l = xmlUTF8MultibyteLen(ctxt, CUR_PTR,
4110
0
                    "invalid character in attribute value\n");
4111
0
            if (l == 0) {
4112
0
                if (chunkSize > 0) {
4113
0
                    xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4114
0
                    chunkSize = 0;
4115
0
                }
4116
0
                xmlSBufAddReplChar(&buf);
4117
0
                NEXTL(1);
4118
0
            } else {
4119
0
                chunkSize += l;
4120
0
                NEXTL(l);
4121
0
            }
4122
4123
0
            inSpace = 0;
4124
0
        } else if (c != '&') {
4125
0
            if (c > 0x20) {
4126
0
                if (c == quote)
4127
0
                    break;
4128
4129
0
                if (c == '<')
4130
0
                    xmlFatalErr(ctxt, XML_ERR_LT_IN_ATTRIBUTE, NULL);
4131
4132
0
                chunkSize += 1;
4133
0
                inSpace = 0;
4134
0
            } else if (!IS_BYTE_CHAR(c)) {
4135
0
                xmlFatalErrMsg(ctxt, XML_ERR_INVALID_CHAR,
4136
0
                        "invalid character in attribute value\n");
4137
0
                if (chunkSize > 0) {
4138
0
                    xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4139
0
                    chunkSize = 0;
4140
0
                }
4141
0
                xmlSBufAddReplChar(&buf);
4142
0
                inSpace = 0;
4143
0
            } else {
4144
                /* Whitespace */
4145
0
                if ((normalize) && (inSpace)) {
4146
                    /* Skip char */
4147
0
                    if (chunkSize > 0) {
4148
0
                        xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4149
0
                        chunkSize = 0;
4150
0
                    }
4151
0
                    attvalFlags |= XML_ATTVAL_NORM_CHANGE;
4152
0
                } else if (c < 0x20) {
4153
                    /* Convert to space */
4154
0
                    if (chunkSize > 0) {
4155
0
                        xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4156
0
                        chunkSize = 0;
4157
0
                    }
4158
4159
0
                    xmlSBufAddCString(&buf, " ", 1);
4160
0
                } else {
4161
0
                    chunkSize += 1;
4162
0
                }
4163
4164
0
                inSpace = 1;
4165
4166
0
                if ((c == 0xD) && (NXT(1) == 0xA))
4167
0
                    CUR_PTR++;
4168
0
            }
4169
4170
0
            NEXTL(1);
4171
0
        } else if (NXT(1) == '#') {
4172
0
            int val;
4173
4174
0
            if (chunkSize > 0) {
4175
0
                xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4176
0
                chunkSize = 0;
4177
0
            }
4178
4179
0
            val = xmlParseCharRef(ctxt);
4180
0
            if (val == 0)
4181
0
                goto error;
4182
4183
0
            if ((val == '&') && (!replaceEntities)) {
4184
                /*
4185
                 * The reparsing will be done in xmlNodeParseContent()
4186
                 * called from SAX2.c
4187
                 */
4188
0
                xmlSBufAddCString(&buf, "&#38;", 5);
4189
0
                inSpace = 0;
4190
0
            } else if (val == ' ') {
4191
0
                if ((normalize) && (inSpace))
4192
0
                    attvalFlags |= XML_ATTVAL_NORM_CHANGE;
4193
0
                else
4194
0
                    xmlSBufAddCString(&buf, " ", 1);
4195
0
                inSpace = 1;
4196
0
            } else {
4197
0
                xmlSBufAddChar(&buf, val);
4198
0
                inSpace = 0;
4199
0
            }
4200
0
        } else {
4201
0
            const xmlChar *name;
4202
0
            xmlEntityPtr ent;
4203
4204
0
            if (chunkSize > 0) {
4205
0
                xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4206
0
                chunkSize = 0;
4207
0
            }
4208
4209
0
            name = xmlParseEntityRefInternal(ctxt);
4210
0
            if (name == NULL) {
4211
                /*
4212
                 * Probably a literal '&' which wasn't escaped.
4213
                 * TODO: Handle gracefully in recovery mode.
4214
                 */
4215
0
                continue;
4216
0
            }
4217
4218
0
            ent = xmlLookupGeneralEntity(ctxt, name, /* isAttr */ 1);
4219
0
            if (ent == NULL)
4220
0
                continue;
4221
4222
0
            if (ent->etype == XML_INTERNAL_PREDEFINED_ENTITY) {
4223
0
                if ((ent->content[0] == '&') && (!replaceEntities))
4224
0
                    xmlSBufAddCString(&buf, "&#38;", 5);
4225
0
                else
4226
0
                    xmlSBufAddString(&buf, ent->content, ent->length);
4227
0
                inSpace = 0;
4228
0
            } else if (replaceEntities) {
4229
0
                if (xmlExpandEntityInAttValue(ctxt, &buf,
4230
0
                        ent->content, ent, normalize, &inSpace, ctxt->inputNr,
4231
0
                        /* check */ 1) > 0)
4232
0
                    attvalFlags |= XML_ATTVAL_NORM_CHANGE;
4233
0
            } else {
4234
0
                if ((ent->flags & entFlags) != entFlags)
4235
0
                    xmlCheckEntityInAttValue(ctxt, ent, ctxt->inputNr);
4236
4237
0
                if (xmlParserEntityCheck(ctxt, ent->expandedSize)) {
4238
0
                    ent->content[0] = 0;
4239
0
                    goto error;
4240
0
                }
4241
4242
                /*
4243
                 * Just output the reference
4244
                 */
4245
0
                xmlSBufAddCString(&buf, "&", 1);
4246
0
                xmlSBufAddString(&buf, ent->name, xmlStrlen(ent->name));
4247
0
                xmlSBufAddCString(&buf, ";", 1);
4248
4249
0
                inSpace = 0;
4250
0
            }
4251
0
  }
4252
0
    }
4253
4254
0
    if ((buf.mem == NULL) && (outFlags != NULL)) {
4255
0
        ret = (xmlChar *) CUR_PTR - chunkSize;
4256
4257
0
        if (attlen != NULL)
4258
0
            *attlen = chunkSize;
4259
0
        if ((normalize) && (inSpace) && (chunkSize > 0)) {
4260
0
            attvalFlags |= XML_ATTVAL_NORM_CHANGE;
4261
0
            *attlen -= 1;
4262
0
        }
4263
4264
        /* Report potential error */
4265
0
        xmlSBufCleanup(&buf, ctxt, "AttValue length too long");
4266
0
    } else {
4267
0
        if (chunkSize > 0)
4268
0
            xmlSBufAddString(&buf, CUR_PTR - chunkSize, chunkSize);
4269
4270
0
        if ((normalize) && (inSpace) && (buf.size > 0)) {
4271
0
            attvalFlags |= XML_ATTVAL_NORM_CHANGE;
4272
0
            buf.size--;
4273
0
        }
4274
4275
0
        ret = xmlSBufFinish(&buf, attlen, ctxt, "AttValue length too long");
4276
0
        attvalFlags |= XML_ATTVAL_ALLOC;
4277
4278
0
        if (ret != NULL) {
4279
0
            if (attlen != NULL)
4280
0
                *attlen = buf.size;
4281
0
        }
4282
0
    }
4283
4284
0
    if (outFlags != NULL)
4285
0
        *outFlags = attvalFlags;
4286
4287
0
    NEXTL(1);
4288
4289
0
    return(ret);
4290
4291
0
error:
4292
0
    xmlSBufCleanup(&buf, ctxt, "AttValue length too long");
4293
0
    return(NULL);
4294
0
}
4295
4296
/**
4297
 * Parse a value for an attribute
4298
 * Note: the parser won't do substitution of entities here, this
4299
 * will be handled later in #xmlStringGetNodeList
4300
 *
4301
 * @deprecated Internal function, don't use.
4302
 *
4303
 *     [10] AttValue ::= '"' ([^<&"] | Reference)* '"' |
4304
 *                       "'" ([^<&'] | Reference)* "'"
4305
 *
4306
 * 3.3.3 Attribute-Value Normalization:
4307
 *
4308
 * Before the value of an attribute is passed to the application or
4309
 * checked for validity, the XML processor must normalize it as follows:
4310
 *
4311
 * - a character reference is processed by appending the referenced
4312
 *   character to the attribute value
4313
 * - an entity reference is processed by recursively processing the
4314
 *   replacement text of the entity
4315
 * - a whitespace character (\#x20, \#xD, \#xA, \#x9) is processed by
4316
 *   appending \#x20 to the normalized value, except that only a single
4317
 *   \#x20 is appended for a "#xD#xA" sequence that is part of an external
4318
 *   parsed entity or the literal entity value of an internal parsed entity
4319
 * - other characters are processed by appending them to the normalized value
4320
 *
4321
 * If the declared value is not CDATA, then the XML processor must further
4322
 * process the normalized attribute value by discarding any leading and
4323
 * trailing space (\#x20) characters, and by replacing sequences of space
4324
 * (\#x20) characters by a single space (\#x20) character.
4325
 * All attributes for which no declaration has been read should be treated
4326
 * by a non-validating parser as if declared CDATA.
4327
 *
4328
 * @param ctxt  an XML parser context
4329
 * @returns the AttValue parsed or NULL. The value has to be freed by the
4330
 * caller.
4331
 */
4332
xmlChar *
4333
0
xmlParseAttValue(xmlParserCtxt *ctxt) {
4334
0
    if ((ctxt == NULL) || (ctxt->input == NULL)) return(NULL);
4335
0
    return(xmlParseAttValueInternal(ctxt, NULL, NULL, 0, 0));
4336
0
}
4337
4338
/**
4339
 * Parse an XML Literal
4340
 *
4341
 * @deprecated Internal function, don't use.
4342
 *
4343
 *     [11] SystemLiteral ::= ('"' [^"]* '"') | ("'" [^']* "'")
4344
 *
4345
 * @param ctxt  an XML parser context
4346
 * @returns the SystemLiteral parsed or NULL
4347
 */
4348
4349
xmlChar *
4350
0
xmlParseSystemLiteral(xmlParserCtxt *ctxt) {
4351
0
    xmlChar *buf = NULL;
4352
0
    int len = 0;
4353
0
    int size = XML_PARSER_BUFFER_SIZE;
4354
0
    int cur, l;
4355
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
4356
0
                    XML_MAX_TEXT_LENGTH :
4357
0
                    XML_MAX_NAME_LENGTH;
4358
0
    xmlChar stop;
4359
4360
0
    if (RAW == '"') {
4361
0
        NEXT;
4362
0
  stop = '"';
4363
0
    } else if (RAW == '\'') {
4364
0
        NEXT;
4365
0
  stop = '\'';
4366
0
    } else {
4367
0
  xmlFatalErr(ctxt, XML_ERR_LITERAL_NOT_STARTED, NULL);
4368
0
  return(NULL);
4369
0
    }
4370
4371
0
    buf = xmlMalloc(size);
4372
0
    if (buf == NULL) {
4373
0
        xmlErrMemory(ctxt);
4374
0
  return(NULL);
4375
0
    }
4376
0
    cur = xmlCurrentCharRecover(ctxt, &l);
4377
0
    while ((IS_CHAR(cur)) && (cur != stop)) { /* checked */
4378
0
  if (len + 5 >= size) {
4379
0
      xmlChar *tmp;
4380
0
            int newSize;
4381
4382
0
            newSize = xmlGrowCapacity(size, 1, 1, maxLength);
4383
0
            if (newSize < 0) {
4384
0
                xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "SystemLiteral");
4385
0
                xmlFree(buf);
4386
0
                return(NULL);
4387
0
            }
4388
0
      tmp = xmlRealloc(buf, newSize);
4389
0
      if (tmp == NULL) {
4390
0
          xmlFree(buf);
4391
0
    xmlErrMemory(ctxt);
4392
0
    return(NULL);
4393
0
      }
4394
0
      buf = tmp;
4395
0
            size = newSize;
4396
0
  }
4397
0
  COPY_BUF(buf, len, cur);
4398
0
  NEXTL(l);
4399
0
  cur = xmlCurrentCharRecover(ctxt, &l);
4400
0
    }
4401
0
    buf[len] = 0;
4402
0
    if (!IS_CHAR(cur)) {
4403
0
  xmlFatalErr(ctxt, XML_ERR_LITERAL_NOT_FINISHED, NULL);
4404
0
    } else {
4405
0
  NEXT;
4406
0
    }
4407
0
    return(buf);
4408
0
}
4409
4410
/**
4411
 * Parse an XML public literal
4412
 *
4413
 * @deprecated Internal function, don't use.
4414
 *
4415
 *     [12] PubidLiteral ::= '"' PubidChar* '"' | "'" (PubidChar - "'")* "'"
4416
 *
4417
 * @param ctxt  an XML parser context
4418
 * @returns the PubidLiteral parsed or NULL.
4419
 */
4420
4421
xmlChar *
4422
0
xmlParsePubidLiteral(xmlParserCtxt *ctxt) {
4423
0
    xmlChar *buf = NULL;
4424
0
    int len = 0;
4425
0
    int size = XML_PARSER_BUFFER_SIZE;
4426
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
4427
0
                    XML_MAX_TEXT_LENGTH :
4428
0
                    XML_MAX_NAME_LENGTH;
4429
0
    xmlChar cur;
4430
0
    xmlChar stop;
4431
4432
0
    if (RAW == '"') {
4433
0
        NEXT;
4434
0
  stop = '"';
4435
0
    } else if (RAW == '\'') {
4436
0
        NEXT;
4437
0
  stop = '\'';
4438
0
    } else {
4439
0
  xmlFatalErr(ctxt, XML_ERR_LITERAL_NOT_STARTED, NULL);
4440
0
  return(NULL);
4441
0
    }
4442
0
    buf = xmlMalloc(size);
4443
0
    if (buf == NULL) {
4444
0
  xmlErrMemory(ctxt);
4445
0
  return(NULL);
4446
0
    }
4447
0
    cur = CUR;
4448
0
    while ((IS_PUBIDCHAR_CH(cur)) && (cur != stop) &&
4449
0
           (PARSER_STOPPED(ctxt) == 0)) { /* checked */
4450
0
  if (len + 1 >= size) {
4451
0
      xmlChar *tmp;
4452
0
            int newSize;
4453
4454
0
      newSize = xmlGrowCapacity(size, 1, 1, maxLength);
4455
0
            if (newSize < 0) {
4456
0
                xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "Public ID");
4457
0
                xmlFree(buf);
4458
0
                return(NULL);
4459
0
            }
4460
0
      tmp = xmlRealloc(buf, newSize);
4461
0
      if (tmp == NULL) {
4462
0
    xmlErrMemory(ctxt);
4463
0
    xmlFree(buf);
4464
0
    return(NULL);
4465
0
      }
4466
0
      buf = tmp;
4467
0
            size = newSize;
4468
0
  }
4469
0
  buf[len++] = cur;
4470
0
  NEXT;
4471
0
  cur = CUR;
4472
0
    }
4473
0
    buf[len] = 0;
4474
0
    if (cur != stop) {
4475
0
  xmlFatalErr(ctxt, XML_ERR_LITERAL_NOT_FINISHED, NULL);
4476
0
    } else {
4477
0
  NEXTL(1);
4478
0
    }
4479
0
    return(buf);
4480
0
}
4481
4482
static void xmlParseCharDataComplex(xmlParserCtxtPtr ctxt, int partial);
4483
4484
/*
4485
 * used for the test in the inner loop of the char data testing
4486
 */
4487
static const unsigned char test_char_data[256] = {
4488
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4489
    0x00, 0x09, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x9, CR/LF separated */
4490
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4491
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4492
    0x20, 0x21, 0x22, 0x23, 0x24, 0x25, 0x00, 0x27, /* & */
4493
    0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x2D, 0x2E, 0x2F,
4494
    0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37,
4495
    0x38, 0x39, 0x3A, 0x3B, 0x00, 0x3D, 0x3E, 0x3F, /* < */
4496
    0x40, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47,
4497
    0x48, 0x49, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F,
4498
    0x50, 0x51, 0x52, 0x53, 0x54, 0x55, 0x56, 0x57,
4499
    0x58, 0x59, 0x5A, 0x5B, 0x5C, 0x00, 0x5E, 0x5F, /* ] */
4500
    0x60, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67,
4501
    0x68, 0x69, 0x6A, 0x6B, 0x6C, 0x6D, 0x6E, 0x6F,
4502
    0x70, 0x71, 0x72, 0x73, 0x74, 0x75, 0x76, 0x77,
4503
    0x78, 0x79, 0x7A, 0x7B, 0x7C, 0x7D, 0x7E, 0x7F,
4504
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* non-ascii */
4505
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4506
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4507
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4508
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4509
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4510
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4511
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4512
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4513
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4514
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4515
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4516
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4517
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4518
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
4519
    0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00
4520
};
4521
4522
static void
4523
xmlCharacters(xmlParserCtxtPtr ctxt, const xmlChar *buf, int size,
4524
0
              int isBlank) {
4525
0
    int checkBlanks;
4526
4527
0
    if ((ctxt->sax == NULL) || (ctxt->disableSAX))
4528
0
        return;
4529
4530
0
    checkBlanks = (!ctxt->keepBlanks) ||
4531
0
                  (ctxt->sax->ignorableWhitespace != ctxt->sax->characters);
4532
4533
    /*
4534
     * Calling areBlanks with only parts of a text node
4535
     * is fundamentally broken, making the NOBLANKS option
4536
     * essentially unusable.
4537
     */
4538
0
    if ((checkBlanks) &&
4539
0
        (areBlanks(ctxt, buf, size, isBlank))) {
4540
0
        if ((ctxt->sax->ignorableWhitespace != NULL) &&
4541
0
            (ctxt->keepBlanks))
4542
0
            ctxt->sax->ignorableWhitespace(ctxt->userData, buf, size);
4543
0
    } else {
4544
0
        if (ctxt->sax->characters != NULL)
4545
0
            ctxt->sax->characters(ctxt->userData, buf, size);
4546
4547
        /*
4548
         * The old code used to update this value for "complex" data
4549
         * even if checkBlanks was false. This was probably a bug.
4550
         */
4551
0
        if ((checkBlanks) && (*ctxt->space == -1))
4552
0
            *ctxt->space = -2;
4553
0
    }
4554
0
}
4555
4556
/**
4557
 * Parse character data. Always makes progress if the first char isn't
4558
 * '<' or '&'.
4559
 *
4560
 * The right angle bracket (>) may be represented using the string "&gt;",
4561
 * and must, for compatibility, be escaped using "&gt;" or a character
4562
 * reference when it appears in the string "]]>" in content, when that
4563
 * string is not marking the end of a CDATA section.
4564
 *
4565
 *     [14] CharData ::= [^<&]* - ([^<&]* ']]>' [^<&]*)
4566
 * @param ctxt  an XML parser context
4567
 * @param partial  buffer may contain partial UTF-8 sequences
4568
 */
4569
static void
4570
0
xmlParseCharDataInternal(xmlParserCtxtPtr ctxt, int partial) {
4571
0
    const xmlChar *in;
4572
0
    int line = ctxt->input->line;
4573
0
    int col = ctxt->input->col;
4574
0
    int ccol;
4575
0
    int terminate = 0;
4576
4577
0
    GROW;
4578
    /*
4579
     * Accelerated common case where input don't need to be
4580
     * modified before passing it to the handler.
4581
     */
4582
0
    in = ctxt->input->cur;
4583
0
    do {
4584
0
get_more_space:
4585
0
        while (*in == 0x20) { in++; ctxt->input->col++; }
4586
0
        if (*in == 0xA) {
4587
0
            do {
4588
0
                ctxt->input->line++; ctxt->input->col = 1;
4589
0
                in++;
4590
0
            } while (*in == 0xA);
4591
0
            goto get_more_space;
4592
0
        }
4593
0
        if (*in == '<') {
4594
0
            while (in > ctxt->input->cur) {
4595
0
                const xmlChar *tmp = ctxt->input->cur;
4596
0
                size_t nbchar = in - tmp;
4597
4598
0
                if (nbchar > XML_MAX_ITEMS)
4599
0
                    nbchar = XML_MAX_ITEMS;
4600
0
                ctxt->input->cur += nbchar;
4601
4602
0
                xmlCharacters(ctxt, tmp, nbchar, 1);
4603
0
            }
4604
0
            return;
4605
0
        }
4606
4607
0
get_more:
4608
0
        ccol = ctxt->input->col;
4609
0
        while (test_char_data[*in]) {
4610
0
            in++;
4611
0
            ccol++;
4612
0
        }
4613
0
        ctxt->input->col = ccol;
4614
0
        if (*in == 0xA) {
4615
0
            do {
4616
0
                ctxt->input->line++; ctxt->input->col = 1;
4617
0
                in++;
4618
0
            } while (*in == 0xA);
4619
0
            goto get_more;
4620
0
        }
4621
0
        if (*in == ']') {
4622
0
            size_t avail = ctxt->input->end - in;
4623
4624
0
            if (partial && avail < 2) {
4625
0
                terminate = 1;
4626
0
                goto invoke_callback;
4627
0
            }
4628
0
            if (in[1] == ']') {
4629
0
                if (partial && avail < 3) {
4630
0
                    terminate = 1;
4631
0
                    goto invoke_callback;
4632
0
                }
4633
0
                if (in[2] == '>')
4634
0
                    xmlFatalErr(ctxt, XML_ERR_MISPLACED_CDATA_END, NULL);
4635
0
            }
4636
4637
0
            in++;
4638
0
            ctxt->input->col++;
4639
0
            goto get_more;
4640
0
        }
4641
4642
0
invoke_callback:
4643
0
        while (in > ctxt->input->cur) {
4644
0
            const xmlChar *tmp = ctxt->input->cur;
4645
0
            size_t nbchar = in - tmp;
4646
4647
0
            if (nbchar > XML_MAX_ITEMS)
4648
0
                nbchar = XML_MAX_ITEMS;
4649
0
            ctxt->input->cur += nbchar;
4650
4651
0
            xmlCharacters(ctxt, tmp, nbchar, 0);
4652
4653
0
            line = ctxt->input->line;
4654
0
            col = ctxt->input->col;
4655
0
        }
4656
0
        ctxt->input->cur = in;
4657
0
        if (*in == 0xD) {
4658
0
            in++;
4659
0
            if (*in == 0xA) {
4660
0
                ctxt->input->cur = in;
4661
0
                in++;
4662
0
                ctxt->input->line++; ctxt->input->col = 1;
4663
0
                continue; /* while */
4664
0
            }
4665
0
            in--;
4666
0
        }
4667
0
        if (*in == '<') {
4668
0
            return;
4669
0
        }
4670
0
        if (*in == '&') {
4671
0
            return;
4672
0
        }
4673
0
        if (terminate) {
4674
0
            return;
4675
0
        }
4676
0
        SHRINK;
4677
0
        GROW;
4678
0
        in = ctxt->input->cur;
4679
0
    } while (((*in >= 0x20) && (*in <= 0x7F)) ||
4680
0
             (*in == 0x09) || (*in == 0x0a));
4681
0
    ctxt->input->line = line;
4682
0
    ctxt->input->col = col;
4683
0
    xmlParseCharDataComplex(ctxt, partial);
4684
0
}
4685
4686
/**
4687
 * Always makes progress if the first char isn't '<' or '&'.
4688
 *
4689
 * parse a CharData section.this is the fallback function
4690
 * of #xmlParseCharData when the parsing requires handling
4691
 * of non-ASCII characters.
4692
 *
4693
 * @param ctxt  an XML parser context
4694
 * @param partial  whether the input can end with truncated UTF-8
4695
 */
4696
static void
4697
0
xmlParseCharDataComplex(xmlParserCtxtPtr ctxt, int partial) {
4698
0
    xmlChar buf[XML_PARSER_BIG_BUFFER_SIZE + 5];
4699
0
    int nbchar = 0;
4700
0
    int cur, l;
4701
4702
0
    cur = xmlCurrentCharRecover(ctxt, &l);
4703
0
    while ((cur != '<') && /* checked */
4704
0
           (cur != '&') &&
4705
0
     (IS_CHAR(cur))) {
4706
0
        if (cur == ']') {
4707
0
            size_t avail = ctxt->input->end - ctxt->input->cur;
4708
4709
0
            if (partial && avail < 2)
4710
0
                break;
4711
0
            if (NXT(1) == ']') {
4712
0
                if (partial && avail < 3)
4713
0
                    break;
4714
0
                if (NXT(2) == '>')
4715
0
                    xmlFatalErr(ctxt, XML_ERR_MISPLACED_CDATA_END, NULL);
4716
0
            }
4717
0
        }
4718
4719
0
  COPY_BUF(buf, nbchar, cur);
4720
  /* move current position before possible calling of ctxt->sax->characters */
4721
0
  NEXTL(l);
4722
0
  if (nbchar >= XML_PARSER_BIG_BUFFER_SIZE) {
4723
0
      buf[nbchar] = 0;
4724
4725
0
            xmlCharacters(ctxt, buf, nbchar, 0);
4726
0
      nbchar = 0;
4727
0
            SHRINK;
4728
0
  }
4729
0
  cur = xmlCurrentCharRecover(ctxt, &l);
4730
0
    }
4731
0
    if (nbchar != 0) {
4732
0
        buf[nbchar] = 0;
4733
4734
0
        xmlCharacters(ctxt, buf, nbchar, 0);
4735
0
    }
4736
    /*
4737
     * cur == 0 can mean
4738
     *
4739
     * - End of buffer.
4740
     * - An actual 0 character.
4741
     * - An incomplete UTF-8 sequence. This is allowed if partial is set.
4742
     */
4743
0
    if (ctxt->input->cur < ctxt->input->end) {
4744
0
        if ((cur == 0) && (CUR != 0)) {
4745
0
            if (partial == 0) {
4746
0
                xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
4747
0
                        "Incomplete UTF-8 sequence starting with %02X\n", CUR);
4748
0
                NEXTL(1);
4749
0
            }
4750
0
        } else if ((cur != '<') && (cur != '&') && (cur != ']')) {
4751
            /* Generate the error and skip the offending character */
4752
0
            xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
4753
0
                              "PCDATA invalid Char value %d\n", cur);
4754
0
            NEXTL(l);
4755
0
        }
4756
0
    }
4757
0
}
4758
4759
/**
4760
 * @deprecated Internal function, don't use.
4761
 * @param ctxt  an XML parser context
4762
 * @param cdata  unused
4763
 */
4764
void
4765
0
xmlParseCharData(xmlParserCtxt *ctxt, ATTRIBUTE_UNUSED int cdata) {
4766
0
    xmlParseCharDataInternal(ctxt, 0);
4767
0
}
4768
4769
/**
4770
 * Parse an External ID or a Public ID
4771
 *
4772
 * @deprecated Internal function, don't use.
4773
 *
4774
 * NOTE: Productions [75] and [83] interact badly since [75] can generate
4775
 * `'PUBLIC' S PubidLiteral S SystemLiteral`
4776
 *
4777
 *     [75] ExternalID ::= 'SYSTEM' S SystemLiteral
4778
 *                       | 'PUBLIC' S PubidLiteral S SystemLiteral
4779
 *
4780
 *     [83] PublicID ::= 'PUBLIC' S PubidLiteral
4781
 *
4782
 * @param ctxt  an XML parser context
4783
 * @param publicId  a xmlChar** receiving PubidLiteral
4784
 * @param strict  indicate whether we should restrict parsing to only
4785
 *          production [75], see NOTE below
4786
 * @returns the function returns SystemLiteral and in the second
4787
 *                case publicID receives PubidLiteral, is strict is off
4788
 *                it is possible to return NULL and have publicID set.
4789
 */
4790
4791
xmlChar *
4792
0
xmlParseExternalID(xmlParserCtxt *ctxt, xmlChar **publicId, int strict) {
4793
0
    xmlChar *URI = NULL;
4794
4795
0
    *publicId = NULL;
4796
0
    if (CMP6(CUR_PTR, 'S', 'Y', 'S', 'T', 'E', 'M')) {
4797
0
        SKIP(6);
4798
0
  if (SKIP_BLANKS == 0) {
4799
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
4800
0
                     "Space required after 'SYSTEM'\n");
4801
0
  }
4802
0
  URI = xmlParseSystemLiteral(ctxt);
4803
0
  if (URI == NULL) {
4804
0
      xmlFatalErr(ctxt, XML_ERR_URI_REQUIRED, NULL);
4805
0
        }
4806
0
    } else if (CMP6(CUR_PTR, 'P', 'U', 'B', 'L', 'I', 'C')) {
4807
0
        SKIP(6);
4808
0
  if (SKIP_BLANKS == 0) {
4809
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
4810
0
        "Space required after 'PUBLIC'\n");
4811
0
  }
4812
0
  *publicId = xmlParsePubidLiteral(ctxt);
4813
0
  if (*publicId == NULL) {
4814
0
      xmlFatalErr(ctxt, XML_ERR_PUBID_REQUIRED, NULL);
4815
0
  }
4816
0
  if (strict) {
4817
      /*
4818
       * We don't handle [83] so "S SystemLiteral" is required.
4819
       */
4820
0
      if (SKIP_BLANKS == 0) {
4821
0
    xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
4822
0
      "Space required after the Public Identifier\n");
4823
0
      }
4824
0
  } else {
4825
      /*
4826
       * We handle [83] so we return immediately, if
4827
       * "S SystemLiteral" is not detected. We skip blanks if no
4828
             * system literal was found, but this is harmless since we must
4829
             * be at the end of a NotationDecl.
4830
       */
4831
0
      if (SKIP_BLANKS == 0) return(NULL);
4832
0
      if ((CUR != '\'') && (CUR != '"')) return(NULL);
4833
0
  }
4834
0
  URI = xmlParseSystemLiteral(ctxt);
4835
0
  if (URI == NULL) {
4836
0
      xmlFatalErr(ctxt, XML_ERR_URI_REQUIRED, NULL);
4837
0
        }
4838
0
    }
4839
0
    return(URI);
4840
0
}
4841
4842
/**
4843
 * Skip an XML (SGML) comment <!-- .... -->
4844
 *  The spec says that "For compatibility, the string "--" (double-hyphen)
4845
 *  must not occur within comments. "
4846
 * This is the slow routine in case the accelerator for ascii didn't work
4847
 *
4848
 *     [15] Comment ::= '<!--' ((Char - '-') | ('-' (Char - '-')))* '-->'
4849
 * @param ctxt  an XML parser context
4850
 * @param buf  the already parsed part of the buffer
4851
 * @param len  number of bytes in the buffer
4852
 * @param size  allocated size of the buffer
4853
 */
4854
static void
4855
xmlParseCommentComplex(xmlParserCtxtPtr ctxt, xmlChar *buf,
4856
0
                       size_t len, size_t size) {
4857
0
    int q, ql;
4858
0
    int r, rl;
4859
0
    int cur, l;
4860
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
4861
0
                    XML_MAX_HUGE_LENGTH :
4862
0
                    XML_MAX_TEXT_LENGTH;
4863
4864
0
    if (buf == NULL) {
4865
0
        len = 0;
4866
0
  size = XML_PARSER_BUFFER_SIZE;
4867
0
  buf = xmlMalloc(size);
4868
0
  if (buf == NULL) {
4869
0
      xmlErrMemory(ctxt);
4870
0
      return;
4871
0
  }
4872
0
    }
4873
0
    q = xmlCurrentCharRecover(ctxt, &ql);
4874
0
    if (q == 0)
4875
0
        goto not_terminated;
4876
0
    if (!IS_CHAR(q)) {
4877
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
4878
0
                          "xmlParseComment: invalid xmlChar value %d\n",
4879
0
                    q);
4880
0
  xmlFree (buf);
4881
0
  return;
4882
0
    }
4883
0
    NEXTL(ql);
4884
0
    r = xmlCurrentCharRecover(ctxt, &rl);
4885
0
    if (r == 0)
4886
0
        goto not_terminated;
4887
0
    if (!IS_CHAR(r)) {
4888
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
4889
0
                          "xmlParseComment: invalid xmlChar value %d\n",
4890
0
                    r);
4891
0
  xmlFree (buf);
4892
0
  return;
4893
0
    }
4894
0
    NEXTL(rl);
4895
0
    cur = xmlCurrentCharRecover(ctxt, &l);
4896
0
    if (cur == 0)
4897
0
        goto not_terminated;
4898
0
    while (IS_CHAR(cur) && /* checked */
4899
0
           ((cur != '>') ||
4900
0
      (r != '-') || (q != '-'))) {
4901
0
  if ((r == '-') && (q == '-')) {
4902
0
      xmlFatalErr(ctxt, XML_ERR_HYPHEN_IN_COMMENT, NULL);
4903
0
  }
4904
0
  if (len + 5 >= size) {
4905
0
      xmlChar *tmp;
4906
0
            int newSize;
4907
4908
0
      newSize = xmlGrowCapacity(size, 1, 1, maxLength);
4909
0
            if (newSize < 0) {
4910
0
                xmlFatalErrMsgStr(ctxt, XML_ERR_COMMENT_NOT_FINISHED,
4911
0
                             "Comment too big found", NULL);
4912
0
                xmlFree (buf);
4913
0
                return;
4914
0
            }
4915
0
      tmp = xmlRealloc(buf, newSize);
4916
0
      if (tmp == NULL) {
4917
0
    xmlErrMemory(ctxt);
4918
0
    xmlFree(buf);
4919
0
    return;
4920
0
      }
4921
0
      buf = tmp;
4922
0
            size = newSize;
4923
0
  }
4924
0
  COPY_BUF(buf, len, q);
4925
4926
0
  q = r;
4927
0
  ql = rl;
4928
0
  r = cur;
4929
0
  rl = l;
4930
4931
0
  NEXTL(l);
4932
0
  cur = xmlCurrentCharRecover(ctxt, &l);
4933
4934
0
    }
4935
0
    buf[len] = 0;
4936
0
    if (cur == 0) {
4937
0
  xmlFatalErrMsgStr(ctxt, XML_ERR_COMMENT_NOT_FINISHED,
4938
0
                       "Comment not terminated \n<!--%.50s\n", buf);
4939
0
    } else if (!IS_CHAR(cur)) {
4940
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
4941
0
                          "xmlParseComment: invalid xmlChar value %d\n",
4942
0
                    cur);
4943
0
    } else {
4944
0
        NEXT;
4945
0
  if ((ctxt->sax != NULL) && (ctxt->sax->comment != NULL) &&
4946
0
      (!ctxt->disableSAX))
4947
0
      ctxt->sax->comment(ctxt->userData, buf);
4948
0
    }
4949
0
    xmlFree(buf);
4950
0
    return;
4951
0
not_terminated:
4952
0
    xmlFatalErrMsgStr(ctxt, XML_ERR_COMMENT_NOT_FINISHED,
4953
0
       "Comment not terminated\n", NULL);
4954
0
    xmlFree(buf);
4955
0
}
4956
4957
/**
4958
 * Parse an XML (SGML) comment. Always consumes '<!'.
4959
 *
4960
 * @deprecated Internal function, don't use.
4961
 *
4962
 *  The spec says that "For compatibility, the string "--" (double-hyphen)
4963
 *  must not occur within comments. "
4964
 *
4965
 *     [15] Comment ::= '<!--' ((Char - '-') | ('-' (Char - '-')))* '-->'
4966
 * @param ctxt  an XML parser context
4967
 */
4968
void
4969
0
xmlParseComment(xmlParserCtxt *ctxt) {
4970
0
    xmlChar *buf = NULL;
4971
0
    size_t size = XML_PARSER_BUFFER_SIZE;
4972
0
    size_t len = 0;
4973
0
    size_t maxLength = (ctxt->options & XML_PARSE_HUGE) ?
4974
0
                       XML_MAX_HUGE_LENGTH :
4975
0
                       XML_MAX_TEXT_LENGTH;
4976
0
    const xmlChar *in;
4977
0
    size_t nbchar = 0;
4978
0
    int ccol;
4979
4980
    /*
4981
     * Check that there is a comment right here.
4982
     */
4983
0
    if ((RAW != '<') || (NXT(1) != '!'))
4984
0
        return;
4985
0
    SKIP(2);
4986
0
    if ((RAW != '-') || (NXT(1) != '-'))
4987
0
        return;
4988
0
    SKIP(2);
4989
0
    GROW;
4990
4991
    /*
4992
     * Accelerated common case where input don't need to be
4993
     * modified before passing it to the handler.
4994
     */
4995
0
    in = ctxt->input->cur;
4996
0
    do {
4997
0
  if (*in == 0xA) {
4998
0
      do {
4999
0
    ctxt->input->line++; ctxt->input->col = 1;
5000
0
    in++;
5001
0
      } while (*in == 0xA);
5002
0
  }
5003
0
get_more:
5004
0
        ccol = ctxt->input->col;
5005
0
  while (((*in > '-') && (*in <= 0x7F)) ||
5006
0
         ((*in >= 0x20) && (*in < '-')) ||
5007
0
         (*in == 0x09)) {
5008
0
        in++;
5009
0
        ccol++;
5010
0
  }
5011
0
  ctxt->input->col = ccol;
5012
0
  if (*in == 0xA) {
5013
0
      do {
5014
0
    ctxt->input->line++; ctxt->input->col = 1;
5015
0
    in++;
5016
0
      } while (*in == 0xA);
5017
0
      goto get_more;
5018
0
  }
5019
0
  nbchar = in - ctxt->input->cur;
5020
  /*
5021
   * save current set of data
5022
   */
5023
0
  if (nbchar > 0) {
5024
0
            if (nbchar > maxLength - len) {
5025
0
                xmlFatalErrMsgStr(ctxt, XML_ERR_COMMENT_NOT_FINISHED,
5026
0
                                  "Comment too big found", NULL);
5027
0
                xmlFree(buf);
5028
0
                return;
5029
0
            }
5030
0
            if (buf == NULL) {
5031
0
                if ((*in == '-') && (in[1] == '-'))
5032
0
                    size = nbchar + 1;
5033
0
                else
5034
0
                    size = XML_PARSER_BUFFER_SIZE + nbchar;
5035
0
                buf = xmlMalloc(size);
5036
0
                if (buf == NULL) {
5037
0
                    xmlErrMemory(ctxt);
5038
0
                    return;
5039
0
                }
5040
0
                len = 0;
5041
0
            } else if (len + nbchar + 1 >= size) {
5042
0
                xmlChar *new_buf;
5043
0
                size += len + nbchar + XML_PARSER_BUFFER_SIZE;
5044
0
                new_buf = xmlRealloc(buf, size);
5045
0
                if (new_buf == NULL) {
5046
0
                    xmlErrMemory(ctxt);
5047
0
                    xmlFree(buf);
5048
0
                    return;
5049
0
                }
5050
0
                buf = new_buf;
5051
0
            }
5052
0
            memcpy(&buf[len], ctxt->input->cur, nbchar);
5053
0
            len += nbchar;
5054
0
            buf[len] = 0;
5055
0
  }
5056
0
  ctxt->input->cur = in;
5057
0
  if (*in == 0xA) {
5058
0
      in++;
5059
0
      ctxt->input->line++; ctxt->input->col = 1;
5060
0
  }
5061
0
  if (*in == 0xD) {
5062
0
      in++;
5063
0
      if (*in == 0xA) {
5064
0
    ctxt->input->cur = in;
5065
0
    in++;
5066
0
    ctxt->input->line++; ctxt->input->col = 1;
5067
0
    goto get_more;
5068
0
      }
5069
0
      in--;
5070
0
  }
5071
0
  SHRINK;
5072
0
  GROW;
5073
0
  in = ctxt->input->cur;
5074
0
  if (*in == '-') {
5075
0
      if (in[1] == '-') {
5076
0
          if (in[2] == '>') {
5077
0
        SKIP(3);
5078
0
        if ((ctxt->sax != NULL) && (ctxt->sax->comment != NULL) &&
5079
0
            (!ctxt->disableSAX)) {
5080
0
      if (buf != NULL)
5081
0
          ctxt->sax->comment(ctxt->userData, buf);
5082
0
      else
5083
0
          ctxt->sax->comment(ctxt->userData, BAD_CAST "");
5084
0
        }
5085
0
        if (buf != NULL)
5086
0
            xmlFree(buf);
5087
0
        return;
5088
0
    }
5089
0
    if (buf != NULL) {
5090
0
        xmlFatalErrMsgStr(ctxt, XML_ERR_HYPHEN_IN_COMMENT,
5091
0
                          "Double hyphen within comment: "
5092
0
                                      "<!--%.50s\n",
5093
0
              buf);
5094
0
    } else
5095
0
        xmlFatalErrMsgStr(ctxt, XML_ERR_HYPHEN_IN_COMMENT,
5096
0
                          "Double hyphen within comment\n", NULL);
5097
0
    in++;
5098
0
    ctxt->input->col++;
5099
0
      }
5100
0
      in++;
5101
0
      ctxt->input->col++;
5102
0
      goto get_more;
5103
0
  }
5104
0
    } while (((*in >= 0x20) && (*in <= 0x7F)) || (*in == 0x09) || (*in == 0x0a));
5105
0
    xmlParseCommentComplex(ctxt, buf, len, size);
5106
0
}
5107
5108
5109
/**
5110
 * Parse the name of a PI
5111
 *
5112
 * @deprecated Internal function, don't use.
5113
 *
5114
 *     [17] PITarget ::= Name - (('X' | 'x') ('M' | 'm') ('L' | 'l'))
5115
 *
5116
 * @param ctxt  an XML parser context
5117
 * @returns the PITarget name or NULL
5118
 */
5119
5120
const xmlChar *
5121
0
xmlParsePITarget(xmlParserCtxt *ctxt) {
5122
0
    const xmlChar *name;
5123
5124
0
    name = xmlParseName(ctxt);
5125
0
    if ((name != NULL) &&
5126
0
        ((name[0] == 'x') || (name[0] == 'X')) &&
5127
0
        ((name[1] == 'm') || (name[1] == 'M')) &&
5128
0
        ((name[2] == 'l') || (name[2] == 'L'))) {
5129
0
  int i;
5130
0
  if ((name[0] == 'x') && (name[1] == 'm') &&
5131
0
      (name[2] == 'l') && (name[3] == 0)) {
5132
0
      xmlFatalErrMsg(ctxt, XML_ERR_RESERVED_XML_NAME,
5133
0
     "XML declaration allowed only at the start of the document\n");
5134
0
      return(name);
5135
0
  } else if (name[3] == 0) {
5136
0
      xmlFatalErr(ctxt, XML_ERR_RESERVED_XML_NAME, NULL);
5137
0
      return(name);
5138
0
  }
5139
0
  for (i = 0;;i++) {
5140
0
      if (xmlW3CPIs[i] == NULL) break;
5141
0
      if (xmlStrEqual(name, (const xmlChar *)xmlW3CPIs[i]))
5142
0
          return(name);
5143
0
  }
5144
0
  xmlWarningMsg(ctxt, XML_ERR_RESERVED_XML_NAME,
5145
0
          "xmlParsePITarget: invalid name prefix 'xml'\n",
5146
0
          NULL, NULL);
5147
0
    }
5148
0
    if ((name != NULL) && (xmlStrchr(name, ':') != NULL)) {
5149
0
  xmlNsErr(ctxt, XML_NS_ERR_COLON,
5150
0
     "colons are forbidden from PI names '%s'\n", name, NULL, NULL);
5151
0
    }
5152
0
    return(name);
5153
0
}
5154
5155
#ifdef LIBXML_CATALOG_ENABLED
5156
/**
5157
 * Parse an XML Catalog Processing Instruction.
5158
 *
5159
 * <?oasis-xml-catalog catalog="http://example.com/catalog.xml"?>
5160
 *
5161
 * Occurs only if allowed by the user and if happening in the Misc
5162
 * part of the document before any doctype information
5163
 * This will add the given catalog to the parsing context in order
5164
 * to be used if there is a resolution need further down in the document
5165
 *
5166
 * @param ctxt  an XML parser context
5167
 * @param catalog  the PI value string
5168
 */
5169
5170
static void
5171
0
xmlParseCatalogPI(xmlParserCtxtPtr ctxt, const xmlChar *catalog) {
5172
0
    xmlChar *URL = NULL;
5173
0
    const xmlChar *tmp, *base;
5174
0
    xmlChar marker;
5175
5176
0
    tmp = catalog;
5177
0
    while (IS_BLANK_CH(*tmp)) tmp++;
5178
0
    if (xmlStrncmp(tmp, BAD_CAST"catalog", 7))
5179
0
  goto error;
5180
0
    tmp += 7;
5181
0
    while (IS_BLANK_CH(*tmp)) tmp++;
5182
0
    if (*tmp != '=') {
5183
0
  return;
5184
0
    }
5185
0
    tmp++;
5186
0
    while (IS_BLANK_CH(*tmp)) tmp++;
5187
0
    marker = *tmp;
5188
0
    if ((marker != '\'') && (marker != '"'))
5189
0
  goto error;
5190
0
    tmp++;
5191
0
    base = tmp;
5192
0
    while ((*tmp != 0) && (*tmp != marker)) tmp++;
5193
0
    if (*tmp == 0)
5194
0
  goto error;
5195
0
    URL = xmlStrndup(base, tmp - base);
5196
0
    tmp++;
5197
0
    while (IS_BLANK_CH(*tmp)) tmp++;
5198
0
    if (*tmp != 0)
5199
0
  goto error;
5200
5201
0
    if (URL != NULL) {
5202
        /*
5203
         * Unfortunately, the catalog API doesn't report OOM errors.
5204
         * xmlGetLastError isn't very helpful since we don't know
5205
         * where the last error came from. We'd have to reset it
5206
         * before this call and restore it afterwards.
5207
         */
5208
0
  ctxt->catalogs = xmlCatalogAddLocal(ctxt->catalogs, URL);
5209
0
  xmlFree(URL);
5210
0
    }
5211
0
    return;
5212
5213
0
error:
5214
0
    xmlWarningMsg(ctxt, XML_WAR_CATALOG_PI,
5215
0
            "Catalog PI syntax error: %s\n",
5216
0
      catalog, NULL);
5217
0
    if (URL != NULL)
5218
0
  xmlFree(URL);
5219
0
}
5220
#endif
5221
5222
/**
5223
 * Parse an XML Processing Instruction.
5224
 *
5225
 * @deprecated Internal function, don't use.
5226
 *
5227
 *     [16] PI ::= '<?' PITarget (S (Char* - (Char* '?>' Char*)))? '?>'
5228
 *
5229
 * The processing is transferred to SAX once parsed.
5230
 *
5231
 * @param ctxt  an XML parser context
5232
 */
5233
5234
void
5235
0
xmlParsePI(xmlParserCtxt *ctxt) {
5236
0
    xmlChar *buf = NULL;
5237
0
    size_t len = 0;
5238
0
    size_t size = XML_PARSER_BUFFER_SIZE;
5239
0
    size_t maxLength = (ctxt->options & XML_PARSE_HUGE) ?
5240
0
                       XML_MAX_HUGE_LENGTH :
5241
0
                       XML_MAX_TEXT_LENGTH;
5242
0
    int cur, l;
5243
0
    const xmlChar *target;
5244
5245
0
    if ((RAW == '<') && (NXT(1) == '?')) {
5246
  /*
5247
   * this is a Processing Instruction.
5248
   */
5249
0
  SKIP(2);
5250
5251
  /*
5252
   * Parse the target name and check for special support like
5253
   * namespace.
5254
   */
5255
0
        target = xmlParsePITarget(ctxt);
5256
0
  if (target != NULL) {
5257
0
      if ((RAW == '?') && (NXT(1) == '>')) {
5258
0
    SKIP(2);
5259
5260
    /*
5261
     * SAX: PI detected.
5262
     */
5263
0
    if ((ctxt->sax) && (!ctxt->disableSAX) &&
5264
0
        (ctxt->sax->processingInstruction != NULL))
5265
0
        ctxt->sax->processingInstruction(ctxt->userData,
5266
0
                                         target, NULL);
5267
0
    return;
5268
0
      }
5269
0
      buf = xmlMalloc(size);
5270
0
      if (buf == NULL) {
5271
0
    xmlErrMemory(ctxt);
5272
0
    return;
5273
0
      }
5274
0
      if (SKIP_BLANKS == 0) {
5275
0
    xmlFatalErrMsgStr(ctxt, XML_ERR_SPACE_REQUIRED,
5276
0
        "ParsePI: PI %s space expected\n", target);
5277
0
      }
5278
0
      cur = xmlCurrentCharRecover(ctxt, &l);
5279
0
      while (IS_CHAR(cur) && /* checked */
5280
0
       ((cur != '?') || (NXT(1) != '>'))) {
5281
0
    if (len + 5 >= size) {
5282
0
        xmlChar *tmp;
5283
0
                    int newSize;
5284
5285
0
                    newSize = xmlGrowCapacity(size, 1, 1, maxLength);
5286
0
                    if (newSize < 0) {
5287
0
                        xmlFatalErrMsgStr(ctxt, XML_ERR_PI_NOT_FINISHED,
5288
0
                                          "PI %s too big found", target);
5289
0
                        xmlFree(buf);
5290
0
                        return;
5291
0
                    }
5292
0
        tmp = xmlRealloc(buf, newSize);
5293
0
        if (tmp == NULL) {
5294
0
      xmlErrMemory(ctxt);
5295
0
      xmlFree(buf);
5296
0
      return;
5297
0
        }
5298
0
        buf = tmp;
5299
0
                    size = newSize;
5300
0
    }
5301
0
    COPY_BUF(buf, len, cur);
5302
0
    NEXTL(l);
5303
0
    cur = xmlCurrentCharRecover(ctxt, &l);
5304
0
      }
5305
0
      buf[len] = 0;
5306
0
      if (cur != '?') {
5307
0
    xmlFatalErrMsgStr(ctxt, XML_ERR_PI_NOT_FINISHED,
5308
0
          "ParsePI: PI %s never end ...\n", target);
5309
0
      } else {
5310
0
    SKIP(2);
5311
5312
0
#ifdef LIBXML_CATALOG_ENABLED
5313
0
    if ((ctxt->inSubset == 0) &&
5314
0
        (xmlStrEqual(target, XML_CATALOG_PI))) {
5315
0
        xmlCatalogAllow allow = xmlCatalogGetDefaults();
5316
5317
0
        if ((ctxt->options & XML_PARSE_CATALOG_PI) &&
5318
0
                        ((allow == XML_CATA_ALLOW_DOCUMENT) ||
5319
0
       (allow == XML_CATA_ALLOW_ALL)))
5320
0
      xmlParseCatalogPI(ctxt, buf);
5321
0
    }
5322
0
#endif
5323
5324
    /*
5325
     * SAX: PI detected.
5326
     */
5327
0
    if ((ctxt->sax) && (!ctxt->disableSAX) &&
5328
0
        (ctxt->sax->processingInstruction != NULL))
5329
0
        ctxt->sax->processingInstruction(ctxt->userData,
5330
0
                                         target, buf);
5331
0
      }
5332
0
      xmlFree(buf);
5333
0
  } else {
5334
0
      xmlFatalErr(ctxt, XML_ERR_PI_NOT_STARTED, NULL);
5335
0
  }
5336
0
    }
5337
0
}
5338
5339
/**
5340
 * Parse a notation declaration. Always consumes '<!'.
5341
 *
5342
 * @deprecated Internal function, don't use.
5343
 *
5344
 *     [82] NotationDecl ::= '<!NOTATION' S Name S (ExternalID |  PublicID)
5345
 *                           S? '>'
5346
 *
5347
 * Hence there is actually 3 choices:
5348
 *
5349
 *     'PUBLIC' S PubidLiteral
5350
 *     'PUBLIC' S PubidLiteral S SystemLiteral
5351
 *     'SYSTEM' S SystemLiteral
5352
 *
5353
 * See the NOTE on #xmlParseExternalID.
5354
 *
5355
 * @param ctxt  an XML parser context
5356
 */
5357
5358
void
5359
0
xmlParseNotationDecl(xmlParserCtxt *ctxt) {
5360
0
    const xmlChar *name;
5361
0
    xmlChar *Pubid;
5362
0
    xmlChar *Systemid;
5363
5364
0
    if ((CUR != '<') || (NXT(1) != '!'))
5365
0
        return;
5366
0
    SKIP(2);
5367
5368
0
    if (CMP8(CUR_PTR, 'N', 'O', 'T', 'A', 'T', 'I', 'O', 'N')) {
5369
0
#ifdef LIBXML_VALID_ENABLED
5370
0
  int oldInputNr = ctxt->inputNr;
5371
0
#endif
5372
5373
0
  SKIP(8);
5374
0
  if (SKIP_BLANKS_PE == 0) {
5375
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5376
0
         "Space required after '<!NOTATION'\n");
5377
0
      return;
5378
0
  }
5379
5380
0
        name = xmlParseName(ctxt);
5381
0
  if (name == NULL) {
5382
0
      xmlFatalErr(ctxt, XML_ERR_NOTATION_NOT_STARTED, NULL);
5383
0
      return;
5384
0
  }
5385
0
  if (xmlStrchr(name, ':') != NULL) {
5386
0
      xmlNsErr(ctxt, XML_NS_ERR_COLON,
5387
0
         "colons are forbidden from notation names '%s'\n",
5388
0
         name, NULL, NULL);
5389
0
  }
5390
0
  if (SKIP_BLANKS_PE == 0) {
5391
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5392
0
         "Space required after the NOTATION name'\n");
5393
0
      return;
5394
0
  }
5395
5396
  /*
5397
   * Parse the IDs.
5398
   */
5399
0
  Systemid = xmlParseExternalID(ctxt, &Pubid, 0);
5400
0
  SKIP_BLANKS_PE;
5401
5402
0
  if (RAW == '>') {
5403
0
#ifdef LIBXML_VALID_ENABLED
5404
0
      if ((ctxt->validate) && (ctxt->inputNr > oldInputNr)) {
5405
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
5406
0
                           "Notation declaration doesn't start and stop"
5407
0
                                 " in the same entity\n",
5408
0
                                 NULL, NULL);
5409
0
      }
5410
0
#endif
5411
0
      NEXT;
5412
0
      if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
5413
0
    (ctxt->sax->notationDecl != NULL))
5414
0
    ctxt->sax->notationDecl(ctxt->userData, name, Pubid, Systemid);
5415
0
  } else {
5416
0
      xmlFatalErr(ctxt, XML_ERR_NOTATION_NOT_FINISHED, NULL);
5417
0
  }
5418
0
  if (Systemid != NULL) xmlFree(Systemid);
5419
0
  if (Pubid != NULL) xmlFree(Pubid);
5420
0
    }
5421
0
}
5422
5423
/**
5424
 * Parse an entity declaration. Always consumes '<!'.
5425
 *
5426
 * @deprecated Internal function, don't use.
5427
 *
5428
 *     [70] EntityDecl ::= GEDecl | PEDecl
5429
 *
5430
 *     [71] GEDecl ::= '<!ENTITY' S Name S EntityDef S? '>'
5431
 *
5432
 *     [72] PEDecl ::= '<!ENTITY' S '%' S Name S PEDef S? '>'
5433
 *
5434
 *     [73] EntityDef ::= EntityValue | (ExternalID NDataDecl?)
5435
 *
5436
 *     [74] PEDef ::= EntityValue | ExternalID
5437
 *
5438
 *     [76] NDataDecl ::= S 'NDATA' S Name
5439
 *
5440
 * [ VC: Notation Declared ]
5441
 * The Name must match the declared name of a notation.
5442
 *
5443
 * @param ctxt  an XML parser context
5444
 */
5445
5446
void
5447
0
xmlParseEntityDecl(xmlParserCtxt *ctxt) {
5448
0
    const xmlChar *name = NULL;
5449
0
    xmlChar *value = NULL;
5450
0
    xmlChar *URI = NULL, *literal = NULL;
5451
0
    const xmlChar *ndata = NULL;
5452
0
    int isParameter = 0;
5453
0
    xmlChar *orig = NULL;
5454
5455
0
    if ((CUR != '<') || (NXT(1) != '!'))
5456
0
        return;
5457
0
    SKIP(2);
5458
5459
    /* GROW; done in the caller */
5460
0
    if (CMP6(CUR_PTR, 'E', 'N', 'T', 'I', 'T', 'Y')) {
5461
0
#ifdef LIBXML_VALID_ENABLED
5462
0
  int oldInputNr = ctxt->inputNr;
5463
0
#endif
5464
5465
0
  SKIP(6);
5466
0
  if (SKIP_BLANKS_PE == 0) {
5467
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5468
0
         "Space required after '<!ENTITY'\n");
5469
0
  }
5470
5471
0
  if (RAW == '%') {
5472
0
      NEXT;
5473
0
      if (SKIP_BLANKS_PE == 0) {
5474
0
    xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5475
0
             "Space required after '%%'\n");
5476
0
      }
5477
0
      isParameter = 1;
5478
0
  }
5479
5480
0
        name = xmlParseName(ctxt);
5481
0
  if (name == NULL) {
5482
0
      xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
5483
0
                     "xmlParseEntityDecl: no name\n");
5484
0
            return;
5485
0
  }
5486
0
  if (xmlStrchr(name, ':') != NULL) {
5487
0
      xmlNsErr(ctxt, XML_NS_ERR_COLON,
5488
0
         "colons are forbidden from entities names '%s'\n",
5489
0
         name, NULL, NULL);
5490
0
  }
5491
0
  if (SKIP_BLANKS_PE == 0) {
5492
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5493
0
         "Space required after the entity name\n");
5494
0
  }
5495
5496
  /*
5497
   * handle the various case of definitions...
5498
   */
5499
0
  if (isParameter) {
5500
0
      if ((RAW == '"') || (RAW == '\'')) {
5501
0
          value = xmlParseEntityValue(ctxt, &orig);
5502
0
    if (value) {
5503
0
        if ((ctxt->sax != NULL) &&
5504
0
      (!ctxt->disableSAX) && (ctxt->sax->entityDecl != NULL))
5505
0
      ctxt->sax->entityDecl(ctxt->userData, name,
5506
0
                        XML_INTERNAL_PARAMETER_ENTITY,
5507
0
            NULL, NULL, value);
5508
0
    }
5509
0
      } else {
5510
0
          URI = xmlParseExternalID(ctxt, &literal, 1);
5511
0
    if ((URI == NULL) && (literal == NULL)) {
5512
0
        xmlFatalErr(ctxt, XML_ERR_VALUE_REQUIRED, NULL);
5513
0
    }
5514
0
    if (URI) {
5515
0
                    if (xmlStrchr(URI, '#')) {
5516
0
                        xmlFatalErr(ctxt, XML_ERR_URI_FRAGMENT, NULL);
5517
0
                    } else {
5518
0
                        if ((ctxt->sax != NULL) &&
5519
0
                            (!ctxt->disableSAX) &&
5520
0
                            (ctxt->sax->entityDecl != NULL))
5521
0
                            ctxt->sax->entityDecl(ctxt->userData, name,
5522
0
                                        XML_EXTERNAL_PARAMETER_ENTITY,
5523
0
                                        literal, URI, NULL);
5524
0
                    }
5525
0
    }
5526
0
      }
5527
0
  } else {
5528
0
      if ((RAW == '"') || (RAW == '\'')) {
5529
0
          value = xmlParseEntityValue(ctxt, &orig);
5530
0
    if ((ctxt->sax != NULL) &&
5531
0
        (!ctxt->disableSAX) && (ctxt->sax->entityDecl != NULL))
5532
0
        ctxt->sax->entityDecl(ctxt->userData, name,
5533
0
        XML_INTERNAL_GENERAL_ENTITY,
5534
0
        NULL, NULL, value);
5535
    /*
5536
     * For expat compatibility in SAX mode.
5537
     */
5538
0
    if ((ctxt->myDoc == NULL) ||
5539
0
        (xmlStrEqual(ctxt->myDoc->version, SAX_COMPAT_MODE))) {
5540
0
        if (ctxt->myDoc == NULL) {
5541
0
      ctxt->myDoc = xmlNewDoc(SAX_COMPAT_MODE);
5542
0
      if (ctxt->myDoc == NULL) {
5543
0
          xmlErrMemory(ctxt);
5544
0
          goto done;
5545
0
      }
5546
0
      ctxt->myDoc->properties = XML_DOC_INTERNAL;
5547
0
        }
5548
0
        if (ctxt->myDoc->intSubset == NULL) {
5549
0
      ctxt->myDoc->intSubset = xmlNewDtd(ctxt->myDoc,
5550
0
              BAD_CAST "fake", NULL, NULL);
5551
0
                        if (ctxt->myDoc->intSubset == NULL) {
5552
0
                            xmlErrMemory(ctxt);
5553
0
                            goto done;
5554
0
                        }
5555
0
                    }
5556
5557
0
        xmlSAX2EntityDecl(ctxt, name, XML_INTERNAL_GENERAL_ENTITY,
5558
0
                    NULL, NULL, value);
5559
0
    }
5560
0
      } else {
5561
0
          URI = xmlParseExternalID(ctxt, &literal, 1);
5562
0
    if ((URI == NULL) && (literal == NULL)) {
5563
0
        xmlFatalErr(ctxt, XML_ERR_VALUE_REQUIRED, NULL);
5564
0
    }
5565
0
    if (URI) {
5566
0
                    if (xmlStrchr(URI, '#')) {
5567
0
                        xmlFatalErr(ctxt, XML_ERR_URI_FRAGMENT, NULL);
5568
0
                    }
5569
0
    }
5570
0
    if ((RAW != '>') && (SKIP_BLANKS_PE == 0)) {
5571
0
        xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5572
0
           "Space required before 'NDATA'\n");
5573
0
    }
5574
0
    if (CMP5(CUR_PTR, 'N', 'D', 'A', 'T', 'A')) {
5575
0
        SKIP(5);
5576
0
        if (SKIP_BLANKS_PE == 0) {
5577
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5578
0
               "Space required after 'NDATA'\n");
5579
0
        }
5580
0
        ndata = xmlParseName(ctxt);
5581
0
        if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
5582
0
            (ctxt->sax->unparsedEntityDecl != NULL))
5583
0
      ctxt->sax->unparsedEntityDecl(ctxt->userData, name,
5584
0
            literal, URI, ndata);
5585
0
    } else {
5586
0
        if ((ctxt->sax != NULL) &&
5587
0
            (!ctxt->disableSAX) && (ctxt->sax->entityDecl != NULL))
5588
0
      ctxt->sax->entityDecl(ctxt->userData, name,
5589
0
            XML_EXTERNAL_GENERAL_PARSED_ENTITY,
5590
0
            literal, URI, NULL);
5591
        /*
5592
         * For expat compatibility in SAX mode.
5593
         * assuming the entity replacement was asked for
5594
         */
5595
0
        if ((ctxt->replaceEntities != 0) &&
5596
0
      ((ctxt->myDoc == NULL) ||
5597
0
      (xmlStrEqual(ctxt->myDoc->version, SAX_COMPAT_MODE)))) {
5598
0
      if (ctxt->myDoc == NULL) {
5599
0
          ctxt->myDoc = xmlNewDoc(SAX_COMPAT_MODE);
5600
0
          if (ctxt->myDoc == NULL) {
5601
0
              xmlErrMemory(ctxt);
5602
0
        goto done;
5603
0
          }
5604
0
          ctxt->myDoc->properties = XML_DOC_INTERNAL;
5605
0
      }
5606
5607
0
      if (ctxt->myDoc->intSubset == NULL) {
5608
0
          ctxt->myDoc->intSubset = xmlNewDtd(ctxt->myDoc,
5609
0
            BAD_CAST "fake", NULL, NULL);
5610
0
                            if (ctxt->myDoc->intSubset == NULL) {
5611
0
                                xmlErrMemory(ctxt);
5612
0
                                goto done;
5613
0
                            }
5614
0
                        }
5615
0
      xmlSAX2EntityDecl(ctxt, name,
5616
0
                  XML_EXTERNAL_GENERAL_PARSED_ENTITY,
5617
0
                  literal, URI, NULL);
5618
0
        }
5619
0
    }
5620
0
      }
5621
0
  }
5622
0
  SKIP_BLANKS_PE;
5623
0
  if (RAW != '>') {
5624
0
      xmlFatalErrMsgStr(ctxt, XML_ERR_ENTITY_NOT_FINISHED,
5625
0
              "xmlParseEntityDecl: entity %s not terminated\n", name);
5626
0
  } else {
5627
0
#ifdef LIBXML_VALID_ENABLED
5628
0
      if ((ctxt->validate) && (ctxt->inputNr > oldInputNr)) {
5629
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
5630
0
                           "Entity declaration doesn't start and stop in"
5631
0
                                 " the same entity\n",
5632
0
                                 NULL, NULL);
5633
0
      }
5634
0
#endif
5635
0
      NEXT;
5636
0
  }
5637
0
  if (orig != NULL) {
5638
      /*
5639
       * Ugly mechanism to save the raw entity value.
5640
       */
5641
0
      xmlEntityPtr cur = NULL;
5642
5643
0
      if (isParameter) {
5644
0
          if ((ctxt->sax != NULL) &&
5645
0
        (ctxt->sax->getParameterEntity != NULL))
5646
0
        cur = ctxt->sax->getParameterEntity(ctxt->userData, name);
5647
0
      } else {
5648
0
          if ((ctxt->sax != NULL) &&
5649
0
        (ctxt->sax->getEntity != NULL))
5650
0
        cur = ctxt->sax->getEntity(ctxt->userData, name);
5651
0
    if ((cur == NULL) && (ctxt->userData==ctxt)) {
5652
0
        cur = xmlSAX2GetEntity(ctxt, name);
5653
0
    }
5654
0
      }
5655
0
            if ((cur != NULL) && (cur->orig == NULL)) {
5656
0
    cur->orig = orig;
5657
0
                orig = NULL;
5658
0
      }
5659
0
  }
5660
5661
0
done:
5662
0
  if (value != NULL) xmlFree(value);
5663
0
  if (URI != NULL) xmlFree(URI);
5664
0
  if (literal != NULL) xmlFree(literal);
5665
0
        if (orig != NULL) xmlFree(orig);
5666
0
    }
5667
0
}
5668
5669
/**
5670
 * Parse an attribute default declaration
5671
 *
5672
 * @deprecated Internal function, don't use.
5673
 *
5674
 *     [60] DefaultDecl ::= '#REQUIRED' | '#IMPLIED' | (('#FIXED' S)? AttValue)
5675
 *
5676
 * [ VC: Required Attribute ]
5677
 * if the default declaration is the keyword \#REQUIRED, then the
5678
 * attribute must be specified for all elements of the type in the
5679
 * attribute-list declaration.
5680
 *
5681
 * [ VC: Attribute Default Legal ]
5682
 * The declared default value must meet the lexical constraints of
5683
 * the declared attribute type c.f. #xmlValidateAttributeDecl
5684
 *
5685
 * [ VC: Fixed Attribute Default ]
5686
 * if an attribute has a default value declared with the \#FIXED
5687
 * keyword, instances of that attribute must match the default value.
5688
 *
5689
 * [ WFC: No < in Attribute Values ]
5690
 * handled in #xmlParseAttValue
5691
 *
5692
 * @param ctxt  an XML parser context
5693
 * @param value  Receive a possible fixed default value for the attribute
5694
 * @returns XML_ATTRIBUTE_NONE, XML_ATTRIBUTE_REQUIRED, XML_ATTRIBUTE_IMPLIED
5695
 *          or XML_ATTRIBUTE_FIXED.
5696
 */
5697
5698
int
5699
0
xmlParseDefaultDecl(xmlParserCtxt *ctxt, xmlChar **value) {
5700
0
    int val;
5701
0
    xmlChar *ret;
5702
5703
0
    *value = NULL;
5704
0
    if (CMP9(CUR_PTR, '#', 'R', 'E', 'Q', 'U', 'I', 'R', 'E', 'D')) {
5705
0
  SKIP(9);
5706
0
  return(XML_ATTRIBUTE_REQUIRED);
5707
0
    }
5708
0
    if (CMP8(CUR_PTR, '#', 'I', 'M', 'P', 'L', 'I', 'E', 'D')) {
5709
0
  SKIP(8);
5710
0
  return(XML_ATTRIBUTE_IMPLIED);
5711
0
    }
5712
0
    val = XML_ATTRIBUTE_NONE;
5713
0
    if (CMP6(CUR_PTR, '#', 'F', 'I', 'X', 'E', 'D')) {
5714
0
  SKIP(6);
5715
0
  val = XML_ATTRIBUTE_FIXED;
5716
0
  if (SKIP_BLANKS_PE == 0) {
5717
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5718
0
         "Space required after '#FIXED'\n");
5719
0
  }
5720
0
    }
5721
0
    ret = xmlParseAttValue(ctxt);
5722
0
    if (ret == NULL) {
5723
0
  xmlFatalErrMsg(ctxt, (xmlParserErrors)ctxt->errNo,
5724
0
           "Attribute default value declaration error\n");
5725
0
    } else
5726
0
        *value = ret;
5727
0
    return(val);
5728
0
}
5729
5730
/**
5731
 * Parse an Notation attribute type.
5732
 *
5733
 * @deprecated Internal function, don't use.
5734
 *
5735
 * Note: the leading 'NOTATION' S part has already being parsed...
5736
 *
5737
 *     [58] NotationType ::= 'NOTATION' S '(' S? Name (S? '|' S? Name)* S? ')'
5738
 *
5739
 * [ VC: Notation Attributes ]
5740
 * Values of this type must match one of the notation names included
5741
 * in the declaration; all notation names in the declaration must be declared.
5742
 *
5743
 * @param ctxt  an XML parser context
5744
 * @returns the notation attribute tree built while parsing
5745
 */
5746
5747
xmlEnumeration *
5748
0
xmlParseNotationType(xmlParserCtxt *ctxt) {
5749
0
    const xmlChar *name;
5750
0
    xmlEnumerationPtr ret = NULL, last = NULL, cur, tmp;
5751
5752
0
    if (RAW != '(') {
5753
0
  xmlFatalErr(ctxt, XML_ERR_NOTATION_NOT_STARTED, NULL);
5754
0
  return(NULL);
5755
0
    }
5756
0
    do {
5757
0
        NEXT;
5758
0
  SKIP_BLANKS_PE;
5759
0
        name = xmlParseName(ctxt);
5760
0
  if (name == NULL) {
5761
0
      xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
5762
0
         "Name expected in NOTATION declaration\n");
5763
0
            xmlFreeEnumeration(ret);
5764
0
      return(NULL);
5765
0
  }
5766
0
        tmp = NULL;
5767
0
#ifdef LIBXML_VALID_ENABLED
5768
0
        if (ctxt->validate) {
5769
0
            tmp = ret;
5770
0
            while (tmp != NULL) {
5771
0
                if (xmlStrEqual(name, tmp->name)) {
5772
0
                    xmlValidityError(ctxt, XML_DTD_DUP_TOKEN,
5773
0
              "standalone: attribute notation value token %s duplicated\n",
5774
0
                                     name, NULL);
5775
0
                    if (!xmlDictOwns(ctxt->dict, name))
5776
0
                        xmlFree((xmlChar *) name);
5777
0
                    break;
5778
0
                }
5779
0
                tmp = tmp->next;
5780
0
            }
5781
0
        }
5782
0
#endif /* LIBXML_VALID_ENABLED */
5783
0
  if (tmp == NULL) {
5784
0
      cur = xmlCreateEnumeration(name);
5785
0
      if (cur == NULL) {
5786
0
                xmlErrMemory(ctxt);
5787
0
                xmlFreeEnumeration(ret);
5788
0
                return(NULL);
5789
0
            }
5790
0
      if (last == NULL) ret = last = cur;
5791
0
      else {
5792
0
    last->next = cur;
5793
0
    last = cur;
5794
0
      }
5795
0
  }
5796
0
  SKIP_BLANKS_PE;
5797
0
    } while (RAW == '|');
5798
0
    if (RAW != ')') {
5799
0
  xmlFatalErr(ctxt, XML_ERR_NOTATION_NOT_FINISHED, NULL);
5800
0
        xmlFreeEnumeration(ret);
5801
0
  return(NULL);
5802
0
    }
5803
0
    NEXT;
5804
0
    return(ret);
5805
0
}
5806
5807
/**
5808
 * Parse an Enumeration attribute type.
5809
 *
5810
 * @deprecated Internal function, don't use.
5811
 *
5812
 *     [59] Enumeration ::= '(' S? Nmtoken (S? '|' S? Nmtoken)* S? ')'
5813
 *
5814
 * [ VC: Enumeration ]
5815
 * Values of this type must match one of the Nmtoken tokens in
5816
 * the declaration
5817
 *
5818
 * @param ctxt  an XML parser context
5819
 * @returns the enumeration attribute tree built while parsing
5820
 */
5821
5822
xmlEnumeration *
5823
0
xmlParseEnumerationType(xmlParserCtxt *ctxt) {
5824
0
    xmlChar *name;
5825
0
    xmlEnumerationPtr ret = NULL, last = NULL, cur, tmp;
5826
5827
0
    if (RAW != '(') {
5828
0
  xmlFatalErr(ctxt, XML_ERR_ATTLIST_NOT_STARTED, NULL);
5829
0
  return(NULL);
5830
0
    }
5831
0
    do {
5832
0
        NEXT;
5833
0
  SKIP_BLANKS_PE;
5834
0
        name = xmlParseNmtoken(ctxt);
5835
0
  if (name == NULL) {
5836
0
      xmlFatalErr(ctxt, XML_ERR_NMTOKEN_REQUIRED, NULL);
5837
0
      return(ret);
5838
0
  }
5839
0
        tmp = NULL;
5840
0
#ifdef LIBXML_VALID_ENABLED
5841
0
        if (ctxt->validate) {
5842
0
            tmp = ret;
5843
0
            while (tmp != NULL) {
5844
0
                if (xmlStrEqual(name, tmp->name)) {
5845
0
                    xmlValidityError(ctxt, XML_DTD_DUP_TOKEN,
5846
0
              "standalone: attribute enumeration value token %s duplicated\n",
5847
0
                                     name, NULL);
5848
0
                    if (!xmlDictOwns(ctxt->dict, name))
5849
0
                        xmlFree(name);
5850
0
                    break;
5851
0
                }
5852
0
                tmp = tmp->next;
5853
0
            }
5854
0
        }
5855
0
#endif /* LIBXML_VALID_ENABLED */
5856
0
  if (tmp == NULL) {
5857
0
      cur = xmlCreateEnumeration(name);
5858
0
      if (!xmlDictOwns(ctxt->dict, name))
5859
0
    xmlFree(name);
5860
0
      if (cur == NULL) {
5861
0
                xmlErrMemory(ctxt);
5862
0
                xmlFreeEnumeration(ret);
5863
0
                return(NULL);
5864
0
            }
5865
0
      if (last == NULL) ret = last = cur;
5866
0
      else {
5867
0
    last->next = cur;
5868
0
    last = cur;
5869
0
      }
5870
0
  }
5871
0
  SKIP_BLANKS_PE;
5872
0
    } while (RAW == '|');
5873
0
    if (RAW != ')') {
5874
0
  xmlFatalErr(ctxt, XML_ERR_ATTLIST_NOT_FINISHED, NULL);
5875
0
  return(ret);
5876
0
    }
5877
0
    NEXT;
5878
0
    return(ret);
5879
0
}
5880
5881
/**
5882
 * Parse an Enumerated attribute type.
5883
 *
5884
 * @deprecated Internal function, don't use.
5885
 *
5886
 *     [57] EnumeratedType ::= NotationType | Enumeration
5887
 *
5888
 *     [58] NotationType ::= 'NOTATION' S '(' S? Name (S? '|' S? Name)* S? ')'
5889
 *
5890
 * @param ctxt  an XML parser context
5891
 * @param tree  the enumeration tree built while parsing
5892
 * @returns XML_ATTRIBUTE_ENUMERATION or XML_ATTRIBUTE_NOTATION
5893
 */
5894
5895
int
5896
0
xmlParseEnumeratedType(xmlParserCtxt *ctxt, xmlEnumeration **tree) {
5897
0
    if (CMP8(CUR_PTR, 'N', 'O', 'T', 'A', 'T', 'I', 'O', 'N')) {
5898
0
  SKIP(8);
5899
0
  if (SKIP_BLANKS_PE == 0) {
5900
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
5901
0
         "Space required after 'NOTATION'\n");
5902
0
      return(0);
5903
0
  }
5904
0
  *tree = xmlParseNotationType(ctxt);
5905
0
  if (*tree == NULL) return(0);
5906
0
  return(XML_ATTRIBUTE_NOTATION);
5907
0
    }
5908
0
    *tree = xmlParseEnumerationType(ctxt);
5909
0
    if (*tree == NULL) return(0);
5910
0
    return(XML_ATTRIBUTE_ENUMERATION);
5911
0
}
5912
5913
/**
5914
 * Parse the Attribute list def for an element
5915
 *
5916
 * @deprecated Internal function, don't use.
5917
 *
5918
 *     [54] AttType ::= StringType | TokenizedType | EnumeratedType
5919
 *
5920
 *     [55] StringType ::= 'CDATA'
5921
 *
5922
 *     [56] TokenizedType ::= 'ID' | 'IDREF' | 'IDREFS' | 'ENTITY' |
5923
 *                            'ENTITIES' | 'NMTOKEN' | 'NMTOKENS'
5924
 *
5925
 * Validity constraints for attribute values syntax are checked in
5926
 * #xmlValidateAttributeValue
5927
 *
5928
 * [ VC: ID ]
5929
 * Values of type ID must match the Name production. A name must not
5930
 * appear more than once in an XML document as a value of this type;
5931
 * i.e., ID values must uniquely identify the elements which bear them.
5932
 *
5933
 * [ VC: One ID per Element Type ]
5934
 * No element type may have more than one ID attribute specified.
5935
 *
5936
 * [ VC: ID Attribute Default ]
5937
 * An ID attribute must have a declared default of \#IMPLIED or \#REQUIRED.
5938
 *
5939
 * [ VC: IDREF ]
5940
 * Values of type IDREF must match the Name production, and values
5941
 * of type IDREFS must match Names; each IDREF Name must match the value
5942
 * of an ID attribute on some element in the XML document; i.e. IDREF
5943
 * values must match the value of some ID attribute.
5944
 *
5945
 * [ VC: Entity Name ]
5946
 * Values of type ENTITY must match the Name production, values
5947
 * of type ENTITIES must match Names; each Entity Name must match the
5948
 * name of an unparsed entity declared in the DTD.
5949
 *
5950
 * [ VC: Name Token ]
5951
 * Values of type NMTOKEN must match the Nmtoken production; values
5952
 * of type NMTOKENS must match Nmtokens.
5953
 *
5954
 * @param ctxt  an XML parser context
5955
 * @param tree  the enumeration tree built while parsing
5956
 * @returns the attribute type
5957
 */
5958
int
5959
0
xmlParseAttributeType(xmlParserCtxt *ctxt, xmlEnumeration **tree) {
5960
0
    if (CMP5(CUR_PTR, 'C', 'D', 'A', 'T', 'A')) {
5961
0
  SKIP(5);
5962
0
  return(XML_ATTRIBUTE_CDATA);
5963
0
     } else if (CMP6(CUR_PTR, 'I', 'D', 'R', 'E', 'F', 'S')) {
5964
0
  SKIP(6);
5965
0
  return(XML_ATTRIBUTE_IDREFS);
5966
0
     } else if (CMP5(CUR_PTR, 'I', 'D', 'R', 'E', 'F')) {
5967
0
  SKIP(5);
5968
0
  return(XML_ATTRIBUTE_IDREF);
5969
0
     } else if ((RAW == 'I') && (NXT(1) == 'D')) {
5970
0
        SKIP(2);
5971
0
  return(XML_ATTRIBUTE_ID);
5972
0
     } else if (CMP6(CUR_PTR, 'E', 'N', 'T', 'I', 'T', 'Y')) {
5973
0
  SKIP(6);
5974
0
  return(XML_ATTRIBUTE_ENTITY);
5975
0
     } else if (CMP8(CUR_PTR, 'E', 'N', 'T', 'I', 'T', 'I', 'E', 'S')) {
5976
0
  SKIP(8);
5977
0
  return(XML_ATTRIBUTE_ENTITIES);
5978
0
     } else if (CMP8(CUR_PTR, 'N', 'M', 'T', 'O', 'K', 'E', 'N', 'S')) {
5979
0
  SKIP(8);
5980
0
  return(XML_ATTRIBUTE_NMTOKENS);
5981
0
     } else if (CMP7(CUR_PTR, 'N', 'M', 'T', 'O', 'K', 'E', 'N')) {
5982
0
  SKIP(7);
5983
0
  return(XML_ATTRIBUTE_NMTOKEN);
5984
0
     }
5985
0
     return(xmlParseEnumeratedType(ctxt, tree));
5986
0
}
5987
5988
/**
5989
 * Parse an attribute list declaration for an element. Always consumes '<!'.
5990
 *
5991
 * @deprecated Internal function, don't use.
5992
 *
5993
 *     [52] AttlistDecl ::= '<!ATTLIST' S Name AttDef* S? '>'
5994
 *
5995
 *     [53] AttDef ::= S Name S AttType S DefaultDecl
5996
 * @param ctxt  an XML parser context
5997
 */
5998
void
5999
0
xmlParseAttributeListDecl(xmlParserCtxt *ctxt) {
6000
0
    const xmlChar *elemName;
6001
0
    const xmlChar *attrName;
6002
0
    xmlEnumerationPtr tree;
6003
6004
0
    if ((CUR != '<') || (NXT(1) != '!'))
6005
0
        return;
6006
0
    SKIP(2);
6007
6008
0
    if (CMP7(CUR_PTR, 'A', 'T', 'T', 'L', 'I', 'S', 'T')) {
6009
0
#ifdef LIBXML_VALID_ENABLED
6010
0
  int oldInputNr = ctxt->inputNr;
6011
0
#endif
6012
6013
0
  SKIP(7);
6014
0
  if (SKIP_BLANKS_PE == 0) {
6015
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
6016
0
                     "Space required after '<!ATTLIST'\n");
6017
0
  }
6018
0
        elemName = xmlParseName(ctxt);
6019
0
  if (elemName == NULL) {
6020
0
      xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
6021
0
         "ATTLIST: no name for Element\n");
6022
0
      return;
6023
0
  }
6024
0
  SKIP_BLANKS_PE;
6025
0
  GROW;
6026
0
  while ((RAW != '>') && (PARSER_STOPPED(ctxt) == 0)) {
6027
0
      int type;
6028
0
      int def;
6029
0
      xmlChar *defaultValue = NULL;
6030
6031
0
      GROW;
6032
0
            tree = NULL;
6033
0
      attrName = xmlParseName(ctxt);
6034
0
      if (attrName == NULL) {
6035
0
    xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
6036
0
             "ATTLIST: no name for Attribute\n");
6037
0
    break;
6038
0
      }
6039
0
      GROW;
6040
0
      if (SKIP_BLANKS_PE == 0) {
6041
0
    xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
6042
0
            "Space required after the attribute name\n");
6043
0
    break;
6044
0
      }
6045
6046
0
      type = xmlParseAttributeType(ctxt, &tree);
6047
0
      if (type <= 0) {
6048
0
          break;
6049
0
      }
6050
6051
0
      GROW;
6052
0
      if (SKIP_BLANKS_PE == 0) {
6053
0
    xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
6054
0
             "Space required after the attribute type\n");
6055
0
          if (tree != NULL)
6056
0
        xmlFreeEnumeration(tree);
6057
0
    break;
6058
0
      }
6059
6060
0
      def = xmlParseDefaultDecl(ctxt, &defaultValue);
6061
0
      if (def <= 0) {
6062
0
                if (defaultValue != NULL)
6063
0
        xmlFree(defaultValue);
6064
0
          if (tree != NULL)
6065
0
        xmlFreeEnumeration(tree);
6066
0
          break;
6067
0
      }
6068
0
      if ((type != XML_ATTRIBUTE_CDATA) && (defaultValue != NULL))
6069
0
          xmlAttrNormalizeSpace(defaultValue, defaultValue);
6070
6071
0
      GROW;
6072
0
            if (RAW != '>') {
6073
0
    if (SKIP_BLANKS_PE == 0) {
6074
0
        xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
6075
0
      "Space required after the attribute default value\n");
6076
0
        if (defaultValue != NULL)
6077
0
      xmlFree(defaultValue);
6078
0
        if (tree != NULL)
6079
0
      xmlFreeEnumeration(tree);
6080
0
        break;
6081
0
    }
6082
0
      }
6083
0
      if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
6084
0
    (ctxt->sax->attributeDecl != NULL))
6085
0
    ctxt->sax->attributeDecl(ctxt->userData, elemName, attrName,
6086
0
                          type, def, defaultValue, tree);
6087
0
      else if (tree != NULL)
6088
0
    xmlFreeEnumeration(tree);
6089
6090
0
      if ((ctxt->sax2) && (defaultValue != NULL) &&
6091
0
          (def != XML_ATTRIBUTE_IMPLIED) &&
6092
0
    (def != XML_ATTRIBUTE_REQUIRED)) {
6093
0
    xmlAddDefAttrs(ctxt, elemName, attrName, defaultValue);
6094
0
      }
6095
0
      if (ctxt->sax2) {
6096
0
    xmlAddSpecialAttr(ctxt, elemName, attrName, type);
6097
0
      }
6098
0
      if (defaultValue != NULL)
6099
0
          xmlFree(defaultValue);
6100
0
      GROW;
6101
0
  }
6102
0
  if (RAW == '>') {
6103
0
#ifdef LIBXML_VALID_ENABLED
6104
0
      if ((ctxt->validate) && (ctxt->inputNr > oldInputNr)) {
6105
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6106
0
                                 "Attribute list declaration doesn't start and"
6107
0
                                 " stop in the same entity\n",
6108
0
                                 NULL, NULL);
6109
0
      }
6110
0
#endif
6111
0
      NEXT;
6112
0
  }
6113
0
    }
6114
0
}
6115
6116
/**
6117
 * Handle PEs and check that we don't pop the entity that started
6118
 * a balanced group.
6119
 *
6120
 * @param ctxt  parser context
6121
 * @param openInputNr  input nr of the entity with opening '('
6122
 */
6123
static void
6124
0
xmlSkipBlankCharsPEBalanced(xmlParserCtxt *ctxt, int openInputNr) {
6125
0
    SKIP_BLANKS;
6126
0
    GROW;
6127
6128
0
    (void) openInputNr;
6129
6130
0
    if (!PARSER_EXTERNAL(ctxt) && !PARSER_IN_PE(ctxt))
6131
0
        return;
6132
6133
0
    while (!PARSER_STOPPED(ctxt)) {
6134
0
        if (ctxt->input->cur >= ctxt->input->end) {
6135
0
#ifdef LIBXML_VALID_ENABLED
6136
0
            if ((ctxt->validate) && (ctxt->inputNr <= openInputNr)) {
6137
0
                xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6138
0
                                 "Element content declaration doesn't start "
6139
0
                                 "and stop in the same entity\n",
6140
0
                                 NULL, NULL);
6141
0
            }
6142
0
#endif
6143
0
            if (PARSER_IN_PE(ctxt))
6144
0
                xmlPopPE(ctxt);
6145
0
            else
6146
0
                break;
6147
0
        } else if (RAW == '%') {
6148
0
            xmlParsePERefInternal(ctxt, 0);
6149
0
        } else {
6150
0
            break;
6151
0
        }
6152
6153
0
        SKIP_BLANKS;
6154
0
        GROW;
6155
0
    }
6156
0
}
6157
6158
/**
6159
 * Parse the declaration for a Mixed Element content
6160
 * The leading '(' and spaces have been skipped in #xmlParseElementContentDecl
6161
 *
6162
 * @deprecated Internal function, don't use.
6163
 *
6164
 *     [51] Mixed ::= '(' S? '#PCDATA' (S? '|' S? Name)* S? ')*' |
6165
 *                    '(' S? '#PCDATA' S? ')'
6166
 *
6167
 * [ VC: Proper Group/PE Nesting ] applies to [51] too (see [49])
6168
 *
6169
 * [ VC: No Duplicate Types ]
6170
 * The same name must not appear more than once in a single
6171
 * mixed-content declaration.
6172
 *
6173
 * @param ctxt  an XML parser context
6174
 * @param openInputNr  the input used for the current entity, needed for
6175
 * boundary checks
6176
 * @returns the list of the xmlElementContent describing the element choices
6177
 */
6178
xmlElementContent *
6179
0
xmlParseElementMixedContentDecl(xmlParserCtxt *ctxt, int openInputNr) {
6180
0
    xmlElementContentPtr ret = NULL, cur = NULL, n;
6181
0
    const xmlChar *elem = NULL;
6182
6183
0
    GROW;
6184
0
    if (CMP7(CUR_PTR, '#', 'P', 'C', 'D', 'A', 'T', 'A')) {
6185
0
  SKIP(7);
6186
0
        xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6187
0
  if (RAW == ')') {
6188
0
#ifdef LIBXML_VALID_ENABLED
6189
0
      if ((ctxt->validate) && (ctxt->inputNr > openInputNr)) {
6190
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6191
0
                                 "Element content declaration doesn't start "
6192
0
                                 "and stop in the same entity\n",
6193
0
                                 NULL, NULL);
6194
0
      }
6195
0
#endif
6196
0
      NEXT;
6197
0
      ret = xmlNewDocElementContent(ctxt->myDoc, NULL, XML_ELEMENT_CONTENT_PCDATA);
6198
0
      if (ret == NULL)
6199
0
                goto mem_error;
6200
0
      if (RAW == '*') {
6201
0
    ret->ocur = XML_ELEMENT_CONTENT_MULT;
6202
0
    NEXT;
6203
0
      }
6204
0
      return(ret);
6205
0
  }
6206
0
  if ((RAW == '(') || (RAW == '|')) {
6207
0
      ret = cur = xmlNewDocElementContent(ctxt->myDoc, NULL, XML_ELEMENT_CONTENT_PCDATA);
6208
0
      if (ret == NULL)
6209
0
                goto mem_error;
6210
0
  }
6211
0
  while ((RAW == '|') && (PARSER_STOPPED(ctxt) == 0)) {
6212
0
      NEXT;
6213
0
            n = xmlNewDocElementContent(ctxt->myDoc, NULL, XML_ELEMENT_CONTENT_OR);
6214
0
            if (n == NULL)
6215
0
                goto mem_error;
6216
0
      if (elem == NULL) {
6217
0
    n->c1 = cur;
6218
0
    if (cur != NULL)
6219
0
        cur->parent = n;
6220
0
    ret = cur = n;
6221
0
      } else {
6222
0
          cur->c2 = n;
6223
0
    n->parent = cur;
6224
0
    n->c1 = xmlNewDocElementContent(ctxt->myDoc, elem, XML_ELEMENT_CONTENT_ELEMENT);
6225
0
                if (n->c1 == NULL)
6226
0
                    goto mem_error;
6227
0
    n->c1->parent = n;
6228
0
    cur = n;
6229
0
      }
6230
0
            xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6231
0
      elem = xmlParseName(ctxt);
6232
0
      if (elem == NULL) {
6233
0
    xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
6234
0
      "xmlParseElementMixedContentDecl : Name expected\n");
6235
0
    xmlFreeDocElementContent(ctxt->myDoc, ret);
6236
0
    return(NULL);
6237
0
      }
6238
0
            xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6239
0
  }
6240
0
  if ((RAW == ')') && (NXT(1) == '*')) {
6241
0
      if (elem != NULL) {
6242
0
    cur->c2 = xmlNewDocElementContent(ctxt->myDoc, elem,
6243
0
                                   XML_ELEMENT_CONTENT_ELEMENT);
6244
0
    if (cur->c2 == NULL)
6245
0
                    goto mem_error;
6246
0
    cur->c2->parent = cur;
6247
0
            }
6248
0
            if (ret != NULL)
6249
0
                ret->ocur = XML_ELEMENT_CONTENT_MULT;
6250
0
#ifdef LIBXML_VALID_ENABLED
6251
0
      if ((ctxt->validate) && (ctxt->inputNr > openInputNr)) {
6252
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6253
0
                                 "Element content declaration doesn't start "
6254
0
                                 "and stop in the same entity\n",
6255
0
                                 NULL, NULL);
6256
0
      }
6257
0
#endif
6258
0
      SKIP(2);
6259
0
  } else {
6260
0
      xmlFreeDocElementContent(ctxt->myDoc, ret);
6261
0
      xmlFatalErr(ctxt, XML_ERR_MIXED_NOT_STARTED, NULL);
6262
0
      return(NULL);
6263
0
  }
6264
6265
0
    } else {
6266
0
  xmlFatalErr(ctxt, XML_ERR_PCDATA_REQUIRED, NULL);
6267
0
    }
6268
0
    return(ret);
6269
6270
0
mem_error:
6271
0
    xmlErrMemory(ctxt);
6272
0
    xmlFreeDocElementContent(ctxt->myDoc, ret);
6273
0
    return(NULL);
6274
0
}
6275
6276
/**
6277
 * Parse the declaration for a Mixed Element content
6278
 * The leading '(' and spaces have been skipped in #xmlParseElementContentDecl
6279
 *
6280
 *     [47] children ::= (choice | seq) ('?' | '*' | '+')?
6281
 *
6282
 *     [48] cp ::= (Name | choice | seq) ('?' | '*' | '+')?
6283
 *
6284
 *     [49] choice ::= '(' S? cp ( S? '|' S? cp )* S? ')'
6285
 *
6286
 *     [50] seq ::= '(' S? cp ( S? ',' S? cp )* S? ')'
6287
 *
6288
 * [ VC: Proper Group/PE Nesting ] applies to [49] and [50]
6289
 * TODO Parameter-entity replacement text must be properly nested
6290
 *  with parenthesized groups. That is to say, if either of the
6291
 *  opening or closing parentheses in a choice, seq, or Mixed
6292
 *  construct is contained in the replacement text for a parameter
6293
 *  entity, both must be contained in the same replacement text. For
6294
 *  interoperability, if a parameter-entity reference appears in a
6295
 *  choice, seq, or Mixed construct, its replacement text should not
6296
 *  be empty, and neither the first nor last non-blank character of
6297
 *  the replacement text should be a connector (| or ,).
6298
 *
6299
 * @param ctxt  an XML parser context
6300
 * @param openInputNr  the input used for the current entity, needed for
6301
 * boundary checks
6302
 * @param depth  the level of recursion
6303
 * @returns the tree of xmlElementContent describing the element
6304
 *          hierarchy.
6305
 */
6306
static xmlElementContentPtr
6307
xmlParseElementChildrenContentDeclPriv(xmlParserCtxtPtr ctxt, int openInputNr,
6308
0
                                       int depth) {
6309
0
    int maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 2048 : 256;
6310
0
    xmlElementContentPtr ret = NULL, cur = NULL, last = NULL, op = NULL;
6311
0
    const xmlChar *elem;
6312
0
    xmlChar type = 0;
6313
6314
0
    if (depth > maxDepth) {
6315
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_RESOURCE_LIMIT,
6316
0
                "xmlParseElementChildrenContentDecl : depth %d too deep, "
6317
0
                "use XML_PARSE_HUGE\n", depth);
6318
0
  return(NULL);
6319
0
    }
6320
0
    xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6321
0
    if (RAW == '(') {
6322
0
        int newInputNr = ctxt->inputNr;
6323
6324
        /* Recurse on first child */
6325
0
  NEXT;
6326
0
        cur = ret = xmlParseElementChildrenContentDeclPriv(ctxt, newInputNr,
6327
0
                                                           depth + 1);
6328
0
        if (cur == NULL)
6329
0
            return(NULL);
6330
0
    } else {
6331
0
  elem = xmlParseName(ctxt);
6332
0
  if (elem == NULL) {
6333
0
      xmlFatalErr(ctxt, XML_ERR_ELEMCONTENT_NOT_STARTED, NULL);
6334
0
      return(NULL);
6335
0
  }
6336
0
        cur = ret = xmlNewDocElementContent(ctxt->myDoc, elem, XML_ELEMENT_CONTENT_ELEMENT);
6337
0
  if (cur == NULL) {
6338
0
      xmlErrMemory(ctxt);
6339
0
      return(NULL);
6340
0
  }
6341
0
  GROW;
6342
0
  if (RAW == '?') {
6343
0
      cur->ocur = XML_ELEMENT_CONTENT_OPT;
6344
0
      NEXT;
6345
0
  } else if (RAW == '*') {
6346
0
      cur->ocur = XML_ELEMENT_CONTENT_MULT;
6347
0
      NEXT;
6348
0
  } else if (RAW == '+') {
6349
0
      cur->ocur = XML_ELEMENT_CONTENT_PLUS;
6350
0
      NEXT;
6351
0
  } else {
6352
0
      cur->ocur = XML_ELEMENT_CONTENT_ONCE;
6353
0
  }
6354
0
  GROW;
6355
0
    }
6356
0
    while (!PARSER_STOPPED(ctxt)) {
6357
0
        xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6358
0
        if (RAW == ')')
6359
0
            break;
6360
        /*
6361
   * Each loop we parse one separator and one element.
6362
   */
6363
0
        if (RAW == ',') {
6364
0
      if (type == 0) type = CUR;
6365
6366
      /*
6367
       * Detect "Name | Name , Name" error
6368
       */
6369
0
      else if (type != CUR) {
6370
0
    xmlFatalErrMsgInt(ctxt, XML_ERR_SEPARATOR_REQUIRED,
6371
0
        "xmlParseElementChildrenContentDecl : '%c' expected\n",
6372
0
                      type);
6373
0
    if ((last != NULL) && (last != ret))
6374
0
        xmlFreeDocElementContent(ctxt->myDoc, last);
6375
0
    if (ret != NULL)
6376
0
        xmlFreeDocElementContent(ctxt->myDoc, ret);
6377
0
    return(NULL);
6378
0
      }
6379
0
      NEXT;
6380
6381
0
      op = xmlNewDocElementContent(ctxt->myDoc, NULL, XML_ELEMENT_CONTENT_SEQ);
6382
0
      if (op == NULL) {
6383
0
                xmlErrMemory(ctxt);
6384
0
    if ((last != NULL) && (last != ret))
6385
0
        xmlFreeDocElementContent(ctxt->myDoc, last);
6386
0
          xmlFreeDocElementContent(ctxt->myDoc, ret);
6387
0
    return(NULL);
6388
0
      }
6389
0
      if (last == NULL) {
6390
0
    op->c1 = ret;
6391
0
    if (ret != NULL)
6392
0
        ret->parent = op;
6393
0
    ret = cur = op;
6394
0
      } else {
6395
0
          cur->c2 = op;
6396
0
    if (op != NULL)
6397
0
        op->parent = cur;
6398
0
    op->c1 = last;
6399
0
    if (last != NULL)
6400
0
        last->parent = op;
6401
0
    cur =op;
6402
0
    last = NULL;
6403
0
      }
6404
0
  } else if (RAW == '|') {
6405
0
      if (type == 0) type = CUR;
6406
6407
      /*
6408
       * Detect "Name , Name | Name" error
6409
       */
6410
0
      else if (type != CUR) {
6411
0
    xmlFatalErrMsgInt(ctxt, XML_ERR_SEPARATOR_REQUIRED,
6412
0
        "xmlParseElementChildrenContentDecl : '%c' expected\n",
6413
0
          type);
6414
0
    if ((last != NULL) && (last != ret))
6415
0
        xmlFreeDocElementContent(ctxt->myDoc, last);
6416
0
    if (ret != NULL)
6417
0
        xmlFreeDocElementContent(ctxt->myDoc, ret);
6418
0
    return(NULL);
6419
0
      }
6420
0
      NEXT;
6421
6422
0
      op = xmlNewDocElementContent(ctxt->myDoc, NULL, XML_ELEMENT_CONTENT_OR);
6423
0
      if (op == NULL) {
6424
0
                xmlErrMemory(ctxt);
6425
0
    if ((last != NULL) && (last != ret))
6426
0
        xmlFreeDocElementContent(ctxt->myDoc, last);
6427
0
    if (ret != NULL)
6428
0
        xmlFreeDocElementContent(ctxt->myDoc, ret);
6429
0
    return(NULL);
6430
0
      }
6431
0
      if (last == NULL) {
6432
0
    op->c1 = ret;
6433
0
    if (ret != NULL)
6434
0
        ret->parent = op;
6435
0
    ret = cur = op;
6436
0
      } else {
6437
0
          cur->c2 = op;
6438
0
    if (op != NULL)
6439
0
        op->parent = cur;
6440
0
    op->c1 = last;
6441
0
    if (last != NULL)
6442
0
        last->parent = op;
6443
0
    cur =op;
6444
0
    last = NULL;
6445
0
      }
6446
0
  } else {
6447
0
      xmlFatalErr(ctxt, XML_ERR_ELEMCONTENT_NOT_FINISHED, NULL);
6448
0
      if ((last != NULL) && (last != ret))
6449
0
          xmlFreeDocElementContent(ctxt->myDoc, last);
6450
0
      if (ret != NULL)
6451
0
    xmlFreeDocElementContent(ctxt->myDoc, ret);
6452
0
      return(NULL);
6453
0
  }
6454
0
        xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6455
0
        if (RAW == '(') {
6456
0
            int newInputNr = ctxt->inputNr;
6457
6458
      /* Recurse on second child */
6459
0
      NEXT;
6460
0
      last = xmlParseElementChildrenContentDeclPriv(ctxt, newInputNr,
6461
0
                                                          depth + 1);
6462
0
            if (last == NULL) {
6463
0
    if (ret != NULL)
6464
0
        xmlFreeDocElementContent(ctxt->myDoc, ret);
6465
0
    return(NULL);
6466
0
            }
6467
0
  } else {
6468
0
      elem = xmlParseName(ctxt);
6469
0
      if (elem == NULL) {
6470
0
    xmlFatalErr(ctxt, XML_ERR_ELEMCONTENT_NOT_STARTED, NULL);
6471
0
    if (ret != NULL)
6472
0
        xmlFreeDocElementContent(ctxt->myDoc, ret);
6473
0
    return(NULL);
6474
0
      }
6475
0
      last = xmlNewDocElementContent(ctxt->myDoc, elem, XML_ELEMENT_CONTENT_ELEMENT);
6476
0
      if (last == NULL) {
6477
0
                xmlErrMemory(ctxt);
6478
0
    if (ret != NULL)
6479
0
        xmlFreeDocElementContent(ctxt->myDoc, ret);
6480
0
    return(NULL);
6481
0
      }
6482
0
      if (RAW == '?') {
6483
0
    last->ocur = XML_ELEMENT_CONTENT_OPT;
6484
0
    NEXT;
6485
0
      } else if (RAW == '*') {
6486
0
    last->ocur = XML_ELEMENT_CONTENT_MULT;
6487
0
    NEXT;
6488
0
      } else if (RAW == '+') {
6489
0
    last->ocur = XML_ELEMENT_CONTENT_PLUS;
6490
0
    NEXT;
6491
0
      } else {
6492
0
    last->ocur = XML_ELEMENT_CONTENT_ONCE;
6493
0
      }
6494
0
  }
6495
0
    }
6496
0
    if ((cur != NULL) && (last != NULL)) {
6497
0
        cur->c2 = last;
6498
0
  if (last != NULL)
6499
0
      last->parent = cur;
6500
0
    }
6501
0
#ifdef LIBXML_VALID_ENABLED
6502
0
    if ((ctxt->validate) && (ctxt->inputNr > openInputNr)) {
6503
0
        xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6504
0
                         "Element content declaration doesn't start "
6505
0
                         "and stop in the same entity\n",
6506
0
                         NULL, NULL);
6507
0
    }
6508
0
#endif
6509
0
    NEXT;
6510
0
    if (RAW == '?') {
6511
0
  if (ret != NULL) {
6512
0
      if ((ret->ocur == XML_ELEMENT_CONTENT_PLUS) ||
6513
0
          (ret->ocur == XML_ELEMENT_CONTENT_MULT))
6514
0
          ret->ocur = XML_ELEMENT_CONTENT_MULT;
6515
0
      else
6516
0
          ret->ocur = XML_ELEMENT_CONTENT_OPT;
6517
0
  }
6518
0
  NEXT;
6519
0
    } else if (RAW == '*') {
6520
0
  if (ret != NULL) {
6521
0
      ret->ocur = XML_ELEMENT_CONTENT_MULT;
6522
0
      cur = ret;
6523
      /*
6524
       * Some normalization:
6525
       * (a | b* | c?)* == (a | b | c)*
6526
       */
6527
0
      while ((cur != NULL) && (cur->type == XML_ELEMENT_CONTENT_OR)) {
6528
0
    if ((cur->c1 != NULL) &&
6529
0
              ((cur->c1->ocur == XML_ELEMENT_CONTENT_OPT) ||
6530
0
         (cur->c1->ocur == XML_ELEMENT_CONTENT_MULT)))
6531
0
        cur->c1->ocur = XML_ELEMENT_CONTENT_ONCE;
6532
0
    if ((cur->c2 != NULL) &&
6533
0
              ((cur->c2->ocur == XML_ELEMENT_CONTENT_OPT) ||
6534
0
         (cur->c2->ocur == XML_ELEMENT_CONTENT_MULT)))
6535
0
        cur->c2->ocur = XML_ELEMENT_CONTENT_ONCE;
6536
0
    cur = cur->c2;
6537
0
      }
6538
0
  }
6539
0
  NEXT;
6540
0
    } else if (RAW == '+') {
6541
0
  if (ret != NULL) {
6542
0
      int found = 0;
6543
6544
0
      if ((ret->ocur == XML_ELEMENT_CONTENT_OPT) ||
6545
0
          (ret->ocur == XML_ELEMENT_CONTENT_MULT))
6546
0
          ret->ocur = XML_ELEMENT_CONTENT_MULT;
6547
0
      else
6548
0
          ret->ocur = XML_ELEMENT_CONTENT_PLUS;
6549
      /*
6550
       * Some normalization:
6551
       * (a | b*)+ == (a | b)*
6552
       * (a | b?)+ == (a | b)*
6553
       */
6554
0
      while ((cur != NULL) && (cur->type == XML_ELEMENT_CONTENT_OR)) {
6555
0
    if ((cur->c1 != NULL) &&
6556
0
              ((cur->c1->ocur == XML_ELEMENT_CONTENT_OPT) ||
6557
0
         (cur->c1->ocur == XML_ELEMENT_CONTENT_MULT))) {
6558
0
        cur->c1->ocur = XML_ELEMENT_CONTENT_ONCE;
6559
0
        found = 1;
6560
0
    }
6561
0
    if ((cur->c2 != NULL) &&
6562
0
              ((cur->c2->ocur == XML_ELEMENT_CONTENT_OPT) ||
6563
0
         (cur->c2->ocur == XML_ELEMENT_CONTENT_MULT))) {
6564
0
        cur->c2->ocur = XML_ELEMENT_CONTENT_ONCE;
6565
0
        found = 1;
6566
0
    }
6567
0
    cur = cur->c2;
6568
0
      }
6569
0
      if (found)
6570
0
    ret->ocur = XML_ELEMENT_CONTENT_MULT;
6571
0
  }
6572
0
  NEXT;
6573
0
    }
6574
0
    return(ret);
6575
0
}
6576
6577
/**
6578
 * Parse the declaration for a Mixed Element content
6579
 * The leading '(' and spaces have been skipped in #xmlParseElementContentDecl
6580
 *
6581
 * @deprecated Internal function, don't use.
6582
 *
6583
 *     [47] children ::= (choice | seq) ('?' | '*' | '+')?
6584
 *
6585
 *     [48] cp ::= (Name | choice | seq) ('?' | '*' | '+')?
6586
 *
6587
 *     [49] choice ::= '(' S? cp ( S? '|' S? cp )* S? ')'
6588
 *
6589
 *     [50] seq ::= '(' S? cp ( S? ',' S? cp )* S? ')'
6590
 *
6591
 * [ VC: Proper Group/PE Nesting ] applies to [49] and [50]
6592
 * TODO Parameter-entity replacement text must be properly nested
6593
 *  with parenthesized groups. That is to say, if either of the
6594
 *  opening or closing parentheses in a choice, seq, or Mixed
6595
 *  construct is contained in the replacement text for a parameter
6596
 *  entity, both must be contained in the same replacement text. For
6597
 *  interoperability, if a parameter-entity reference appears in a
6598
 *  choice, seq, or Mixed construct, its replacement text should not
6599
 *  be empty, and neither the first nor last non-blank character of
6600
 *  the replacement text should be a connector (| or ,).
6601
 *
6602
 * @param ctxt  an XML parser context
6603
 * @param inputchk  the input used for the current entity, needed for boundary checks
6604
 * @returns the tree of xmlElementContent describing the element
6605
 *          hierarchy.
6606
 */
6607
xmlElementContent *
6608
0
xmlParseElementChildrenContentDecl(xmlParserCtxt *ctxt, int inputchk) {
6609
    /* stub left for API/ABI compat */
6610
0
    return(xmlParseElementChildrenContentDeclPriv(ctxt, inputchk, 1));
6611
0
}
6612
6613
/**
6614
 * Parse the declaration for an Element content either Mixed or Children,
6615
 * the cases EMPTY and ANY are handled directly in #xmlParseElementDecl
6616
 *
6617
 * @deprecated Internal function, don't use.
6618
 *
6619
 *     [46] contentspec ::= 'EMPTY' | 'ANY' | Mixed | children
6620
 *
6621
 * @param ctxt  an XML parser context
6622
 * @param name  the name of the element being defined.
6623
 * @param result  the Element Content pointer will be stored here if any
6624
 * @returns an xmlElementTypeVal value or -1 on error
6625
 */
6626
6627
int
6628
xmlParseElementContentDecl(xmlParserCtxt *ctxt, const xmlChar *name,
6629
0
                           xmlElementContent **result) {
6630
6631
0
    xmlElementContentPtr tree = NULL;
6632
0
    int openInputNr = ctxt->inputNr;
6633
0
    int res;
6634
6635
0
    *result = NULL;
6636
6637
0
    if (RAW != '(') {
6638
0
  xmlFatalErrMsgStr(ctxt, XML_ERR_ELEMCONTENT_NOT_STARTED,
6639
0
    "xmlParseElementContentDecl : %s '(' expected\n", name);
6640
0
  return(-1);
6641
0
    }
6642
0
    NEXT;
6643
0
    xmlSkipBlankCharsPEBalanced(ctxt, openInputNr);
6644
0
    if (CMP7(CUR_PTR, '#', 'P', 'C', 'D', 'A', 'T', 'A')) {
6645
0
        tree = xmlParseElementMixedContentDecl(ctxt, openInputNr);
6646
0
  res = XML_ELEMENT_TYPE_MIXED;
6647
0
    } else {
6648
0
        tree = xmlParseElementChildrenContentDeclPriv(ctxt, openInputNr, 1);
6649
0
  res = XML_ELEMENT_TYPE_ELEMENT;
6650
0
    }
6651
0
    if (tree == NULL)
6652
0
        return(-1);
6653
0
    SKIP_BLANKS_PE;
6654
0
    *result = tree;
6655
0
    return(res);
6656
0
}
6657
6658
/**
6659
 * Parse an element declaration. Always consumes '<!'.
6660
 *
6661
 * @deprecated Internal function, don't use.
6662
 *
6663
 *     [45] elementdecl ::= '<!ELEMENT' S Name S contentspec S? '>'
6664
 *
6665
 * [ VC: Unique Element Type Declaration ]
6666
 * No element type may be declared more than once
6667
 *
6668
 * @param ctxt  an XML parser context
6669
 * @returns the type of the element, or -1 in case of error
6670
 */
6671
int
6672
0
xmlParseElementDecl(xmlParserCtxt *ctxt) {
6673
0
    const xmlChar *name;
6674
0
    int ret = -1;
6675
0
    xmlElementContentPtr content  = NULL;
6676
6677
0
    if ((CUR != '<') || (NXT(1) != '!'))
6678
0
        return(ret);
6679
0
    SKIP(2);
6680
6681
    /* GROW; done in the caller */
6682
0
    if (CMP7(CUR_PTR, 'E', 'L', 'E', 'M', 'E', 'N', 'T')) {
6683
0
#ifdef LIBXML_VALID_ENABLED
6684
0
  int oldInputNr = ctxt->inputNr;
6685
0
#endif
6686
6687
0
  SKIP(7);
6688
0
  if (SKIP_BLANKS_PE == 0) {
6689
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
6690
0
               "Space required after 'ELEMENT'\n");
6691
0
      return(-1);
6692
0
  }
6693
0
        name = xmlParseName(ctxt);
6694
0
  if (name == NULL) {
6695
0
      xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
6696
0
         "xmlParseElementDecl: no name for Element\n");
6697
0
      return(-1);
6698
0
  }
6699
0
  if (SKIP_BLANKS_PE == 0) {
6700
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
6701
0
         "Space required after the element name\n");
6702
0
  }
6703
0
  if (CMP5(CUR_PTR, 'E', 'M', 'P', 'T', 'Y')) {
6704
0
      SKIP(5);
6705
      /*
6706
       * Element must always be empty.
6707
       */
6708
0
      ret = XML_ELEMENT_TYPE_EMPTY;
6709
0
  } else if ((RAW == 'A') && (NXT(1) == 'N') &&
6710
0
             (NXT(2) == 'Y')) {
6711
0
      SKIP(3);
6712
      /*
6713
       * Element is a generic container.
6714
       */
6715
0
      ret = XML_ELEMENT_TYPE_ANY;
6716
0
  } else if (RAW == '(') {
6717
0
      ret = xmlParseElementContentDecl(ctxt, name, &content);
6718
0
            if (ret <= 0)
6719
0
                return(-1);
6720
0
  } else {
6721
      /*
6722
       * [ WFC: PEs in Internal Subset ] error handling.
6723
       */
6724
0
            xmlFatalErrMsg(ctxt, XML_ERR_ELEMCONTENT_NOT_STARTED,
6725
0
                  "xmlParseElementDecl: 'EMPTY', 'ANY' or '(' expected\n");
6726
0
      return(-1);
6727
0
  }
6728
6729
0
  SKIP_BLANKS_PE;
6730
6731
0
  if (RAW != '>') {
6732
0
      xmlFatalErr(ctxt, XML_ERR_GT_REQUIRED, NULL);
6733
0
      if (content != NULL) {
6734
0
    xmlFreeDocElementContent(ctxt->myDoc, content);
6735
0
      }
6736
0
  } else {
6737
0
#ifdef LIBXML_VALID_ENABLED
6738
0
      if ((ctxt->validate) && (ctxt->inputNr > oldInputNr)) {
6739
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6740
0
                                 "Element declaration doesn't start and stop in"
6741
0
                                 " the same entity\n",
6742
0
                                 NULL, NULL);
6743
0
      }
6744
0
#endif
6745
6746
0
      NEXT;
6747
0
      if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
6748
0
    (ctxt->sax->elementDecl != NULL)) {
6749
0
    if (content != NULL)
6750
0
        content->parent = NULL;
6751
0
          ctxt->sax->elementDecl(ctxt->userData, name, ret,
6752
0
                           content);
6753
0
    if ((content != NULL) && (content->parent == NULL)) {
6754
        /*
6755
         * this is a trick: if xmlAddElementDecl is called,
6756
         * instead of copying the full tree it is plugged directly
6757
         * if called from the parser. Avoid duplicating the
6758
         * interfaces or change the API/ABI
6759
         */
6760
0
        xmlFreeDocElementContent(ctxt->myDoc, content);
6761
0
    }
6762
0
      } else if (content != NULL) {
6763
0
    xmlFreeDocElementContent(ctxt->myDoc, content);
6764
0
      }
6765
0
  }
6766
0
    }
6767
0
    return(ret);
6768
0
}
6769
6770
/**
6771
 * Parse a conditional section. Always consumes '<!['.
6772
 *
6773
 *     [61] conditionalSect ::= includeSect | ignoreSect
6774
 *     [62] includeSect ::= '<![' S? 'INCLUDE' S? '[' extSubsetDecl ']]>'
6775
 *     [63] ignoreSect ::= '<![' S? 'IGNORE' S? '[' ignoreSectContents* ']]>'
6776
 *     [64] ignoreSectContents ::= Ignore ('<![' ignoreSectContents ']]>'
6777
 *                                 Ignore)*
6778
 *     [65] Ignore ::= Char* - (Char* ('<![' | ']]>') Char*)
6779
 * @param ctxt  an XML parser context
6780
 */
6781
6782
static void
6783
0
xmlParseConditionalSections(xmlParserCtxtPtr ctxt) {
6784
0
    size_t depth = 0;
6785
0
    int isFreshPE = 0;
6786
0
    int oldInputNr = ctxt->inputNr;
6787
0
    int declInputNr = ctxt->inputNr;
6788
6789
0
    while (!PARSER_STOPPED(ctxt)) {
6790
0
        if (ctxt->input->cur >= ctxt->input->end) {
6791
0
            if (ctxt->inputNr <= oldInputNr) {
6792
0
                xmlFatalErr(ctxt, XML_ERR_EXT_SUBSET_NOT_FINISHED, NULL);
6793
0
                return;
6794
0
            }
6795
6796
0
            xmlPopPE(ctxt);
6797
0
            declInputNr = ctxt->inputNr;
6798
0
        } else if ((RAW == '<') && (NXT(1) == '!') && (NXT(2) == '[')) {
6799
0
            SKIP(3);
6800
0
            SKIP_BLANKS_PE;
6801
6802
0
            isFreshPE = 0;
6803
6804
0
            if (CMP7(CUR_PTR, 'I', 'N', 'C', 'L', 'U', 'D', 'E')) {
6805
0
                SKIP(7);
6806
0
                SKIP_BLANKS_PE;
6807
0
                if (RAW != '[') {
6808
0
                    xmlFatalErr(ctxt, XML_ERR_CONDSEC_INVALID, NULL);
6809
0
                    return;
6810
0
                }
6811
0
#ifdef LIBXML_VALID_ENABLED
6812
0
                if ((ctxt->validate) && (ctxt->inputNr > declInputNr)) {
6813
0
        xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6814
0
                                     "All markup of the conditional section is"
6815
0
                                     " not in the same entity\n",
6816
0
                                     NULL, NULL);
6817
0
                }
6818
0
#endif
6819
0
                NEXT;
6820
6821
0
                depth++;
6822
0
            } else if (CMP6(CUR_PTR, 'I', 'G', 'N', 'O', 'R', 'E')) {
6823
0
                size_t ignoreDepth = 0;
6824
6825
0
                SKIP(6);
6826
0
                SKIP_BLANKS_PE;
6827
0
                if (RAW != '[') {
6828
0
                    xmlFatalErr(ctxt, XML_ERR_CONDSEC_INVALID, NULL);
6829
0
                    return;
6830
0
                }
6831
0
#ifdef LIBXML_VALID_ENABLED
6832
0
                if ((ctxt->validate) && (ctxt->inputNr > declInputNr)) {
6833
0
        xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6834
0
                                     "All markup of the conditional section is"
6835
0
                                     " not in the same entity\n",
6836
0
                                     NULL, NULL);
6837
0
                }
6838
0
#endif
6839
0
                NEXT;
6840
6841
0
                while (PARSER_STOPPED(ctxt) == 0) {
6842
0
                    if (RAW == 0) {
6843
0
                        xmlFatalErr(ctxt, XML_ERR_CONDSEC_NOT_FINISHED, NULL);
6844
0
                        return;
6845
0
                    }
6846
0
                    if ((RAW == '<') && (NXT(1) == '!') && (NXT(2) == '[')) {
6847
0
                        SKIP(3);
6848
0
                        ignoreDepth++;
6849
                        /* Check for integer overflow */
6850
0
                        if (ignoreDepth == 0) {
6851
0
                            xmlErrMemory(ctxt);
6852
0
                            return;
6853
0
                        }
6854
0
                    } else if ((RAW == ']') && (NXT(1) == ']') &&
6855
0
                               (NXT(2) == '>')) {
6856
0
                        SKIP(3);
6857
0
                        if (ignoreDepth == 0)
6858
0
                            break;
6859
0
                        ignoreDepth--;
6860
0
                    } else {
6861
0
                        NEXT;
6862
0
                    }
6863
0
                }
6864
6865
0
#ifdef LIBXML_VALID_ENABLED
6866
0
                if ((ctxt->validate) && (ctxt->inputNr > declInputNr)) {
6867
0
        xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6868
0
                                     "All markup of the conditional section is"
6869
0
                                     " not in the same entity\n",
6870
0
                                     NULL, NULL);
6871
0
                }
6872
0
#endif
6873
0
            } else {
6874
0
                xmlFatalErr(ctxt, XML_ERR_CONDSEC_INVALID_KEYWORD, NULL);
6875
0
                return;
6876
0
            }
6877
0
        } else if ((depth > 0) &&
6878
0
                   (RAW == ']') && (NXT(1) == ']') && (NXT(2) == '>')) {
6879
0
            if (isFreshPE) {
6880
0
                xmlFatalErrMsg(ctxt, XML_ERR_CONDSEC_INVALID,
6881
0
                               "Parameter entity must match "
6882
0
                               "extSubsetDecl\n");
6883
0
                return;
6884
0
            }
6885
6886
0
            depth--;
6887
0
#ifdef LIBXML_VALID_ENABLED
6888
0
            if ((ctxt->validate) && (ctxt->inputNr > declInputNr)) {
6889
0
    xmlValidityError(ctxt, XML_ERR_ENTITY_BOUNDARY,
6890
0
                                 "All markup of the conditional section is not"
6891
0
                                 " in the same entity\n",
6892
0
                                 NULL, NULL);
6893
0
            }
6894
0
#endif
6895
0
            SKIP(3);
6896
0
        } else if ((RAW == '<') && ((NXT(1) == '!') || (NXT(1) == '?'))) {
6897
0
            isFreshPE = 0;
6898
0
            xmlParseMarkupDecl(ctxt);
6899
0
        } else if (RAW == '%') {
6900
0
            xmlParsePERefInternal(ctxt, 1);
6901
0
            if (ctxt->inputNr > declInputNr) {
6902
0
                isFreshPE = 1;
6903
0
                declInputNr = ctxt->inputNr;
6904
0
            }
6905
0
        } else {
6906
0
            xmlFatalErr(ctxt, XML_ERR_EXT_SUBSET_NOT_FINISHED, NULL);
6907
0
            return;
6908
0
        }
6909
6910
0
        if (depth == 0)
6911
0
            break;
6912
6913
0
        SKIP_BLANKS;
6914
0
        SHRINK;
6915
0
        GROW;
6916
0
    }
6917
0
}
6918
6919
/**
6920
 * Parse markup declarations. Always consumes '<!' or '<?'.
6921
 *
6922
 * @deprecated Internal function, don't use.
6923
 *
6924
 *     [29] markupdecl ::= elementdecl | AttlistDecl | EntityDecl |
6925
 *                         NotationDecl | PI | Comment
6926
 *
6927
 * [ VC: Proper Declaration/PE Nesting ]
6928
 * Parameter-entity replacement text must be properly nested with
6929
 * markup declarations. That is to say, if either the first character
6930
 * or the last character of a markup declaration (markupdecl above) is
6931
 * contained in the replacement text for a parameter-entity reference,
6932
 * both must be contained in the same replacement text.
6933
 *
6934
 * [ WFC: PEs in Internal Subset ]
6935
 * In the internal DTD subset, parameter-entity references can occur
6936
 * only where markup declarations can occur, not within markup declarations.
6937
 * (This does not apply to references that occur in external parameter
6938
 * entities or to the external subset.)
6939
 *
6940
 * @param ctxt  an XML parser context
6941
 */
6942
void
6943
0
xmlParseMarkupDecl(xmlParserCtxt *ctxt) {
6944
0
    GROW;
6945
0
    if (CUR == '<') {
6946
0
        if (NXT(1) == '!') {
6947
0
      switch (NXT(2)) {
6948
0
          case 'E':
6949
0
        if (NXT(3) == 'L')
6950
0
      xmlParseElementDecl(ctxt);
6951
0
        else if (NXT(3) == 'N')
6952
0
      xmlParseEntityDecl(ctxt);
6953
0
                    else
6954
0
                        SKIP(2);
6955
0
        break;
6956
0
          case 'A':
6957
0
        xmlParseAttributeListDecl(ctxt);
6958
0
        break;
6959
0
          case 'N':
6960
0
        xmlParseNotationDecl(ctxt);
6961
0
        break;
6962
0
          case '-':
6963
0
        xmlParseComment(ctxt);
6964
0
        break;
6965
0
    default:
6966
0
                    xmlFatalErr(ctxt,
6967
0
                                ctxt->inSubset == 2 ?
6968
0
                                    XML_ERR_EXT_SUBSET_NOT_FINISHED :
6969
0
                                    XML_ERR_INT_SUBSET_NOT_FINISHED,
6970
0
                                NULL);
6971
0
                    SKIP(2);
6972
0
        break;
6973
0
      }
6974
0
  } else if (NXT(1) == '?') {
6975
0
      xmlParsePI(ctxt);
6976
0
  }
6977
0
    }
6978
0
}
6979
6980
/**
6981
 * Parse an XML declaration header for external entities
6982
 *
6983
 * @deprecated Internal function, don't use.
6984
 *
6985
 *     [77] TextDecl ::= '<?xml' VersionInfo? EncodingDecl S? '?>'
6986
 * @param ctxt  an XML parser context
6987
 */
6988
6989
void
6990
0
xmlParseTextDecl(xmlParserCtxt *ctxt) {
6991
0
    xmlChar *version;
6992
6993
    /*
6994
     * We know that '<?xml' is here.
6995
     */
6996
0
    if ((CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) && (IS_BLANK_CH(NXT(5)))) {
6997
0
  SKIP(5);
6998
0
    } else {
6999
0
  xmlFatalErr(ctxt, XML_ERR_XMLDECL_NOT_STARTED, NULL);
7000
0
  return;
7001
0
    }
7002
7003
0
    if (SKIP_BLANKS == 0) {
7004
0
  xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
7005
0
           "Space needed after '<?xml'\n");
7006
0
    }
7007
7008
    /*
7009
     * We may have the VersionInfo here.
7010
     */
7011
0
    version = xmlParseVersionInfo(ctxt);
7012
0
    if (version == NULL) {
7013
0
  version = xmlCharStrdup(XML_DEFAULT_VERSION);
7014
0
        if (version == NULL) {
7015
0
            xmlErrMemory(ctxt);
7016
0
            return;
7017
0
        }
7018
0
    } else {
7019
0
  if (SKIP_BLANKS == 0) {
7020
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
7021
0
               "Space needed here\n");
7022
0
  }
7023
0
    }
7024
0
    ctxt->input->version = version;
7025
7026
    /*
7027
     * We must have the encoding declaration
7028
     */
7029
0
    xmlParseEncodingDecl(ctxt);
7030
7031
0
    SKIP_BLANKS;
7032
0
    if ((RAW == '?') && (NXT(1) == '>')) {
7033
0
        SKIP(2);
7034
0
    } else if (RAW == '>') {
7035
        /* Deprecated old WD ... */
7036
0
  xmlFatalErr(ctxt, XML_ERR_XMLDECL_NOT_FINISHED, NULL);
7037
0
  NEXT;
7038
0
    } else {
7039
0
        int c;
7040
7041
0
  xmlFatalErr(ctxt, XML_ERR_XMLDECL_NOT_FINISHED, NULL);
7042
0
        while ((PARSER_STOPPED(ctxt) == 0) && ((c = CUR) != 0)) {
7043
0
            NEXT;
7044
0
            if (c == '>')
7045
0
                break;
7046
0
        }
7047
0
    }
7048
0
}
7049
7050
/**
7051
 * Parse Markup declarations from an external subset
7052
 *
7053
 * @deprecated Internal function, don't use.
7054
 *
7055
 *     [30] extSubset ::= textDecl? extSubsetDecl
7056
 *
7057
 *     [31] extSubsetDecl ::= (markupdecl | conditionalSect |
7058
 *                             PEReference | S) *
7059
 * @param ctxt  an XML parser context
7060
 * @param publicId  the public identifier
7061
 * @param systemId  the system identifier (URL)
7062
 */
7063
void
7064
xmlParseExternalSubset(xmlParserCtxt *ctxt, const xmlChar *publicId,
7065
0
                       const xmlChar *systemId) {
7066
0
    int oldInputNr;
7067
7068
0
    xmlCtxtInitializeLate(ctxt);
7069
7070
0
    xmlDetectEncoding(ctxt);
7071
7072
0
    if (CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) {
7073
0
  xmlParseTextDecl(ctxt);
7074
0
    }
7075
0
    if (ctxt->myDoc == NULL) {
7076
0
        ctxt->myDoc = xmlNewDoc(BAD_CAST "1.0");
7077
0
  if (ctxt->myDoc == NULL) {
7078
0
      xmlErrMemory(ctxt);
7079
0
      return;
7080
0
  }
7081
0
  ctxt->myDoc->properties = XML_DOC_INTERNAL;
7082
0
    }
7083
0
    if ((ctxt->myDoc->intSubset == NULL) &&
7084
0
        (xmlCreateIntSubset(ctxt->myDoc, NULL, publicId, systemId) == NULL)) {
7085
0
        xmlErrMemory(ctxt);
7086
0
    }
7087
7088
0
    ctxt->inSubset = 2;
7089
0
    oldInputNr = ctxt->inputNr;
7090
7091
0
    SKIP_BLANKS;
7092
0
    while (!PARSER_STOPPED(ctxt)) {
7093
0
        if (ctxt->input->cur >= ctxt->input->end) {
7094
0
            if (ctxt->inputNr <= oldInputNr) {
7095
0
                xmlParserCheckEOF(ctxt, XML_ERR_EXT_SUBSET_NOT_FINISHED);
7096
0
                break;
7097
0
            }
7098
7099
0
            xmlPopPE(ctxt);
7100
0
        } else if ((RAW == '<') && (NXT(1) == '!') && (NXT(2) == '[')) {
7101
0
            xmlParseConditionalSections(ctxt);
7102
0
        } else if ((RAW == '<') && ((NXT(1) == '!') || (NXT(1) == '?'))) {
7103
0
            xmlParseMarkupDecl(ctxt);
7104
0
        } else if (RAW == '%') {
7105
0
            xmlParsePERefInternal(ctxt, 1);
7106
0
        } else {
7107
0
            xmlFatalErr(ctxt, XML_ERR_EXT_SUBSET_NOT_FINISHED, NULL);
7108
7109
0
            while (ctxt->inputNr > oldInputNr)
7110
0
                xmlPopPE(ctxt);
7111
0
            break;
7112
0
        }
7113
0
        SKIP_BLANKS;
7114
0
        SHRINK;
7115
0
        GROW;
7116
0
    }
7117
0
}
7118
7119
/**
7120
 * Parse and handle entity references in content, depending on the SAX
7121
 * interface, this may end-up in a call to character() if this is a
7122
 * CharRef, a predefined entity, if there is no reference() callback.
7123
 * or if the parser was asked to switch to that mode.
7124
 *
7125
 * @deprecated Internal function, don't use.
7126
 *
7127
 * Always consumes '&'.
7128
 *
7129
 *     [67] Reference ::= EntityRef | CharRef
7130
 * @param ctxt  an XML parser context
7131
 */
7132
void
7133
0
xmlParseReference(xmlParserCtxt *ctxt) {
7134
0
    xmlEntityPtr ent = NULL;
7135
0
    const xmlChar *name;
7136
0
    xmlChar *val;
7137
7138
0
    if (RAW != '&')
7139
0
        return;
7140
7141
    /*
7142
     * Simple case of a CharRef
7143
     */
7144
0
    if (NXT(1) == '#') {
7145
0
  int i = 0;
7146
0
  xmlChar out[16];
7147
0
  int value = xmlParseCharRef(ctxt);
7148
7149
0
  if (value == 0)
7150
0
      return;
7151
7152
        /*
7153
         * Just encode the value in UTF-8
7154
         */
7155
0
        COPY_BUF(out, i, value);
7156
0
        out[i] = 0;
7157
0
        if ((ctxt->sax != NULL) && (ctxt->sax->characters != NULL) &&
7158
0
            (!ctxt->disableSAX))
7159
0
            ctxt->sax->characters(ctxt->userData, out, i);
7160
0
  return;
7161
0
    }
7162
7163
    /*
7164
     * We are seeing an entity reference
7165
     */
7166
0
    name = xmlParseEntityRefInternal(ctxt);
7167
0
    if (name == NULL)
7168
0
        return;
7169
0
    ent = xmlLookupGeneralEntity(ctxt, name, /* isAttr */ 0);
7170
0
    if (ent == NULL) {
7171
        /*
7172
         * Create a reference for undeclared entities.
7173
         */
7174
0
        if ((ctxt->replaceEntities == 0) &&
7175
0
            (ctxt->sax != NULL) &&
7176
0
            (ctxt->disableSAX == 0) &&
7177
0
            (ctxt->sax->reference != NULL)) {
7178
0
            ctxt->sax->reference(ctxt->userData, name);
7179
0
        }
7180
0
        return;
7181
0
    }
7182
0
    if (!ctxt->wellFormed)
7183
0
  return;
7184
7185
    /* special case of predefined entities */
7186
0
    if ((ent->name == NULL) ||
7187
0
        (ent->etype == XML_INTERNAL_PREDEFINED_ENTITY)) {
7188
0
  val = ent->content;
7189
0
  if (val == NULL) return;
7190
  /*
7191
   * inline the entity.
7192
   */
7193
0
  if ((ctxt->sax != NULL) && (ctxt->sax->characters != NULL) &&
7194
0
      (!ctxt->disableSAX))
7195
0
      ctxt->sax->characters(ctxt->userData, val, xmlStrlen(val));
7196
0
  return;
7197
0
    }
7198
7199
    /*
7200
     * Some users try to parse entities on their own and used to set
7201
     * the renamed "checked" member. Fix the flags to cover this
7202
     * case.
7203
     */
7204
0
    if (((ent->flags & XML_ENT_PARSED) == 0) && (ent->children != NULL))
7205
0
        ent->flags |= XML_ENT_PARSED;
7206
7207
    /*
7208
     * The first reference to the entity trigger a parsing phase
7209
     * where the ent->children is filled with the result from
7210
     * the parsing.
7211
     * Note: external parsed entities will not be loaded, it is not
7212
     * required for a non-validating parser, unless the parsing option
7213
     * of validating, or substituting entities were given. Doing so is
7214
     * far more secure as the parser will only process data coming from
7215
     * the document entity by default.
7216
     *
7217
     * FIXME: This doesn't work correctly since entities can be
7218
     * expanded with different namespace declarations in scope.
7219
     * For example:
7220
     *
7221
     * <!DOCTYPE doc [
7222
     *   <!ENTITY ent "<ns:elem/>">
7223
     * ]>
7224
     * <doc>
7225
     *   <decl1 xmlns:ns="urn:ns1">
7226
     *     &ent;
7227
     *   </decl1>
7228
     *   <decl2 xmlns:ns="urn:ns2">
7229
     *     &ent;
7230
     *   </decl2>
7231
     * </doc>
7232
     *
7233
     * Proposed fix:
7234
     *
7235
     * - Ignore current namespace declarations when parsing the
7236
     *   entity. If a prefix can't be resolved, don't report an error
7237
     *   but mark it as unresolved.
7238
     * - Try to resolve these prefixes when expanding the entity.
7239
     *   This will require a specialized version of xmlStaticCopyNode
7240
     *   which can also make use of the namespace hash table to avoid
7241
     *   quadratic behavior.
7242
     *
7243
     * Alternatively, we could simply reparse the entity on each
7244
     * expansion like we already do with custom SAX callbacks.
7245
     * External entity content should be cached in this case.
7246
     */
7247
0
    if ((ent->etype == XML_INTERNAL_GENERAL_ENTITY) ||
7248
0
        (((ctxt->options & XML_PARSE_NO_XXE) == 0) &&
7249
0
         ((ctxt->replaceEntities) ||
7250
0
          (ctxt->validate)))) {
7251
0
        if ((ent->flags & XML_ENT_PARSED) == 0) {
7252
0
            xmlCtxtParseEntity(ctxt, ent);
7253
0
        } else if (ent->children == NULL) {
7254
            /*
7255
             * Probably running in SAX mode and the callbacks don't
7256
             * build the entity content. Parse the entity again.
7257
             *
7258
             * This will also be triggered in normal tree builder mode
7259
             * if an entity happens to be empty, causing unnecessary
7260
             * reloads. It's hard to come up with a reliable check in
7261
             * which mode we're running.
7262
             */
7263
0
            xmlCtxtParseEntity(ctxt, ent);
7264
0
        }
7265
0
    }
7266
7267
    /*
7268
     * We also check for amplification if entities aren't substituted.
7269
     * They might be expanded later.
7270
     */
7271
0
    if (xmlParserEntityCheck(ctxt, ent->expandedSize))
7272
0
        return;
7273
7274
0
    if ((ctxt->sax == NULL) || (ctxt->disableSAX))
7275
0
        return;
7276
7277
0
    if (ctxt->replaceEntities == 0) {
7278
  /*
7279
   * Create a reference
7280
   */
7281
0
        if (ctxt->sax->reference != NULL)
7282
0
      ctxt->sax->reference(ctxt->userData, ent->name);
7283
0
    } else if ((ent->children != NULL) && (ctxt->node != NULL)) {
7284
0
        xmlNodePtr copy, cur;
7285
7286
        /*
7287
         * Seems we are generating the DOM content, copy the tree
7288
   */
7289
0
        cur = ent->children;
7290
7291
        /*
7292
         * Handle first text node with SAX to coalesce text efficiently
7293
         */
7294
0
        if ((cur->type == XML_TEXT_NODE) ||
7295
0
            (cur->type == XML_CDATA_SECTION_NODE)) {
7296
0
            int len = xmlStrlen(cur->content);
7297
7298
0
            if ((cur->type == XML_TEXT_NODE) ||
7299
0
                (ctxt->options & XML_PARSE_NOCDATA)) {
7300
0
                if (ctxt->sax->characters != NULL)
7301
0
                    ctxt->sax->characters(ctxt->userData, cur->content, len);
7302
0
            } else {
7303
0
                if (ctxt->sax->cdataBlock != NULL)
7304
0
                    ctxt->sax->cdataBlock(ctxt->userData, cur->content, len);
7305
0
            }
7306
7307
0
            cur = cur->next;
7308
0
        }
7309
7310
0
        while (cur != NULL) {
7311
0
            xmlNodePtr last;
7312
7313
            /*
7314
             * Handle last text node with SAX to coalesce text efficiently
7315
             */
7316
0
            if ((cur->next == NULL) &&
7317
0
                ((cur->type == XML_TEXT_NODE) ||
7318
0
                 (cur->type == XML_CDATA_SECTION_NODE))) {
7319
0
                int len = xmlStrlen(cur->content);
7320
7321
0
                if ((cur->type == XML_TEXT_NODE) ||
7322
0
                    (ctxt->options & XML_PARSE_NOCDATA)) {
7323
0
                    if (ctxt->sax->characters != NULL)
7324
0
                        ctxt->sax->characters(ctxt->userData, cur->content,
7325
0
                                              len);
7326
0
                } else {
7327
0
                    if (ctxt->sax->cdataBlock != NULL)
7328
0
                        ctxt->sax->cdataBlock(ctxt->userData, cur->content,
7329
0
                                              len);
7330
0
                }
7331
7332
0
                break;
7333
0
            }
7334
7335
            /*
7336
             * Reset coalesce buffer stats only for non-text nodes.
7337
             */
7338
0
            ctxt->nodemem = 0;
7339
0
            ctxt->nodelen = 0;
7340
7341
0
            copy = xmlDocCopyNode(cur, ctxt->myDoc, 1);
7342
7343
0
            if (copy == NULL) {
7344
0
                xmlErrMemory(ctxt);
7345
0
                break;
7346
0
            }
7347
7348
0
            if (ctxt->parseMode == XML_PARSE_READER) {
7349
                /* Needed for reader */
7350
0
                copy->extra = cur->extra;
7351
                /* Maybe needed for reader */
7352
0
                copy->_private = cur->_private;
7353
0
            }
7354
7355
0
            copy->parent = ctxt->node;
7356
0
            last = ctxt->node->last;
7357
0
            if (last == NULL) {
7358
0
                ctxt->node->children = copy;
7359
0
            } else {
7360
0
                last->next = copy;
7361
0
                copy->prev = last;
7362
0
            }
7363
0
            ctxt->node->last = copy;
7364
7365
0
            cur = cur->next;
7366
0
        }
7367
0
    }
7368
0
}
7369
7370
static void
7371
0
xmlHandleUndeclaredEntity(xmlParserCtxtPtr ctxt, const xmlChar *name) {
7372
    /*
7373
     * [ WFC: Entity Declared ]
7374
     * In a document without any DTD, a document with only an
7375
     * internal DTD subset which contains no parameter entity
7376
     * references, or a document with "standalone='yes'", the
7377
     * Name given in the entity reference must match that in an
7378
     * entity declaration, except that well-formed documents
7379
     * need not declare any of the following entities: amp, lt,
7380
     * gt, apos, quot.
7381
     * The declaration of a parameter entity must precede any
7382
     * reference to it.
7383
     * Similarly, the declaration of a general entity must
7384
     * precede any reference to it which appears in a default
7385
     * value in an attribute-list declaration. Note that if
7386
     * entities are declared in the external subset or in
7387
     * external parameter entities, a non-validating processor
7388
     * is not obligated to read and process their declarations;
7389
     * for such documents, the rule that an entity must be
7390
     * declared is a well-formedness constraint only if
7391
     * standalone='yes'.
7392
     */
7393
0
    if ((ctxt->standalone == 1) ||
7394
0
        ((ctxt->hasExternalSubset == 0) &&
7395
0
         (ctxt->hasPErefs == 0))) {
7396
0
        xmlFatalErrMsgStr(ctxt, XML_ERR_UNDECLARED_ENTITY,
7397
0
                          "Entity '%s' not defined\n", name);
7398
0
#ifdef LIBXML_VALID_ENABLED
7399
0
    } else if (ctxt->validate) {
7400
        /*
7401
         * [ VC: Entity Declared ]
7402
         * In a document with an external subset or external
7403
         * parameter entities with "standalone='no'", ...
7404
         * ... The declaration of a parameter entity must
7405
         * precede any reference to it...
7406
         */
7407
0
        xmlValidityError(ctxt, XML_ERR_UNDECLARED_ENTITY,
7408
0
                         "Entity '%s' not defined\n", name, NULL);
7409
0
#endif
7410
0
    } else if ((ctxt->loadsubset & ~XML_SKIP_IDS) ||
7411
0
               ((ctxt->replaceEntities) &&
7412
0
                ((ctxt->options & XML_PARSE_NO_XXE) == 0))) {
7413
        /*
7414
         * Also raise a non-fatal error
7415
         *
7416
         * - if the external subset is loaded and all entity declarations
7417
         *   should be available, or
7418
         * - entity substition was requested without restricting
7419
         *   external entity access.
7420
         */
7421
0
        xmlErrMsgStr(ctxt, XML_WAR_UNDECLARED_ENTITY,
7422
0
                     "Entity '%s' not defined\n", name);
7423
0
    } else {
7424
0
        xmlWarningMsg(ctxt, XML_WAR_UNDECLARED_ENTITY,
7425
0
                      "Entity '%s' not defined\n", name, NULL);
7426
0
    }
7427
7428
0
    ctxt->valid = 0;
7429
0
}
7430
7431
static xmlEntityPtr
7432
0
xmlLookupGeneralEntity(xmlParserCtxtPtr ctxt, const xmlChar *name, int inAttr) {
7433
0
    xmlEntityPtr ent = NULL;
7434
7435
    /*
7436
     * Predefined entities override any extra definition
7437
     */
7438
0
    if ((ctxt->options & XML_PARSE_OLDSAX) == 0) {
7439
0
        ent = xmlGetPredefinedEntity(name);
7440
0
        if (ent != NULL)
7441
0
            return(ent);
7442
0
    }
7443
7444
    /*
7445
     * Ask first SAX for entity resolution, otherwise try the
7446
     * entities which may have stored in the parser context.
7447
     */
7448
0
    if (ctxt->sax != NULL) {
7449
0
  if (ctxt->sax->getEntity != NULL)
7450
0
      ent = ctxt->sax->getEntity(ctxt->userData, name);
7451
0
  if ((ctxt->wellFormed == 1 ) && (ent == NULL) &&
7452
0
      (ctxt->options & XML_PARSE_OLDSAX))
7453
0
      ent = xmlGetPredefinedEntity(name);
7454
0
  if ((ctxt->wellFormed == 1 ) && (ent == NULL) &&
7455
0
      (ctxt->userData==ctxt)) {
7456
0
      ent = xmlSAX2GetEntity(ctxt, name);
7457
0
  }
7458
0
    }
7459
7460
0
    if (ent == NULL) {
7461
0
        xmlHandleUndeclaredEntity(ctxt, name);
7462
0
    }
7463
7464
    /*
7465
     * [ WFC: Parsed Entity ]
7466
     * An entity reference must not contain the name of an
7467
     * unparsed entity
7468
     */
7469
0
    else if (ent->etype == XML_EXTERNAL_GENERAL_UNPARSED_ENTITY) {
7470
0
  xmlFatalErrMsgStr(ctxt, XML_ERR_UNPARSED_ENTITY,
7471
0
     "Entity reference to unparsed entity %s\n", name);
7472
0
        ent = NULL;
7473
0
    }
7474
7475
    /*
7476
     * [ WFC: No External Entity References ]
7477
     * Attribute values cannot contain direct or indirect
7478
     * entity references to external entities.
7479
     */
7480
0
    else if (ent->etype == XML_EXTERNAL_GENERAL_PARSED_ENTITY) {
7481
0
        if (inAttr) {
7482
0
            xmlFatalErrMsgStr(ctxt, XML_ERR_ENTITY_IS_EXTERNAL,
7483
0
                 "Attribute references external entity '%s'\n", name);
7484
0
            ent = NULL;
7485
0
        }
7486
0
    }
7487
7488
0
    return(ent);
7489
0
}
7490
7491
/**
7492
 * Parse an entity reference. Always consumes '&'.
7493
 *
7494
 *     [68] EntityRef ::= '&' Name ';'
7495
 *
7496
 * @param ctxt  an XML parser context
7497
 * @returns the name, or NULL in case of error.
7498
 */
7499
static const xmlChar *
7500
0
xmlParseEntityRefInternal(xmlParserCtxtPtr ctxt) {
7501
0
    const xmlChar *name;
7502
7503
0
    GROW;
7504
7505
0
    if (RAW != '&')
7506
0
        return(NULL);
7507
0
    NEXT;
7508
0
    name = xmlParseName(ctxt);
7509
0
    if (name == NULL) {
7510
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
7511
0
           "xmlParseEntityRef: no name\n");
7512
0
        return(NULL);
7513
0
    }
7514
0
    if (RAW != ';') {
7515
0
  xmlFatalErr(ctxt, XML_ERR_ENTITYREF_SEMICOL_MISSING, NULL);
7516
0
  return(NULL);
7517
0
    }
7518
0
    NEXT;
7519
7520
0
    return(name);
7521
0
}
7522
7523
/**
7524
 * @deprecated Internal function, don't use.
7525
 *
7526
 * @param ctxt  an XML parser context
7527
 * @returns the xmlEntity if found, or NULL otherwise.
7528
 */
7529
xmlEntity *
7530
0
xmlParseEntityRef(xmlParserCtxt *ctxt) {
7531
0
    const xmlChar *name;
7532
7533
0
    if (ctxt == NULL)
7534
0
        return(NULL);
7535
7536
0
    name = xmlParseEntityRefInternal(ctxt);
7537
0
    if (name == NULL)
7538
0
        return(NULL);
7539
7540
0
    return(xmlLookupGeneralEntity(ctxt, name, /* inAttr */ 0));
7541
0
}
7542
7543
/**
7544
 * Parse ENTITY references declarations, but this version parses it from
7545
 * a string value.
7546
 *
7547
 *     [68] EntityRef ::= '&' Name ';'
7548
 *
7549
 * [ WFC: Entity Declared ]
7550
 * In a document without any DTD, a document with only an internal DTD
7551
 * subset which contains no parameter entity references, or a document
7552
 * with "standalone='yes'", the Name given in the entity reference
7553
 * must match that in an entity declaration, except that well-formed
7554
 * documents need not declare any of the following entities: amp, lt,
7555
 * gt, apos, quot.  The declaration of a parameter entity must precede
7556
 * any reference to it.  Similarly, the declaration of a general entity
7557
 * must precede any reference to it which appears in a default value in an
7558
 * attribute-list declaration. Note that if entities are declared in the
7559
 * external subset or in external parameter entities, a non-validating
7560
 * processor is not obligated to read and process their declarations;
7561
 * for such documents, the rule that an entity must be declared is a
7562
 * well-formedness constraint only if standalone='yes'.
7563
 *
7564
 * [ WFC: Parsed Entity ]
7565
 * An entity reference must not contain the name of an unparsed entity
7566
 *
7567
 * @param ctxt  an XML parser context
7568
 * @param str  a pointer to an index in the string
7569
 * @returns the xmlEntity if found, or NULL otherwise. The str pointer
7570
 * is updated to the current location in the string.
7571
 */
7572
static xmlChar *
7573
0
xmlParseStringEntityRef(xmlParserCtxtPtr ctxt, const xmlChar ** str) {
7574
0
    xmlChar *name;
7575
0
    const xmlChar *ptr;
7576
0
    xmlChar cur;
7577
7578
0
    if ((str == NULL) || (*str == NULL))
7579
0
        return(NULL);
7580
0
    ptr = *str;
7581
0
    cur = *ptr;
7582
0
    if (cur != '&')
7583
0
  return(NULL);
7584
7585
0
    ptr++;
7586
0
    name = xmlParseStringName(ctxt, &ptr);
7587
0
    if (name == NULL) {
7588
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
7589
0
           "xmlParseStringEntityRef: no name\n");
7590
0
  *str = ptr;
7591
0
  return(NULL);
7592
0
    }
7593
0
    if (*ptr != ';') {
7594
0
  xmlFatalErr(ctxt, XML_ERR_ENTITYREF_SEMICOL_MISSING, NULL);
7595
0
        xmlFree(name);
7596
0
  *str = ptr;
7597
0
  return(NULL);
7598
0
    }
7599
0
    ptr++;
7600
7601
0
    *str = ptr;
7602
0
    return(name);
7603
0
}
7604
7605
/**
7606
 * Parse a parameter entity reference. Always consumes '%'.
7607
 *
7608
 * The entity content is handled directly by pushing it's content as
7609
 * a new input stream.
7610
 *
7611
 *     [69] PEReference ::= '%' Name ';'
7612
 *
7613
 * [ WFC: No Recursion ]
7614
 * A parsed entity must not contain a recursive
7615
 * reference to itself, either directly or indirectly.
7616
 *
7617
 * [ WFC: Entity Declared ]
7618
 * In a document without any DTD, a document with only an internal DTD
7619
 * subset which contains no parameter entity references, or a document
7620
 * with "standalone='yes'", ...  ... The declaration of a parameter
7621
 * entity must precede any reference to it...
7622
 *
7623
 * [ VC: Entity Declared ]
7624
 * In a document with an external subset or external parameter entities
7625
 * with "standalone='no'", ...  ... The declaration of a parameter entity
7626
 * must precede any reference to it...
7627
 *
7628
 * [ WFC: In DTD ]
7629
 * Parameter-entity references may only appear in the DTD.
7630
 * NOTE: misleading but this is handled.
7631
 *
7632
 * @param ctxt  an XML parser context
7633
 * @param markupDecl  whether the PERef starts a markup declaration
7634
 */
7635
static void
7636
0
xmlParsePERefInternal(xmlParserCtxt *ctxt, int markupDecl) {
7637
0
    const xmlChar *name;
7638
0
    xmlEntityPtr entity = NULL;
7639
0
    xmlParserInputPtr input;
7640
7641
0
    if (RAW != '%')
7642
0
        return;
7643
0
    NEXT;
7644
0
    name = xmlParseName(ctxt);
7645
0
    if (name == NULL) {
7646
0
  xmlFatalErrMsg(ctxt, XML_ERR_PEREF_NO_NAME, "PEReference: no name\n");
7647
0
  return;
7648
0
    }
7649
0
    if (RAW != ';') {
7650
0
  xmlFatalErr(ctxt, XML_ERR_PEREF_SEMICOL_MISSING, NULL);
7651
0
        return;
7652
0
    }
7653
7654
0
    NEXT;
7655
7656
    /* Must be set before xmlHandleUndeclaredEntity */
7657
0
    ctxt->hasPErefs = 1;
7658
7659
    /*
7660
     * Request the entity from SAX
7661
     */
7662
0
    if ((ctxt->sax != NULL) &&
7663
0
  (ctxt->sax->getParameterEntity != NULL))
7664
0
  entity = ctxt->sax->getParameterEntity(ctxt->userData, name);
7665
7666
0
    if (entity == NULL) {
7667
0
        xmlHandleUndeclaredEntity(ctxt, name);
7668
0
    } else {
7669
  /*
7670
   * Internal checking in case the entity quest barfed
7671
   */
7672
0
  if ((entity->etype != XML_INTERNAL_PARAMETER_ENTITY) &&
7673
0
      (entity->etype != XML_EXTERNAL_PARAMETER_ENTITY)) {
7674
0
      xmlWarningMsg(ctxt, XML_WAR_UNDECLARED_ENTITY,
7675
0
      "Internal: %%%s; is not a parameter entity\n",
7676
0
        name, NULL);
7677
0
  } else {
7678
0
      if ((entity->etype == XML_EXTERNAL_PARAMETER_ENTITY) &&
7679
0
                ((ctxt->options & XML_PARSE_NO_XXE) ||
7680
0
     (((ctxt->loadsubset & ~XML_SKIP_IDS) == 0) &&
7681
0
      (ctxt->replaceEntities == 0) &&
7682
0
      (ctxt->validate == 0))))
7683
0
    return;
7684
7685
0
            if (entity->flags & XML_ENT_EXPANDING) {
7686
0
                xmlFatalErr(ctxt, XML_ERR_ENTITY_LOOP, NULL);
7687
0
                return;
7688
0
            }
7689
7690
0
      input = xmlNewEntityInputStream(ctxt, entity);
7691
0
      if (xmlCtxtPushInput(ctxt, input) < 0) {
7692
0
                xmlFreeInputStream(input);
7693
0
    return;
7694
0
            }
7695
7696
0
            entity->flags |= XML_ENT_EXPANDING;
7697
7698
0
            if (markupDecl)
7699
0
                input->flags |= XML_INPUT_MARKUP_DECL;
7700
7701
0
            GROW;
7702
7703
0
      if (entity->etype == XML_EXTERNAL_PARAMETER_ENTITY) {
7704
0
                xmlDetectEncoding(ctxt);
7705
7706
0
                if ((CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) &&
7707
0
                    (IS_BLANK_CH(NXT(5)))) {
7708
0
                    xmlParseTextDecl(ctxt);
7709
0
                }
7710
0
            }
7711
0
  }
7712
0
    }
7713
0
}
7714
7715
/**
7716
 * Parse a parameter entity reference.
7717
 *
7718
 * @deprecated Internal function, don't use.
7719
 *
7720
 * @param ctxt  an XML parser context
7721
 */
7722
void
7723
0
xmlParsePEReference(xmlParserCtxt *ctxt) {
7724
0
    xmlParsePERefInternal(ctxt, 0);
7725
0
}
7726
7727
/**
7728
 * Load the content of an entity.
7729
 *
7730
 * @param ctxt  an XML parser context
7731
 * @param entity  an unloaded system entity
7732
 * @returns 0 in case of success and -1 in case of failure
7733
 */
7734
static int
7735
0
xmlLoadEntityContent(xmlParserCtxtPtr ctxt, xmlEntityPtr entity) {
7736
0
    xmlParserInputPtr oldinput, input = NULL;
7737
0
    xmlParserInputPtr *oldinputTab;
7738
0
    xmlChar *oldencoding;
7739
0
    xmlChar *content = NULL;
7740
0
    xmlResourceType rtype;
7741
0
    size_t length, i;
7742
0
    int oldinputNr, oldinputMax;
7743
0
    int ret = -1;
7744
0
    int res;
7745
7746
0
    if ((ctxt == NULL) || (entity == NULL) ||
7747
0
        ((entity->etype != XML_EXTERNAL_PARAMETER_ENTITY) &&
7748
0
   (entity->etype != XML_EXTERNAL_GENERAL_PARSED_ENTITY)) ||
7749
0
  (entity->content != NULL)) {
7750
0
  xmlFatalErr(ctxt, XML_ERR_ARGUMENT,
7751
0
              "xmlLoadEntityContent parameter error");
7752
0
        return(-1);
7753
0
    }
7754
7755
0
    if (entity->etype == XML_EXTERNAL_PARAMETER_ENTITY)
7756
0
        rtype = XML_RESOURCE_PARAMETER_ENTITY;
7757
0
    else
7758
0
        rtype = XML_RESOURCE_GENERAL_ENTITY;
7759
7760
0
    input = xmlLoadResource(ctxt, (char *) entity->URI,
7761
0
                            (char *) entity->ExternalID, rtype);
7762
0
    if (input == NULL)
7763
0
        return(-1);
7764
7765
0
    oldinput = ctxt->input;
7766
0
    oldinputNr = ctxt->inputNr;
7767
0
    oldinputMax = ctxt->inputMax;
7768
0
    oldinputTab = ctxt->inputTab;
7769
0
    oldencoding = ctxt->encoding;
7770
7771
0
    ctxt->input = NULL;
7772
0
    ctxt->inputNr = 0;
7773
0
    ctxt->inputMax = 1;
7774
0
    ctxt->encoding = NULL;
7775
0
    ctxt->inputTab = xmlMalloc(sizeof(xmlParserInputPtr));
7776
0
    if (ctxt->inputTab == NULL) {
7777
0
        xmlErrMemory(ctxt);
7778
0
        xmlFreeInputStream(input);
7779
0
        goto error;
7780
0
    }
7781
7782
0
    xmlBufResetInput(input->buf->buffer, input);
7783
7784
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
7785
0
        xmlFreeInputStream(input);
7786
0
        goto error;
7787
0
    }
7788
7789
0
    xmlDetectEncoding(ctxt);
7790
7791
    /*
7792
     * Parse a possible text declaration first
7793
     */
7794
0
    if ((CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) && (IS_BLANK_CH(NXT(5)))) {
7795
0
  xmlParseTextDecl(ctxt);
7796
        /*
7797
         * An XML-1.0 document can't reference an entity not XML-1.0
7798
         */
7799
0
        if ((xmlStrEqual(ctxt->version, BAD_CAST "1.0")) &&
7800
0
            (!xmlStrEqual(ctxt->input->version, BAD_CAST "1.0"))) {
7801
0
            xmlFatalErrMsg(ctxt, XML_ERR_VERSION_MISMATCH,
7802
0
                           "Version mismatch between document and entity\n");
7803
0
        }
7804
0
    }
7805
7806
0
    length = input->cur - input->base;
7807
0
    xmlBufShrink(input->buf->buffer, length);
7808
0
    xmlSaturatedAdd(&ctxt->sizeentities, length);
7809
7810
0
    while ((res = xmlParserInputBufferGrow(input->buf, 4096)) > 0)
7811
0
        ;
7812
7813
0
    xmlBufResetInput(input->buf->buffer, input);
7814
7815
0
    if (res < 0) {
7816
0
        xmlCtxtErrIO(ctxt, input->buf->error, NULL);
7817
0
        goto error;
7818
0
    }
7819
7820
0
    length = xmlBufUse(input->buf->buffer);
7821
0
    if (length > INT_MAX) {
7822
0
        xmlErrMemory(ctxt);
7823
0
        goto error;
7824
0
    }
7825
7826
0
    content = xmlStrndup(xmlBufContent(input->buf->buffer), length);
7827
0
    if (content == NULL) {
7828
0
        xmlErrMemory(ctxt);
7829
0
        goto error;
7830
0
    }
7831
7832
0
    for (i = 0; i < length; ) {
7833
0
        int clen = length - i;
7834
0
        int c = xmlGetUTF8Char(content + i, &clen);
7835
7836
0
        if ((c < 0) || (!IS_CHAR(c))) {
7837
0
            xmlFatalErrMsgInt(ctxt, XML_ERR_INVALID_CHAR,
7838
0
                              "xmlLoadEntityContent: invalid char value %d\n",
7839
0
                              content[i]);
7840
0
            goto error;
7841
0
        }
7842
0
        i += clen;
7843
0
    }
7844
7845
0
    xmlSaturatedAdd(&ctxt->sizeentities, length);
7846
0
    entity->content = content;
7847
0
    entity->length = length;
7848
0
    content = NULL;
7849
0
    ret = 0;
7850
7851
0
error:
7852
0
    while (ctxt->inputNr > 0)
7853
0
        xmlFreeInputStream(xmlCtxtPopInput(ctxt));
7854
0
    xmlFree(ctxt->inputTab);
7855
0
    xmlFree(ctxt->encoding);
7856
7857
0
    ctxt->input = oldinput;
7858
0
    ctxt->inputNr = oldinputNr;
7859
0
    ctxt->inputMax = oldinputMax;
7860
0
    ctxt->inputTab = oldinputTab;
7861
0
    ctxt->encoding = oldencoding;
7862
7863
0
    xmlFree(content);
7864
7865
0
    return(ret);
7866
0
}
7867
7868
/**
7869
 * Parse PEReference declarations
7870
 *
7871
 *     [69] PEReference ::= '%' Name ';'
7872
 *
7873
 * [ WFC: No Recursion ]
7874
 * A parsed entity must not contain a recursive
7875
 * reference to itself, either directly or indirectly.
7876
 *
7877
 * [ WFC: Entity Declared ]
7878
 * In a document without any DTD, a document with only an internal DTD
7879
 * subset which contains no parameter entity references, or a document
7880
 * with "standalone='yes'", ...  ... The declaration of a parameter
7881
 * entity must precede any reference to it...
7882
 *
7883
 * [ VC: Entity Declared ]
7884
 * In a document with an external subset or external parameter entities
7885
 * with "standalone='no'", ...  ... The declaration of a parameter entity
7886
 * must precede any reference to it...
7887
 *
7888
 * [ WFC: In DTD ]
7889
 * Parameter-entity references may only appear in the DTD.
7890
 * NOTE: misleading but this is handled.
7891
 *
7892
 * @param ctxt  an XML parser context
7893
 * @param str  a pointer to an index in the string
7894
 * @returns the string of the entity content.
7895
 *         str is updated to the current value of the index
7896
 */
7897
static xmlEntityPtr
7898
0
xmlParseStringPEReference(xmlParserCtxtPtr ctxt, const xmlChar **str) {
7899
0
    const xmlChar *ptr;
7900
0
    xmlChar cur;
7901
0
    xmlChar *name;
7902
0
    xmlEntityPtr entity = NULL;
7903
7904
0
    if ((str == NULL) || (*str == NULL)) return(NULL);
7905
0
    ptr = *str;
7906
0
    cur = *ptr;
7907
0
    if (cur != '%')
7908
0
        return(NULL);
7909
0
    ptr++;
7910
0
    name = xmlParseStringName(ctxt, &ptr);
7911
0
    if (name == NULL) {
7912
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
7913
0
           "xmlParseStringPEReference: no name\n");
7914
0
  *str = ptr;
7915
0
  return(NULL);
7916
0
    }
7917
0
    cur = *ptr;
7918
0
    if (cur != ';') {
7919
0
  xmlFatalErr(ctxt, XML_ERR_ENTITYREF_SEMICOL_MISSING, NULL);
7920
0
  xmlFree(name);
7921
0
  *str = ptr;
7922
0
  return(NULL);
7923
0
    }
7924
0
    ptr++;
7925
7926
    /* Must be set before xmlHandleUndeclaredEntity */
7927
0
    ctxt->hasPErefs = 1;
7928
7929
    /*
7930
     * Request the entity from SAX
7931
     */
7932
0
    if ((ctxt->sax != NULL) &&
7933
0
  (ctxt->sax->getParameterEntity != NULL))
7934
0
  entity = ctxt->sax->getParameterEntity(ctxt->userData, name);
7935
7936
0
    if (entity == NULL) {
7937
0
        xmlHandleUndeclaredEntity(ctxt, name);
7938
0
    } else {
7939
  /*
7940
   * Internal checking in case the entity quest barfed
7941
   */
7942
0
  if ((entity->etype != XML_INTERNAL_PARAMETER_ENTITY) &&
7943
0
      (entity->etype != XML_EXTERNAL_PARAMETER_ENTITY)) {
7944
0
      xmlWarningMsg(ctxt, XML_WAR_UNDECLARED_ENTITY,
7945
0
        "%%%s; is not a parameter entity\n",
7946
0
        name, NULL);
7947
0
  }
7948
0
    }
7949
7950
0
    xmlFree(name);
7951
0
    *str = ptr;
7952
0
    return(entity);
7953
0
}
7954
7955
/**
7956
 * Parse a DOCTYPE declaration
7957
 *
7958
 * @deprecated Internal function, don't use.
7959
 *
7960
 *     [28] doctypedecl ::= '<!DOCTYPE' S Name (S ExternalID)? S?
7961
 *                          ('[' (markupdecl | PEReference | S)* ']' S?)? '>'
7962
 *
7963
 * [ VC: Root Element Type ]
7964
 * The Name in the document type declaration must match the element
7965
 * type of the root element.
7966
 *
7967
 * @param ctxt  an XML parser context
7968
 */
7969
7970
void
7971
0
xmlParseDocTypeDecl(xmlParserCtxt *ctxt) {
7972
0
    const xmlChar *name = NULL;
7973
0
    xmlChar *publicId = NULL;
7974
0
    xmlChar *URI = NULL;
7975
7976
    /*
7977
     * We know that '<!DOCTYPE' has been detected.
7978
     */
7979
0
    SKIP(9);
7980
7981
0
    if (SKIP_BLANKS == 0) {
7982
0
        xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
7983
0
                       "Space required after 'DOCTYPE'\n");
7984
0
    }
7985
7986
    /*
7987
     * Parse the DOCTYPE name.
7988
     */
7989
0
    name = xmlParseName(ctxt);
7990
0
    if (name == NULL) {
7991
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
7992
0
           "xmlParseDocTypeDecl : no DOCTYPE name !\n");
7993
0
    }
7994
0
    ctxt->intSubName = name;
7995
7996
0
    SKIP_BLANKS;
7997
7998
    /*
7999
     * Check for public and system identifier (URI)
8000
     */
8001
0
    URI = xmlParseExternalID(ctxt, &publicId, 1);
8002
8003
0
    if ((URI != NULL) || (publicId != NULL)) {
8004
0
        ctxt->hasExternalSubset = 1;
8005
0
    }
8006
0
    ctxt->extSubURI = URI;
8007
0
    ctxt->extSubSystem = publicId;
8008
8009
0
    SKIP_BLANKS;
8010
8011
    /*
8012
     * Create and update the internal subset.
8013
     */
8014
0
    if ((ctxt->sax != NULL) && (ctxt->sax->internalSubset != NULL) &&
8015
0
  (!ctxt->disableSAX))
8016
0
  ctxt->sax->internalSubset(ctxt->userData, name, publicId, URI);
8017
8018
0
    if ((RAW != '[') && (RAW != '>')) {
8019
0
  xmlFatalErr(ctxt, XML_ERR_DOCTYPE_NOT_FINISHED, NULL);
8020
0
    }
8021
0
}
8022
8023
/**
8024
 * Parse the internal subset declaration
8025
 *
8026
 *     [28 end] ('[' (markupdecl | PEReference | S)* ']' S?)? '>'
8027
 * @param ctxt  an XML parser context
8028
 */
8029
8030
static void
8031
0
xmlParseInternalSubset(xmlParserCtxtPtr ctxt) {
8032
    /*
8033
     * Is there any DTD definition ?
8034
     */
8035
0
    if (RAW == '[') {
8036
0
        int oldInputNr = ctxt->inputNr;
8037
8038
0
        NEXT;
8039
  /*
8040
   * Parse the succession of Markup declarations and
8041
   * PEReferences.
8042
   * Subsequence (markupdecl | PEReference | S)*
8043
   */
8044
0
  SKIP_BLANKS;
8045
0
        while (1) {
8046
0
            if (PARSER_STOPPED(ctxt)) {
8047
0
                return;
8048
0
            } else if (ctxt->input->cur >= ctxt->input->end) {
8049
0
                if (ctxt->inputNr <= oldInputNr) {
8050
0
                xmlFatalErr(ctxt, XML_ERR_INT_SUBSET_NOT_FINISHED, NULL);
8051
0
                    return;
8052
0
                }
8053
0
                xmlPopPE(ctxt);
8054
0
            } else if ((RAW == ']') && (ctxt->inputNr <= oldInputNr)) {
8055
0
                NEXT;
8056
0
                SKIP_BLANKS;
8057
0
                break;
8058
0
            } else if ((PARSER_EXTERNAL(ctxt)) &&
8059
0
                       (RAW == '<') && (NXT(1) == '!') && (NXT(2) == '[')) {
8060
                /*
8061
                 * Conditional sections are allowed in external entities
8062
                 * included by PE References in the internal subset.
8063
                 */
8064
0
                xmlParseConditionalSections(ctxt);
8065
0
            } else if ((RAW == '<') && ((NXT(1) == '!') || (NXT(1) == '?'))) {
8066
0
                xmlParseMarkupDecl(ctxt);
8067
0
            } else if (RAW == '%') {
8068
0
                xmlParsePERefInternal(ctxt, 1);
8069
0
            } else {
8070
0
                xmlFatalErr(ctxt, XML_ERR_INT_SUBSET_NOT_FINISHED, NULL);
8071
8072
0
                while (ctxt->inputNr > oldInputNr)
8073
0
                    xmlPopPE(ctxt);
8074
0
                return;
8075
0
            }
8076
0
            SKIP_BLANKS;
8077
0
            SHRINK;
8078
0
            GROW;
8079
0
        }
8080
0
    }
8081
8082
    /*
8083
     * We should be at the end of the DOCTYPE declaration.
8084
     */
8085
0
    if (RAW != '>') {
8086
0
        xmlFatalErr(ctxt, XML_ERR_DOCTYPE_NOT_FINISHED, NULL);
8087
0
        return;
8088
0
    }
8089
0
    NEXT;
8090
0
}
8091
8092
#ifdef LIBXML_SAX1_ENABLED
8093
/**
8094
 * Parse an attribute
8095
 *
8096
 * @deprecated Internal function, don't use.
8097
 *
8098
 *     [41] Attribute ::= Name Eq AttValue
8099
 *
8100
 * [ WFC: No External Entity References ]
8101
 * Attribute values cannot contain direct or indirect entity references
8102
 * to external entities.
8103
 *
8104
 * [ WFC: No < in Attribute Values ]
8105
 * The replacement text of any entity referred to directly or indirectly in
8106
 * an attribute value (other than "&lt;") must not contain a <.
8107
 *
8108
 * [ VC: Attribute Value Type ]
8109
 * The attribute must have been declared; the value must be of the type
8110
 * declared for it.
8111
 *
8112
 *     [25] Eq ::= S? '=' S?
8113
 *
8114
 * With namespace:
8115
 *
8116
 *     [NS 11] Attribute ::= QName Eq AttValue
8117
 *
8118
 * Also the case QName == xmlns:??? is handled independently as a namespace
8119
 * definition.
8120
 *
8121
 * @param ctxt  an XML parser context
8122
 * @param value  a xmlChar ** used to store the value of the attribute
8123
 * @returns the attribute name, and the value in *value.
8124
 */
8125
8126
const xmlChar *
8127
0
xmlParseAttribute(xmlParserCtxt *ctxt, xmlChar **value) {
8128
0
    const xmlChar *name;
8129
0
    xmlChar *val;
8130
8131
0
    *value = NULL;
8132
0
    GROW;
8133
0
    name = xmlParseName(ctxt);
8134
0
    if (name == NULL) {
8135
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
8136
0
                 "error parsing attribute name\n");
8137
0
        return(NULL);
8138
0
    }
8139
8140
    /*
8141
     * read the value
8142
     */
8143
0
    SKIP_BLANKS;
8144
0
    if (RAW == '=') {
8145
0
        NEXT;
8146
0
  SKIP_BLANKS;
8147
0
  val = xmlParseAttValue(ctxt);
8148
0
    } else {
8149
0
  xmlFatalErrMsgStr(ctxt, XML_ERR_ATTRIBUTE_WITHOUT_VALUE,
8150
0
         "Specification mandates value for attribute %s\n", name);
8151
0
  return(name);
8152
0
    }
8153
8154
    /*
8155
     * Check that xml:lang conforms to the specification
8156
     * No more registered as an error, just generate a warning now
8157
     * since this was deprecated in XML second edition
8158
     */
8159
0
    if ((ctxt->pedantic) && (xmlStrEqual(name, BAD_CAST "xml:lang"))) {
8160
0
  if (!xmlCheckLanguageID(val)) {
8161
0
      xmlWarningMsg(ctxt, XML_WAR_LANG_VALUE,
8162
0
              "Malformed value for xml:lang : %s\n",
8163
0
        val, NULL);
8164
0
  }
8165
0
    }
8166
8167
    /*
8168
     * Check that xml:space conforms to the specification
8169
     */
8170
0
    if (xmlStrEqual(name, BAD_CAST "xml:space")) {
8171
0
  if (xmlStrEqual(val, BAD_CAST "default"))
8172
0
      *(ctxt->space) = 0;
8173
0
  else if (xmlStrEqual(val, BAD_CAST "preserve"))
8174
0
      *(ctxt->space) = 1;
8175
0
  else {
8176
0
    xmlWarningMsg(ctxt, XML_WAR_SPACE_VALUE,
8177
0
"Invalid value \"%s\" for xml:space : \"default\" or \"preserve\" expected\n",
8178
0
                                 val, NULL);
8179
0
  }
8180
0
    }
8181
8182
0
    *value = val;
8183
0
    return(name);
8184
0
}
8185
8186
/**
8187
 * Parse a start tag. Always consumes '<'.
8188
 *
8189
 * @deprecated Internal function, don't use.
8190
 *
8191
 *     [40] STag ::= '<' Name (S Attribute)* S? '>'
8192
 *
8193
 * [ WFC: Unique Att Spec ]
8194
 * No attribute name may appear more than once in the same start-tag or
8195
 * empty-element tag.
8196
 *
8197
 *     [44] EmptyElemTag ::= '<' Name (S Attribute)* S? '/>'
8198
 *
8199
 * [ WFC: Unique Att Spec ]
8200
 * No attribute name may appear more than once in the same start-tag or
8201
 * empty-element tag.
8202
 *
8203
 * With namespace:
8204
 *
8205
 *     [NS 8] STag ::= '<' QName (S Attribute)* S? '>'
8206
 *
8207
 *     [NS 10] EmptyElement ::= '<' QName (S Attribute)* S? '/>'
8208
 *
8209
 * @param ctxt  an XML parser context
8210
 * @returns the element name parsed
8211
 */
8212
8213
const xmlChar *
8214
0
xmlParseStartTag(xmlParserCtxt *ctxt) {
8215
0
    const xmlChar *name;
8216
0
    const xmlChar *attname;
8217
0
    xmlChar *attvalue;
8218
0
    const xmlChar **atts = ctxt->atts;
8219
0
    int nbatts = 0;
8220
0
    int maxatts = ctxt->maxatts;
8221
0
    int i;
8222
8223
0
    if (RAW != '<') return(NULL);
8224
0
    NEXT1;
8225
8226
0
    name = xmlParseName(ctxt);
8227
0
    if (name == NULL) {
8228
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
8229
0
       "xmlParseStartTag: invalid element name\n");
8230
0
        return(NULL);
8231
0
    }
8232
8233
    /*
8234
     * Now parse the attributes, it ends up with the ending
8235
     *
8236
     * (S Attribute)* S?
8237
     */
8238
0
    SKIP_BLANKS;
8239
0
    GROW;
8240
8241
0
    while (((RAW != '>') &&
8242
0
     ((RAW != '/') || (NXT(1) != '>')) &&
8243
0
     (IS_BYTE_CHAR(RAW))) && (PARSER_STOPPED(ctxt) == 0)) {
8244
0
  attname = xmlParseAttribute(ctxt, &attvalue);
8245
0
        if (attname == NULL)
8246
0
      break;
8247
0
        if (attvalue != NULL) {
8248
      /*
8249
       * [ WFC: Unique Att Spec ]
8250
       * No attribute name may appear more than once in the same
8251
       * start-tag or empty-element tag.
8252
       */
8253
0
      for (i = 0; i < nbatts;i += 2) {
8254
0
          if (xmlStrEqual(atts[i], attname)) {
8255
0
        xmlErrAttributeDup(ctxt, NULL, attname);
8256
0
        goto failed;
8257
0
    }
8258
0
      }
8259
      /*
8260
       * Add the pair to atts
8261
       */
8262
0
      if (nbatts + 4 > maxatts) {
8263
0
          const xmlChar **n;
8264
0
                int newSize;
8265
8266
0
                newSize = xmlGrowCapacity(maxatts, sizeof(n[0]) * 2,
8267
0
                                          11, XML_MAX_ATTRS);
8268
0
                if (newSize < 0) {
8269
0
        xmlErrMemory(ctxt);
8270
0
        goto failed;
8271
0
    }
8272
0
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
8273
0
                if (newSize < 2)
8274
0
                    newSize = 2;
8275
0
#endif
8276
0
          n = xmlRealloc(atts, newSize * sizeof(n[0]) * 2);
8277
0
    if (n == NULL) {
8278
0
        xmlErrMemory(ctxt);
8279
0
        goto failed;
8280
0
    }
8281
0
    atts = n;
8282
0
                maxatts = newSize * 2;
8283
0
    ctxt->atts = atts;
8284
0
    ctxt->maxatts = maxatts;
8285
0
      }
8286
8287
0
      atts[nbatts++] = attname;
8288
0
      atts[nbatts++] = attvalue;
8289
0
      atts[nbatts] = NULL;
8290
0
      atts[nbatts + 1] = NULL;
8291
8292
0
            attvalue = NULL;
8293
0
  }
8294
8295
0
failed:
8296
8297
0
        if (attvalue != NULL)
8298
0
            xmlFree(attvalue);
8299
8300
0
  GROW
8301
0
  if ((RAW == '>') || (((RAW == '/') && (NXT(1) == '>'))))
8302
0
      break;
8303
0
  if (SKIP_BLANKS == 0) {
8304
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
8305
0
         "attributes construct error\n");
8306
0
  }
8307
0
  SHRINK;
8308
0
        GROW;
8309
0
    }
8310
8311
    /*
8312
     * SAX: Start of Element !
8313
     */
8314
0
    if ((ctxt->sax != NULL) && (ctxt->sax->startElement != NULL) &&
8315
0
  (!ctxt->disableSAX)) {
8316
0
  if (nbatts > 0)
8317
0
      ctxt->sax->startElement(ctxt->userData, name, atts);
8318
0
  else
8319
0
      ctxt->sax->startElement(ctxt->userData, name, NULL);
8320
0
    }
8321
8322
0
    if (atts != NULL) {
8323
        /* Free only the content strings */
8324
0
        for (i = 1;i < nbatts;i+=2)
8325
0
      if (atts[i] != NULL)
8326
0
         xmlFree((xmlChar *) atts[i]);
8327
0
    }
8328
0
    return(name);
8329
0
}
8330
8331
/**
8332
 * Parse an end tag. Always consumes '</'.
8333
 *
8334
 *     [42] ETag ::= '</' Name S? '>'
8335
 *
8336
 * With namespace
8337
 *
8338
 *     [NS 9] ETag ::= '</' QName S? '>'
8339
 * @param ctxt  an XML parser context
8340
 * @param line  line of the start tag
8341
 */
8342
8343
static void
8344
0
xmlParseEndTag1(xmlParserCtxtPtr ctxt, int line) {
8345
0
    const xmlChar *name;
8346
8347
0
    GROW;
8348
0
    if ((RAW != '<') || (NXT(1) != '/')) {
8349
0
  xmlFatalErrMsg(ctxt, XML_ERR_LTSLASH_REQUIRED,
8350
0
           "xmlParseEndTag: '</' not found\n");
8351
0
  return;
8352
0
    }
8353
0
    SKIP(2);
8354
8355
0
    name = xmlParseNameAndCompare(ctxt,ctxt->name);
8356
8357
    /*
8358
     * We should definitely be at the ending "S? '>'" part
8359
     */
8360
0
    GROW;
8361
0
    SKIP_BLANKS;
8362
0
    if ((!IS_BYTE_CHAR(RAW)) || (RAW != '>')) {
8363
0
  xmlFatalErr(ctxt, XML_ERR_GT_REQUIRED, NULL);
8364
0
    } else
8365
0
  NEXT1;
8366
8367
    /*
8368
     * [ WFC: Element Type Match ]
8369
     * The Name in an element's end-tag must match the element type in the
8370
     * start-tag.
8371
     *
8372
     */
8373
0
    if (name != (xmlChar*)1) {
8374
0
        if (name == NULL) name = BAD_CAST "unparsable";
8375
0
        xmlFatalErrMsgStrIntStr(ctxt, XML_ERR_TAG_NAME_MISMATCH,
8376
0
         "Opening and ending tag mismatch: %s line %d and %s\n",
8377
0
                    ctxt->name, line, name);
8378
0
    }
8379
8380
    /*
8381
     * SAX: End of Tag
8382
     */
8383
0
    if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL) &&
8384
0
  (!ctxt->disableSAX))
8385
0
        ctxt->sax->endElement(ctxt->userData, ctxt->name);
8386
8387
0
    namePop(ctxt);
8388
0
    spacePop(ctxt);
8389
0
}
8390
8391
/**
8392
 * Parse an end of tag
8393
 *
8394
 * @deprecated Internal function, don't use.
8395
 *
8396
 *     [42] ETag ::= '</' Name S? '>'
8397
 *
8398
 * With namespace
8399
 *
8400
 *     [NS 9] ETag ::= '</' QName S? '>'
8401
 * @param ctxt  an XML parser context
8402
 */
8403
8404
void
8405
0
xmlParseEndTag(xmlParserCtxt *ctxt) {
8406
0
    xmlParseEndTag1(ctxt, 0);
8407
0
}
8408
#endif /* LIBXML_SAX1_ENABLED */
8409
8410
/************************************************************************
8411
 *                  *
8412
 *          SAX 2 specific operations       *
8413
 *                  *
8414
 ************************************************************************/
8415
8416
/**
8417
 * Parse an XML Namespace QName
8418
 *
8419
 *     [6]  QName  ::= (Prefix ':')? LocalPart
8420
 *     [7]  Prefix  ::= NCName
8421
 *     [8]  LocalPart  ::= NCName
8422
 *
8423
 * @param ctxt  an XML parser context
8424
 * @param prefix  pointer to store the prefix part
8425
 * @returns the Name parsed or NULL
8426
 */
8427
8428
static xmlHashedString
8429
0
xmlParseQNameHashed(xmlParserCtxtPtr ctxt, xmlHashedString *prefix) {
8430
0
    xmlHashedString l, p;
8431
0
    int start, isNCName = 0;
8432
8433
0
    l.name = NULL;
8434
0
    p.name = NULL;
8435
8436
0
    GROW;
8437
0
    start = CUR_PTR - BASE_PTR;
8438
8439
0
    l = xmlParseNCName(ctxt);
8440
0
    if (l.name != NULL) {
8441
0
        isNCName = 1;
8442
0
        if (CUR == ':') {
8443
0
            NEXT;
8444
0
            p = l;
8445
0
            l = xmlParseNCName(ctxt);
8446
0
        }
8447
0
    }
8448
0
    if ((l.name == NULL) || (CUR == ':')) {
8449
0
        xmlChar *tmp;
8450
8451
0
        l.name = NULL;
8452
0
        p.name = NULL;
8453
0
        if ((isNCName == 0) && (CUR != ':'))
8454
0
            return(l);
8455
0
        tmp = xmlParseNmtoken(ctxt);
8456
0
        if (tmp != NULL)
8457
0
            xmlFree(tmp);
8458
0
        l = xmlDictLookupHashed(ctxt->dict, BASE_PTR + start,
8459
0
                                CUR_PTR - (BASE_PTR + start));
8460
0
        if (l.name == NULL) {
8461
0
            xmlErrMemory(ctxt);
8462
0
            return(l);
8463
0
        }
8464
0
        xmlNsErr(ctxt, XML_NS_ERR_QNAME,
8465
0
                 "Failed to parse QName '%s'\n", l.name, NULL, NULL);
8466
0
    }
8467
8468
0
    *prefix = p;
8469
0
    return(l);
8470
0
}
8471
8472
/**
8473
 * Parse an XML Namespace QName
8474
 *
8475
 *     [6]  QName  ::= (Prefix ':')? LocalPart
8476
 *     [7]  Prefix  ::= NCName
8477
 *     [8]  LocalPart  ::= NCName
8478
 *
8479
 * @param ctxt  an XML parser context
8480
 * @param prefix  pointer to store the prefix part
8481
 * @returns the Name parsed or NULL
8482
 */
8483
8484
static const xmlChar *
8485
0
xmlParseQName(xmlParserCtxtPtr ctxt, const xmlChar **prefix) {
8486
0
    xmlHashedString n, p;
8487
8488
0
    n = xmlParseQNameHashed(ctxt, &p);
8489
0
    if (n.name == NULL)
8490
0
        return(NULL);
8491
0
    *prefix = p.name;
8492
0
    return(n.name);
8493
0
}
8494
8495
/**
8496
 * Parse an XML name and compares for match
8497
 * (specialized for endtag parsing)
8498
 *
8499
 * @param ctxt  an XML parser context
8500
 * @param name  the localname
8501
 * @param prefix  the prefix, if any.
8502
 * @returns NULL for an illegal name, (xmlChar*) 1 for success
8503
 * and the name for mismatch
8504
 */
8505
8506
static const xmlChar *
8507
xmlParseQNameAndCompare(xmlParserCtxtPtr ctxt, xmlChar const *name,
8508
0
                        xmlChar const *prefix) {
8509
0
    const xmlChar *cmp;
8510
0
    const xmlChar *in;
8511
0
    const xmlChar *ret;
8512
0
    const xmlChar *prefix2;
8513
8514
0
    if (prefix == NULL) return(xmlParseNameAndCompare(ctxt, name));
8515
8516
0
    GROW;
8517
0
    in = ctxt->input->cur;
8518
8519
0
    cmp = prefix;
8520
0
    while (*in != 0 && *in == *cmp) {
8521
0
  ++in;
8522
0
  ++cmp;
8523
0
    }
8524
0
    if ((*cmp == 0) && (*in == ':')) {
8525
0
        in++;
8526
0
  cmp = name;
8527
0
  while (*in != 0 && *in == *cmp) {
8528
0
      ++in;
8529
0
      ++cmp;
8530
0
  }
8531
0
  if (*cmp == 0 && (*in == '>' || IS_BLANK_CH (*in))) {
8532
      /* success */
8533
0
            ctxt->input->col += in - ctxt->input->cur;
8534
0
      ctxt->input->cur = in;
8535
0
      return((const xmlChar*) 1);
8536
0
  }
8537
0
    }
8538
    /*
8539
     * all strings coms from the dictionary, equality can be done directly
8540
     */
8541
0
    ret = xmlParseQName (ctxt, &prefix2);
8542
0
    if (ret == NULL)
8543
0
        return(NULL);
8544
0
    if ((ret == name) && (prefix == prefix2))
8545
0
  return((const xmlChar*) 1);
8546
0
    return ret;
8547
0
}
8548
8549
/**
8550
 * Parse an attribute in the new SAX2 framework.
8551
 *
8552
 * @param ctxt  an XML parser context
8553
 * @param pref  the element prefix
8554
 * @param elem  the element name
8555
 * @param hprefix  resulting attribute prefix
8556
 * @param value  resulting value of the attribute
8557
 * @param len  resulting length of the attribute
8558
 * @param alloc  resulting indicator if the attribute was allocated
8559
 * @returns the attribute name, and the value in *value, .
8560
 */
8561
8562
static xmlHashedString
8563
xmlParseAttribute2(xmlParserCtxtPtr ctxt,
8564
                   const xmlChar * pref, const xmlChar * elem,
8565
                   xmlHashedString * hprefix, xmlChar ** value,
8566
                   int *len, int *alloc)
8567
0
{
8568
0
    xmlHashedString hname;
8569
0
    const xmlChar *prefix, *name;
8570
0
    xmlChar *val = NULL, *internal_val = NULL;
8571
0
    int special = 0;
8572
0
    int isNamespace;
8573
0
    int flags;
8574
8575
0
    *value = NULL;
8576
0
    GROW;
8577
0
    hname = xmlParseQNameHashed(ctxt, hprefix);
8578
0
    if (hname.name == NULL) {
8579
0
        xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
8580
0
                       "error parsing attribute name\n");
8581
0
        return(hname);
8582
0
    }
8583
0
    name = hname.name;
8584
0
    prefix = hprefix->name;
8585
8586
    /*
8587
     * get the type if needed
8588
     */
8589
0
    if (ctxt->attsSpecial != NULL) {
8590
0
        special = XML_PTR_TO_INT(xmlHashQLookup2(ctxt->attsSpecial, pref, elem,
8591
0
                                              prefix, name));
8592
0
    }
8593
8594
    /*
8595
     * read the value
8596
     */
8597
0
    SKIP_BLANKS;
8598
0
    if (RAW != '=') {
8599
0
        xmlFatalErrMsgStr(ctxt, XML_ERR_ATTRIBUTE_WITHOUT_VALUE,
8600
0
                          "Specification mandates value for attribute %s\n",
8601
0
                          name);
8602
0
        goto error;
8603
0
    }
8604
8605
8606
0
    NEXT;
8607
0
    SKIP_BLANKS;
8608
0
    flags = 0;
8609
0
    isNamespace = (((prefix == NULL) && (name == ctxt->str_xmlns)) ||
8610
0
                   (prefix == ctxt->str_xmlns));
8611
0
    val = xmlParseAttValueInternal(ctxt, len, &flags, special,
8612
0
                                   isNamespace);
8613
0
    if (val == NULL)
8614
0
        goto error;
8615
8616
0
    *alloc = (flags & XML_ATTVAL_ALLOC) != 0;
8617
8618
0
#ifdef LIBXML_VALID_ENABLED
8619
0
    if ((ctxt->validate) &&
8620
0
        (ctxt->standalone == 1) &&
8621
0
        (special & XML_SPECIAL_EXTERNAL) &&
8622
0
        (flags & XML_ATTVAL_NORM_CHANGE)) {
8623
0
        xmlValidityError(ctxt, XML_DTD_NOT_STANDALONE,
8624
0
                         "standalone: normalization of attribute %s on %s "
8625
0
                         "by external subset declaration\n",
8626
0
                         name, elem);
8627
0
    }
8628
0
#endif
8629
8630
0
    if (prefix == ctxt->str_xml) {
8631
        /*
8632
         * Check that xml:lang conforms to the specification
8633
         * No more registered as an error, just generate a warning now
8634
         * since this was deprecated in XML second edition
8635
         */
8636
0
        if ((ctxt->pedantic) && (xmlStrEqual(name, BAD_CAST "lang"))) {
8637
0
            internal_val = xmlStrndup(val, *len);
8638
0
            if (internal_val == NULL)
8639
0
                goto mem_error;
8640
0
            if (!xmlCheckLanguageID(internal_val)) {
8641
0
                xmlWarningMsg(ctxt, XML_WAR_LANG_VALUE,
8642
0
                              "Malformed value for xml:lang : %s\n",
8643
0
                              internal_val, NULL);
8644
0
            }
8645
0
        }
8646
8647
        /*
8648
         * Check that xml:space conforms to the specification
8649
         */
8650
0
        if (xmlStrEqual(name, BAD_CAST "space")) {
8651
0
            internal_val = xmlStrndup(val, *len);
8652
0
            if (internal_val == NULL)
8653
0
                goto mem_error;
8654
0
            if (xmlStrEqual(internal_val, BAD_CAST "default"))
8655
0
                *(ctxt->space) = 0;
8656
0
            else if (xmlStrEqual(internal_val, BAD_CAST "preserve"))
8657
0
                *(ctxt->space) = 1;
8658
0
            else {
8659
0
                xmlWarningMsg(ctxt, XML_WAR_SPACE_VALUE,
8660
0
                              "Invalid value \"%s\" for xml:space : \"default\" or \"preserve\" expected\n",
8661
0
                              internal_val, NULL);
8662
0
            }
8663
0
        }
8664
0
        if (internal_val) {
8665
0
            xmlFree(internal_val);
8666
0
        }
8667
0
    }
8668
8669
0
    *value = val;
8670
0
    return (hname);
8671
8672
0
mem_error:
8673
0
    xmlErrMemory(ctxt);
8674
0
error:
8675
0
    if ((val != NULL) && (*alloc != 0))
8676
0
        xmlFree(val);
8677
0
    return(hname);
8678
0
}
8679
8680
/**
8681
 * Inserts a new attribute into the hash table.
8682
 *
8683
 * @param ctxt  parser context
8684
 * @param size  size of the hash table
8685
 * @param name  attribute name
8686
 * @param uri  namespace uri
8687
 * @param hashValue  combined hash value of name and uri
8688
 * @param aindex  attribute index (this is a multiple of 5)
8689
 * @returns INT_MAX if no existing attribute was found, the attribute
8690
 * index if an attribute was found, -1 if a memory allocation failed.
8691
 */
8692
static int
8693
xmlAttrHashInsert(xmlParserCtxtPtr ctxt, unsigned size, const xmlChar *name,
8694
0
                  const xmlChar *uri, unsigned hashValue, int aindex) {
8695
0
    xmlAttrHashBucket *table = ctxt->attrHash;
8696
0
    xmlAttrHashBucket *bucket;
8697
0
    unsigned hindex;
8698
8699
0
    hindex = hashValue & (size - 1);
8700
0
    bucket = &table[hindex];
8701
8702
0
    while (bucket->index >= 0) {
8703
0
        const xmlChar **atts = &ctxt->atts[bucket->index];
8704
8705
0
        if (name == atts[0]) {
8706
0
            int nsIndex = XML_PTR_TO_INT(atts[2]);
8707
8708
0
            if ((nsIndex == NS_INDEX_EMPTY) ? (uri == NULL) :
8709
0
                (nsIndex == NS_INDEX_XML) ? (uri == ctxt->str_xml_ns) :
8710
0
                (uri == ctxt->nsTab[nsIndex * 2 + 1]))
8711
0
                return(bucket->index);
8712
0
        }
8713
8714
0
        hindex++;
8715
0
        bucket++;
8716
0
        if (hindex >= size) {
8717
0
            hindex = 0;
8718
0
            bucket = table;
8719
0
        }
8720
0
    }
8721
8722
0
    bucket->index = aindex;
8723
8724
0
    return(INT_MAX);
8725
0
}
8726
8727
static int
8728
xmlAttrHashInsertQName(xmlParserCtxtPtr ctxt, unsigned size,
8729
                       const xmlChar *name, const xmlChar *prefix,
8730
0
                       unsigned hashValue, int aindex) {
8731
0
    xmlAttrHashBucket *table = ctxt->attrHash;
8732
0
    xmlAttrHashBucket *bucket;
8733
0
    unsigned hindex;
8734
8735
0
    hindex = hashValue & (size - 1);
8736
0
    bucket = &table[hindex];
8737
8738
0
    while (bucket->index >= 0) {
8739
0
        const xmlChar **atts = &ctxt->atts[bucket->index];
8740
8741
0
        if ((name == atts[0]) && (prefix == atts[1]))
8742
0
            return(bucket->index);
8743
8744
0
        hindex++;
8745
0
        bucket++;
8746
0
        if (hindex >= size) {
8747
0
            hindex = 0;
8748
0
            bucket = table;
8749
0
        }
8750
0
    }
8751
8752
0
    bucket->index = aindex;
8753
8754
0
    return(INT_MAX);
8755
0
}
8756
/**
8757
 * Parse a start tag. Always consumes '<'.
8758
 *
8759
 * This routine is called when running SAX2 parsing
8760
 *
8761
 *     [40] STag ::= '<' Name (S Attribute)* S? '>'
8762
 *
8763
 * [ WFC: Unique Att Spec ]
8764
 * No attribute name may appear more than once in the same start-tag or
8765
 * empty-element tag.
8766
 *
8767
 *     [44] EmptyElemTag ::= '<' Name (S Attribute)* S? '/>'
8768
 *
8769
 * [ WFC: Unique Att Spec ]
8770
 * No attribute name may appear more than once in the same start-tag or
8771
 * empty-element tag.
8772
 *
8773
 * With namespace:
8774
 *
8775
 *     [NS 8] STag ::= '<' QName (S Attribute)* S? '>'
8776
 *
8777
 *     [NS 10] EmptyElement ::= '<' QName (S Attribute)* S? '/>'
8778
 *
8779
 * @param ctxt  an XML parser context
8780
 * @param pref  resulting namespace prefix
8781
 * @param URI  resulting namespace URI
8782
 * @param nbNsPtr  resulting number of namespace declarations
8783
 * @returns the element name parsed
8784
 */
8785
8786
static const xmlChar *
8787
xmlParseStartTag2(xmlParserCtxtPtr ctxt, const xmlChar **pref,
8788
0
                  const xmlChar **URI, int *nbNsPtr) {
8789
0
    xmlHashedString hlocalname;
8790
0
    xmlHashedString hprefix;
8791
0
    xmlHashedString hattname;
8792
0
    xmlHashedString haprefix;
8793
0
    const xmlChar *localname;
8794
0
    const xmlChar *prefix;
8795
0
    const xmlChar *attname;
8796
0
    const xmlChar *aprefix;
8797
0
    const xmlChar *uri;
8798
0
    xmlChar *attvalue = NULL;
8799
0
    const xmlChar **atts = ctxt->atts;
8800
0
    unsigned attrHashSize = 0;
8801
0
    int maxatts = ctxt->maxatts;
8802
0
    int nratts, nbatts, nbdef;
8803
0
    int i, j, nbNs, nbTotalDef, attval, nsIndex, maxAtts;
8804
0
    int alloc = 0;
8805
0
    int numNsErr = 0;
8806
0
    int numDupErr = 0;
8807
8808
0
    if (RAW != '<') return(NULL);
8809
0
    NEXT1;
8810
8811
0
    nbatts = 0;
8812
0
    nratts = 0;
8813
0
    nbdef = 0;
8814
0
    nbNs = 0;
8815
0
    nbTotalDef = 0;
8816
0
    attval = 0;
8817
8818
0
    if (xmlParserNsStartElement(ctxt->nsdb) < 0) {
8819
0
        xmlErrMemory(ctxt);
8820
0
        return(NULL);
8821
0
    }
8822
8823
0
    hlocalname = xmlParseQNameHashed(ctxt, &hprefix);
8824
0
    if (hlocalname.name == NULL) {
8825
0
  xmlFatalErrMsg(ctxt, XML_ERR_NAME_REQUIRED,
8826
0
           "StartTag: invalid element name\n");
8827
0
        return(NULL);
8828
0
    }
8829
0
    localname = hlocalname.name;
8830
0
    prefix = hprefix.name;
8831
8832
    /*
8833
     * Now parse the attributes, it ends up with the ending
8834
     *
8835
     * (S Attribute)* S?
8836
     */
8837
0
    SKIP_BLANKS;
8838
0
    GROW;
8839
8840
    /*
8841
     * The ctxt->atts array will be ultimately passed to the SAX callback
8842
     * containing five xmlChar pointers for each attribute:
8843
     *
8844
     * [0] attribute name
8845
     * [1] attribute prefix
8846
     * [2] namespace URI
8847
     * [3] attribute value
8848
     * [4] end of attribute value
8849
     *
8850
     * To save memory, we reuse this array temporarily and store integers
8851
     * in these pointer variables.
8852
     *
8853
     * [0] attribute name
8854
     * [1] attribute prefix
8855
     * [2] hash value of attribute prefix, and later namespace index
8856
     * [3] for non-allocated values: ptrdiff_t offset into input buffer
8857
     * [4] for non-allocated values: ptrdiff_t offset into input buffer
8858
     *
8859
     * The ctxt->attallocs array contains an additional unsigned int for
8860
     * each attribute, containing the hash value of the attribute name
8861
     * and the alloc flag in bit 31.
8862
     */
8863
8864
0
    while (((RAW != '>') &&
8865
0
     ((RAW != '/') || (NXT(1) != '>')) &&
8866
0
     (IS_BYTE_CHAR(RAW))) && (PARSER_STOPPED(ctxt) == 0)) {
8867
0
  int len = -1;
8868
8869
0
  hattname = xmlParseAttribute2(ctxt, prefix, localname,
8870
0
                                          &haprefix, &attvalue, &len,
8871
0
                                          &alloc);
8872
0
        if (hattname.name == NULL)
8873
0
      break;
8874
0
        if (attvalue == NULL)
8875
0
            goto next_attr;
8876
0
        attname = hattname.name;
8877
0
        aprefix = haprefix.name;
8878
0
  if (len < 0) len = xmlStrlen(attvalue);
8879
8880
0
        if ((attname == ctxt->str_xmlns) && (aprefix == NULL)) {
8881
0
            xmlHashedString huri;
8882
0
            xmlURIPtr parsedUri;
8883
8884
0
            huri = xmlDictLookupHashed(ctxt->dict, attvalue, len);
8885
0
            uri = huri.name;
8886
0
            if (uri == NULL) {
8887
0
                xmlErrMemory(ctxt);
8888
0
                goto next_attr;
8889
0
            }
8890
0
            if (*uri != 0) {
8891
0
                if (xmlParseURISafe((const char *) uri, &parsedUri) < 0) {
8892
0
                    xmlErrMemory(ctxt);
8893
0
                    goto next_attr;
8894
0
                }
8895
0
                if (parsedUri == NULL) {
8896
0
                    xmlNsErr(ctxt, XML_WAR_NS_URI,
8897
0
                             "xmlns: '%s' is not a valid URI\n",
8898
0
                                       uri, NULL, NULL);
8899
0
                } else {
8900
0
                    if (parsedUri->scheme == NULL) {
8901
0
                        xmlNsWarn(ctxt, XML_WAR_NS_URI_RELATIVE,
8902
0
                                  "xmlns: URI %s is not absolute\n",
8903
0
                                  uri, NULL, NULL);
8904
0
                    }
8905
0
                    xmlFreeURI(parsedUri);
8906
0
                }
8907
0
                if (uri == ctxt->str_xml_ns) {
8908
0
                    if (attname != ctxt->str_xml) {
8909
0
                        xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8910
0
                     "xml namespace URI cannot be the default namespace\n",
8911
0
                                 NULL, NULL, NULL);
8912
0
                    }
8913
0
                    goto next_attr;
8914
0
                }
8915
0
                if ((len == 29) &&
8916
0
                    (xmlStrEqual(uri,
8917
0
                             BAD_CAST "http://www.w3.org/2000/xmlns/"))) {
8918
0
                    xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8919
0
                         "reuse of the xmlns namespace name is forbidden\n",
8920
0
                             NULL, NULL, NULL);
8921
0
                    goto next_attr;
8922
0
                }
8923
0
            }
8924
8925
0
            if (xmlParserNsPush(ctxt, NULL, &huri, NULL, 0) > 0)
8926
0
                nbNs++;
8927
0
        } else if (aprefix == ctxt->str_xmlns) {
8928
0
            xmlHashedString huri;
8929
0
            xmlURIPtr parsedUri;
8930
8931
0
            huri = xmlDictLookupHashed(ctxt->dict, attvalue, len);
8932
0
            uri = huri.name;
8933
0
            if (uri == NULL) {
8934
0
                xmlErrMemory(ctxt);
8935
0
                goto next_attr;
8936
0
            }
8937
8938
0
            if (attname == ctxt->str_xml) {
8939
0
                if (uri != ctxt->str_xml_ns) {
8940
0
                    xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8941
0
                             "xml namespace prefix mapped to wrong URI\n",
8942
0
                             NULL, NULL, NULL);
8943
0
                }
8944
                /*
8945
                 * Do not keep a namespace definition node
8946
                 */
8947
0
                goto next_attr;
8948
0
            }
8949
0
            if (uri == ctxt->str_xml_ns) {
8950
0
                if (attname != ctxt->str_xml) {
8951
0
                    xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8952
0
                             "xml namespace URI mapped to wrong prefix\n",
8953
0
                             NULL, NULL, NULL);
8954
0
                }
8955
0
                goto next_attr;
8956
0
            }
8957
0
            if (attname == ctxt->str_xmlns) {
8958
0
                xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8959
0
                         "redefinition of the xmlns prefix is forbidden\n",
8960
0
                         NULL, NULL, NULL);
8961
0
                goto next_attr;
8962
0
            }
8963
0
            if ((len == 29) &&
8964
0
                (xmlStrEqual(uri,
8965
0
                             BAD_CAST "http://www.w3.org/2000/xmlns/"))) {
8966
0
                xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8967
0
                         "reuse of the xmlns namespace name is forbidden\n",
8968
0
                         NULL, NULL, NULL);
8969
0
                goto next_attr;
8970
0
            }
8971
0
            if ((uri == NULL) || (uri[0] == 0)) {
8972
0
                xmlNsErr(ctxt, XML_NS_ERR_XML_NAMESPACE,
8973
0
                         "xmlns:%s: Empty XML namespace is not allowed\n",
8974
0
                              attname, NULL, NULL);
8975
0
                goto next_attr;
8976
0
            } else {
8977
0
                if (xmlParseURISafe((const char *) uri, &parsedUri) < 0) {
8978
0
                    xmlErrMemory(ctxt);
8979
0
                    goto next_attr;
8980
0
                }
8981
0
                if (parsedUri == NULL) {
8982
0
                    xmlNsErr(ctxt, XML_WAR_NS_URI,
8983
0
                         "xmlns:%s: '%s' is not a valid URI\n",
8984
0
                                       attname, uri, NULL);
8985
0
                } else {
8986
0
                    if ((ctxt->pedantic) && (parsedUri->scheme == NULL)) {
8987
0
                        xmlNsWarn(ctxt, XML_WAR_NS_URI_RELATIVE,
8988
0
                                  "xmlns:%s: URI %s is not absolute\n",
8989
0
                                  attname, uri, NULL);
8990
0
                    }
8991
0
                    xmlFreeURI(parsedUri);
8992
0
                }
8993
0
            }
8994
8995
0
            if (xmlParserNsPush(ctxt, &hattname, &huri, NULL, 0) > 0)
8996
0
                nbNs++;
8997
0
        } else {
8998
            /*
8999
             * Populate attributes array, see above for repurposing
9000
             * of xmlChar pointers.
9001
             */
9002
0
            if ((atts == NULL) || (nbatts + 5 > maxatts)) {
9003
0
                int res = xmlCtxtGrowAttrs(ctxt);
9004
9005
0
                maxatts = ctxt->maxatts;
9006
0
                atts = ctxt->atts;
9007
9008
0
                if (res < 0)
9009
0
                    goto next_attr;
9010
0
            }
9011
0
            ctxt->attallocs[nratts++] = (hattname.hashValue & 0x7FFFFFFF) |
9012
0
                                        ((unsigned) alloc << 31);
9013
0
            atts[nbatts++] = attname;
9014
0
            atts[nbatts++] = aprefix;
9015
0
            atts[nbatts++] = XML_INT_TO_PTR(haprefix.hashValue);
9016
0
            if (alloc) {
9017
0
                atts[nbatts++] = attvalue;
9018
0
                attvalue += len;
9019
0
                atts[nbatts++] = attvalue;
9020
0
            } else {
9021
                /*
9022
                 * attvalue points into the input buffer which can be
9023
                 * reallocated. Store differences to input->base instead.
9024
                 * The pointers will be reconstructed later.
9025
                 */
9026
0
                atts[nbatts++] = XML_INT_TO_PTR(attvalue - BASE_PTR);
9027
0
                attvalue += len;
9028
0
                atts[nbatts++] = XML_INT_TO_PTR(attvalue - BASE_PTR);
9029
0
            }
9030
            /*
9031
             * tag if some deallocation is needed
9032
             */
9033
0
            if (alloc != 0) attval = 1;
9034
0
            attvalue = NULL; /* moved into atts */
9035
0
        }
9036
9037
0
next_attr:
9038
0
        if ((attvalue != NULL) && (alloc != 0)) {
9039
0
            xmlFree(attvalue);
9040
0
            attvalue = NULL;
9041
0
        }
9042
9043
0
  GROW
9044
0
  if ((RAW == '>') || (((RAW == '/') && (NXT(1) == '>'))))
9045
0
      break;
9046
0
  if (SKIP_BLANKS == 0) {
9047
0
      xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
9048
0
         "attributes construct error\n");
9049
0
      break;
9050
0
  }
9051
0
        GROW;
9052
0
    }
9053
9054
    /*
9055
     * Namespaces from default attributes
9056
     */
9057
0
    if (ctxt->attsDefault != NULL) {
9058
0
        xmlDefAttrsPtr defaults;
9059
9060
0
  defaults = xmlHashLookup2(ctxt->attsDefault, localname, prefix);
9061
0
  if (defaults != NULL) {
9062
0
      for (i = 0; i < defaults->nbAttrs; i++) {
9063
0
                xmlDefAttr *attr = &defaults->attrs[i];
9064
9065
0
          attname = attr->name.name;
9066
0
    aprefix = attr->prefix.name;
9067
9068
0
    if ((attname == ctxt->str_xmlns) && (aprefix == NULL)) {
9069
0
                    xmlParserEntityCheck(ctxt, attr->expandedSize);
9070
9071
0
                    if (xmlParserNsPush(ctxt, NULL, &attr->value, NULL, 1) > 0)
9072
0
                        nbNs++;
9073
0
    } else if (aprefix == ctxt->str_xmlns) {
9074
0
                    xmlParserEntityCheck(ctxt, attr->expandedSize);
9075
9076
0
                    if (xmlParserNsPush(ctxt, &attr->name, &attr->value,
9077
0
                                      NULL, 1) > 0)
9078
0
                        nbNs++;
9079
0
    } else {
9080
0
                    if (nratts + nbTotalDef >= XML_MAX_ATTRS) {
9081
0
                        xmlFatalErr(ctxt, XML_ERR_RESOURCE_LIMIT,
9082
0
                                    "Maximum number of attributes exceeded");
9083
0
                        break;
9084
0
                    }
9085
0
                    nbTotalDef += 1;
9086
0
                }
9087
0
      }
9088
0
  }
9089
0
    }
9090
9091
    /*
9092
     * Resolve attribute namespaces
9093
     */
9094
0
    for (i = 0; i < nbatts; i += 5) {
9095
0
        attname = atts[i];
9096
0
        aprefix = atts[i+1];
9097
9098
        /*
9099
  * The default namespace does not apply to attribute names.
9100
  */
9101
0
  if (aprefix == NULL) {
9102
0
            nsIndex = NS_INDEX_EMPTY;
9103
0
        } else if (aprefix == ctxt->str_xml) {
9104
0
            nsIndex = NS_INDEX_XML;
9105
0
        } else {
9106
0
            haprefix.name = aprefix;
9107
0
            haprefix.hashValue = (size_t) atts[i+2];
9108
0
            nsIndex = xmlParserNsLookup(ctxt, &haprefix, NULL);
9109
9110
0
      if ((nsIndex == INT_MAX) || (nsIndex < ctxt->nsdb->minNsIndex)) {
9111
0
                xmlNsErr(ctxt, XML_NS_ERR_UNDEFINED_NAMESPACE,
9112
0
        "Namespace prefix %s for %s on %s is not defined\n",
9113
0
        aprefix, attname, localname);
9114
0
                nsIndex = NS_INDEX_EMPTY;
9115
0
            }
9116
0
        }
9117
9118
0
        atts[i+2] = XML_INT_TO_PTR(nsIndex);
9119
0
    }
9120
9121
    /*
9122
     * Maximum number of attributes including default attributes.
9123
     */
9124
0
    maxAtts = nratts + nbTotalDef;
9125
9126
    /*
9127
     * Verify that attribute names are unique.
9128
     */
9129
0
    if (maxAtts > 1) {
9130
0
        attrHashSize = 4;
9131
0
        while (attrHashSize / 2 < (unsigned) maxAtts)
9132
0
            attrHashSize *= 2;
9133
9134
0
        if (attrHashSize > ctxt->attrHashMax) {
9135
0
            xmlAttrHashBucket *tmp;
9136
9137
0
            tmp = xmlRealloc(ctxt->attrHash, attrHashSize * sizeof(tmp[0]));
9138
0
            if (tmp == NULL) {
9139
0
                xmlErrMemory(ctxt);
9140
0
                goto done;
9141
0
            }
9142
9143
0
            ctxt->attrHash = tmp;
9144
0
            ctxt->attrHashMax = attrHashSize;
9145
0
        }
9146
9147
0
        memset(ctxt->attrHash, -1, attrHashSize * sizeof(ctxt->attrHash[0]));
9148
9149
0
        for (i = 0, j = 0; j < nratts; i += 5, j++) {
9150
0
            const xmlChar *nsuri;
9151
0
            unsigned hashValue, nameHashValue, uriHashValue;
9152
0
            int res;
9153
9154
0
            attname = atts[i];
9155
0
            aprefix = atts[i+1];
9156
0
            nsIndex = XML_PTR_TO_INT(atts[i+2]);
9157
            /* Hash values always have bit 31 set, see dict.c */
9158
0
            nameHashValue = ctxt->attallocs[j] | 0x80000000;
9159
9160
0
            if (nsIndex == NS_INDEX_EMPTY) {
9161
                /*
9162
                 * Prefix with empty namespace means an undeclared
9163
                 * prefix which was already reported above.
9164
                 */
9165
0
                if (aprefix != NULL)
9166
0
                    continue;
9167
0
                nsuri = NULL;
9168
0
                uriHashValue = URI_HASH_EMPTY;
9169
0
            } else if (nsIndex == NS_INDEX_XML) {
9170
0
                nsuri = ctxt->str_xml_ns;
9171
0
                uriHashValue = URI_HASH_XML;
9172
0
            } else {
9173
0
                nsuri = ctxt->nsTab[nsIndex * 2 + 1];
9174
0
                uriHashValue = ctxt->nsdb->extra[nsIndex].uriHashValue;
9175
0
            }
9176
9177
0
            hashValue = xmlDictCombineHash(nameHashValue, uriHashValue);
9178
0
            res = xmlAttrHashInsert(ctxt, attrHashSize, attname, nsuri,
9179
0
                                    hashValue, i);
9180
0
            if (res < 0)
9181
0
                continue;
9182
9183
            /*
9184
             * [ WFC: Unique Att Spec ]
9185
             * No attribute name may appear more than once in the same
9186
             * start-tag or empty-element tag.
9187
             * As extended by the Namespace in XML REC.
9188
             */
9189
0
            if (res < INT_MAX) {
9190
0
                if (aprefix == atts[res+1]) {
9191
0
                    xmlErrAttributeDup(ctxt, aprefix, attname);
9192
0
                    numDupErr += 1;
9193
0
                } else {
9194
0
                    xmlNsErr(ctxt, XML_NS_ERR_ATTRIBUTE_REDEFINED,
9195
0
                             "Namespaced Attribute %s in '%s' redefined\n",
9196
0
                             attname, nsuri, NULL);
9197
0
                    numNsErr += 1;
9198
0
                }
9199
0
            }
9200
0
        }
9201
0
    }
9202
9203
    /*
9204
     * Default attributes
9205
     */
9206
0
    if (ctxt->attsDefault != NULL) {
9207
0
        xmlDefAttrsPtr defaults;
9208
9209
0
  defaults = xmlHashLookup2(ctxt->attsDefault, localname, prefix);
9210
0
  if (defaults != NULL) {
9211
0
      for (i = 0; i < defaults->nbAttrs; i++) {
9212
0
                xmlDefAttr *attr = &defaults->attrs[i];
9213
0
                const xmlChar *nsuri = NULL;
9214
0
                unsigned hashValue, uriHashValue = 0;
9215
0
                int res;
9216
9217
0
          attname = attr->name.name;
9218
0
    aprefix = attr->prefix.name;
9219
9220
0
    if ((attname == ctxt->str_xmlns) && (aprefix == NULL))
9221
0
                    continue;
9222
0
    if (aprefix == ctxt->str_xmlns)
9223
0
                    continue;
9224
9225
0
                if (aprefix == NULL) {
9226
0
                    nsIndex = NS_INDEX_EMPTY;
9227
0
                    nsuri = NULL;
9228
0
                    uriHashValue = URI_HASH_EMPTY;
9229
0
                } else if (aprefix == ctxt->str_xml) {
9230
0
                    nsIndex = NS_INDEX_XML;
9231
0
                    nsuri = ctxt->str_xml_ns;
9232
0
                    uriHashValue = URI_HASH_XML;
9233
0
                } else {
9234
0
                    nsIndex = xmlParserNsLookup(ctxt, &attr->prefix, NULL);
9235
0
                    if ((nsIndex == INT_MAX) ||
9236
0
                        (nsIndex < ctxt->nsdb->minNsIndex)) {
9237
0
                        xmlNsErr(ctxt, XML_NS_ERR_UNDEFINED_NAMESPACE,
9238
0
                                 "Namespace prefix %s for %s on %s is not "
9239
0
                                 "defined\n",
9240
0
                                 aprefix, attname, localname);
9241
0
                        nsIndex = NS_INDEX_EMPTY;
9242
0
                        nsuri = NULL;
9243
0
                        uriHashValue = URI_HASH_EMPTY;
9244
0
                    } else {
9245
0
                        nsuri = ctxt->nsTab[nsIndex * 2 + 1];
9246
0
                        uriHashValue = ctxt->nsdb->extra[nsIndex].uriHashValue;
9247
0
                    }
9248
0
                }
9249
9250
                /*
9251
                 * Check whether the attribute exists
9252
                 */
9253
0
                if (maxAtts > 1) {
9254
0
                    hashValue = xmlDictCombineHash(attr->name.hashValue,
9255
0
                                                   uriHashValue);
9256
0
                    res = xmlAttrHashInsert(ctxt, attrHashSize, attname, nsuri,
9257
0
                                            hashValue, nbatts);
9258
0
                    if (res < 0)
9259
0
                        continue;
9260
0
                    if (res < INT_MAX) {
9261
0
                        if (aprefix == atts[res+1])
9262
0
                            continue;
9263
0
                        xmlNsErr(ctxt, XML_NS_ERR_ATTRIBUTE_REDEFINED,
9264
0
                                 "Namespaced Attribute %s in '%s' redefined\n",
9265
0
                                 attname, nsuri, NULL);
9266
0
                    }
9267
0
                }
9268
9269
0
                xmlParserEntityCheck(ctxt, attr->expandedSize);
9270
9271
0
                if ((atts == NULL) || (nbatts + 5 > maxatts)) {
9272
0
                    res = xmlCtxtGrowAttrs(ctxt);
9273
9274
0
                    maxatts = ctxt->maxatts;
9275
0
                    atts = ctxt->atts;
9276
9277
0
                    if (res < 0) {
9278
0
                        localname = NULL;
9279
0
                        goto done;
9280
0
                    }
9281
0
                }
9282
9283
0
                atts[nbatts++] = attname;
9284
0
                atts[nbatts++] = aprefix;
9285
0
                atts[nbatts++] = XML_INT_TO_PTR(nsIndex);
9286
0
                atts[nbatts++] = attr->value.name;
9287
0
                atts[nbatts++] = attr->valueEnd;
9288
9289
0
#ifdef LIBXML_VALID_ENABLED
9290
                /*
9291
                 * This should be moved to valid.c, but we don't keep track
9292
                 * whether an attribute was defaulted.
9293
                 */
9294
0
                if ((ctxt->validate) &&
9295
0
                    (ctxt->standalone == 1) &&
9296
0
                    (attr->external != 0)) {
9297
0
                    xmlValidityError(ctxt, XML_DTD_STANDALONE_DEFAULTED,
9298
0
                            "standalone: attribute %s on %s defaulted "
9299
0
                            "from external subset\n",
9300
0
                            attname, localname);
9301
0
                }
9302
0
#endif
9303
0
                nbdef++;
9304
0
      }
9305
0
  }
9306
0
    }
9307
9308
    /*
9309
     * Using a single hash table for nsUri/localName pairs cannot
9310
     * detect duplicate QNames reliably. The following example will
9311
     * only result in two namespace errors.
9312
     *
9313
     * <doc xmlns:a="a" xmlns:b="a">
9314
     *   <elem a:a="" b:a="" b:a=""/>
9315
     * </doc>
9316
     *
9317
     * If we saw more than one namespace error but no duplicate QNames
9318
     * were found, we have to scan for duplicate QNames.
9319
     */
9320
0
    if ((numDupErr == 0) && (numNsErr > 1)) {
9321
0
        memset(ctxt->attrHash, -1,
9322
0
               attrHashSize * sizeof(ctxt->attrHash[0]));
9323
9324
0
        for (i = 0, j = 0; j < nratts; i += 5, j++) {
9325
0
            unsigned hashValue, nameHashValue, prefixHashValue;
9326
0
            int res;
9327
9328
0
            aprefix = atts[i+1];
9329
0
            if (aprefix == NULL)
9330
0
                continue;
9331
9332
0
            attname = atts[i];
9333
            /* Hash values always have bit 31 set, see dict.c */
9334
0
            nameHashValue = ctxt->attallocs[j] | 0x80000000;
9335
0
            prefixHashValue = xmlDictComputeHash(ctxt->dict, aprefix);
9336
9337
0
            hashValue = xmlDictCombineHash(nameHashValue, prefixHashValue);
9338
0
            res = xmlAttrHashInsertQName(ctxt, attrHashSize, attname,
9339
0
                                         aprefix, hashValue, i);
9340
0
            if (res < INT_MAX)
9341
0
                xmlErrAttributeDup(ctxt, aprefix, attname);
9342
0
        }
9343
0
    }
9344
9345
    /*
9346
     * Reconstruct attribute pointers
9347
     */
9348
0
    for (i = 0, j = 0; i < nbatts; i += 5, j++) {
9349
        /* namespace URI */
9350
0
        nsIndex = XML_PTR_TO_INT(atts[i+2]);
9351
0
        if (nsIndex == INT_MAX)
9352
0
            atts[i+2] = NULL;
9353
0
        else if (nsIndex == INT_MAX - 1)
9354
0
            atts[i+2] = ctxt->str_xml_ns;
9355
0
        else
9356
0
            atts[i+2] = ctxt->nsTab[nsIndex * 2 + 1];
9357
9358
0
        if ((j < nratts) && (ctxt->attallocs[j] & 0x80000000) == 0) {
9359
0
            atts[i+3] = BASE_PTR + XML_PTR_TO_INT(atts[i+3]);  /* value */
9360
0
            atts[i+4] = BASE_PTR + XML_PTR_TO_INT(atts[i+4]);  /* valuend */
9361
0
        }
9362
0
    }
9363
9364
0
    uri = xmlParserNsLookupUri(ctxt, &hprefix);
9365
0
    if ((prefix != NULL) && (uri == NULL)) {
9366
0
  xmlNsErr(ctxt, XML_NS_ERR_UNDEFINED_NAMESPACE,
9367
0
           "Namespace prefix %s on %s is not defined\n",
9368
0
     prefix, localname, NULL);
9369
0
    }
9370
0
    *pref = prefix;
9371
0
    *URI = uri;
9372
9373
    /*
9374
     * SAX callback
9375
     */
9376
0
    if ((ctxt->sax != NULL) && (ctxt->sax->startElementNs != NULL) &&
9377
0
  (!ctxt->disableSAX)) {
9378
0
  if (nbNs > 0)
9379
0
      ctxt->sax->startElementNs(ctxt->userData, localname, prefix, uri,
9380
0
                          nbNs, ctxt->nsTab + 2 * (ctxt->nsNr - nbNs),
9381
0
        nbatts / 5, nbdef, atts);
9382
0
  else
9383
0
      ctxt->sax->startElementNs(ctxt->userData, localname, prefix, uri,
9384
0
                          0, NULL, nbatts / 5, nbdef, atts);
9385
0
    }
9386
9387
0
done:
9388
    /*
9389
     * Free allocated attribute values
9390
     */
9391
0
    if (attval != 0) {
9392
0
  for (i = 0, j = 0; j < nratts; i += 5, j++)
9393
0
      if (ctxt->attallocs[j] & 0x80000000)
9394
0
          xmlFree((xmlChar *) atts[i+3]);
9395
0
    }
9396
9397
0
    *nbNsPtr = nbNs;
9398
0
    return(localname);
9399
0
}
9400
9401
/**
9402
 * Parse an end tag. Always consumes '</'.
9403
 *
9404
 *     [42] ETag ::= '</' Name S? '>'
9405
 *
9406
 * With namespace
9407
 *
9408
 *     [NS 9] ETag ::= '</' QName S? '>'
9409
 * @param ctxt  an XML parser context
9410
 * @param tag  the corresponding start tag
9411
 */
9412
9413
static void
9414
0
xmlParseEndTag2(xmlParserCtxtPtr ctxt, const xmlStartTag *tag) {
9415
0
    const xmlChar *name;
9416
9417
0
    GROW;
9418
0
    if ((RAW != '<') || (NXT(1) != '/')) {
9419
0
  xmlFatalErr(ctxt, XML_ERR_LTSLASH_REQUIRED, NULL);
9420
0
  return;
9421
0
    }
9422
0
    SKIP(2);
9423
9424
0
    if (tag->prefix == NULL)
9425
0
        name = xmlParseNameAndCompare(ctxt, ctxt->name);
9426
0
    else
9427
0
        name = xmlParseQNameAndCompare(ctxt, ctxt->name, tag->prefix);
9428
9429
    /*
9430
     * We should definitely be at the ending "S? '>'" part
9431
     */
9432
0
    GROW;
9433
0
    SKIP_BLANKS;
9434
0
    if ((!IS_BYTE_CHAR(RAW)) || (RAW != '>')) {
9435
0
  xmlFatalErr(ctxt, XML_ERR_GT_REQUIRED, NULL);
9436
0
    } else
9437
0
  NEXT1;
9438
9439
    /*
9440
     * [ WFC: Element Type Match ]
9441
     * The Name in an element's end-tag must match the element type in the
9442
     * start-tag.
9443
     *
9444
     */
9445
0
    if (name != (xmlChar*)1) {
9446
0
        if (name == NULL) name = BAD_CAST "unparsable";
9447
0
        xmlFatalErrMsgStrIntStr(ctxt, XML_ERR_TAG_NAME_MISMATCH,
9448
0
         "Opening and ending tag mismatch: %s line %d and %s\n",
9449
0
                    ctxt->name, tag->line, name);
9450
0
    }
9451
9452
    /*
9453
     * SAX: End of Tag
9454
     */
9455
0
    if ((ctxt->sax != NULL) && (ctxt->sax->endElementNs != NULL) &&
9456
0
  (!ctxt->disableSAX))
9457
0
  ctxt->sax->endElementNs(ctxt->userData, ctxt->name, tag->prefix,
9458
0
                                tag->URI);
9459
9460
0
    spacePop(ctxt);
9461
0
    if (tag->nsNr != 0)
9462
0
  xmlParserNsPop(ctxt, tag->nsNr);
9463
0
}
9464
9465
/**
9466
 * Parse escaped pure raw content. Always consumes '<!['.
9467
 *
9468
 * @deprecated Internal function, don't use.
9469
 *
9470
 *     [18] CDSect ::= CDStart CData CDEnd
9471
 *
9472
 *     [19] CDStart ::= '<![CDATA['
9473
 *
9474
 *     [20] Data ::= (Char* - (Char* ']]>' Char*))
9475
 *
9476
 *     [21] CDEnd ::= ']]>'
9477
 * @param ctxt  an XML parser context
9478
 */
9479
void
9480
0
xmlParseCDSect(xmlParserCtxt *ctxt) {
9481
0
    xmlChar *buf = NULL;
9482
0
    int len = 0;
9483
0
    int size = XML_PARSER_BUFFER_SIZE;
9484
0
    int r, rl;
9485
0
    int s, sl;
9486
0
    int cur, l;
9487
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
9488
0
                    XML_MAX_HUGE_LENGTH :
9489
0
                    XML_MAX_TEXT_LENGTH;
9490
9491
0
    if ((CUR != '<') || (NXT(1) != '!') || (NXT(2) != '['))
9492
0
        return;
9493
0
    SKIP(3);
9494
9495
0
    if (!CMP6(CUR_PTR, 'C', 'D', 'A', 'T', 'A', '['))
9496
0
        return;
9497
0
    SKIP(6);
9498
9499
0
    r = xmlCurrentCharRecover(ctxt, &rl);
9500
0
    if (!IS_CHAR(r)) {
9501
0
  xmlFatalErr(ctxt, XML_ERR_CDATA_NOT_FINISHED, NULL);
9502
0
        goto out;
9503
0
    }
9504
0
    NEXTL(rl);
9505
0
    s = xmlCurrentCharRecover(ctxt, &sl);
9506
0
    if (!IS_CHAR(s)) {
9507
0
  xmlFatalErr(ctxt, XML_ERR_CDATA_NOT_FINISHED, NULL);
9508
0
        goto out;
9509
0
    }
9510
0
    NEXTL(sl);
9511
0
    cur = xmlCurrentCharRecover(ctxt, &l);
9512
0
    buf = xmlMalloc(size);
9513
0
    if (buf == NULL) {
9514
0
  xmlErrMemory(ctxt);
9515
0
        goto out;
9516
0
    }
9517
0
    while (IS_CHAR(cur) &&
9518
0
           ((r != ']') || (s != ']') || (cur != '>'))) {
9519
0
  if (len + 5 >= size) {
9520
0
      xmlChar *tmp;
9521
0
            int newSize;
9522
9523
0
            newSize = xmlGrowCapacity(size, 1, 1, maxLength);
9524
0
            if (newSize < 0) {
9525
0
                xmlFatalErrMsg(ctxt, XML_ERR_CDATA_NOT_FINISHED,
9526
0
                               "CData section too big found\n");
9527
0
                goto out;
9528
0
            }
9529
0
      tmp = xmlRealloc(buf, newSize);
9530
0
      if (tmp == NULL) {
9531
0
    xmlErrMemory(ctxt);
9532
0
                goto out;
9533
0
      }
9534
0
      buf = tmp;
9535
0
      size = newSize;
9536
0
  }
9537
0
  COPY_BUF(buf, len, r);
9538
0
  r = s;
9539
0
  rl = sl;
9540
0
  s = cur;
9541
0
  sl = l;
9542
0
  NEXTL(l);
9543
0
  cur = xmlCurrentCharRecover(ctxt, &l);
9544
0
    }
9545
0
    buf[len] = 0;
9546
0
    if (cur != '>') {
9547
0
  xmlFatalErrMsgStr(ctxt, XML_ERR_CDATA_NOT_FINISHED,
9548
0
                       "CData section not finished\n%.50s\n", buf);
9549
0
        goto out;
9550
0
    }
9551
0
    NEXTL(l);
9552
9553
    /*
9554
     * OK the buffer is to be consumed as cdata.
9555
     */
9556
0
    if ((ctxt->sax != NULL) && (!ctxt->disableSAX)) {
9557
0
        if ((ctxt->sax->cdataBlock != NULL) &&
9558
0
            ((ctxt->options & XML_PARSE_NOCDATA) == 0)) {
9559
0
            ctxt->sax->cdataBlock(ctxt->userData, buf, len);
9560
0
        } else if (ctxt->sax->characters != NULL) {
9561
0
            ctxt->sax->characters(ctxt->userData, buf, len);
9562
0
        }
9563
0
    }
9564
9565
0
out:
9566
0
    xmlFree(buf);
9567
0
}
9568
9569
/**
9570
 * Parse a content sequence. Stops at EOF or '</'. Leaves checking of
9571
 * unexpected EOF to the caller.
9572
 *
9573
 * @param ctxt  an XML parser context
9574
 */
9575
9576
static void
9577
0
xmlParseContentInternal(xmlParserCtxtPtr ctxt) {
9578
0
    int oldNameNr = ctxt->nameNr;
9579
0
    int oldSpaceNr = ctxt->spaceNr;
9580
0
    int oldNodeNr = ctxt->nodeNr;
9581
9582
0
    GROW;
9583
0
    while ((ctxt->input->cur < ctxt->input->end) &&
9584
0
     (PARSER_STOPPED(ctxt) == 0)) {
9585
0
  const xmlChar *cur = ctxt->input->cur;
9586
9587
  /*
9588
   * First case : a Processing Instruction.
9589
   */
9590
0
  if ((*cur == '<') && (cur[1] == '?')) {
9591
0
      xmlParsePI(ctxt);
9592
0
  }
9593
9594
  /*
9595
   * Second case : a CDSection
9596
   */
9597
  /* 2.6.0 test was *cur not RAW */
9598
0
  else if (CMP9(CUR_PTR, '<', '!', '[', 'C', 'D', 'A', 'T', 'A', '[')) {
9599
0
      xmlParseCDSect(ctxt);
9600
0
  }
9601
9602
  /*
9603
   * Third case :  a comment
9604
   */
9605
0
  else if ((*cur == '<') && (NXT(1) == '!') &&
9606
0
     (NXT(2) == '-') && (NXT(3) == '-')) {
9607
0
      xmlParseComment(ctxt);
9608
0
  }
9609
9610
  /*
9611
   * Fourth case :  a sub-element.
9612
   */
9613
0
  else if (*cur == '<') {
9614
0
            if (NXT(1) == '/') {
9615
0
                if (ctxt->nameNr <= oldNameNr)
9616
0
                    break;
9617
0
          xmlParseElementEnd(ctxt);
9618
0
            } else {
9619
0
          xmlParseElementStart(ctxt);
9620
0
            }
9621
0
  }
9622
9623
  /*
9624
   * Fifth case : a reference. If if has not been resolved,
9625
   *    parsing returns it's Name, create the node
9626
   */
9627
9628
0
  else if (*cur == '&') {
9629
0
      xmlParseReference(ctxt);
9630
0
  }
9631
9632
  /*
9633
   * Last case, text. Note that References are handled directly.
9634
   */
9635
0
  else {
9636
0
      xmlParseCharDataInternal(ctxt, 0);
9637
0
  }
9638
9639
0
  SHRINK;
9640
0
  GROW;
9641
0
    }
9642
9643
0
    if ((ctxt->nameNr > oldNameNr) &&
9644
0
        (ctxt->input->cur >= ctxt->input->end) &&
9645
0
        (ctxt->wellFormed)) {
9646
0
        const xmlChar *name = ctxt->nameTab[ctxt->nameNr - 1];
9647
0
        int line = ctxt->pushTab[ctxt->nameNr - 1].line;
9648
0
        xmlFatalErrMsgStrIntStr(ctxt, XML_ERR_TAG_NOT_FINISHED,
9649
0
                "Premature end of data in tag %s line %d\n",
9650
0
                name, line, NULL);
9651
0
    }
9652
9653
    /*
9654
     * Clean up in error case
9655
     */
9656
9657
0
    while (ctxt->nodeNr > oldNodeNr)
9658
0
        nodePop(ctxt);
9659
9660
0
    while (ctxt->nameNr > oldNameNr) {
9661
0
        xmlStartTag *tag = &ctxt->pushTab[ctxt->nameNr - 1];
9662
9663
0
        if (tag->nsNr != 0)
9664
0
            xmlParserNsPop(ctxt, tag->nsNr);
9665
9666
0
        namePop(ctxt);
9667
0
    }
9668
9669
0
    while (ctxt->spaceNr > oldSpaceNr)
9670
0
        spacePop(ctxt);
9671
0
}
9672
9673
/**
9674
 * Parse XML element content. This is useful if you're only interested
9675
 * in custom SAX callbacks. If you want a node list, use
9676
 * #xmlCtxtParseContent.
9677
 *
9678
 * @param ctxt  an XML parser context
9679
 */
9680
void
9681
0
xmlParseContent(xmlParserCtxt *ctxt) {
9682
0
    if ((ctxt == NULL) || (ctxt->input == NULL))
9683
0
        return;
9684
9685
0
    xmlCtxtInitializeLate(ctxt);
9686
9687
0
    xmlParseContentInternal(ctxt);
9688
9689
0
    xmlParserCheckEOF(ctxt, XML_ERR_NOT_WELL_BALANCED);
9690
0
}
9691
9692
/**
9693
 * Parse an XML element
9694
 *
9695
 * @deprecated Internal function, don't use.
9696
 *
9697
 *     [39] element ::= EmptyElemTag | STag content ETag
9698
 *
9699
 * [ WFC: Element Type Match ]
9700
 * The Name in an element's end-tag must match the element type in the
9701
 * start-tag.
9702
 *
9703
 * @param ctxt  an XML parser context
9704
 */
9705
9706
void
9707
0
xmlParseElement(xmlParserCtxt *ctxt) {
9708
0
    if (xmlParseElementStart(ctxt) != 0)
9709
0
        return;
9710
9711
0
    xmlParseContentInternal(ctxt);
9712
9713
0
    if (ctxt->input->cur >= ctxt->input->end) {
9714
0
        if (ctxt->wellFormed) {
9715
0
            const xmlChar *name = ctxt->nameTab[ctxt->nameNr - 1];
9716
0
            int line = ctxt->pushTab[ctxt->nameNr - 1].line;
9717
0
            xmlFatalErrMsgStrIntStr(ctxt, XML_ERR_TAG_NOT_FINISHED,
9718
0
                    "Premature end of data in tag %s line %d\n",
9719
0
                    name, line, NULL);
9720
0
        }
9721
0
        return;
9722
0
    }
9723
9724
0
    xmlParseElementEnd(ctxt);
9725
0
}
9726
9727
/**
9728
 * Parse the start of an XML element. Returns -1 in case of error, 0 if an
9729
 * opening tag was parsed, 1 if an empty element was parsed.
9730
 *
9731
 * Always consumes '<'.
9732
 *
9733
 * @param ctxt  an XML parser context
9734
 */
9735
static int
9736
0
xmlParseElementStart(xmlParserCtxtPtr ctxt) {
9737
0
    int maxDepth = (ctxt->options & XML_PARSE_HUGE) ? 2048 : 256;
9738
0
    const xmlChar *name;
9739
0
    const xmlChar *prefix = NULL;
9740
0
    const xmlChar *URI = NULL;
9741
0
    xmlParserNodeInfo node_info;
9742
0
    int line;
9743
0
    xmlNodePtr cur;
9744
0
    int nbNs = 0;
9745
9746
0
    if (ctxt->nameNr > maxDepth) {
9747
0
        xmlFatalErrMsgInt(ctxt, XML_ERR_RESOURCE_LIMIT,
9748
0
                "Excessive depth in document: %d use XML_PARSE_HUGE option\n",
9749
0
                ctxt->nameNr);
9750
0
  return(-1);
9751
0
    }
9752
9753
    /* Capture start position */
9754
0
    if (ctxt->record_info) {
9755
0
        node_info.begin_pos = ctxt->input->consumed +
9756
0
                          (CUR_PTR - ctxt->input->base);
9757
0
  node_info.begin_line = ctxt->input->line;
9758
0
    }
9759
9760
0
    if (ctxt->spaceNr == 0)
9761
0
  spacePush(ctxt, -1);
9762
0
    else if (*ctxt->space == -2)
9763
0
  spacePush(ctxt, -1);
9764
0
    else
9765
0
  spacePush(ctxt, *ctxt->space);
9766
9767
0
    line = ctxt->input->line;
9768
0
#ifdef LIBXML_SAX1_ENABLED
9769
0
    if (ctxt->sax2)
9770
0
#endif /* LIBXML_SAX1_ENABLED */
9771
0
        name = xmlParseStartTag2(ctxt, &prefix, &URI, &nbNs);
9772
0
#ifdef LIBXML_SAX1_ENABLED
9773
0
    else
9774
0
  name = xmlParseStartTag(ctxt);
9775
0
#endif /* LIBXML_SAX1_ENABLED */
9776
0
    if (name == NULL) {
9777
0
  spacePop(ctxt);
9778
0
        return(-1);
9779
0
    }
9780
0
    nameNsPush(ctxt, name, prefix, URI, line, nbNs);
9781
0
    cur = ctxt->node;
9782
9783
0
#ifdef LIBXML_VALID_ENABLED
9784
    /*
9785
     * [ VC: Root Element Type ]
9786
     * The Name in the document type declaration must match the element
9787
     * type of the root element.
9788
     */
9789
0
    if (ctxt->validate && ctxt->wellFormed && ctxt->myDoc &&
9790
0
        ctxt->node && (ctxt->node == ctxt->myDoc->children))
9791
0
        ctxt->valid &= xmlValidateRoot(&ctxt->vctxt, ctxt->myDoc);
9792
0
#endif /* LIBXML_VALID_ENABLED */
9793
9794
    /*
9795
     * Check for an Empty Element.
9796
     */
9797
0
    if ((RAW == '/') && (NXT(1) == '>')) {
9798
0
        SKIP(2);
9799
0
  if (ctxt->sax2) {
9800
0
      if ((ctxt->sax != NULL) && (ctxt->sax->endElementNs != NULL) &&
9801
0
    (!ctxt->disableSAX))
9802
0
    ctxt->sax->endElementNs(ctxt->userData, name, prefix, URI);
9803
0
#ifdef LIBXML_SAX1_ENABLED
9804
0
  } else {
9805
0
      if ((ctxt->sax != NULL) && (ctxt->sax->endElement != NULL) &&
9806
0
    (!ctxt->disableSAX))
9807
0
    ctxt->sax->endElement(ctxt->userData, name);
9808
0
#endif /* LIBXML_SAX1_ENABLED */
9809
0
  }
9810
0
  namePop(ctxt);
9811
0
  spacePop(ctxt);
9812
0
  if (nbNs > 0)
9813
0
      xmlParserNsPop(ctxt, nbNs);
9814
0
  if (cur != NULL && ctxt->record_info) {
9815
0
            node_info.node = cur;
9816
0
            node_info.end_pos = ctxt->input->consumed +
9817
0
                                (CUR_PTR - ctxt->input->base);
9818
0
            node_info.end_line = ctxt->input->line;
9819
0
            xmlParserAddNodeInfo(ctxt, &node_info);
9820
0
  }
9821
0
  return(1);
9822
0
    }
9823
0
    if (RAW == '>') {
9824
0
        NEXT1;
9825
0
        if (cur != NULL && ctxt->record_info) {
9826
0
            node_info.node = cur;
9827
0
            node_info.end_pos = 0;
9828
0
            node_info.end_line = 0;
9829
0
            xmlParserAddNodeInfo(ctxt, &node_info);
9830
0
        }
9831
0
    } else {
9832
0
        xmlFatalErrMsgStrIntStr(ctxt, XML_ERR_GT_REQUIRED,
9833
0
         "Couldn't find end of Start Tag %s line %d\n",
9834
0
                    name, line, NULL);
9835
9836
  /*
9837
   * end of parsing of this node.
9838
   */
9839
0
  nodePop(ctxt);
9840
0
  namePop(ctxt);
9841
0
  spacePop(ctxt);
9842
0
  if (nbNs > 0)
9843
0
      xmlParserNsPop(ctxt, nbNs);
9844
0
  return(-1);
9845
0
    }
9846
9847
0
    return(0);
9848
0
}
9849
9850
/**
9851
 * Parse the end of an XML element. Always consumes '</'.
9852
 *
9853
 * @param ctxt  an XML parser context
9854
 */
9855
static void
9856
0
xmlParseElementEnd(xmlParserCtxtPtr ctxt) {
9857
0
    xmlNodePtr cur = ctxt->node;
9858
9859
0
    if (ctxt->nameNr <= 0) {
9860
0
        if ((RAW == '<') && (NXT(1) == '/'))
9861
0
            SKIP(2);
9862
0
        return;
9863
0
    }
9864
9865
    /*
9866
     * parse the end of tag: '</' should be here.
9867
     */
9868
0
    if (ctxt->sax2) {
9869
0
  xmlParseEndTag2(ctxt, &ctxt->pushTab[ctxt->nameNr - 1]);
9870
0
  namePop(ctxt);
9871
0
    }
9872
0
#ifdef LIBXML_SAX1_ENABLED
9873
0
    else
9874
0
  xmlParseEndTag1(ctxt, 0);
9875
0
#endif /* LIBXML_SAX1_ENABLED */
9876
9877
    /*
9878
     * Capture end position
9879
     */
9880
0
    if (cur != NULL && ctxt->record_info) {
9881
0
        xmlParserNodeInfoPtr node_info;
9882
9883
0
        node_info = (xmlParserNodeInfoPtr) xmlParserFindNodeInfo(ctxt, cur);
9884
0
        if (node_info != NULL) {
9885
0
            node_info->end_pos = ctxt->input->consumed +
9886
0
                                 (CUR_PTR - ctxt->input->base);
9887
0
            node_info->end_line = ctxt->input->line;
9888
0
        }
9889
0
    }
9890
0
}
9891
9892
/**
9893
 * Parse the XML version value.
9894
 *
9895
 * @deprecated Internal function, don't use.
9896
 *
9897
 *     [26] VersionNum ::= '1.' [0-9]+
9898
 *
9899
 * In practice allow [0-9].[0-9]+ at that level
9900
 *
9901
 * @param ctxt  an XML parser context
9902
 * @returns the string giving the XML version number, or NULL
9903
 */
9904
xmlChar *
9905
0
xmlParseVersionNum(xmlParserCtxt *ctxt) {
9906
0
    xmlChar *buf = NULL;
9907
0
    int len = 0;
9908
0
    int size = 10;
9909
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
9910
0
                    XML_MAX_TEXT_LENGTH :
9911
0
                    XML_MAX_NAME_LENGTH;
9912
0
    xmlChar cur;
9913
9914
0
    buf = xmlMalloc(size);
9915
0
    if (buf == NULL) {
9916
0
  xmlErrMemory(ctxt);
9917
0
  return(NULL);
9918
0
    }
9919
0
    cur = CUR;
9920
0
    if (!((cur >= '0') && (cur <= '9'))) {
9921
0
  xmlFree(buf);
9922
0
  return(NULL);
9923
0
    }
9924
0
    buf[len++] = cur;
9925
0
    NEXT;
9926
0
    cur=CUR;
9927
0
    if (cur != '.') {
9928
0
  xmlFree(buf);
9929
0
  return(NULL);
9930
0
    }
9931
0
    buf[len++] = cur;
9932
0
    NEXT;
9933
0
    cur=CUR;
9934
0
    while ((cur >= '0') && (cur <= '9')) {
9935
0
  if (len + 1 >= size) {
9936
0
      xmlChar *tmp;
9937
0
            int newSize;
9938
9939
0
            newSize = xmlGrowCapacity(size, 1, 1, maxLength);
9940
0
            if (newSize < 0) {
9941
0
                xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "VersionNum");
9942
0
                xmlFree(buf);
9943
0
                return(NULL);
9944
0
            }
9945
0
      tmp = xmlRealloc(buf, newSize);
9946
0
      if (tmp == NULL) {
9947
0
    xmlErrMemory(ctxt);
9948
0
          xmlFree(buf);
9949
0
    return(NULL);
9950
0
      }
9951
0
      buf = tmp;
9952
0
            size = newSize;
9953
0
  }
9954
0
  buf[len++] = cur;
9955
0
  NEXT;
9956
0
  cur=CUR;
9957
0
    }
9958
0
    buf[len] = 0;
9959
0
    return(buf);
9960
0
}
9961
9962
/**
9963
 * Parse the XML version.
9964
 *
9965
 * @deprecated Internal function, don't use.
9966
 *
9967
 *     [24] VersionInfo ::= S 'version' Eq (' VersionNum ' | " VersionNum ")
9968
 *
9969
 *     [25] Eq ::= S? '=' S?
9970
 *
9971
 * @param ctxt  an XML parser context
9972
 * @returns the version string, e.g. "1.0"
9973
 */
9974
9975
xmlChar *
9976
0
xmlParseVersionInfo(xmlParserCtxt *ctxt) {
9977
0
    xmlChar *version = NULL;
9978
9979
0
    if (CMP7(CUR_PTR, 'v', 'e', 'r', 's', 'i', 'o', 'n')) {
9980
0
  SKIP(7);
9981
0
  SKIP_BLANKS;
9982
0
  if (RAW != '=') {
9983
0
      xmlFatalErr(ctxt, XML_ERR_EQUAL_REQUIRED, NULL);
9984
0
      return(NULL);
9985
0
        }
9986
0
  NEXT;
9987
0
  SKIP_BLANKS;
9988
0
  if (RAW == '"') {
9989
0
      NEXT;
9990
0
      version = xmlParseVersionNum(ctxt);
9991
0
      if (RAW != '"') {
9992
0
    xmlFatalErr(ctxt, XML_ERR_STRING_NOT_CLOSED, NULL);
9993
0
      } else
9994
0
          NEXT;
9995
0
  } else if (RAW == '\''){
9996
0
      NEXT;
9997
0
      version = xmlParseVersionNum(ctxt);
9998
0
      if (RAW != '\'') {
9999
0
    xmlFatalErr(ctxt, XML_ERR_STRING_NOT_CLOSED, NULL);
10000
0
      } else
10001
0
          NEXT;
10002
0
  } else {
10003
0
      xmlFatalErr(ctxt, XML_ERR_STRING_NOT_STARTED, NULL);
10004
0
  }
10005
0
    }
10006
0
    return(version);
10007
0
}
10008
10009
/**
10010
 * Parse the XML encoding name
10011
 *
10012
 * @deprecated Internal function, don't use.
10013
 *
10014
 *     [81] EncName ::= [A-Za-z] ([A-Za-z0-9._] | '-')*
10015
 *
10016
 * @param ctxt  an XML parser context
10017
 * @returns the encoding name value or NULL
10018
 */
10019
xmlChar *
10020
0
xmlParseEncName(xmlParserCtxt *ctxt) {
10021
0
    xmlChar *buf = NULL;
10022
0
    int len = 0;
10023
0
    int size = 10;
10024
0
    int maxLength = (ctxt->options & XML_PARSE_HUGE) ?
10025
0
                    XML_MAX_TEXT_LENGTH :
10026
0
                    XML_MAX_NAME_LENGTH;
10027
0
    xmlChar cur;
10028
10029
0
    cur = CUR;
10030
0
    if (((cur >= 'a') && (cur <= 'z')) ||
10031
0
        ((cur >= 'A') && (cur <= 'Z'))) {
10032
0
  buf = xmlMalloc(size);
10033
0
  if (buf == NULL) {
10034
0
      xmlErrMemory(ctxt);
10035
0
      return(NULL);
10036
0
  }
10037
10038
0
  buf[len++] = cur;
10039
0
  NEXT;
10040
0
  cur = CUR;
10041
0
  while (((cur >= 'a') && (cur <= 'z')) ||
10042
0
         ((cur >= 'A') && (cur <= 'Z')) ||
10043
0
         ((cur >= '0') && (cur <= '9')) ||
10044
0
         (cur == '.') || (cur == '_') ||
10045
0
         (cur == '-')) {
10046
0
      if (len + 1 >= size) {
10047
0
          xmlChar *tmp;
10048
0
                int newSize;
10049
10050
0
                newSize = xmlGrowCapacity(size, 1, 1, maxLength);
10051
0
                if (newSize < 0) {
10052
0
                    xmlFatalErr(ctxt, XML_ERR_NAME_TOO_LONG, "EncName");
10053
0
                    xmlFree(buf);
10054
0
                    return(NULL);
10055
0
                }
10056
0
    tmp = xmlRealloc(buf, newSize);
10057
0
    if (tmp == NULL) {
10058
0
        xmlErrMemory(ctxt);
10059
0
        xmlFree(buf);
10060
0
        return(NULL);
10061
0
    }
10062
0
    buf = tmp;
10063
0
                size = newSize;
10064
0
      }
10065
0
      buf[len++] = cur;
10066
0
      NEXT;
10067
0
      cur = CUR;
10068
0
        }
10069
0
  buf[len] = 0;
10070
0
    } else {
10071
0
  xmlFatalErr(ctxt, XML_ERR_ENCODING_NAME, NULL);
10072
0
    }
10073
0
    return(buf);
10074
0
}
10075
10076
/**
10077
 * Parse the XML encoding declaration
10078
 *
10079
 * @deprecated Internal function, don't use.
10080
 *
10081
 *     [80] EncodingDecl ::= S 'encoding' Eq ('"' EncName '"' | 
10082
 *                           "'" EncName "'")
10083
 *
10084
 * this setups the conversion filters.
10085
 *
10086
 * @param ctxt  an XML parser context
10087
 * @returns the encoding value or NULL
10088
 */
10089
10090
const xmlChar *
10091
0
xmlParseEncodingDecl(xmlParserCtxt *ctxt) {
10092
0
    xmlChar *encoding = NULL;
10093
10094
0
    SKIP_BLANKS;
10095
0
    if (CMP8(CUR_PTR, 'e', 'n', 'c', 'o', 'd', 'i', 'n', 'g') == 0)
10096
0
        return(NULL);
10097
10098
0
    SKIP(8);
10099
0
    SKIP_BLANKS;
10100
0
    if (RAW != '=') {
10101
0
        xmlFatalErr(ctxt, XML_ERR_EQUAL_REQUIRED, NULL);
10102
0
        return(NULL);
10103
0
    }
10104
0
    NEXT;
10105
0
    SKIP_BLANKS;
10106
0
    if (RAW == '"') {
10107
0
        NEXT;
10108
0
        encoding = xmlParseEncName(ctxt);
10109
0
        if (RAW != '"') {
10110
0
            xmlFatalErr(ctxt, XML_ERR_STRING_NOT_CLOSED, NULL);
10111
0
            xmlFree(encoding);
10112
0
            return(NULL);
10113
0
        } else
10114
0
            NEXT;
10115
0
    } else if (RAW == '\''){
10116
0
        NEXT;
10117
0
        encoding = xmlParseEncName(ctxt);
10118
0
        if (RAW != '\'') {
10119
0
            xmlFatalErr(ctxt, XML_ERR_STRING_NOT_CLOSED, NULL);
10120
0
            xmlFree(encoding);
10121
0
            return(NULL);
10122
0
        } else
10123
0
            NEXT;
10124
0
    } else {
10125
0
        xmlFatalErr(ctxt, XML_ERR_STRING_NOT_STARTED, NULL);
10126
0
    }
10127
10128
0
    if (encoding == NULL)
10129
0
        return(NULL);
10130
10131
0
    xmlSetDeclaredEncoding(ctxt, encoding);
10132
10133
0
    return(ctxt->encoding);
10134
0
}
10135
10136
/**
10137
 * Parse the XML standalone declaration
10138
 *
10139
 * @deprecated Internal function, don't use.
10140
 *
10141
 *     [32] SDDecl ::= S 'standalone' Eq
10142
 *                     (("'" ('yes' | 'no') "'") | ('"' ('yes' | 'no')'"'))
10143
 *
10144
 * [ VC: Standalone Document Declaration ]
10145
 * TODO The standalone document declaration must have the value "no"
10146
 * if any external markup declarations contain declarations of:
10147
 *  - attributes with default values, if elements to which these
10148
 *    attributes apply appear in the document without specifications
10149
 *    of values for these attributes, or
10150
 *  - entities (other than amp, lt, gt, apos, quot), if references
10151
 *    to those entities appear in the document, or
10152
 *  - attributes with values subject to normalization, where the
10153
 *    attribute appears in the document with a value which will change
10154
 *    as a result of normalization, or
10155
 *  - element types with element content, if white space occurs directly
10156
 *    within any instance of those types.
10157
 *
10158
 * @param ctxt  an XML parser context
10159
 * @returns
10160
 *   1 if standalone="yes"
10161
 *   0 if standalone="no"
10162
 *  -2 if standalone attribute is missing or invalid
10163
 *    (A standalone value of -2 means that the XML declaration was found,
10164
 *     but no value was specified for the standalone attribute).
10165
 */
10166
10167
int
10168
0
xmlParseSDDecl(xmlParserCtxt *ctxt) {
10169
0
    int standalone = -2;
10170
10171
0
    SKIP_BLANKS;
10172
0
    if (CMP10(CUR_PTR, 's', 't', 'a', 'n', 'd', 'a', 'l', 'o', 'n', 'e')) {
10173
0
  SKIP(10);
10174
0
        SKIP_BLANKS;
10175
0
  if (RAW != '=') {
10176
0
      xmlFatalErr(ctxt, XML_ERR_EQUAL_REQUIRED, NULL);
10177
0
      return(standalone);
10178
0
        }
10179
0
  NEXT;
10180
0
  SKIP_BLANKS;
10181
0
        if (RAW == '\''){
10182
0
      NEXT;
10183
0
      if ((RAW == 'n') && (NXT(1) == 'o')) {
10184
0
          standalone = 0;
10185
0
                SKIP(2);
10186
0
      } else if ((RAW == 'y') && (NXT(1) == 'e') &&
10187
0
                 (NXT(2) == 's')) {
10188
0
          standalone = 1;
10189
0
    SKIP(3);
10190
0
            } else {
10191
0
    xmlFatalErr(ctxt, XML_ERR_STANDALONE_VALUE, NULL);
10192
0
      }
10193
0
      if (RAW != '\'') {
10194
0
    xmlFatalErr(ctxt, XML_ERR_STRING_NOT_CLOSED, NULL);
10195
0
      } else
10196
0
          NEXT;
10197
0
  } else if (RAW == '"'){
10198
0
      NEXT;
10199
0
      if ((RAW == 'n') && (NXT(1) == 'o')) {
10200
0
          standalone = 0;
10201
0
    SKIP(2);
10202
0
      } else if ((RAW == 'y') && (NXT(1) == 'e') &&
10203
0
                 (NXT(2) == 's')) {
10204
0
          standalone = 1;
10205
0
                SKIP(3);
10206
0
            } else {
10207
0
    xmlFatalErr(ctxt, XML_ERR_STANDALONE_VALUE, NULL);
10208
0
      }
10209
0
      if (RAW != '"') {
10210
0
    xmlFatalErr(ctxt, XML_ERR_STRING_NOT_CLOSED, NULL);
10211
0
      } else
10212
0
          NEXT;
10213
0
  } else {
10214
0
      xmlFatalErr(ctxt, XML_ERR_STRING_NOT_STARTED, NULL);
10215
0
        }
10216
0
    }
10217
0
    return(standalone);
10218
0
}
10219
10220
/**
10221
 * Parse an XML declaration header
10222
 *
10223
 * @deprecated Internal function, don't use.
10224
 *
10225
 *     [23] XMLDecl ::= '<?xml' VersionInfo EncodingDecl? SDDecl? S? '?>'
10226
 * @param ctxt  an XML parser context
10227
 */
10228
10229
void
10230
0
xmlParseXMLDecl(xmlParserCtxt *ctxt) {
10231
0
    xmlChar *version;
10232
10233
    /*
10234
     * This value for standalone indicates that the document has an
10235
     * XML declaration but it does not have a standalone attribute.
10236
     * It will be overwritten later if a standalone attribute is found.
10237
     */
10238
10239
0
    ctxt->standalone = -2;
10240
10241
    /*
10242
     * We know that '<?xml' is here.
10243
     */
10244
0
    SKIP(5);
10245
10246
0
    if (!IS_BLANK_CH(RAW)) {
10247
0
  xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED,
10248
0
                 "Blank needed after '<?xml'\n");
10249
0
    }
10250
0
    SKIP_BLANKS;
10251
10252
    /*
10253
     * We must have the VersionInfo here.
10254
     */
10255
0
    version = xmlParseVersionInfo(ctxt);
10256
0
    if (version == NULL) {
10257
0
  xmlFatalErr(ctxt, XML_ERR_VERSION_MISSING, NULL);
10258
0
    } else {
10259
0
  if (!xmlStrEqual(version, (const xmlChar *) XML_DEFAULT_VERSION)) {
10260
      /*
10261
       * Changed here for XML-1.0 5th edition
10262
       */
10263
0
      if (ctxt->options & XML_PARSE_OLD10) {
10264
0
    xmlFatalErrMsgStr(ctxt, XML_ERR_UNKNOWN_VERSION,
10265
0
                "Unsupported version '%s'\n",
10266
0
                version);
10267
0
      } else {
10268
0
          if ((version[0] == '1') && ((version[1] == '.'))) {
10269
0
        xmlWarningMsg(ctxt, XML_WAR_UNKNOWN_VERSION,
10270
0
                      "Unsupported version '%s'\n",
10271
0
          version, NULL);
10272
0
    } else {
10273
0
        xmlFatalErrMsgStr(ctxt, XML_ERR_UNKNOWN_VERSION,
10274
0
              "Unsupported version '%s'\n",
10275
0
              version);
10276
0
    }
10277
0
      }
10278
0
  }
10279
0
  if (ctxt->version != NULL)
10280
0
      xmlFree(ctxt->version);
10281
0
  ctxt->version = version;
10282
0
    }
10283
10284
    /*
10285
     * We may have the encoding declaration
10286
     */
10287
0
    if (!IS_BLANK_CH(RAW)) {
10288
0
        if ((RAW == '?') && (NXT(1) == '>')) {
10289
0
      SKIP(2);
10290
0
      return;
10291
0
  }
10292
0
  xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED, "Blank needed here\n");
10293
0
    }
10294
0
    xmlParseEncodingDecl(ctxt);
10295
10296
    /*
10297
     * We may have the standalone status.
10298
     */
10299
0
    if ((ctxt->encoding != NULL) && (!IS_BLANK_CH(RAW))) {
10300
0
        if ((RAW == '?') && (NXT(1) == '>')) {
10301
0
      SKIP(2);
10302
0
      return;
10303
0
  }
10304
0
  xmlFatalErrMsg(ctxt, XML_ERR_SPACE_REQUIRED, "Blank needed here\n");
10305
0
    }
10306
10307
    /*
10308
     * We can grow the input buffer freely at that point
10309
     */
10310
0
    GROW;
10311
10312
0
    SKIP_BLANKS;
10313
0
    ctxt->standalone = xmlParseSDDecl(ctxt);
10314
10315
0
    SKIP_BLANKS;
10316
0
    if ((RAW == '?') && (NXT(1) == '>')) {
10317
0
        SKIP(2);
10318
0
    } else if (RAW == '>') {
10319
        /* Deprecated old WD ... */
10320
0
  xmlFatalErr(ctxt, XML_ERR_XMLDECL_NOT_FINISHED, NULL);
10321
0
  NEXT;
10322
0
    } else {
10323
0
        int c;
10324
10325
0
  xmlFatalErr(ctxt, XML_ERR_XMLDECL_NOT_FINISHED, NULL);
10326
0
        while ((PARSER_STOPPED(ctxt) == 0) &&
10327
0
               ((c = CUR) != 0)) {
10328
0
            NEXT;
10329
0
            if (c == '>')
10330
0
                break;
10331
0
        }
10332
0
    }
10333
0
}
10334
10335
/**
10336
 * @since 2.14.0
10337
 *
10338
 * @param ctxt  parser context
10339
 * @returns the version from the XML declaration.
10340
 */
10341
const xmlChar *
10342
0
xmlCtxtGetVersion(xmlParserCtxt *ctxt) {
10343
0
    if (ctxt == NULL)
10344
0
        return(NULL);
10345
10346
0
    return(ctxt->version);
10347
0
}
10348
10349
/**
10350
 * @since 2.14.0
10351
 *
10352
 * @param ctxt  parser context
10353
 * @returns the value from the standalone document declaration.
10354
 */
10355
int
10356
0
xmlCtxtGetStandalone(xmlParserCtxt *ctxt) {
10357
0
    if (ctxt == NULL)
10358
0
        return(0);
10359
10360
0
    return(ctxt->standalone);
10361
0
}
10362
10363
/**
10364
 * Parse an XML Misc* optional field.
10365
 *
10366
 * @deprecated Internal function, don't use.
10367
 *
10368
 *     [27] Misc ::= Comment | PI |  S
10369
 * @param ctxt  an XML parser context
10370
 */
10371
10372
void
10373
0
xmlParseMisc(xmlParserCtxt *ctxt) {
10374
0
    while (PARSER_STOPPED(ctxt) == 0) {
10375
0
        SKIP_BLANKS;
10376
0
        GROW;
10377
0
        if ((RAW == '<') && (NXT(1) == '?')) {
10378
0
      xmlParsePI(ctxt);
10379
0
        } else if (CMP4(CUR_PTR, '<', '!', '-', '-')) {
10380
0
      xmlParseComment(ctxt);
10381
0
        } else {
10382
0
            break;
10383
0
        }
10384
0
    }
10385
0
}
10386
10387
static void
10388
0
xmlFinishDocument(xmlParserCtxtPtr ctxt) {
10389
0
    xmlDocPtr doc;
10390
10391
    /*
10392
     * SAX: end of the document processing.
10393
     */
10394
0
    if ((ctxt->sax) && (ctxt->sax->endDocument != NULL))
10395
0
        ctxt->sax->endDocument(ctxt->userData);
10396
10397
    /*
10398
     * Remove locally kept entity definitions if the tree was not built
10399
     */
10400
0
    doc = ctxt->myDoc;
10401
0
    if ((doc != NULL) &&
10402
0
        (xmlStrEqual(doc->version, SAX_COMPAT_MODE))) {
10403
0
        xmlFreeDoc(doc);
10404
0
        ctxt->myDoc = NULL;
10405
0
    }
10406
0
}
10407
10408
/**
10409
 * Parse an XML document and invoke the SAX handlers. This is useful
10410
 * if you're only interested in custom SAX callbacks. If you want a
10411
 * document tree, use #xmlCtxtParseDocument.
10412
 *
10413
 * @param ctxt  an XML parser context
10414
 * @returns 0, -1 in case of error.
10415
 */
10416
10417
int
10418
0
xmlParseDocument(xmlParserCtxt *ctxt) {
10419
0
    if ((ctxt == NULL) || (ctxt->input == NULL))
10420
0
        return(-1);
10421
10422
0
    GROW;
10423
10424
    /*
10425
     * SAX: detecting the level.
10426
     */
10427
0
    xmlCtxtInitializeLate(ctxt);
10428
10429
0
    if ((ctxt->sax) && (ctxt->sax->setDocumentLocator)) {
10430
0
        ctxt->sax->setDocumentLocator(ctxt->userData,
10431
0
                (xmlSAXLocator *) &xmlDefaultSAXLocator);
10432
0
    }
10433
10434
0
    xmlDetectEncoding(ctxt);
10435
10436
0
    if (CUR == 0) {
10437
0
  xmlFatalErr(ctxt, XML_ERR_DOCUMENT_EMPTY, NULL);
10438
0
  return(-1);
10439
0
    }
10440
10441
0
    GROW;
10442
0
    if ((CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) && (IS_BLANK_CH(NXT(5)))) {
10443
10444
  /*
10445
   * Note that we will switch encoding on the fly.
10446
   */
10447
0
  xmlParseXMLDecl(ctxt);
10448
0
  SKIP_BLANKS;
10449
0
    } else {
10450
0
  ctxt->version = xmlCharStrdup(XML_DEFAULT_VERSION);
10451
0
        if (ctxt->version == NULL) {
10452
0
            xmlErrMemory(ctxt);
10453
0
            return(-1);
10454
0
        }
10455
0
    }
10456
0
    if ((ctxt->sax) && (ctxt->sax->startDocument) && (!ctxt->disableSAX))
10457
0
        ctxt->sax->startDocument(ctxt->userData);
10458
0
    if ((ctxt->myDoc != NULL) && (ctxt->input != NULL) &&
10459
0
        (ctxt->input->buf != NULL) && (ctxt->input->buf->compressed >= 0)) {
10460
0
  ctxt->myDoc->compression = ctxt->input->buf->compressed;
10461
0
    }
10462
10463
    /*
10464
     * The Misc part of the Prolog
10465
     */
10466
0
    xmlParseMisc(ctxt);
10467
10468
    /*
10469
     * Then possibly doc type declaration(s) and more Misc
10470
     * (doctypedecl Misc*)?
10471
     */
10472
0
    GROW;
10473
0
    if (CMP9(CUR_PTR, '<', '!', 'D', 'O', 'C', 'T', 'Y', 'P', 'E')) {
10474
10475
0
  ctxt->inSubset = 1;
10476
0
  xmlParseDocTypeDecl(ctxt);
10477
0
  if (RAW == '[') {
10478
0
      xmlParseInternalSubset(ctxt);
10479
0
  } else if (RAW == '>') {
10480
0
            NEXT;
10481
0
        }
10482
10483
  /*
10484
   * Create and update the external subset.
10485
   */
10486
0
  ctxt->inSubset = 2;
10487
0
  if ((ctxt->sax != NULL) && (ctxt->sax->externalSubset != NULL) &&
10488
0
      (!ctxt->disableSAX))
10489
0
      ctxt->sax->externalSubset(ctxt->userData, ctxt->intSubName,
10490
0
                                ctxt->extSubSystem, ctxt->extSubURI);
10491
0
  ctxt->inSubset = 0;
10492
10493
0
        xmlCleanSpecialAttr(ctxt);
10494
10495
0
  xmlParseMisc(ctxt);
10496
0
    }
10497
10498
    /*
10499
     * Time to start parsing the tree itself
10500
     */
10501
0
    GROW;
10502
0
    if (RAW != '<') {
10503
0
        if (ctxt->wellFormed)
10504
0
            xmlFatalErrMsg(ctxt, XML_ERR_DOCUMENT_EMPTY,
10505
0
                           "Start tag expected, '<' not found\n");
10506
0
    } else {
10507
0
  xmlParseElement(ctxt);
10508
10509
  /*
10510
   * The Misc part at the end
10511
   */
10512
0
  xmlParseMisc(ctxt);
10513
10514
0
        xmlParserCheckEOF(ctxt, XML_ERR_DOCUMENT_END);
10515
0
    }
10516
10517
0
    ctxt->instate = XML_PARSER_EOF;
10518
0
    xmlFinishDocument(ctxt);
10519
10520
0
    if (! ctxt->wellFormed) {
10521
0
  ctxt->valid = 0;
10522
0
  return(-1);
10523
0
    }
10524
10525
0
    return(0);
10526
0
}
10527
10528
/**
10529
 * Parse a general parsed entity
10530
 * An external general parsed entity is well-formed if it matches the
10531
 * production labeled extParsedEnt.
10532
 *
10533
 * @deprecated Internal function, don't use.
10534
 *
10535
 *     [78] extParsedEnt ::= TextDecl? content
10536
 *
10537
 * @param ctxt  an XML parser context
10538
 * @returns 0, -1 in case of error. the parser context is augmented
10539
 *                as a result of the parsing.
10540
 */
10541
10542
int
10543
0
xmlParseExtParsedEnt(xmlParserCtxt *ctxt) {
10544
0
    if ((ctxt == NULL) || (ctxt->input == NULL))
10545
0
        return(-1);
10546
10547
0
    xmlCtxtInitializeLate(ctxt);
10548
10549
0
    if ((ctxt->sax) && (ctxt->sax->setDocumentLocator)) {
10550
0
        ctxt->sax->setDocumentLocator(ctxt->userData,
10551
0
                (xmlSAXLocator *) &xmlDefaultSAXLocator);
10552
0
    }
10553
10554
0
    xmlDetectEncoding(ctxt);
10555
10556
0
    if (CUR == 0) {
10557
0
  xmlFatalErr(ctxt, XML_ERR_DOCUMENT_EMPTY, NULL);
10558
0
    }
10559
10560
    /*
10561
     * Check for the XMLDecl in the Prolog.
10562
     */
10563
0
    GROW;
10564
0
    if ((CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) && (IS_BLANK_CH(NXT(5)))) {
10565
10566
  /*
10567
   * Note that we will switch encoding on the fly.
10568
   */
10569
0
  xmlParseXMLDecl(ctxt);
10570
0
  SKIP_BLANKS;
10571
0
    } else {
10572
0
  ctxt->version = xmlCharStrdup(XML_DEFAULT_VERSION);
10573
0
    }
10574
0
    if ((ctxt->sax) && (ctxt->sax->startDocument) && (!ctxt->disableSAX))
10575
0
        ctxt->sax->startDocument(ctxt->userData);
10576
10577
    /*
10578
     * Doing validity checking on chunk doesn't make sense
10579
     */
10580
0
    ctxt->options &= ~XML_PARSE_DTDVALID;
10581
0
    ctxt->validate = 0;
10582
0
    ctxt->depth = 0;
10583
10584
0
    xmlParseContentInternal(ctxt);
10585
10586
0
    if (ctxt->input->cur < ctxt->input->end)
10587
0
  xmlFatalErr(ctxt, XML_ERR_NOT_WELL_BALANCED, NULL);
10588
10589
    /*
10590
     * SAX: end of the document processing.
10591
     */
10592
0
    if ((ctxt->sax) && (ctxt->sax->endDocument != NULL))
10593
0
        ctxt->sax->endDocument(ctxt->userData);
10594
10595
0
    if (! ctxt->wellFormed) return(-1);
10596
0
    return(0);
10597
0
}
10598
10599
#ifdef LIBXML_PUSH_ENABLED
10600
/************************************************************************
10601
 *                  *
10602
 *    Progressive parsing interfaces        *
10603
 *                  *
10604
 ************************************************************************/
10605
10606
/**
10607
 * Check whether the input buffer contains a character.
10608
 *
10609
 * @param ctxt  an XML parser context
10610
 * @param c  character
10611
 */
10612
static int
10613
0
xmlParseLookupChar(xmlParserCtxtPtr ctxt, int c) {
10614
0
    const xmlChar *cur;
10615
10616
0
    if (ctxt->checkIndex == 0) {
10617
0
        cur = ctxt->input->cur + 1;
10618
0
    } else {
10619
0
        cur = ctxt->input->cur + ctxt->checkIndex;
10620
0
    }
10621
10622
0
    if (memchr(cur, c, ctxt->input->end - cur) == NULL) {
10623
0
        size_t index = ctxt->input->end - ctxt->input->cur;
10624
10625
0
        if (index > LONG_MAX) {
10626
0
            ctxt->checkIndex = 0;
10627
0
            return(1);
10628
0
        }
10629
0
        ctxt->checkIndex = index;
10630
0
        return(0);
10631
0
    } else {
10632
0
        ctxt->checkIndex = 0;
10633
0
        return(1);
10634
0
    }
10635
0
}
10636
10637
/**
10638
 * Check whether the input buffer contains a string.
10639
 *
10640
 * @param ctxt  an XML parser context
10641
 * @param startDelta  delta to apply at the start
10642
 * @param str  string
10643
 * @param strLen  length of string
10644
 */
10645
static const xmlChar *
10646
xmlParseLookupString(xmlParserCtxtPtr ctxt, size_t startDelta,
10647
0
                     const char *str, size_t strLen) {
10648
0
    const xmlChar *cur, *term;
10649
10650
0
    if (ctxt->checkIndex == 0) {
10651
0
        cur = ctxt->input->cur + startDelta;
10652
0
    } else {
10653
0
        cur = ctxt->input->cur + ctxt->checkIndex;
10654
0
    }
10655
10656
0
    term = BAD_CAST strstr((const char *) cur, str);
10657
0
    if (term == NULL) {
10658
0
        const xmlChar *end = ctxt->input->end;
10659
0
        size_t index;
10660
10661
        /* Rescan (strLen - 1) characters. */
10662
0
        if ((size_t) (end - cur) < strLen)
10663
0
            end = cur;
10664
0
        else
10665
0
            end -= strLen - 1;
10666
0
        index = end - ctxt->input->cur;
10667
0
        if (index > LONG_MAX) {
10668
0
            ctxt->checkIndex = 0;
10669
0
            return(ctxt->input->end - strLen);
10670
0
        }
10671
0
        ctxt->checkIndex = index;
10672
0
    } else {
10673
0
        ctxt->checkIndex = 0;
10674
0
    }
10675
10676
0
    return(term);
10677
0
}
10678
10679
/**
10680
 * Check whether the input buffer contains terminated char data.
10681
 *
10682
 * @param ctxt  an XML parser context
10683
 */
10684
static int
10685
0
xmlParseLookupCharData(xmlParserCtxtPtr ctxt) {
10686
0
    const xmlChar *cur = ctxt->input->cur + ctxt->checkIndex;
10687
0
    const xmlChar *end = ctxt->input->end;
10688
0
    size_t index;
10689
10690
0
    while (cur < end) {
10691
0
        if ((*cur == '<') || (*cur == '&')) {
10692
0
            ctxt->checkIndex = 0;
10693
0
            return(1);
10694
0
        }
10695
0
        cur++;
10696
0
    }
10697
10698
0
    index = cur - ctxt->input->cur;
10699
0
    if (index > LONG_MAX) {
10700
0
        ctxt->checkIndex = 0;
10701
0
        return(1);
10702
0
    }
10703
0
    ctxt->checkIndex = index;
10704
0
    return(0);
10705
0
}
10706
10707
/**
10708
 * Check whether there's enough data in the input buffer to finish parsing
10709
 * a start tag. This has to take quotes into account.
10710
 *
10711
 * @param ctxt  an XML parser context
10712
 */
10713
static int
10714
0
xmlParseLookupGt(xmlParserCtxtPtr ctxt) {
10715
0
    const xmlChar *cur;
10716
0
    const xmlChar *end = ctxt->input->end;
10717
0
    int state = ctxt->endCheckState;
10718
0
    size_t index;
10719
10720
0
    if (ctxt->checkIndex == 0)
10721
0
        cur = ctxt->input->cur + 1;
10722
0
    else
10723
0
        cur = ctxt->input->cur + ctxt->checkIndex;
10724
10725
0
    while (cur < end) {
10726
0
        if (state) {
10727
0
            if (*cur == state)
10728
0
                state = 0;
10729
0
        } else if (*cur == '\'' || *cur == '"') {
10730
0
            state = *cur;
10731
0
        } else if (*cur == '>') {
10732
0
            ctxt->checkIndex = 0;
10733
0
            ctxt->endCheckState = 0;
10734
0
            return(1);
10735
0
        }
10736
0
        cur++;
10737
0
    }
10738
10739
0
    index = cur - ctxt->input->cur;
10740
0
    if (index > LONG_MAX) {
10741
0
        ctxt->checkIndex = 0;
10742
0
        ctxt->endCheckState = 0;
10743
0
        return(1);
10744
0
    }
10745
0
    ctxt->checkIndex = index;
10746
0
    ctxt->endCheckState = state;
10747
0
    return(0);
10748
0
}
10749
10750
/**
10751
 * Check whether there's enough data in the input buffer to finish parsing
10752
 * the internal subset.
10753
 *
10754
 * @param ctxt  an XML parser context
10755
 */
10756
static int
10757
0
xmlParseLookupInternalSubset(xmlParserCtxtPtr ctxt) {
10758
    /*
10759
     * Sorry, but progressive parsing of the internal subset is not
10760
     * supported. We first check that the full content of the internal
10761
     * subset is available and parsing is launched only at that point.
10762
     * Internal subset ends with "']' S? '>'" in an unescaped section and
10763
     * not in a ']]>' sequence which are conditional sections.
10764
     */
10765
0
    const xmlChar *cur, *start;
10766
0
    const xmlChar *end = ctxt->input->end;
10767
0
    int state = ctxt->endCheckState;
10768
0
    size_t index;
10769
10770
0
    if (ctxt->checkIndex == 0) {
10771
0
        cur = ctxt->input->cur + 1;
10772
0
    } else {
10773
0
        cur = ctxt->input->cur + ctxt->checkIndex;
10774
0
    }
10775
0
    start = cur;
10776
10777
0
    while (cur < end) {
10778
0
        if (state == '-') {
10779
0
            if ((*cur == '-') &&
10780
0
                (cur[1] == '-') &&
10781
0
                (cur[2] == '>')) {
10782
0
                state = 0;
10783
0
                cur += 3;
10784
0
                start = cur;
10785
0
                continue;
10786
0
            }
10787
0
        }
10788
0
        else if (state == ']') {
10789
0
            if (*cur == '>') {
10790
0
                ctxt->checkIndex = 0;
10791
0
                ctxt->endCheckState = 0;
10792
0
                return(1);
10793
0
            }
10794
0
            if (IS_BLANK_CH(*cur)) {
10795
0
                state = ' ';
10796
0
            } else if (*cur != ']') {
10797
0
                state = 0;
10798
0
                start = cur;
10799
0
                continue;
10800
0
            }
10801
0
        }
10802
0
        else if (state == ' ') {
10803
0
            if (*cur == '>') {
10804
0
                ctxt->checkIndex = 0;
10805
0
                ctxt->endCheckState = 0;
10806
0
                return(1);
10807
0
            }
10808
0
            if (!IS_BLANK_CH(*cur)) {
10809
0
                state = 0;
10810
0
                start = cur;
10811
0
                continue;
10812
0
            }
10813
0
        }
10814
0
        else if (state != 0) {
10815
0
            if (*cur == state) {
10816
0
                state = 0;
10817
0
                start = cur + 1;
10818
0
            }
10819
0
        }
10820
0
        else if (*cur == '<') {
10821
0
            if ((cur[1] == '!') &&
10822
0
                (cur[2] == '-') &&
10823
0
                (cur[3] == '-')) {
10824
0
                state = '-';
10825
0
                cur += 4;
10826
                /* Don't treat <!--> as comment */
10827
0
                start = cur;
10828
0
                continue;
10829
0
            }
10830
0
        }
10831
0
        else if ((*cur == '"') || (*cur == '\'') || (*cur == ']')) {
10832
0
            state = *cur;
10833
0
        }
10834
10835
0
        cur++;
10836
0
    }
10837
10838
    /*
10839
     * Rescan the three last characters to detect "<!--" and "-->"
10840
     * split across chunks.
10841
     */
10842
0
    if ((state == 0) || (state == '-')) {
10843
0
        if (cur - start < 3)
10844
0
            cur = start;
10845
0
        else
10846
0
            cur -= 3;
10847
0
    }
10848
0
    index = cur - ctxt->input->cur;
10849
0
    if (index > LONG_MAX) {
10850
0
        ctxt->checkIndex = 0;
10851
0
        ctxt->endCheckState = 0;
10852
0
        return(1);
10853
0
    }
10854
0
    ctxt->checkIndex = index;
10855
0
    ctxt->endCheckState = state;
10856
0
    return(0);
10857
0
}
10858
10859
/**
10860
 * Try to progress on parsing
10861
 *
10862
 * @param ctxt  an XML parser context
10863
 * @param terminate  last chunk indicator
10864
 * @returns zero if no parsing was possible
10865
 */
10866
static int
10867
0
xmlParseTryOrFinish(xmlParserCtxtPtr ctxt, int terminate) {
10868
0
    int ret = 0;
10869
0
    size_t avail;
10870
0
    xmlChar cur, next;
10871
10872
0
    if (ctxt->input == NULL)
10873
0
        return(0);
10874
10875
0
    if ((ctxt->input != NULL) &&
10876
0
        (ctxt->input->cur - ctxt->input->base > 4096)) {
10877
0
        xmlParserShrink(ctxt);
10878
0
    }
10879
10880
0
    while (ctxt->disableSAX == 0) {
10881
0
        avail = ctxt->input->end - ctxt->input->cur;
10882
0
        if (avail < 1)
10883
0
      goto done;
10884
0
        switch (ctxt->instate) {
10885
0
            case XML_PARSER_EOF:
10886
          /*
10887
     * Document parsing is done !
10888
     */
10889
0
          goto done;
10890
0
            case XML_PARSER_START:
10891
                /*
10892
                 * Very first chars read from the document flow.
10893
                 */
10894
0
                if ((!terminate) && (avail < 4))
10895
0
                    goto done;
10896
10897
                /*
10898
                 * We need more bytes to detect EBCDIC code pages.
10899
                 * See xmlDetectEBCDIC.
10900
                 */
10901
0
                if ((CMP4(CUR_PTR, 0x4C, 0x6F, 0xA7, 0x94)) &&
10902
0
                    (!terminate) && (avail < 200))
10903
0
                    goto done;
10904
10905
0
                xmlDetectEncoding(ctxt);
10906
0
                ctxt->instate = XML_PARSER_XML_DECL;
10907
0
    break;
10908
10909
0
            case XML_PARSER_XML_DECL:
10910
0
    if ((!terminate) && (avail < 2))
10911
0
        goto done;
10912
0
    cur = ctxt->input->cur[0];
10913
0
    next = ctxt->input->cur[1];
10914
0
          if ((cur == '<') && (next == '?')) {
10915
        /* PI or XML decl */
10916
0
        if ((!terminate) &&
10917
0
                        (!xmlParseLookupString(ctxt, 2, "?>", 2)))
10918
0
      goto done;
10919
0
        if ((ctxt->input->cur[2] == 'x') &&
10920
0
      (ctxt->input->cur[3] == 'm') &&
10921
0
      (ctxt->input->cur[4] == 'l') &&
10922
0
      (IS_BLANK_CH(ctxt->input->cur[5]))) {
10923
0
      ret += 5;
10924
0
      xmlParseXMLDecl(ctxt);
10925
0
        } else {
10926
0
      ctxt->version = xmlCharStrdup(XML_DEFAULT_VERSION);
10927
0
                        if (ctxt->version == NULL) {
10928
0
                            xmlErrMemory(ctxt);
10929
0
                            break;
10930
0
                        }
10931
0
        }
10932
0
    } else {
10933
0
        ctxt->version = xmlCharStrdup(XML_DEFAULT_VERSION);
10934
0
        if (ctxt->version == NULL) {
10935
0
            xmlErrMemory(ctxt);
10936
0
      break;
10937
0
        }
10938
0
    }
10939
0
                if ((ctxt->sax) && (ctxt->sax->setDocumentLocator)) {
10940
0
                    ctxt->sax->setDocumentLocator(ctxt->userData,
10941
0
                            (xmlSAXLocator *) &xmlDefaultSAXLocator);
10942
0
                }
10943
0
                if ((ctxt->sax) && (ctxt->sax->startDocument) &&
10944
0
                    (!ctxt->disableSAX))
10945
0
                    ctxt->sax->startDocument(ctxt->userData);
10946
0
                ctxt->instate = XML_PARSER_MISC;
10947
0
    break;
10948
0
            case XML_PARSER_START_TAG: {
10949
0
          const xmlChar *name;
10950
0
    const xmlChar *prefix = NULL;
10951
0
    const xmlChar *URI = NULL;
10952
0
                int line = ctxt->input->line;
10953
0
    int nbNs = 0;
10954
10955
0
    if ((!terminate) && (avail < 2))
10956
0
        goto done;
10957
0
    cur = ctxt->input->cur[0];
10958
0
          if (cur != '<') {
10959
0
        xmlFatalErrMsg(ctxt, XML_ERR_DOCUMENT_EMPTY,
10960
0
                                   "Start tag expected, '<' not found");
10961
0
                    ctxt->instate = XML_PARSER_EOF;
10962
0
                    xmlFinishDocument(ctxt);
10963
0
        goto done;
10964
0
    }
10965
0
    if ((!terminate) && (!xmlParseLookupGt(ctxt)))
10966
0
                    goto done;
10967
0
    if (ctxt->spaceNr == 0)
10968
0
        spacePush(ctxt, -1);
10969
0
    else if (*ctxt->space == -2)
10970
0
        spacePush(ctxt, -1);
10971
0
    else
10972
0
        spacePush(ctxt, *ctxt->space);
10973
0
#ifdef LIBXML_SAX1_ENABLED
10974
0
    if (ctxt->sax2)
10975
0
#endif /* LIBXML_SAX1_ENABLED */
10976
0
        name = xmlParseStartTag2(ctxt, &prefix, &URI, &nbNs);
10977
0
#ifdef LIBXML_SAX1_ENABLED
10978
0
    else
10979
0
        name = xmlParseStartTag(ctxt);
10980
0
#endif /* LIBXML_SAX1_ENABLED */
10981
0
    if (name == NULL) {
10982
0
        spacePop(ctxt);
10983
0
                    ctxt->instate = XML_PARSER_EOF;
10984
0
                    xmlFinishDocument(ctxt);
10985
0
        goto done;
10986
0
    }
10987
0
#ifdef LIBXML_VALID_ENABLED
10988
    /*
10989
     * [ VC: Root Element Type ]
10990
     * The Name in the document type declaration must match
10991
     * the element type of the root element.
10992
     */
10993
0
    if (ctxt->validate && ctxt->wellFormed && ctxt->myDoc &&
10994
0
        ctxt->node && (ctxt->node == ctxt->myDoc->children))
10995
0
        ctxt->valid &= xmlValidateRoot(&ctxt->vctxt, ctxt->myDoc);
10996
0
#endif /* LIBXML_VALID_ENABLED */
10997
10998
    /*
10999
     * Check for an Empty Element.
11000
     */
11001
0
    if ((RAW == '/') && (NXT(1) == '>')) {
11002
0
        SKIP(2);
11003
11004
0
        if (ctxt->sax2) {
11005
0
      if ((ctxt->sax != NULL) &&
11006
0
          (ctxt->sax->endElementNs != NULL) &&
11007
0
          (!ctxt->disableSAX))
11008
0
          ctxt->sax->endElementNs(ctxt->userData, name,
11009
0
                                  prefix, URI);
11010
0
      if (nbNs > 0)
11011
0
          xmlParserNsPop(ctxt, nbNs);
11012
0
#ifdef LIBXML_SAX1_ENABLED
11013
0
        } else {
11014
0
      if ((ctxt->sax != NULL) &&
11015
0
          (ctxt->sax->endElement != NULL) &&
11016
0
          (!ctxt->disableSAX))
11017
0
          ctxt->sax->endElement(ctxt->userData, name);
11018
0
#endif /* LIBXML_SAX1_ENABLED */
11019
0
        }
11020
0
        spacePop(ctxt);
11021
0
    } else if (RAW == '>') {
11022
0
        NEXT;
11023
0
                    nameNsPush(ctxt, name, prefix, URI, line, nbNs);
11024
0
    } else {
11025
0
        xmlFatalErrMsgStr(ctxt, XML_ERR_GT_REQUIRED,
11026
0
           "Couldn't find end of Start Tag %s\n",
11027
0
           name);
11028
0
        nodePop(ctxt);
11029
0
        spacePop(ctxt);
11030
0
                    if (nbNs > 0)
11031
0
                        xmlParserNsPop(ctxt, nbNs);
11032
0
    }
11033
11034
0
                if (ctxt->nameNr == 0)
11035
0
                    ctxt->instate = XML_PARSER_EPILOG;
11036
0
                else
11037
0
                    ctxt->instate = XML_PARSER_CONTENT;
11038
0
                break;
11039
0
      }
11040
0
            case XML_PARSER_CONTENT: {
11041
0
    cur = ctxt->input->cur[0];
11042
11043
0
    if (cur == '<') {
11044
0
                    if ((!terminate) && (avail < 2))
11045
0
                        goto done;
11046
0
        next = ctxt->input->cur[1];
11047
11048
0
                    if (next == '/') {
11049
0
                        ctxt->instate = XML_PARSER_END_TAG;
11050
0
                        break;
11051
0
                    } else if (next == '?') {
11052
0
                        if ((!terminate) &&
11053
0
                            (!xmlParseLookupString(ctxt, 2, "?>", 2)))
11054
0
                            goto done;
11055
0
                        xmlParsePI(ctxt);
11056
0
                        ctxt->instate = XML_PARSER_CONTENT;
11057
0
                        break;
11058
0
                    } else if (next == '!') {
11059
0
                        if ((!terminate) && (avail < 3))
11060
0
                            goto done;
11061
0
                        next = ctxt->input->cur[2];
11062
11063
0
                        if (next == '-') {
11064
0
                            if ((!terminate) && (avail < 4))
11065
0
                                goto done;
11066
0
                            if (ctxt->input->cur[3] == '-') {
11067
0
                                if ((!terminate) &&
11068
0
                                    (!xmlParseLookupString(ctxt, 4, "-->", 3)))
11069
0
                                    goto done;
11070
0
                                xmlParseComment(ctxt);
11071
0
                                ctxt->instate = XML_PARSER_CONTENT;
11072
0
                                break;
11073
0
                            }
11074
0
                        } else if (next == '[') {
11075
0
                            if ((!terminate) && (avail < 9))
11076
0
                                goto done;
11077
0
                            if ((ctxt->input->cur[2] == '[') &&
11078
0
                                (ctxt->input->cur[3] == 'C') &&
11079
0
                                (ctxt->input->cur[4] == 'D') &&
11080
0
                                (ctxt->input->cur[5] == 'A') &&
11081
0
                                (ctxt->input->cur[6] == 'T') &&
11082
0
                                (ctxt->input->cur[7] == 'A') &&
11083
0
                                (ctxt->input->cur[8] == '[')) {
11084
0
                                if ((!terminate) &&
11085
0
                                    (!xmlParseLookupString(ctxt, 9, "]]>", 3)))
11086
0
                                    goto done;
11087
0
                                ctxt->instate = XML_PARSER_CDATA_SECTION;
11088
0
                                xmlParseCDSect(ctxt);
11089
0
                                ctxt->instate = XML_PARSER_CONTENT;
11090
0
                                break;
11091
0
                            }
11092
0
                        }
11093
0
                    }
11094
0
    } else if (cur == '&') {
11095
0
        if ((!terminate) && (!xmlParseLookupChar(ctxt, ';')))
11096
0
      goto done;
11097
0
        xmlParseReference(ctxt);
11098
0
                    break;
11099
0
    } else {
11100
        /* TODO Avoid the extra copy, handle directly !!! */
11101
        /*
11102
         * Goal of the following test is:
11103
         *  - minimize calls to the SAX 'character' callback
11104
         *    when they are mergeable
11105
         *  - handle an problem for isBlank when we only parse
11106
         *    a sequence of blank chars and the next one is
11107
         *    not available to check against '<' presence.
11108
         *  - tries to homogenize the differences in SAX
11109
         *    callbacks between the push and pull versions
11110
         *    of the parser.
11111
         */
11112
0
        if (avail < XML_PARSER_BIG_BUFFER_SIZE) {
11113
0
      if ((!terminate) && (!xmlParseLookupCharData(ctxt)))
11114
0
          goto done;
11115
0
                    }
11116
0
                    ctxt->checkIndex = 0;
11117
0
        xmlParseCharDataInternal(ctxt, !terminate);
11118
0
                    break;
11119
0
    }
11120
11121
0
                ctxt->instate = XML_PARSER_START_TAG;
11122
0
    break;
11123
0
      }
11124
0
            case XML_PARSER_END_TAG:
11125
0
    if ((!terminate) && (!xmlParseLookupChar(ctxt, '>')))
11126
0
        goto done;
11127
0
    if (ctxt->sax2) {
11128
0
              xmlParseEndTag2(ctxt, &ctxt->pushTab[ctxt->nameNr - 1]);
11129
0
        nameNsPop(ctxt);
11130
0
    }
11131
0
#ifdef LIBXML_SAX1_ENABLED
11132
0
      else
11133
0
        xmlParseEndTag1(ctxt, 0);
11134
0
#endif /* LIBXML_SAX1_ENABLED */
11135
0
    if (ctxt->nameNr == 0) {
11136
0
        ctxt->instate = XML_PARSER_EPILOG;
11137
0
    } else {
11138
0
        ctxt->instate = XML_PARSER_CONTENT;
11139
0
    }
11140
0
    break;
11141
0
            case XML_PARSER_MISC:
11142
0
            case XML_PARSER_PROLOG:
11143
0
            case XML_PARSER_EPILOG:
11144
0
    SKIP_BLANKS;
11145
0
                avail = ctxt->input->end - ctxt->input->cur;
11146
0
    if (avail < 1)
11147
0
        goto done;
11148
0
    if (ctxt->input->cur[0] == '<') {
11149
0
                    if ((!terminate) && (avail < 2))
11150
0
                        goto done;
11151
0
                    next = ctxt->input->cur[1];
11152
0
                    if (next == '?') {
11153
0
                        if ((!terminate) &&
11154
0
                            (!xmlParseLookupString(ctxt, 2, "?>", 2)))
11155
0
                            goto done;
11156
0
                        xmlParsePI(ctxt);
11157
0
                        break;
11158
0
                    } else if (next == '!') {
11159
0
                        if ((!terminate) && (avail < 3))
11160
0
                            goto done;
11161
11162
0
                        if (ctxt->input->cur[2] == '-') {
11163
0
                            if ((!terminate) && (avail < 4))
11164
0
                                goto done;
11165
0
                            if (ctxt->input->cur[3] == '-') {
11166
0
                                if ((!terminate) &&
11167
0
                                    (!xmlParseLookupString(ctxt, 4, "-->", 3)))
11168
0
                                    goto done;
11169
0
                                xmlParseComment(ctxt);
11170
0
                                break;
11171
0
                            }
11172
0
                        } else if (ctxt->instate == XML_PARSER_MISC) {
11173
0
                            if ((!terminate) && (avail < 9))
11174
0
                                goto done;
11175
0
                            if ((ctxt->input->cur[2] == 'D') &&
11176
0
                                (ctxt->input->cur[3] == 'O') &&
11177
0
                                (ctxt->input->cur[4] == 'C') &&
11178
0
                                (ctxt->input->cur[5] == 'T') &&
11179
0
                                (ctxt->input->cur[6] == 'Y') &&
11180
0
                                (ctxt->input->cur[7] == 'P') &&
11181
0
                                (ctxt->input->cur[8] == 'E')) {
11182
0
                                if ((!terminate) && (!xmlParseLookupGt(ctxt)))
11183
0
                                    goto done;
11184
0
                                ctxt->inSubset = 1;
11185
0
                                xmlParseDocTypeDecl(ctxt);
11186
0
                                if (RAW == '[') {
11187
0
                                    ctxt->instate = XML_PARSER_DTD;
11188
0
                                } else {
11189
0
                                    if (RAW == '>')
11190
0
                                        NEXT;
11191
                                    /*
11192
                                     * Create and update the external subset.
11193
                                     */
11194
0
                                    ctxt->inSubset = 2;
11195
0
                                    if ((ctxt->sax != NULL) &&
11196
0
                                        (!ctxt->disableSAX) &&
11197
0
                                        (ctxt->sax->externalSubset != NULL))
11198
0
                                        ctxt->sax->externalSubset(
11199
0
                                                ctxt->userData,
11200
0
                                                ctxt->intSubName,
11201
0
                                                ctxt->extSubSystem,
11202
0
                                                ctxt->extSubURI);
11203
0
                                    ctxt->inSubset = 0;
11204
0
                                    xmlCleanSpecialAttr(ctxt);
11205
0
                                    ctxt->instate = XML_PARSER_PROLOG;
11206
0
                                }
11207
0
                                break;
11208
0
                            }
11209
0
                        }
11210
0
                    }
11211
0
                }
11212
11213
0
                if (ctxt->instate == XML_PARSER_EPILOG) {
11214
0
                    if (ctxt->errNo == XML_ERR_OK)
11215
0
                        xmlFatalErr(ctxt, XML_ERR_DOCUMENT_END, NULL);
11216
0
        ctxt->instate = XML_PARSER_EOF;
11217
0
                    xmlFinishDocument(ctxt);
11218
0
                } else {
11219
0
        ctxt->instate = XML_PARSER_START_TAG;
11220
0
    }
11221
0
    break;
11222
0
            case XML_PARSER_DTD: {
11223
0
                if ((!terminate) && (!xmlParseLookupInternalSubset(ctxt)))
11224
0
                    goto done;
11225
0
    xmlParseInternalSubset(ctxt);
11226
0
    ctxt->inSubset = 2;
11227
0
    if ((ctxt->sax != NULL) && (!ctxt->disableSAX) &&
11228
0
        (ctxt->sax->externalSubset != NULL))
11229
0
        ctxt->sax->externalSubset(ctxt->userData, ctxt->intSubName,
11230
0
          ctxt->extSubSystem, ctxt->extSubURI);
11231
0
    ctxt->inSubset = 0;
11232
0
    xmlCleanSpecialAttr(ctxt);
11233
0
    ctxt->instate = XML_PARSER_PROLOG;
11234
0
                break;
11235
0
      }
11236
0
            default:
11237
0
                xmlFatalErrMsg(ctxt, XML_ERR_INTERNAL_ERROR,
11238
0
      "PP: internal error\n");
11239
0
    ctxt->instate = XML_PARSER_EOF;
11240
0
    break;
11241
0
  }
11242
0
    }
11243
0
done:
11244
0
    return(ret);
11245
0
}
11246
11247
/**
11248
 * Parse a chunk of memory in push parser mode.
11249
 *
11250
 * Assumes that the parser context was initialized with
11251
 * #xmlCreatePushParserCtxt.
11252
 *
11253
 * The last chunk, which will often be empty, must be marked with
11254
 * the `terminate` flag. With the default SAX callbacks, the resulting
11255
 * document will be available in ctxt->myDoc. This pointer will not
11256
 * be freed when calling #xmlFreeParserCtxt and must be freed by the
11257
 * caller. If the document isn't well-formed, it will still be returned
11258
 * in ctxt->myDoc.
11259
 *
11260
 * As an exception, #xmlCtxtResetPush will free the document in
11261
 * ctxt->myDoc. So ctxt->myDoc should be set to NULL after extracting
11262
 * the document.
11263
 *
11264
 * Since 2.14.0, #xmlCtxtGetDocument can be used to retrieve the
11265
 * result document.
11266
 *
11267
 * @param ctxt  an XML parser context
11268
 * @param chunk  chunk of memory
11269
 * @param size  size of chunk in bytes
11270
 * @param terminate  last chunk indicator
11271
 * @returns an xmlParserErrors code (0 on success).
11272
 */
11273
int
11274
xmlParseChunk(xmlParserCtxt *ctxt, const char *chunk, int size,
11275
0
              int terminate) {
11276
0
    size_t curBase;
11277
0
    size_t maxLength;
11278
0
    size_t pos;
11279
0
    int end_in_lf = 0;
11280
0
    int res;
11281
11282
0
    if ((ctxt == NULL) || (size < 0))
11283
0
        return(XML_ERR_ARGUMENT);
11284
0
    if ((chunk == NULL) && (size > 0))
11285
0
        return(XML_ERR_ARGUMENT);
11286
0
    if ((ctxt->input == NULL) || (ctxt->input->buf == NULL))
11287
0
        return(XML_ERR_ARGUMENT);
11288
0
    if (ctxt->disableSAX != 0)
11289
0
        return(ctxt->errNo);
11290
11291
0
    ctxt->input->flags |= XML_INPUT_PROGRESSIVE;
11292
0
    if (ctxt->instate == XML_PARSER_START)
11293
0
        xmlCtxtInitializeLate(ctxt);
11294
0
    if ((size > 0) && (chunk != NULL) && (!terminate) &&
11295
0
        (chunk[size - 1] == '\r')) {
11296
0
  end_in_lf = 1;
11297
0
  size--;
11298
0
    }
11299
11300
    /*
11301
     * Also push an empty chunk to make sure that the raw buffer
11302
     * will be flushed if there is an encoder.
11303
     */
11304
0
    pos = ctxt->input->cur - ctxt->input->base;
11305
0
    res = xmlParserInputBufferPush(ctxt->input->buf, size, chunk);
11306
0
    xmlBufUpdateInput(ctxt->input->buf->buffer, ctxt->input, pos);
11307
0
    if (res < 0) {
11308
0
        xmlCtxtErrIO(ctxt, ctxt->input->buf->error, NULL);
11309
0
        return(ctxt->errNo);
11310
0
    }
11311
11312
0
    xmlParseTryOrFinish(ctxt, terminate);
11313
11314
0
    curBase = ctxt->input->cur - ctxt->input->base;
11315
0
    maxLength = (ctxt->options & XML_PARSE_HUGE) ?
11316
0
                XML_MAX_HUGE_LENGTH :
11317
0
                XML_MAX_LOOKUP_LIMIT;
11318
0
    if (curBase > maxLength) {
11319
0
        xmlFatalErr(ctxt, XML_ERR_RESOURCE_LIMIT,
11320
0
                    "Buffer size limit exceeded, try XML_PARSE_HUGE\n");
11321
0
    }
11322
11323
0
    if ((ctxt->errNo != XML_ERR_OK) && (ctxt->disableSAX != 0))
11324
0
        return(ctxt->errNo);
11325
11326
0
    if (end_in_lf == 1) {
11327
0
  pos = ctxt->input->cur - ctxt->input->base;
11328
0
  res = xmlParserInputBufferPush(ctxt->input->buf, 1, "\r");
11329
0
  xmlBufUpdateInput(ctxt->input->buf->buffer, ctxt->input, pos);
11330
0
        if (res < 0) {
11331
0
            xmlCtxtErrIO(ctxt, ctxt->input->buf->error, NULL);
11332
0
            return(ctxt->errNo);
11333
0
        }
11334
0
    }
11335
0
    if (terminate) {
11336
  /*
11337
   * Check for termination
11338
   */
11339
0
        if ((ctxt->instate != XML_PARSER_EOF) &&
11340
0
            (ctxt->instate != XML_PARSER_EPILOG)) {
11341
0
            if (ctxt->nameNr > 0) {
11342
0
                const xmlChar *name = ctxt->nameTab[ctxt->nameNr - 1];
11343
0
                int line = ctxt->pushTab[ctxt->nameNr - 1].line;
11344
0
                xmlFatalErrMsgStrIntStr(ctxt, XML_ERR_TAG_NOT_FINISHED,
11345
0
                        "Premature end of data in tag %s line %d\n",
11346
0
                        name, line, NULL);
11347
0
            } else if (ctxt->instate == XML_PARSER_START) {
11348
0
                xmlFatalErr(ctxt, XML_ERR_DOCUMENT_EMPTY, NULL);
11349
0
            } else {
11350
0
                xmlFatalErrMsg(ctxt, XML_ERR_DOCUMENT_EMPTY,
11351
0
                               "Start tag expected, '<' not found\n");
11352
0
            }
11353
0
        } else {
11354
0
            xmlParserCheckEOF(ctxt, XML_ERR_DOCUMENT_END);
11355
0
        }
11356
0
  if (ctxt->instate != XML_PARSER_EOF) {
11357
0
            ctxt->instate = XML_PARSER_EOF;
11358
0
            xmlFinishDocument(ctxt);
11359
0
  }
11360
0
    }
11361
0
    if (ctxt->wellFormed == 0)
11362
0
  return((xmlParserErrors) ctxt->errNo);
11363
0
    else
11364
0
        return(0);
11365
0
}
11366
11367
/************************************************************************
11368
 *                  *
11369
 *    I/O front end functions to the parser     *
11370
 *                  *
11371
 ************************************************************************/
11372
11373
/**
11374
 * Create a parser context for using the XML parser in push mode.
11375
 * See #xmlParseChunk.
11376
 *
11377
 * Passing an initial chunk is useless and deprecated.
11378
 *
11379
 * The push parser doesn't support recovery mode or the
11380
 * XML_PARSE_NOBLANKS option.
11381
 *
11382
 * `filename` is used as base URI to fetch external entities and for
11383
 * error reports.
11384
 *
11385
 * @param sax  a SAX handler (optional)
11386
 * @param user_data  user data for SAX callbacks (optional)
11387
 * @param chunk  initial chunk (optional, deprecated)
11388
 * @param size  size of initial chunk in bytes
11389
 * @param filename  file name or URI (optional)
11390
 * @returns the new parser context or NULL if a memory allocation
11391
 * failed.
11392
 */
11393
11394
xmlParserCtxt *
11395
xmlCreatePushParserCtxt(xmlSAXHandler *sax, void *user_data,
11396
0
                        const char *chunk, int size, const char *filename) {
11397
0
    xmlParserCtxtPtr ctxt;
11398
0
    xmlParserInputPtr input;
11399
11400
0
    ctxt = xmlNewSAXParserCtxt(sax, user_data);
11401
0
    if (ctxt == NULL)
11402
0
  return(NULL);
11403
11404
0
    ctxt->options &= ~XML_PARSE_NODICT;
11405
0
    ctxt->dictNames = 1;
11406
11407
0
    input = xmlNewPushInput(filename, chunk, size);
11408
0
    if (input == NULL) {
11409
0
  xmlFreeParserCtxt(ctxt);
11410
0
  return(NULL);
11411
0
    }
11412
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
11413
0
        xmlFreeInputStream(input);
11414
0
        xmlFreeParserCtxt(ctxt);
11415
0
        return(NULL);
11416
0
    }
11417
11418
0
    return(ctxt);
11419
0
}
11420
#endif /* LIBXML_PUSH_ENABLED */
11421
11422
/**
11423
 * Blocks further parser processing
11424
 *
11425
 * @param ctxt  an XML parser context
11426
 */
11427
void
11428
0
xmlStopParser(xmlParserCtxt *ctxt) {
11429
0
    if (ctxt == NULL)
11430
0
        return;
11431
11432
    /* This stops the parser */
11433
0
    ctxt->disableSAX = 2;
11434
11435
    /*
11436
     * xmlStopParser is often called from error handlers,
11437
     * so we can't raise an error here to avoid infinite
11438
     * loops. Just make sure that an error condition is
11439
     * reported.
11440
     */
11441
0
    if (ctxt->errNo == XML_ERR_OK) {
11442
0
        ctxt->errNo = XML_ERR_USER_STOP;
11443
0
        ctxt->lastError.code = XML_ERR_USER_STOP;
11444
0
        ctxt->wellFormed = 0;
11445
0
    }
11446
0
}
11447
11448
/**
11449
 * Create a parser context for using the XML parser with an existing
11450
 * I/O stream
11451
 *
11452
 * @param sax  a SAX handler (optional)
11453
 * @param user_data  user data for SAX callbacks (optional)
11454
 * @param ioread  an I/O read function
11455
 * @param ioclose  an I/O close function (optional)
11456
 * @param ioctx  an I/O handler
11457
 * @param enc  the charset encoding if known (deprecated)
11458
 * @returns the new parser context or NULL
11459
 */
11460
xmlParserCtxt *
11461
xmlCreateIOParserCtxt(xmlSAXHandler *sax, void *user_data,
11462
                      xmlInputReadCallback ioread,
11463
                      xmlInputCloseCallback ioclose,
11464
0
                      void *ioctx, xmlCharEncoding enc) {
11465
0
    xmlParserCtxtPtr ctxt;
11466
0
    xmlParserInputPtr input;
11467
0
    const char *encoding;
11468
11469
0
    ctxt = xmlNewSAXParserCtxt(sax, user_data);
11470
0
    if (ctxt == NULL)
11471
0
  return(NULL);
11472
11473
0
    encoding = xmlGetCharEncodingName(enc);
11474
0
    input = xmlCtxtNewInputFromIO(ctxt, NULL, ioread, ioclose, ioctx,
11475
0
                                  encoding, 0);
11476
0
    if (input == NULL) {
11477
0
  xmlFreeParserCtxt(ctxt);
11478
0
        return (NULL);
11479
0
    }
11480
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
11481
0
        xmlFreeInputStream(input);
11482
0
        xmlFreeParserCtxt(ctxt);
11483
0
        return(NULL);
11484
0
    }
11485
11486
0
    return(ctxt);
11487
0
}
11488
11489
#ifdef LIBXML_VALID_ENABLED
11490
/************************************************************************
11491
 *                  *
11492
 *    Front ends when parsing a DTD       *
11493
 *                  *
11494
 ************************************************************************/
11495
11496
/**
11497
 * Parse a DTD.
11498
 *
11499
 * Option XML_PARSE_DTDLOAD should be enabled in the parser context
11500
 * to make external entities work.
11501
 *
11502
 * @since 2.14.0
11503
 *
11504
 * @param ctxt  a parser context
11505
 * @param input  a parser input
11506
 * @param publicId  public ID of the DTD (optional)
11507
 * @param systemId  system ID of the DTD (optional)
11508
 * @returns the resulting xmlDtd or NULL in case of error.
11509
 * `input` will be freed by the function in any case.
11510
 */
11511
xmlDtd *
11512
xmlCtxtParseDtd(xmlParserCtxt *ctxt, xmlParserInput *input,
11513
0
                const xmlChar *publicId, const xmlChar *systemId) {
11514
0
    xmlDtdPtr ret = NULL;
11515
11516
0
    if ((ctxt == NULL) || (input == NULL)) {
11517
0
        xmlFatalErr(ctxt, XML_ERR_ARGUMENT, NULL);
11518
0
        xmlFreeInputStream(input);
11519
0
        return(NULL);
11520
0
    }
11521
11522
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
11523
0
        xmlFreeInputStream(input);
11524
0
        return(NULL);
11525
0
    }
11526
11527
0
    if (publicId == NULL)
11528
0
        publicId = BAD_CAST "none";
11529
0
    if (systemId == NULL)
11530
0
        systemId = BAD_CAST "none";
11531
11532
0
    ctxt->myDoc = xmlNewDoc(BAD_CAST "1.0");
11533
0
    if (ctxt->myDoc == NULL) {
11534
0
        xmlErrMemory(ctxt);
11535
0
        goto error;
11536
0
    }
11537
0
    ctxt->myDoc->properties = XML_DOC_INTERNAL;
11538
0
    ctxt->myDoc->extSubset = xmlNewDtd(ctxt->myDoc, BAD_CAST "none",
11539
0
                                       publicId, systemId);
11540
0
    if (ctxt->myDoc->extSubset == NULL) {
11541
0
        xmlErrMemory(ctxt);
11542
0
        xmlFreeDoc(ctxt->myDoc);
11543
0
        goto error;
11544
0
    }
11545
11546
0
    xmlParseExternalSubset(ctxt, publicId, systemId);
11547
11548
0
    if (ctxt->wellFormed) {
11549
0
        ret = ctxt->myDoc->extSubset;
11550
0
        ctxt->myDoc->extSubset = NULL;
11551
0
        if (ret != NULL) {
11552
0
            xmlNodePtr tmp;
11553
11554
0
            ret->doc = NULL;
11555
0
            tmp = ret->children;
11556
0
            while (tmp != NULL) {
11557
0
                tmp->doc = NULL;
11558
0
                tmp = tmp->next;
11559
0
            }
11560
0
        }
11561
0
    } else {
11562
0
        ret = NULL;
11563
0
    }
11564
0
    xmlFreeDoc(ctxt->myDoc);
11565
0
    ctxt->myDoc = NULL;
11566
11567
0
error:
11568
0
    xmlFreeInputStream(xmlCtxtPopInput(ctxt));
11569
11570
0
    return(ret);
11571
0
}
11572
11573
/**
11574
 * Load and parse a DTD
11575
 *
11576
 * @deprecated Use #xmlCtxtParseDtd.
11577
 *
11578
 * @param sax  the SAX handler block or NULL
11579
 * @param input  an Input Buffer
11580
 * @param enc  the charset encoding if known
11581
 * @returns the resulting xmlDtd or NULL in case of error.
11582
 * `input` will be freed by the function in any case.
11583
 */
11584
11585
xmlDtd *
11586
xmlIOParseDTD(xmlSAXHandler *sax, xmlParserInputBuffer *input,
11587
0
        xmlCharEncoding enc) {
11588
0
    xmlDtdPtr ret = NULL;
11589
0
    xmlParserCtxtPtr ctxt;
11590
0
    xmlParserInputPtr pinput = NULL;
11591
11592
0
    if (input == NULL)
11593
0
  return(NULL);
11594
11595
0
    ctxt = xmlNewSAXParserCtxt(sax, NULL);
11596
0
    if (ctxt == NULL) {
11597
0
        xmlFreeParserInputBuffer(input);
11598
0
  return(NULL);
11599
0
    }
11600
0
    xmlCtxtSetOptions(ctxt, XML_PARSE_DTDLOAD);
11601
11602
    /*
11603
     * generate a parser input from the I/O handler
11604
     */
11605
11606
0
    pinput = xmlNewIOInputStream(ctxt, input, XML_CHAR_ENCODING_NONE);
11607
0
    if (pinput == NULL) {
11608
0
  xmlFreeParserCtxt(ctxt);
11609
0
  return(NULL);
11610
0
    }
11611
11612
0
    if (enc != XML_CHAR_ENCODING_NONE) {
11613
0
        xmlSwitchEncoding(ctxt, enc);
11614
0
    }
11615
11616
0
    ret = xmlCtxtParseDtd(ctxt, pinput, NULL, NULL);
11617
11618
0
    xmlFreeParserCtxt(ctxt);
11619
0
    return(ret);
11620
0
}
11621
11622
/**
11623
 * Load and parse an external subset.
11624
 *
11625
 * @deprecated Use #xmlCtxtParseDtd.
11626
 *
11627
 * @param sax  the SAX handler block
11628
 * @param publicId  public identifier of the DTD (optional)
11629
 * @param systemId  system identifier (URL) of the DTD
11630
 * @returns the resulting xmlDtd or NULL in case of error.
11631
 */
11632
11633
xmlDtd *
11634
xmlSAXParseDTD(xmlSAXHandler *sax, const xmlChar *publicId,
11635
0
               const xmlChar *systemId) {
11636
0
    xmlDtdPtr ret = NULL;
11637
0
    xmlParserCtxtPtr ctxt;
11638
0
    xmlParserInputPtr input = NULL;
11639
0
    xmlChar* systemIdCanonic;
11640
11641
0
    if ((publicId == NULL) && (systemId == NULL)) return(NULL);
11642
11643
0
    ctxt = xmlNewSAXParserCtxt(sax, NULL);
11644
0
    if (ctxt == NULL) {
11645
0
  return(NULL);
11646
0
    }
11647
0
    xmlCtxtSetOptions(ctxt, XML_PARSE_DTDLOAD);
11648
11649
    /*
11650
     * Canonicalise the system ID
11651
     */
11652
0
    systemIdCanonic = xmlCanonicPath(systemId);
11653
0
    if ((systemId != NULL) && (systemIdCanonic == NULL)) {
11654
0
  xmlFreeParserCtxt(ctxt);
11655
0
  return(NULL);
11656
0
    }
11657
11658
    /*
11659
     * Ask the Entity resolver to load the damn thing
11660
     */
11661
11662
0
    if ((ctxt->sax != NULL) && (ctxt->sax->resolveEntity != NULL))
11663
0
  input = ctxt->sax->resolveEntity(ctxt->userData, publicId,
11664
0
                                   systemIdCanonic);
11665
0
    if (input == NULL) {
11666
0
  xmlFreeParserCtxt(ctxt);
11667
0
  if (systemIdCanonic != NULL)
11668
0
      xmlFree(systemIdCanonic);
11669
0
  return(NULL);
11670
0
    }
11671
11672
0
    if (input->filename == NULL)
11673
0
  input->filename = (char *) systemIdCanonic;
11674
0
    else
11675
0
  xmlFree(systemIdCanonic);
11676
11677
0
    ret = xmlCtxtParseDtd(ctxt, input, publicId, systemId);
11678
11679
0
    xmlFreeParserCtxt(ctxt);
11680
0
    return(ret);
11681
0
}
11682
11683
11684
/**
11685
 * Load and parse an external subset.
11686
 *
11687
 * @param publicId  public identifier of the DTD (optional)
11688
 * @param systemId  system identifier (URL) of the DTD
11689
 * @returns the resulting xmlDtd or NULL in case of error.
11690
 */
11691
11692
xmlDtd *
11693
0
xmlParseDTD(const xmlChar *publicId, const xmlChar *systemId) {
11694
0
    return(xmlSAXParseDTD(NULL, publicId, systemId));
11695
0
}
11696
#endif /* LIBXML_VALID_ENABLED */
11697
11698
/************************************************************************
11699
 *                  *
11700
 *    Front ends when parsing an Entity     *
11701
 *                  *
11702
 ************************************************************************/
11703
11704
static xmlNodePtr
11705
xmlCtxtParseContentInternal(xmlParserCtxtPtr ctxt, xmlParserInputPtr input,
11706
0
                            int hasTextDecl, int buildTree) {
11707
0
    xmlNodePtr root = NULL;
11708
0
    xmlNodePtr list = NULL;
11709
0
    xmlChar *rootName = BAD_CAST "#root";
11710
0
    int result;
11711
11712
0
    if (buildTree) {
11713
0
        root = xmlNewDocNode(ctxt->myDoc, NULL, rootName, NULL);
11714
0
        if (root == NULL) {
11715
0
            xmlErrMemory(ctxt);
11716
0
            goto error;
11717
0
        }
11718
0
    }
11719
11720
0
    if (xmlCtxtPushInput(ctxt, input) < 0)
11721
0
        goto error;
11722
11723
0
    nameNsPush(ctxt, rootName, NULL, NULL, 0, 0);
11724
0
    spacePush(ctxt, -1);
11725
11726
0
    if (buildTree)
11727
0
        nodePush(ctxt, root);
11728
11729
0
    if (hasTextDecl) {
11730
0
        xmlDetectEncoding(ctxt);
11731
11732
        /*
11733
         * Parse a possible text declaration first
11734
         */
11735
0
        if ((CMP5(CUR_PTR, '<', '?', 'x', 'm', 'l')) &&
11736
0
            (IS_BLANK_CH(NXT(5)))) {
11737
0
            xmlParseTextDecl(ctxt);
11738
            /*
11739
             * An XML-1.0 document can't reference an entity not XML-1.0
11740
             */
11741
0
            if ((xmlStrEqual(ctxt->version, BAD_CAST "1.0")) &&
11742
0
                (!xmlStrEqual(ctxt->input->version, BAD_CAST "1.0"))) {
11743
0
                xmlFatalErrMsg(ctxt, XML_ERR_VERSION_MISMATCH,
11744
0
                               "Version mismatch between document and "
11745
0
                               "entity\n");
11746
0
            }
11747
0
        }
11748
0
    }
11749
11750
0
    xmlParseContentInternal(ctxt);
11751
11752
0
    if (ctxt->input->cur < ctxt->input->end)
11753
0
  xmlFatalErr(ctxt, XML_ERR_NOT_WELL_BALANCED, NULL);
11754
11755
0
    if ((ctxt->wellFormed) ||
11756
0
        ((ctxt->recovery) && (!xmlCtxtIsCatastrophicError(ctxt)))) {
11757
0
        if (root != NULL) {
11758
0
            xmlNodePtr cur;
11759
11760
            /*
11761
             * Unlink newly created node list.
11762
             */
11763
0
            list = root->children;
11764
0
            root->children = NULL;
11765
0
            root->last = NULL;
11766
0
            for (cur = list; cur != NULL; cur = cur->next)
11767
0
                cur->parent = NULL;
11768
0
        }
11769
0
    }
11770
11771
    /*
11772
     * Read the rest of the stream in case of errors. We want
11773
     * to account for the whole entity size.
11774
     */
11775
0
    do {
11776
0
        ctxt->input->cur = ctxt->input->end;
11777
0
        xmlParserShrink(ctxt);
11778
0
        result = xmlParserGrow(ctxt);
11779
0
    } while (result > 0);
11780
11781
0
    if (buildTree)
11782
0
        nodePop(ctxt);
11783
11784
0
    namePop(ctxt);
11785
0
    spacePop(ctxt);
11786
11787
0
    xmlCtxtPopInput(ctxt);
11788
11789
0
error:
11790
0
    xmlFreeNode(root);
11791
11792
0
    return(list);
11793
0
}
11794
11795
static void
11796
0
xmlCtxtParseEntity(xmlParserCtxtPtr ctxt, xmlEntityPtr ent) {
11797
0
    xmlParserInputPtr input;
11798
0
    xmlNodePtr list;
11799
0
    unsigned long consumed;
11800
0
    int isExternal;
11801
0
    int buildTree;
11802
0
    int oldMinNsIndex;
11803
0
    int oldNodelen, oldNodemem;
11804
11805
0
    isExternal = (ent->etype == XML_EXTERNAL_GENERAL_PARSED_ENTITY);
11806
0
    buildTree = (ctxt->node != NULL);
11807
11808
    /*
11809
     * Recursion check
11810
     */
11811
0
    if (ent->flags & XML_ENT_EXPANDING) {
11812
0
        xmlFatalErr(ctxt, XML_ERR_ENTITY_LOOP, NULL);
11813
0
        goto error;
11814
0
    }
11815
11816
    /*
11817
     * Load entity
11818
     */
11819
0
    input = xmlNewEntityInputStream(ctxt, ent);
11820
0
    if (input == NULL)
11821
0
        goto error;
11822
11823
    /*
11824
     * When building a tree, we need to limit the scope of namespace
11825
     * declarations, so that entities don't reference xmlNs structs
11826
     * from the parent of a reference.
11827
     */
11828
0
    oldMinNsIndex = ctxt->nsdb->minNsIndex;
11829
0
    if (buildTree)
11830
0
        ctxt->nsdb->minNsIndex = ctxt->nsNr;
11831
11832
0
    oldNodelen = ctxt->nodelen;
11833
0
    oldNodemem = ctxt->nodemem;
11834
0
    ctxt->nodelen = 0;
11835
0
    ctxt->nodemem = 0;
11836
11837
    /*
11838
     * Parse content
11839
     *
11840
     * This initiates a recursive call chain:
11841
     *
11842
     * - xmlCtxtParseContentInternal
11843
     * - xmlParseContentInternal
11844
     * - xmlParseReference
11845
     * - xmlCtxtParseEntity
11846
     *
11847
     * The nesting depth is limited by the maximum number of inputs,
11848
     * see xmlCtxtPushInput.
11849
     *
11850
     * It's possible to make this non-recursive (minNsIndex must be
11851
     * stored in the input struct) at the expense of code readability.
11852
     */
11853
11854
0
    ent->flags |= XML_ENT_EXPANDING;
11855
11856
0
    list = xmlCtxtParseContentInternal(ctxt, input, isExternal, buildTree);
11857
11858
0
    ent->flags &= ~XML_ENT_EXPANDING;
11859
11860
0
    ctxt->nsdb->minNsIndex = oldMinNsIndex;
11861
0
    ctxt->nodelen = oldNodelen;
11862
0
    ctxt->nodemem = oldNodemem;
11863
11864
    /*
11865
     * Entity size accounting
11866
     */
11867
0
    consumed = input->consumed;
11868
0
    xmlSaturatedAddSizeT(&consumed, input->end - input->base);
11869
11870
0
    if ((ent->flags & XML_ENT_CHECKED) == 0)
11871
0
        xmlSaturatedAdd(&ent->expandedSize, consumed);
11872
11873
0
    if ((ent->flags & XML_ENT_PARSED) == 0) {
11874
0
        if (isExternal)
11875
0
            xmlSaturatedAdd(&ctxt->sizeentities, consumed);
11876
11877
0
        ent->children = list;
11878
11879
0
        while (list != NULL) {
11880
0
            list->parent = (xmlNodePtr) ent;
11881
11882
            /*
11883
             * Downstream code like the nginx xslt module can set
11884
             * ctxt->myDoc->extSubset to a separate DTD, so the entity
11885
             * might have a different or a NULL document.
11886
             */
11887
0
            if (list->doc != ent->doc)
11888
0
                xmlSetTreeDoc(list, ent->doc);
11889
11890
0
            if (list->next == NULL)
11891
0
                ent->last = list;
11892
0
            list = list->next;
11893
0
        }
11894
0
    } else {
11895
0
        xmlFreeNodeList(list);
11896
0
    }
11897
11898
0
    xmlFreeInputStream(input);
11899
11900
0
error:
11901
0
    ent->flags |= XML_ENT_PARSED | XML_ENT_CHECKED;
11902
0
}
11903
11904
/**
11905
 * Parse an external general entity within an existing parsing context
11906
 * An external general parsed entity is well-formed if it matches the
11907
 * production labeled extParsedEnt.
11908
 *
11909
 *     [78] extParsedEnt ::= TextDecl? content
11910
 *
11911
 * @param ctxt  the existing parsing context
11912
 * @param URL  the URL for the entity to load
11913
 * @param ID  the System ID for the entity to load
11914
 * @param listOut  the return value for the set of parsed nodes
11915
 * @returns 0 if the entity is well formed, -1 in case of args problem and
11916
 *    the parser error code otherwise
11917
 */
11918
11919
int
11920
xmlParseCtxtExternalEntity(xmlParserCtxt *ctxt, const xmlChar *URL,
11921
0
                           const xmlChar *ID, xmlNode **listOut) {
11922
0
    xmlParserInputPtr input;
11923
0
    xmlNodePtr list;
11924
11925
0
    if (listOut != NULL)
11926
0
        *listOut = NULL;
11927
11928
0
    if (ctxt == NULL)
11929
0
        return(XML_ERR_ARGUMENT);
11930
11931
0
    input = xmlLoadResource(ctxt, (char *) URL, (char *) ID,
11932
0
                            XML_RESOURCE_GENERAL_ENTITY);
11933
0
    if (input == NULL)
11934
0
        return(ctxt->errNo);
11935
11936
0
    xmlCtxtInitializeLate(ctxt);
11937
11938
0
    list = xmlCtxtParseContentInternal(ctxt, input, /* hasTextDecl */ 1, 1);
11939
0
    if (listOut != NULL)
11940
0
        *listOut = list;
11941
0
    else
11942
0
        xmlFreeNodeList(list);
11943
11944
0
    xmlFreeInputStream(input);
11945
0
    return(ctxt->errNo);
11946
0
}
11947
11948
#ifdef LIBXML_SAX1_ENABLED
11949
/**
11950
 * Parse an external general entity
11951
 * An external general parsed entity is well-formed if it matches the
11952
 * production labeled extParsedEnt.
11953
 *
11954
 * This function uses deprecated global variables to set parser options
11955
 * which default to XML_PARSE_NODICT.
11956
 *
11957
 * @deprecated Use #xmlParseCtxtExternalEntity.
11958
 *
11959
 *     [78] extParsedEnt ::= TextDecl? content
11960
 *
11961
 * @param doc  the document the chunk pertains to
11962
 * @param sax  the SAX handler block (possibly NULL)
11963
 * @param user_data  The user data returned on SAX callbacks (possibly NULL)
11964
 * @param depth  Used for loop detection, use 0
11965
 * @param URL  the URL for the entity to load
11966
 * @param ID  the System ID for the entity to load
11967
 * @param list  the return value for the set of parsed nodes
11968
 * @returns 0 if the entity is well formed, -1 in case of args problem and
11969
 *    the parser error code otherwise
11970
 */
11971
11972
int
11973
xmlParseExternalEntity(xmlDoc *doc, xmlSAXHandler *sax, void *user_data,
11974
0
    int depth, const xmlChar *URL, const xmlChar *ID, xmlNode **list) {
11975
0
    xmlParserCtxtPtr ctxt;
11976
0
    int ret;
11977
11978
0
    if (list != NULL)
11979
0
        *list = NULL;
11980
11981
0
    if (doc == NULL)
11982
0
        return(XML_ERR_ARGUMENT);
11983
11984
0
    ctxt = xmlNewSAXParserCtxt(sax, user_data);
11985
0
    if (ctxt == NULL)
11986
0
        return(XML_ERR_NO_MEMORY);
11987
11988
0
    ctxt->depth = depth;
11989
0
    ctxt->myDoc = doc;
11990
0
    ret = xmlParseCtxtExternalEntity(ctxt, URL, ID, list);
11991
11992
0
    xmlFreeParserCtxt(ctxt);
11993
0
    return(ret);
11994
0
}
11995
11996
/**
11997
 * Parse a well-balanced chunk of an XML document
11998
 * called by the parser
11999
 * The allowed sequence for the Well Balanced Chunk is the one defined by
12000
 * the content production in the XML grammar:
12001
 *
12002
 *     [43] content ::= (element | CharData | Reference | CDSect | PI |
12003
 *                       Comment)*
12004
 *
12005
 * This function uses deprecated global variables to set parser options
12006
 * which default to XML_PARSE_NODICT.
12007
 *
12008
 * @param doc  the document the chunk pertains to (must not be NULL)
12009
 * @param sax  the SAX handler block (possibly NULL)
12010
 * @param user_data  The user data returned on SAX callbacks (possibly NULL)
12011
 * @param depth  Used for loop detection, use 0
12012
 * @param string  the input string in UTF8 or ISO-Latin (zero terminated)
12013
 * @param lst  the return value for the set of parsed nodes
12014
 * @returns 0 if the chunk is well balanced, -1 in case of args problem and
12015
 *    the parser error code otherwise
12016
 */
12017
12018
int
12019
xmlParseBalancedChunkMemory(xmlDoc *doc, xmlSAXHandler *sax,
12020
0
     void *user_data, int depth, const xmlChar *string, xmlNode **lst) {
12021
0
    return xmlParseBalancedChunkMemoryRecover( doc, sax, user_data,
12022
0
                                                depth, string, lst, 0 );
12023
0
}
12024
#endif /* LIBXML_SAX1_ENABLED */
12025
12026
/**
12027
 * Parse a well-balanced chunk of XML matching the 'content' production.
12028
 *
12029
 * Namespaces in scope of `node` and entities of `node`'s document are
12030
 * recognized. When validating, the DTD of `node`'s document is used.
12031
 *
12032
 * Always consumes `input` even in error case.
12033
 *
12034
 * @since 2.14.0
12035
 *
12036
 * @param ctxt  parser context
12037
 * @param input  parser input
12038
 * @param node  target node or document
12039
 * @param hasTextDecl  whether to parse text declaration
12040
 * @returns a node list or NULL in case of error.
12041
 */
12042
xmlNode *
12043
xmlCtxtParseContent(xmlParserCtxt *ctxt, xmlParserInput *input,
12044
0
                    xmlNode *node, int hasTextDecl) {
12045
0
    xmlDocPtr doc;
12046
0
    xmlNodePtr cur, list = NULL;
12047
0
    int nsnr = 0;
12048
0
    xmlDictPtr oldDict;
12049
0
    int oldOptions, oldDictNames, oldLoadSubset;
12050
12051
0
    if ((ctxt == NULL) || (input == NULL) || (node == NULL)) {
12052
0
        xmlFatalErr(ctxt, XML_ERR_ARGUMENT, NULL);
12053
0
        goto exit;
12054
0
    }
12055
12056
0
    doc = node->doc;
12057
0
    if (doc == NULL) {
12058
0
        xmlFatalErr(ctxt, XML_ERR_ARGUMENT, NULL);
12059
0
        goto exit;
12060
0
    }
12061
12062
0
    switch (node->type) {
12063
0
        case XML_ELEMENT_NODE:
12064
0
        case XML_DOCUMENT_NODE:
12065
0
        case XML_HTML_DOCUMENT_NODE:
12066
0
            break;
12067
12068
0
        case XML_ATTRIBUTE_NODE:
12069
0
        case XML_TEXT_NODE:
12070
0
        case XML_CDATA_SECTION_NODE:
12071
0
        case XML_ENTITY_REF_NODE:
12072
0
        case XML_PI_NODE:
12073
0
        case XML_COMMENT_NODE:
12074
0
            for (cur = node->parent; cur != NULL; cur = cur->parent) {
12075
0
                if ((cur->type == XML_ELEMENT_NODE) ||
12076
0
                    (cur->type == XML_DOCUMENT_NODE) ||
12077
0
                    (cur->type == XML_HTML_DOCUMENT_NODE)) {
12078
0
                    node = cur;
12079
0
                    break;
12080
0
                }
12081
0
            }
12082
0
            break;
12083
12084
0
        default:
12085
0
            xmlFatalErr(ctxt, XML_ERR_ARGUMENT, NULL);
12086
0
            goto exit;
12087
0
    }
12088
12089
0
    xmlCtxtReset(ctxt);
12090
12091
0
    oldDict = ctxt->dict;
12092
0
    oldOptions = ctxt->options;
12093
0
    oldDictNames = ctxt->dictNames;
12094
0
    oldLoadSubset = ctxt->loadsubset;
12095
12096
    /*
12097
     * Use input doc's dict if present, else assure XML_PARSE_NODICT is set.
12098
     */
12099
0
    if (doc->dict != NULL) {
12100
0
        ctxt->dict = doc->dict;
12101
0
    } else {
12102
0
        ctxt->options |= XML_PARSE_NODICT;
12103
0
        ctxt->dictNames = 0;
12104
0
    }
12105
12106
    /*
12107
     * Disable IDs
12108
     */
12109
0
    ctxt->loadsubset |= XML_SKIP_IDS;
12110
0
    ctxt->options |= XML_PARSE_SKIP_IDS;
12111
12112
0
    ctxt->myDoc = doc;
12113
12114
0
#ifdef LIBXML_HTML_ENABLED
12115
0
    if (ctxt->html) {
12116
        /*
12117
         * When parsing in context, it makes no sense to add implied
12118
         * elements like html/body/etc...
12119
         */
12120
0
        ctxt->options |= HTML_PARSE_NOIMPLIED;
12121
12122
0
        list = htmlCtxtParseContentInternal(ctxt, input);
12123
0
    } else
12124
0
#endif
12125
0
    {
12126
0
        xmlCtxtInitializeLate(ctxt);
12127
12128
        /*
12129
         * initialize the SAX2 namespaces stack
12130
         */
12131
0
        cur = node;
12132
0
        while ((cur != NULL) && (cur->type == XML_ELEMENT_NODE)) {
12133
0
            xmlNsPtr ns = cur->nsDef;
12134
0
            xmlHashedString hprefix, huri;
12135
12136
0
            while (ns != NULL) {
12137
0
                hprefix = xmlDictLookupHashed(ctxt->dict, ns->prefix, -1);
12138
0
                huri = xmlDictLookupHashed(ctxt->dict, ns->href, -1);
12139
0
                if (xmlParserNsPush(ctxt, &hprefix, &huri, ns, 1) > 0)
12140
0
                    nsnr++;
12141
0
                ns = ns->next;
12142
0
            }
12143
0
            cur = cur->parent;
12144
0
        }
12145
12146
0
        list = xmlCtxtParseContentInternal(ctxt, input, hasTextDecl, 1);
12147
12148
0
        if (nsnr > 0)
12149
0
            xmlParserNsPop(ctxt, nsnr);
12150
0
    }
12151
12152
0
    ctxt->dict = oldDict;
12153
0
    ctxt->options = oldOptions;
12154
0
    ctxt->dictNames = oldDictNames;
12155
0
    ctxt->loadsubset = oldLoadSubset;
12156
0
    ctxt->myDoc = NULL;
12157
0
    ctxt->node = NULL;
12158
12159
0
exit:
12160
0
    xmlFreeInputStream(input);
12161
0
    return(list);
12162
0
}
12163
12164
/**
12165
 * Parse a well-balanced chunk of an XML document
12166
 * within the context (DTD, namespaces, etc ...) of the given node.
12167
 *
12168
 * The allowed sequence for the data is a Well Balanced Chunk defined by
12169
 * the content production in the XML grammar:
12170
 *
12171
 *     [43] content ::= (element | CharData | Reference | CDSect | PI |
12172
 *                       Comment)*
12173
 *
12174
 * This function assumes the encoding of `node`'s document which is
12175
 * typically not what you want. A better alternative is
12176
 * #xmlCtxtParseContent.
12177
 *
12178
 * @param node  the context node
12179
 * @param data  the input string
12180
 * @param datalen  the input string length in bytes
12181
 * @param options  a combination of xmlParserOption
12182
 * @param listOut  the return value for the set of parsed nodes
12183
 * @returns XML_ERR_OK if the chunk is well balanced, and the parser
12184
 * error code otherwise
12185
 */
12186
xmlParserErrors
12187
xmlParseInNodeContext(xmlNode *node, const char *data, int datalen,
12188
0
                      int options, xmlNode **listOut) {
12189
0
    xmlParserCtxtPtr ctxt;
12190
0
    xmlParserInputPtr input;
12191
0
    xmlDocPtr doc;
12192
0
    xmlNodePtr list;
12193
0
    xmlParserErrors ret;
12194
12195
0
    if (listOut == NULL)
12196
0
        return(XML_ERR_INTERNAL_ERROR);
12197
0
    *listOut = NULL;
12198
12199
0
    if ((node == NULL) || (data == NULL) || (datalen < 0))
12200
0
        return(XML_ERR_INTERNAL_ERROR);
12201
12202
0
    doc = node->doc;
12203
0
    if (doc == NULL)
12204
0
        return(XML_ERR_INTERNAL_ERROR);
12205
12206
0
#ifdef LIBXML_HTML_ENABLED
12207
0
    if (doc->type == XML_HTML_DOCUMENT_NODE) {
12208
0
        ctxt = htmlNewParserCtxt();
12209
0
    }
12210
0
    else
12211
0
#endif
12212
0
        ctxt = xmlNewParserCtxt();
12213
12214
0
    if (ctxt == NULL)
12215
0
        return(XML_ERR_NO_MEMORY);
12216
12217
0
    input = xmlCtxtNewInputFromMemory(ctxt, NULL, data, datalen,
12218
0
                                      (const char *) doc->encoding,
12219
0
                                      XML_INPUT_BUF_STATIC);
12220
0
    if (input == NULL) {
12221
0
        xmlFreeParserCtxt(ctxt);
12222
0
        return(XML_ERR_NO_MEMORY);
12223
0
    }
12224
12225
0
    xmlCtxtUseOptions(ctxt, options);
12226
12227
0
    list = xmlCtxtParseContent(ctxt, input, node, /* hasTextDecl */ 0);
12228
12229
0
    if (list == NULL) {
12230
0
        ret = ctxt->errNo;
12231
0
        if (ret == XML_ERR_ARGUMENT)
12232
0
            ret = XML_ERR_INTERNAL_ERROR;
12233
0
    } else {
12234
0
        ret = XML_ERR_OK;
12235
0
        *listOut = list;
12236
0
    }
12237
12238
0
    xmlFreeParserCtxt(ctxt);
12239
12240
0
    return(ret);
12241
0
}
12242
12243
#ifdef LIBXML_SAX1_ENABLED
12244
/**
12245
 * Parse a well-balanced chunk of an XML document
12246
 *
12247
 * The allowed sequence for the Well Balanced Chunk is the one defined by
12248
 * the content production in the XML grammar:
12249
 *
12250
 *     [43] content ::= (element | CharData | Reference | CDSect | PI |
12251
 *                       Comment)*
12252
 *
12253
 * In case recover is set to 1, the nodelist will not be empty even if
12254
 * the parsed chunk is not well balanced, assuming the parsing succeeded to
12255
 * some extent.
12256
 *
12257
 * This function uses deprecated global variables to set parser options
12258
 * which default to XML_PARSE_NODICT.
12259
 *
12260
 * @param doc  the document the chunk pertains to (must not be NULL)
12261
 * @param sax  the SAX handler block (possibly NULL)
12262
 * @param user_data  The user data returned on SAX callbacks (possibly NULL)
12263
 * @param depth  Used for loop detection, use 0
12264
 * @param string  the input string in UTF8 or ISO-Latin (zero terminated)
12265
 * @param listOut  the return value for the set of parsed nodes
12266
 * @param recover  return nodes even if the data is broken (use 0)
12267
 * @returns 0 if the chunk is well balanced, or thehe parser error code
12268
 * otherwise.
12269
 */
12270
int
12271
xmlParseBalancedChunkMemoryRecover(xmlDoc *doc, xmlSAXHandler *sax,
12272
     void *user_data, int depth, const xmlChar *string, xmlNode **listOut,
12273
0
     int recover) {
12274
0
    xmlParserCtxtPtr ctxt;
12275
0
    xmlParserInputPtr input;
12276
0
    xmlNodePtr list;
12277
0
    int ret;
12278
12279
0
    if (listOut != NULL)
12280
0
        *listOut = NULL;
12281
12282
0
    if (string == NULL)
12283
0
        return(XML_ERR_ARGUMENT);
12284
12285
0
    ctxt = xmlNewSAXParserCtxt(sax, user_data);
12286
0
    if (ctxt == NULL)
12287
0
        return(XML_ERR_NO_MEMORY);
12288
12289
0
    xmlCtxtInitializeLate(ctxt);
12290
12291
0
    ctxt->depth = depth;
12292
0
    ctxt->myDoc = doc;
12293
0
    if (recover) {
12294
0
        ctxt->options |= XML_PARSE_RECOVER;
12295
0
        ctxt->recovery = 1;
12296
0
    }
12297
12298
0
    input = xmlNewStringInputStream(ctxt, string);
12299
0
    if (input == NULL) {
12300
0
        ret = ctxt->errNo;
12301
0
        goto error;
12302
0
    }
12303
12304
0
    list = xmlCtxtParseContentInternal(ctxt, input, /* hasTextDecl */ 0, 1);
12305
0
    if (listOut != NULL)
12306
0
        *listOut = list;
12307
0
    else
12308
0
        xmlFreeNodeList(list);
12309
12310
0
    if (!ctxt->wellFormed)
12311
0
        ret = ctxt->errNo;
12312
0
    else
12313
0
        ret = XML_ERR_OK;
12314
12315
0
error:
12316
0
    xmlFreeInputStream(input);
12317
0
    xmlFreeParserCtxt(ctxt);
12318
0
    return(ret);
12319
0
}
12320
12321
/**
12322
 * Parse an XML external entity out of context and build a tree.
12323
 * It use the given SAX function block to handle the parsing callback.
12324
 * If sax is NULL, fallback to the default DOM tree building routines.
12325
 *
12326
 * @deprecated Don't use.
12327
 *
12328
 *     [78] extParsedEnt ::= TextDecl? content
12329
 *
12330
 * This correspond to a "Well Balanced" chunk
12331
 *
12332
 * This function uses deprecated global variables to set parser options
12333
 * which default to XML_PARSE_NODICT.
12334
 *
12335
 * @param sax  the SAX handler block
12336
 * @param filename  the filename
12337
 * @returns the resulting document tree
12338
 */
12339
12340
xmlDoc *
12341
0
xmlSAXParseEntity(xmlSAXHandler *sax, const char *filename) {
12342
0
    xmlDocPtr ret;
12343
0
    xmlParserCtxtPtr ctxt;
12344
12345
0
    ctxt = xmlCreateFileParserCtxt(filename);
12346
0
    if (ctxt == NULL) {
12347
0
  return(NULL);
12348
0
    }
12349
0
    if (sax != NULL) {
12350
0
        if (sax->initialized == XML_SAX2_MAGIC) {
12351
0
            *ctxt->sax = *sax;
12352
0
        } else {
12353
0
            memset(ctxt->sax, 0, sizeof(*ctxt->sax));
12354
0
            memcpy(ctxt->sax, sax, sizeof(xmlSAXHandlerV1));
12355
0
        }
12356
0
        ctxt->userData = NULL;
12357
0
    }
12358
12359
0
    xmlParseExtParsedEnt(ctxt);
12360
12361
0
    if (ctxt->wellFormed) {
12362
0
  ret = ctxt->myDoc;
12363
0
    } else {
12364
0
        ret = NULL;
12365
0
        xmlFreeDoc(ctxt->myDoc);
12366
0
    }
12367
12368
0
    xmlFreeParserCtxt(ctxt);
12369
12370
0
    return(ret);
12371
0
}
12372
12373
/**
12374
 * Parse an XML external entity out of context and build a tree.
12375
 *
12376
 *     [78] extParsedEnt ::= TextDecl? content
12377
 *
12378
 * This correspond to a "Well Balanced" chunk
12379
 *
12380
 * This function uses deprecated global variables to set parser options
12381
 * which default to XML_PARSE_NODICT.
12382
 *
12383
 * @deprecated Don't use.
12384
 *
12385
 * @param filename  the filename
12386
 * @returns the resulting document tree
12387
 */
12388
12389
xmlDoc *
12390
0
xmlParseEntity(const char *filename) {
12391
0
    return(xmlSAXParseEntity(NULL, filename));
12392
0
}
12393
#endif /* LIBXML_SAX1_ENABLED */
12394
12395
/**
12396
 * Create a parser context for an external entity
12397
 * Automatic support for ZLIB/Compress compressed document is provided
12398
 * by default if found at compile-time.
12399
 *
12400
 * @deprecated Don't use.
12401
 *
12402
 * @param URL  the entity URL
12403
 * @param ID  the entity PUBLIC ID
12404
 * @param base  a possible base for the target URI
12405
 * @returns the new parser context or NULL
12406
 */
12407
xmlParserCtxt *
12408
xmlCreateEntityParserCtxt(const xmlChar *URL, const xmlChar *ID,
12409
0
                    const xmlChar *base) {
12410
0
    xmlParserCtxtPtr ctxt;
12411
0
    xmlParserInputPtr input;
12412
0
    xmlChar *uri = NULL;
12413
12414
0
    ctxt = xmlNewParserCtxt();
12415
0
    if (ctxt == NULL)
12416
0
  return(NULL);
12417
12418
0
    if (base != NULL) {
12419
0
        if (xmlBuildURISafe(URL, base, &uri) < 0)
12420
0
            goto error;
12421
0
        if (uri != NULL)
12422
0
            URL = uri;
12423
0
    }
12424
12425
0
    input = xmlLoadResource(ctxt, (char *) URL, (char *) ID,
12426
0
                            XML_RESOURCE_UNKNOWN);
12427
0
    if (input == NULL)
12428
0
        goto error;
12429
12430
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
12431
0
        xmlFreeInputStream(input);
12432
0
        goto error;
12433
0
    }
12434
12435
0
    xmlFree(uri);
12436
0
    return(ctxt);
12437
12438
0
error:
12439
0
    xmlFree(uri);
12440
0
    xmlFreeParserCtxt(ctxt);
12441
0
    return(NULL);
12442
0
}
12443
12444
/************************************************************************
12445
 *                  *
12446
 *    Front ends when parsing from a file     *
12447
 *                  *
12448
 ************************************************************************/
12449
12450
/**
12451
 * Create a parser context for a file or URL content.
12452
 * Automatic support for ZLIB/Compress compressed document is provided
12453
 * by default if found at compile-time and for file accesses
12454
 *
12455
 * @deprecated Use #xmlNewParserCtxt and #xmlCtxtReadFile.
12456
 *
12457
 * @param filename  the filename or URL
12458
 * @param options  a combination of xmlParserOption
12459
 * @returns the new parser context or NULL
12460
 */
12461
xmlParserCtxt *
12462
xmlCreateURLParserCtxt(const char *filename, int options)
12463
0
{
12464
0
    xmlParserCtxtPtr ctxt;
12465
0
    xmlParserInputPtr input;
12466
12467
0
    ctxt = xmlNewParserCtxt();
12468
0
    if (ctxt == NULL)
12469
0
  return(NULL);
12470
12471
0
    xmlCtxtUseOptions(ctxt, options);
12472
12473
0
    input = xmlLoadResource(ctxt, filename, NULL, XML_RESOURCE_MAIN_DOCUMENT);
12474
0
    if (input == NULL) {
12475
0
  xmlFreeParserCtxt(ctxt);
12476
0
  return(NULL);
12477
0
    }
12478
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
12479
0
        xmlFreeInputStream(input);
12480
0
        xmlFreeParserCtxt(ctxt);
12481
0
        return(NULL);
12482
0
    }
12483
12484
0
    return(ctxt);
12485
0
}
12486
12487
/**
12488
 * Create a parser context for a file content.
12489
 * Automatic support for ZLIB/Compress compressed document is provided
12490
 * by default if found at compile-time.
12491
 *
12492
 * @deprecated Use #xmlNewParserCtxt and #xmlCtxtReadFile.
12493
 *
12494
 * @param filename  the filename
12495
 * @returns the new parser context or NULL
12496
 */
12497
xmlParserCtxt *
12498
xmlCreateFileParserCtxt(const char *filename)
12499
0
{
12500
0
    return(xmlCreateURLParserCtxt(filename, 0));
12501
0
}
12502
12503
#ifdef LIBXML_SAX1_ENABLED
12504
/**
12505
 * Parse an XML file and build a tree. Automatic support for ZLIB/Compress
12506
 * compressed document is provided by default if found at compile-time.
12507
 * It use the given SAX function block to handle the parsing callback.
12508
 * If sax is NULL, fallback to the default DOM tree building routines.
12509
 *
12510
 * This function uses deprecated global variables to set parser options
12511
 * which default to XML_PARSE_NODICT.
12512
 *
12513
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadFile.
12514
 *
12515
 * User data (void *) is stored within the parser context in the
12516
 * context's _private member, so it is available nearly everywhere in libxml
12517
 *
12518
 * @param sax  the SAX handler block
12519
 * @param filename  the filename
12520
 * @param recovery  work in recovery mode, i.e. tries to read no Well Formed
12521
 *             documents
12522
 * @param data  the userdata
12523
 * @returns the resulting document tree
12524
 */
12525
12526
xmlDoc *
12527
xmlSAXParseFileWithData(xmlSAXHandler *sax, const char *filename,
12528
0
                        int recovery, void *data) {
12529
0
    xmlDocPtr ret = NULL;
12530
0
    xmlParserCtxtPtr ctxt;
12531
0
    xmlParserInputPtr input;
12532
12533
0
    ctxt = xmlNewSAXParserCtxt(sax, NULL);
12534
0
    if (ctxt == NULL)
12535
0
  return(NULL);
12536
12537
0
    if (data != NULL)
12538
0
  ctxt->_private = data;
12539
12540
0
    if (recovery) {
12541
0
        ctxt->options |= XML_PARSE_RECOVER;
12542
0
        ctxt->recovery = 1;
12543
0
    }
12544
12545
0
    if ((filename != NULL) && (filename[0] == '-') && (filename[1] == 0))
12546
0
        input = xmlCtxtNewInputFromFd(ctxt, filename, STDIN_FILENO, NULL, 0);
12547
0
    else
12548
0
        input = xmlCtxtNewInputFromUrl(ctxt, filename, NULL, NULL, 0);
12549
12550
0
    if (input != NULL)
12551
0
        ret = xmlCtxtParseDocument(ctxt, input);
12552
12553
0
    xmlFreeParserCtxt(ctxt);
12554
0
    return(ret);
12555
0
}
12556
12557
/**
12558
 * Parse an XML file and build a tree. Automatic support for ZLIB/Compress
12559
 * compressed document is provided by default if found at compile-time.
12560
 * It use the given SAX function block to handle the parsing callback.
12561
 * If sax is NULL, fallback to the default DOM tree building routines.
12562
 *
12563
 * This function uses deprecated global variables to set parser options
12564
 * which default to XML_PARSE_NODICT.
12565
 *
12566
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadFile.
12567
 *
12568
 * @param sax  the SAX handler block
12569
 * @param filename  the filename
12570
 * @param recovery  work in recovery mode, i.e. tries to read no Well Formed
12571
 *             documents
12572
 * @returns the resulting document tree
12573
 */
12574
12575
xmlDoc *
12576
xmlSAXParseFile(xmlSAXHandler *sax, const char *filename,
12577
0
                          int recovery) {
12578
0
    return(xmlSAXParseFileWithData(sax,filename,recovery,NULL));
12579
0
}
12580
12581
/**
12582
 * Parse an XML in-memory document and build a tree.
12583
 * In the case the document is not Well Formed, a attempt to build a
12584
 * tree is tried anyway
12585
 *
12586
 * This function uses deprecated global variables to set parser options
12587
 * which default to XML_PARSE_NODICT | XML_PARSE_RECOVER.
12588
 *
12589
 * @deprecated Use #xmlReadDoc with XML_PARSE_RECOVER.
12590
 *
12591
 * @param cur  a pointer to an array of xmlChar
12592
 * @returns the resulting document tree or NULL in case of failure
12593
 */
12594
12595
xmlDoc *
12596
0
xmlRecoverDoc(const xmlChar *cur) {
12597
0
    return(xmlSAXParseDoc(NULL, cur, 1));
12598
0
}
12599
12600
/**
12601
 * Parse an XML file and build a tree. Automatic support for ZLIB/Compress
12602
 * compressed document is provided by default if found at compile-time.
12603
 *
12604
 * This function uses deprecated global variables to set parser options
12605
 * which default to XML_PARSE_NODICT.
12606
 *
12607
 * @deprecated Use #xmlReadFile.
12608
 *
12609
 * @param filename  the filename
12610
 * @returns the resulting document tree if the file was wellformed,
12611
 * NULL otherwise.
12612
 */
12613
12614
xmlDoc *
12615
0
xmlParseFile(const char *filename) {
12616
0
    return(xmlSAXParseFile(NULL, filename, 0));
12617
0
}
12618
12619
/**
12620
 * Parse an XML file and build a tree. Automatic support for ZLIB/Compress
12621
 * compressed document is provided by default if found at compile-time.
12622
 * In the case the document is not Well Formed, it attempts to build
12623
 * a tree anyway
12624
 *
12625
 * This function uses deprecated global variables to set parser options
12626
 * which default to XML_PARSE_NODICT | XML_PARSE_RECOVER.
12627
 *
12628
 * @deprecated Use #xmlReadFile with XML_PARSE_RECOVER.
12629
 *
12630
 * @param filename  the filename
12631
 * @returns the resulting document tree or NULL in case of failure
12632
 */
12633
12634
xmlDoc *
12635
0
xmlRecoverFile(const char *filename) {
12636
0
    return(xmlSAXParseFile(NULL, filename, 1));
12637
0
}
12638
12639
12640
/**
12641
 * Setup the parser context to parse a new buffer; Clears any prior
12642
 * contents from the parser context. The buffer parameter must not be
12643
 * NULL, but the filename parameter can be
12644
 *
12645
 * @deprecated Don't use.
12646
 *
12647
 * @param ctxt  an XML parser context
12648
 * @param buffer  a xmlChar * buffer
12649
 * @param filename  a file name
12650
 */
12651
void
12652
xmlSetupParserForBuffer(xmlParserCtxt *ctxt, const xmlChar* buffer,
12653
                             const char* filename)
12654
0
{
12655
0
    xmlParserInputPtr input;
12656
12657
0
    if ((ctxt == NULL) || (buffer == NULL))
12658
0
        return;
12659
12660
0
    xmlCtxtReset(ctxt);
12661
12662
0
    input = xmlCtxtNewInputFromString(ctxt, filename, (const char *) buffer,
12663
0
                                      NULL, 0);
12664
0
    if (input == NULL)
12665
0
        return;
12666
0
    if (xmlCtxtPushInput(ctxt, input) < 0)
12667
0
        xmlFreeInputStream(input);
12668
0
}
12669
12670
/**
12671
 * Parse an XML file and call the given SAX handler routines.
12672
 * Automatic support for ZLIB/Compress compressed document is provided
12673
 *
12674
 * This function uses deprecated global variables to set parser options
12675
 * which default to XML_PARSE_NODICT.
12676
 *
12677
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadFile.
12678
 *
12679
 * @param sax  a SAX handler
12680
 * @param user_data  The user data returned on SAX callbacks
12681
 * @param filename  a file name
12682
 * @returns 0 in case of success or a error number otherwise
12683
 */
12684
int
12685
xmlSAXUserParseFile(xmlSAXHandler *sax, void *user_data,
12686
0
                    const char *filename) {
12687
0
    int ret = 0;
12688
0
    xmlParserCtxtPtr ctxt;
12689
12690
0
    ctxt = xmlCreateFileParserCtxt(filename);
12691
0
    if (ctxt == NULL) return -1;
12692
0
    if (sax != NULL) {
12693
0
        if (sax->initialized == XML_SAX2_MAGIC) {
12694
0
            *ctxt->sax = *sax;
12695
0
        } else {
12696
0
            memset(ctxt->sax, 0, sizeof(*ctxt->sax));
12697
0
            memcpy(ctxt->sax, sax, sizeof(xmlSAXHandlerV1));
12698
0
        }
12699
0
  ctxt->userData = user_data;
12700
0
    }
12701
12702
0
    xmlParseDocument(ctxt);
12703
12704
0
    if (ctxt->wellFormed)
12705
0
  ret = 0;
12706
0
    else {
12707
0
        if (ctxt->errNo != 0)
12708
0
      ret = ctxt->errNo;
12709
0
  else
12710
0
      ret = -1;
12711
0
    }
12712
0
    if (ctxt->myDoc != NULL) {
12713
0
        xmlFreeDoc(ctxt->myDoc);
12714
0
  ctxt->myDoc = NULL;
12715
0
    }
12716
0
    xmlFreeParserCtxt(ctxt);
12717
12718
0
    return ret;
12719
0
}
12720
#endif /* LIBXML_SAX1_ENABLED */
12721
12722
/************************************************************************
12723
 *                  *
12724
 *    Front ends when parsing from memory     *
12725
 *                  *
12726
 ************************************************************************/
12727
12728
/**
12729
 * Create a parser context for an XML in-memory document. The input buffer
12730
 * must not contain a terminating null byte.
12731
 *
12732
 * @param buffer  a pointer to a char array
12733
 * @param size  the size of the array
12734
 * @returns the new parser context or NULL
12735
 */
12736
xmlParserCtxt *
12737
0
xmlCreateMemoryParserCtxt(const char *buffer, int size) {
12738
0
    xmlParserCtxtPtr ctxt;
12739
0
    xmlParserInputPtr input;
12740
12741
0
    if (size < 0)
12742
0
  return(NULL);
12743
12744
0
    ctxt = xmlNewParserCtxt();
12745
0
    if (ctxt == NULL)
12746
0
  return(NULL);
12747
12748
0
    input = xmlCtxtNewInputFromMemory(ctxt, NULL, buffer, size, NULL, 0);
12749
0
    if (input == NULL) {
12750
0
  xmlFreeParserCtxt(ctxt);
12751
0
  return(NULL);
12752
0
    }
12753
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
12754
0
        xmlFreeInputStream(input);
12755
0
        xmlFreeParserCtxt(ctxt);
12756
0
        return(NULL);
12757
0
    }
12758
12759
0
    return(ctxt);
12760
0
}
12761
12762
#ifdef LIBXML_SAX1_ENABLED
12763
/**
12764
 * Parse an XML in-memory block and use the given SAX function block
12765
 * to handle the parsing callback. If sax is NULL, fallback to the default
12766
 * DOM tree building routines.
12767
 *
12768
 * This function uses deprecated global variables to set parser options
12769
 * which default to XML_PARSE_NODICT.
12770
 *
12771
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadMemory.
12772
 *
12773
 * User data (void *) is stored within the parser context in the
12774
 * context's _private member, so it is available nearly everywhere in libxml
12775
 *
12776
 * @param sax  the SAX handler block
12777
 * @param buffer  an pointer to a char array
12778
 * @param size  the size of the array
12779
 * @param recovery  work in recovery mode, i.e. tries to read no Well Formed
12780
 *             documents
12781
 * @param data  the userdata
12782
 * @returns the resulting document tree
12783
 */
12784
12785
xmlDoc *
12786
xmlSAXParseMemoryWithData(xmlSAXHandler *sax, const char *buffer,
12787
0
                          int size, int recovery, void *data) {
12788
0
    xmlDocPtr ret = NULL;
12789
0
    xmlParserCtxtPtr ctxt;
12790
0
    xmlParserInputPtr input;
12791
12792
0
    if (size < 0)
12793
0
        return(NULL);
12794
12795
0
    ctxt = xmlNewSAXParserCtxt(sax, NULL);
12796
0
    if (ctxt == NULL)
12797
0
        return(NULL);
12798
12799
0
    if (data != NULL)
12800
0
  ctxt->_private=data;
12801
12802
0
    if (recovery) {
12803
0
        ctxt->options |= XML_PARSE_RECOVER;
12804
0
        ctxt->recovery = 1;
12805
0
    }
12806
12807
0
    input = xmlCtxtNewInputFromMemory(ctxt, NULL, buffer, size, NULL,
12808
0
                                      XML_INPUT_BUF_STATIC);
12809
12810
0
    if (input != NULL)
12811
0
        ret = xmlCtxtParseDocument(ctxt, input);
12812
12813
0
    xmlFreeParserCtxt(ctxt);
12814
0
    return(ret);
12815
0
}
12816
12817
/**
12818
 * Parse an XML in-memory block and use the given SAX function block
12819
 * to handle the parsing callback. If sax is NULL, fallback to the default
12820
 * DOM tree building routines.
12821
 *
12822
 * This function uses deprecated global variables to set parser options
12823
 * which default to XML_PARSE_NODICT.
12824
 *
12825
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadMemory.
12826
 *
12827
 * @param sax  the SAX handler block
12828
 * @param buffer  an pointer to a char array
12829
 * @param size  the size of the array
12830
 * @param recovery  work in recovery mode, i.e. tries to read not Well Formed
12831
 *             documents
12832
 * @returns the resulting document tree
12833
 */
12834
xmlDoc *
12835
xmlSAXParseMemory(xmlSAXHandler *sax, const char *buffer,
12836
0
            int size, int recovery) {
12837
0
    return xmlSAXParseMemoryWithData(sax, buffer, size, recovery, NULL);
12838
0
}
12839
12840
/**
12841
 * Parse an XML in-memory block and build a tree.
12842
 *
12843
 * This function uses deprecated global variables to set parser options
12844
 * which default to XML_PARSE_NODICT.
12845
 *
12846
 * @deprecated Use #xmlReadMemory.
12847
 *
12848
 * @param buffer  an pointer to a char array
12849
 * @param size  the size of the array
12850
 * @returns the resulting document tree
12851
 */
12852
12853
0
xmlDoc *xmlParseMemory(const char *buffer, int size) {
12854
0
   return(xmlSAXParseMemory(NULL, buffer, size, 0));
12855
0
}
12856
12857
/**
12858
 * Parse an XML in-memory block and build a tree.
12859
 * In the case the document is not Well Formed, an attempt to
12860
 * build a tree is tried anyway
12861
 *
12862
 * This function uses deprecated global variables to set parser options
12863
 * which default to XML_PARSE_NODICT | XML_PARSE_RECOVER.
12864
 *
12865
 * @deprecated Use #xmlReadMemory with XML_PARSE_RECOVER.
12866
 *
12867
 * @param buffer  an pointer to a char array
12868
 * @param size  the size of the array
12869
 * @returns the resulting document tree or NULL in case of error
12870
 */
12871
12872
0
xmlDoc *xmlRecoverMemory(const char *buffer, int size) {
12873
0
   return(xmlSAXParseMemory(NULL, buffer, size, 1));
12874
0
}
12875
12876
/**
12877
 * Parse an XML in-memory buffer and call the given SAX handler routines.
12878
 *
12879
 * This function uses deprecated global variables to set parser options
12880
 * which default to XML_PARSE_NODICT.
12881
 *
12882
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadMemory.
12883
 *
12884
 * @param sax  a SAX handler
12885
 * @param user_data  The user data returned on SAX callbacks
12886
 * @param buffer  an in-memory XML document input
12887
 * @param size  the length of the XML document in bytes
12888
 * @returns 0 in case of success or a error number otherwise
12889
 */
12890
int xmlSAXUserParseMemory(xmlSAXHandler *sax, void *user_data,
12891
0
        const char *buffer, int size) {
12892
0
    int ret = 0;
12893
0
    xmlParserCtxtPtr ctxt;
12894
12895
0
    ctxt = xmlCreateMemoryParserCtxt(buffer, size);
12896
0
    if (ctxt == NULL) return -1;
12897
0
    if (sax != NULL) {
12898
0
        if (sax->initialized == XML_SAX2_MAGIC) {
12899
0
            *ctxt->sax = *sax;
12900
0
        } else {
12901
0
            memset(ctxt->sax, 0, sizeof(*ctxt->sax));
12902
0
            memcpy(ctxt->sax, sax, sizeof(xmlSAXHandlerV1));
12903
0
        }
12904
0
  ctxt->userData = user_data;
12905
0
    }
12906
12907
0
    xmlParseDocument(ctxt);
12908
12909
0
    if (ctxt->wellFormed)
12910
0
  ret = 0;
12911
0
    else {
12912
0
        if (ctxt->errNo != 0)
12913
0
      ret = ctxt->errNo;
12914
0
  else
12915
0
      ret = -1;
12916
0
    }
12917
0
    if (ctxt->myDoc != NULL) {
12918
0
        xmlFreeDoc(ctxt->myDoc);
12919
0
  ctxt->myDoc = NULL;
12920
0
    }
12921
0
    xmlFreeParserCtxt(ctxt);
12922
12923
0
    return ret;
12924
0
}
12925
#endif /* LIBXML_SAX1_ENABLED */
12926
12927
/**
12928
 * Creates a parser context for an XML in-memory document.
12929
 *
12930
 * @param str  a pointer to an array of xmlChar
12931
 * @returns the new parser context or NULL
12932
 */
12933
xmlParserCtxt *
12934
0
xmlCreateDocParserCtxt(const xmlChar *str) {
12935
0
    xmlParserCtxtPtr ctxt;
12936
0
    xmlParserInputPtr input;
12937
12938
0
    ctxt = xmlNewParserCtxt();
12939
0
    if (ctxt == NULL)
12940
0
  return(NULL);
12941
12942
0
    input = xmlCtxtNewInputFromString(ctxt, NULL, (const char *) str, NULL, 0);
12943
0
    if (input == NULL) {
12944
0
  xmlFreeParserCtxt(ctxt);
12945
0
  return(NULL);
12946
0
    }
12947
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
12948
0
        xmlFreeInputStream(input);
12949
0
        xmlFreeParserCtxt(ctxt);
12950
0
        return(NULL);
12951
0
    }
12952
12953
0
    return(ctxt);
12954
0
}
12955
12956
#ifdef LIBXML_SAX1_ENABLED
12957
/**
12958
 * Parse an XML in-memory document and build a tree.
12959
 * It use the given SAX function block to handle the parsing callback.
12960
 * If sax is NULL, fallback to the default DOM tree building routines.
12961
 *
12962
 * This function uses deprecated global variables to set parser options
12963
 * which default to XML_PARSE_NODICT.
12964
 *
12965
 * @deprecated Use #xmlNewSAXParserCtxt and #xmlCtxtReadDoc.
12966
 *
12967
 * @param sax  the SAX handler block
12968
 * @param cur  a pointer to an array of xmlChar
12969
 * @param recovery  work in recovery mode, i.e. tries to read no Well Formed
12970
 *             documents
12971
 * @returns the resulting document tree
12972
 */
12973
12974
xmlDoc *
12975
0
xmlSAXParseDoc(xmlSAXHandler *sax, const xmlChar *cur, int recovery) {
12976
0
    xmlDocPtr ret;
12977
0
    xmlParserCtxtPtr ctxt;
12978
0
    xmlSAXHandlerPtr oldsax = NULL;
12979
12980
0
    if (cur == NULL) return(NULL);
12981
12982
12983
0
    ctxt = xmlCreateDocParserCtxt(cur);
12984
0
    if (ctxt == NULL) return(NULL);
12985
0
    if (sax != NULL) {
12986
0
        oldsax = ctxt->sax;
12987
0
        ctxt->sax = sax;
12988
0
        ctxt->userData = NULL;
12989
0
    }
12990
12991
0
    xmlParseDocument(ctxt);
12992
0
    if ((ctxt->wellFormed) || recovery) ret = ctxt->myDoc;
12993
0
    else {
12994
0
       ret = NULL;
12995
0
       xmlFreeDoc(ctxt->myDoc);
12996
0
       ctxt->myDoc = NULL;
12997
0
    }
12998
0
    if (sax != NULL)
12999
0
  ctxt->sax = oldsax;
13000
0
    xmlFreeParserCtxt(ctxt);
13001
13002
0
    return(ret);
13003
0
}
13004
13005
/**
13006
 * Parse an XML in-memory document and build a tree.
13007
 *
13008
 * This function uses deprecated global variables to set parser options
13009
 * which default to XML_PARSE_NODICT.
13010
 *
13011
 * @deprecated Use #xmlReadDoc.
13012
 *
13013
 * @param cur  a pointer to an array of xmlChar
13014
 * @returns the resulting document tree
13015
 */
13016
13017
xmlDoc *
13018
0
xmlParseDoc(const xmlChar *cur) {
13019
0
    return(xmlSAXParseDoc(NULL, cur, 0));
13020
0
}
13021
#endif /* LIBXML_SAX1_ENABLED */
13022
13023
/************************************************************************
13024
 *                  *
13025
 *  New set (2.6.0) of simpler and more flexible APIs   *
13026
 *                  *
13027
 ************************************************************************/
13028
13029
/**
13030
 * Reset a parser context
13031
 *
13032
 * @param ctxt  an XML parser context
13033
 */
13034
void
13035
xmlCtxtReset(xmlParserCtxt *ctxt)
13036
0
{
13037
0
    xmlParserInputPtr input;
13038
13039
0
    if (ctxt == NULL)
13040
0
        return;
13041
13042
0
    while ((input = xmlCtxtPopInput(ctxt)) != NULL) { /* Non consuming */
13043
0
        xmlFreeInputStream(input);
13044
0
    }
13045
0
    ctxt->inputNr = 0;
13046
0
    ctxt->input = NULL;
13047
13048
0
    ctxt->spaceNr = 0;
13049
0
    if (ctxt->spaceTab != NULL) {
13050
0
  ctxt->spaceTab[0] = -1;
13051
0
  ctxt->space = &ctxt->spaceTab[0];
13052
0
    } else {
13053
0
        ctxt->space = NULL;
13054
0
    }
13055
13056
13057
0
    ctxt->nodeNr = 0;
13058
0
    ctxt->node = NULL;
13059
13060
0
    ctxt->nameNr = 0;
13061
0
    ctxt->name = NULL;
13062
13063
0
    ctxt->nsNr = 0;
13064
0
    xmlParserNsReset(ctxt->nsdb);
13065
13066
0
    if (ctxt->version != NULL) {
13067
0
        xmlFree(ctxt->version);
13068
0
        ctxt->version = NULL;
13069
0
    }
13070
0
    if (ctxt->encoding != NULL) {
13071
0
        xmlFree(ctxt->encoding);
13072
0
        ctxt->encoding = NULL;
13073
0
    }
13074
0
    if (ctxt->extSubURI != NULL) {
13075
0
        xmlFree(ctxt->extSubURI);
13076
0
        ctxt->extSubURI = NULL;
13077
0
    }
13078
0
    if (ctxt->extSubSystem != NULL) {
13079
0
        xmlFree(ctxt->extSubSystem);
13080
0
        ctxt->extSubSystem = NULL;
13081
0
    }
13082
0
    if (ctxt->directory != NULL) {
13083
0
        xmlFree(ctxt->directory);
13084
0
        ctxt->directory = NULL;
13085
0
    }
13086
13087
0
    if (ctxt->myDoc != NULL)
13088
0
        xmlFreeDoc(ctxt->myDoc);
13089
0
    ctxt->myDoc = NULL;
13090
13091
0
    ctxt->standalone = -1;
13092
0
    ctxt->hasExternalSubset = 0;
13093
0
    ctxt->hasPErefs = 0;
13094
0
    ctxt->html = ctxt->html ? 1 : 0;
13095
0
    ctxt->instate = XML_PARSER_START;
13096
13097
0
    ctxt->wellFormed = 1;
13098
0
    ctxt->nsWellFormed = 1;
13099
0
    ctxt->disableSAX = 0;
13100
0
    ctxt->valid = 1;
13101
0
    ctxt->record_info = 0;
13102
0
    ctxt->checkIndex = 0;
13103
0
    ctxt->endCheckState = 0;
13104
0
    ctxt->inSubset = 0;
13105
0
    ctxt->errNo = XML_ERR_OK;
13106
0
    ctxt->depth = 0;
13107
0
    ctxt->catalogs = NULL;
13108
0
    ctxt->sizeentities = 0;
13109
0
    ctxt->sizeentcopy = 0;
13110
0
    xmlInitNodeInfoSeq(&ctxt->node_seq);
13111
13112
0
    if (ctxt->attsDefault != NULL) {
13113
0
        xmlHashFree(ctxt->attsDefault, xmlHashDefaultDeallocator);
13114
0
        ctxt->attsDefault = NULL;
13115
0
    }
13116
0
    if (ctxt->attsSpecial != NULL) {
13117
0
        xmlHashFree(ctxt->attsSpecial, NULL);
13118
0
        ctxt->attsSpecial = NULL;
13119
0
    }
13120
13121
0
#ifdef LIBXML_CATALOG_ENABLED
13122
0
    if (ctxt->catalogs != NULL)
13123
0
  xmlCatalogFreeLocal(ctxt->catalogs);
13124
0
#endif
13125
0
    ctxt->nbErrors = 0;
13126
0
    ctxt->nbWarnings = 0;
13127
0
    if (ctxt->lastError.code != XML_ERR_OK)
13128
0
        xmlResetError(&ctxt->lastError);
13129
0
}
13130
13131
/**
13132
 * Reset a push parser context
13133
 *
13134
 * @param ctxt  an XML parser context
13135
 * @param chunk  a pointer to an array of chars
13136
 * @param size  number of chars in the array
13137
 * @param filename  an optional file name or URI
13138
 * @param encoding  the document encoding, or NULL
13139
 * @returns 0 in case of success and 1 in case of error
13140
 */
13141
int
13142
xmlCtxtResetPush(xmlParserCtxt *ctxt, const char *chunk,
13143
                 int size, const char *filename, const char *encoding)
13144
0
{
13145
0
    xmlParserInputPtr input;
13146
13147
0
    if (ctxt == NULL)
13148
0
        return(1);
13149
13150
0
    xmlCtxtReset(ctxt);
13151
13152
0
    input = xmlNewPushInput(filename, chunk, size);
13153
0
    if (input == NULL)
13154
0
        return(1);
13155
13156
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
13157
0
        xmlFreeInputStream(input);
13158
0
        return(1);
13159
0
    }
13160
13161
0
    if (encoding != NULL)
13162
0
        xmlSwitchEncodingName(ctxt, encoding);
13163
13164
0
    return(0);
13165
0
}
13166
13167
static int
13168
xmlCtxtSetOptionsInternal(xmlParserCtxtPtr ctxt, int options, int keepMask)
13169
0
{
13170
0
    int allMask;
13171
13172
0
    if (ctxt == NULL)
13173
0
        return(-1);
13174
13175
    /*
13176
     * XInclude options aren't handled by the parser.
13177
     *
13178
     * XML_PARSE_XINCLUDE
13179
     * XML_PARSE_NOXINCNODE
13180
     * XML_PARSE_NOBASEFIX
13181
     */
13182
0
    allMask = XML_PARSE_RECOVER |
13183
0
              XML_PARSE_NOENT |
13184
0
              XML_PARSE_DTDLOAD |
13185
0
              XML_PARSE_DTDATTR |
13186
0
              XML_PARSE_DTDVALID |
13187
0
              XML_PARSE_NOERROR |
13188
0
              XML_PARSE_NOWARNING |
13189
0
              XML_PARSE_PEDANTIC |
13190
0
              XML_PARSE_NOBLANKS |
13191
0
#ifdef LIBXML_SAX1_ENABLED
13192
0
              XML_PARSE_SAX1 |
13193
0
#endif
13194
0
              XML_PARSE_NONET |
13195
0
              XML_PARSE_NODICT |
13196
0
              XML_PARSE_NSCLEAN |
13197
0
              XML_PARSE_NOCDATA |
13198
0
              XML_PARSE_COMPACT |
13199
0
              XML_PARSE_OLD10 |
13200
0
              XML_PARSE_HUGE |
13201
0
              XML_PARSE_OLDSAX |
13202
0
              XML_PARSE_IGNORE_ENC |
13203
0
              XML_PARSE_BIG_LINES |
13204
0
              XML_PARSE_NO_XXE |
13205
0
              XML_PARSE_UNZIP |
13206
0
              XML_PARSE_NO_SYS_CATALOG |
13207
0
              XML_PARSE_CATALOG_PI;
13208
13209
0
    ctxt->options = (ctxt->options & keepMask) | (options & allMask);
13210
13211
    /*
13212
     * For some options, struct members are historically the source
13213
     * of truth. The values are initalized from global variables and
13214
     * old code could also modify them directly. Several older API
13215
     * functions that don't take an options argument rely on these
13216
     * deprecated mechanisms.
13217
     *
13218
     * Once public access to struct members and the globals are
13219
     * disabled, we can use the options bitmask as source of
13220
     * truth, making all these struct members obsolete.
13221
     *
13222
     * The XML_DETECT_IDS flags is misnamed. It simply enables
13223
     * loading of the external subset.
13224
     */
13225
0
    ctxt->recovery = (options & XML_PARSE_RECOVER) ? 1 : 0;
13226
0
    ctxt->replaceEntities = (options & XML_PARSE_NOENT) ? 1 : 0;
13227
0
    ctxt->loadsubset = (options & XML_PARSE_DTDLOAD) ? XML_DETECT_IDS : 0;
13228
0
    ctxt->loadsubset |= (options & XML_PARSE_DTDATTR) ? XML_COMPLETE_ATTRS : 0;
13229
0
    ctxt->loadsubset |= (options & XML_PARSE_SKIP_IDS) ? XML_SKIP_IDS : 0;
13230
0
    ctxt->validate = (options & XML_PARSE_DTDVALID) ? 1 : 0;
13231
0
    ctxt->pedantic = (options & XML_PARSE_PEDANTIC) ? 1 : 0;
13232
0
    ctxt->keepBlanks = (options & XML_PARSE_NOBLANKS) ? 0 : 1;
13233
0
    ctxt->dictNames = (options & XML_PARSE_NODICT) ? 0 : 1;
13234
13235
0
    return(options & ~allMask);
13236
0
}
13237
13238
/**
13239
 * Applies the options to the parser context. Unset options are
13240
 * cleared.
13241
 *
13242
 * @since 2.13.0
13243
 *
13244
 * With older versions, you can use #xmlCtxtUseOptions.
13245
 *
13246
 * @param ctxt  an XML parser context
13247
 * @param options  a bitmask of xmlParserOption values
13248
 * @returns 0 in case of success, the set of unknown or unimplemented options
13249
 *         in case of error.
13250
 */
13251
int
13252
xmlCtxtSetOptions(xmlParserCtxt *ctxt, int options)
13253
0
{
13254
0
#ifdef LIBXML_HTML_ENABLED
13255
0
    if ((ctxt != NULL) && (ctxt->html))
13256
0
        return(htmlCtxtSetOptions(ctxt, options));
13257
0
#endif
13258
13259
0
    return(xmlCtxtSetOptionsInternal(ctxt, options, 0));
13260
0
}
13261
13262
/**
13263
 * Get the current options of the parser context.
13264
 *
13265
 * @since 2.14.0
13266
 *
13267
 * @param ctxt  an XML parser context
13268
 * @returns the current options set in the parser context, or -1 if ctxt is NULL.
13269
 */
13270
int
13271
xmlCtxtGetOptions(xmlParserCtxt *ctxt)
13272
0
{
13273
0
    if (ctxt == NULL)
13274
0
        return(-1);
13275
13276
0
    return(ctxt->options);
13277
0
}
13278
13279
/**
13280
 * Applies the options to the parser context. The following options
13281
 * are never cleared and can only be enabled:
13282
 *
13283
 * - XML_PARSE_NOERROR
13284
 * - XML_PARSE_NOWARNING
13285
 * - XML_PARSE_NONET
13286
 * - XML_PARSE_NSCLEAN
13287
 * - XML_PARSE_NOCDATA
13288
 * - XML_PARSE_COMPACT
13289
 * - XML_PARSE_OLD10
13290
 * - XML_PARSE_HUGE
13291
 * - XML_PARSE_OLDSAX
13292
 * - XML_PARSE_IGNORE_ENC
13293
 * - XML_PARSE_BIG_LINES
13294
 *
13295
 * @deprecated Use #xmlCtxtSetOptions.
13296
 *
13297
 * @param ctxt  an XML parser context
13298
 * @param options  a combination of xmlParserOption
13299
 * @returns 0 in case of success, the set of unknown or unimplemented options
13300
 *         in case of error.
13301
 */
13302
int
13303
xmlCtxtUseOptions(xmlParserCtxt *ctxt, int options)
13304
0
{
13305
0
    int keepMask;
13306
13307
0
#ifdef LIBXML_HTML_ENABLED
13308
0
    if ((ctxt != NULL) && (ctxt->html))
13309
0
        return(htmlCtxtUseOptions(ctxt, options));
13310
0
#endif
13311
13312
    /*
13313
     * For historic reasons, some options can only be enabled.
13314
     */
13315
0
    keepMask = XML_PARSE_NOERROR |
13316
0
               XML_PARSE_NOWARNING |
13317
0
               XML_PARSE_NONET |
13318
0
               XML_PARSE_NSCLEAN |
13319
0
               XML_PARSE_NOCDATA |
13320
0
               XML_PARSE_COMPACT |
13321
0
               XML_PARSE_OLD10 |
13322
0
               XML_PARSE_HUGE |
13323
0
               XML_PARSE_OLDSAX |
13324
0
               XML_PARSE_IGNORE_ENC |
13325
0
               XML_PARSE_BIG_LINES;
13326
13327
0
    return(xmlCtxtSetOptionsInternal(ctxt, options, keepMask));
13328
0
}
13329
13330
/**
13331
 * To protect against exponential entity expansion ("billion laughs"), the
13332
 * size of serialized output is (roughly) limited to the input size
13333
 * multiplied by this factor. The default value is 5.
13334
 *
13335
 * When working with documents making heavy use of entity expansion, it can
13336
 * be necessary to increase the value. For security reasons, this should only
13337
 * be considered when processing trusted input.
13338
 *
13339
 * @param ctxt  an XML parser context
13340
 * @param maxAmpl  maximum amplification factor
13341
 */
13342
void
13343
xmlCtxtSetMaxAmplification(xmlParserCtxt *ctxt, unsigned maxAmpl)
13344
0
{
13345
0
    if (ctxt == NULL)
13346
0
        return;
13347
0
    if (maxAmpl == 0)
13348
0
        return;
13349
0
    ctxt->maxAmpl = maxAmpl;
13350
0
}
13351
13352
/**
13353
 * Parse an XML document and return the resulting document tree.
13354
 * Takes ownership of the input object.
13355
 *
13356
 * @since 2.13.0
13357
 *
13358
 * @param ctxt  an XML parser context
13359
 * @param input  parser input
13360
 * @returns the resulting document tree or NULL
13361
 */
13362
xmlDoc *
13363
xmlCtxtParseDocument(xmlParserCtxt *ctxt, xmlParserInput *input)
13364
0
{
13365
0
    xmlDocPtr ret = NULL;
13366
13367
0
    if ((ctxt == NULL) || (input == NULL)) {
13368
0
        xmlFatalErr(ctxt, XML_ERR_ARGUMENT, NULL);
13369
0
        xmlFreeInputStream(input);
13370
0
        return(NULL);
13371
0
    }
13372
13373
    /* assert(ctxt->inputNr == 0); */
13374
0
    while (ctxt->inputNr > 0)
13375
0
        xmlFreeInputStream(xmlCtxtPopInput(ctxt));
13376
13377
0
    if (xmlCtxtPushInput(ctxt, input) < 0) {
13378
0
        xmlFreeInputStream(input);
13379
0
        return(NULL);
13380
0
    }
13381
13382
0
    xmlParseDocument(ctxt);
13383
13384
0
    ret = xmlCtxtGetDocument(ctxt);
13385
13386
    /* assert(ctxt->inputNr == 1); */
13387
0
    while (ctxt->inputNr > 0)
13388
0
        xmlFreeInputStream(xmlCtxtPopInput(ctxt));
13389
13390
0
    return(ret);
13391
0
}
13392
13393
/**
13394
 * Convenience function to parse an XML document from a
13395
 * zero-terminated string.
13396
 *
13397
 * See #xmlCtxtReadDoc for details.
13398
 *
13399
 * @param cur  a pointer to a zero terminated string
13400
 * @param URL  base URL (optional)
13401
 * @param encoding  the document encoding (optional)
13402
 * @param options  a combination of xmlParserOption
13403
 * @returns the resulting document tree
13404
 */
13405
xmlDoc *
13406
xmlReadDoc(const xmlChar *cur, const char *URL, const char *encoding,
13407
           int options)
13408
0
{
13409
0
    xmlParserCtxtPtr ctxt;
13410
0
    xmlParserInputPtr input;
13411
0
    xmlDocPtr doc = NULL;
13412
13413
0
    ctxt = xmlNewParserCtxt();
13414
0
    if (ctxt == NULL)
13415
0
        return(NULL);
13416
13417
0
    xmlCtxtUseOptions(ctxt, options);
13418
13419
0
    input = xmlCtxtNewInputFromString(ctxt, URL, (const char *) cur, encoding,
13420
0
                                      XML_INPUT_BUF_STATIC);
13421
13422
0
    if (input != NULL)
13423
0
        doc = xmlCtxtParseDocument(ctxt, input);
13424
13425
0
    xmlFreeParserCtxt(ctxt);
13426
0
    return(doc);
13427
0
}
13428
13429
/**
13430
 * Convenience function to parse an XML file from the filesystem
13431
 * or a global, user-defined resource loader.
13432
 *
13433
 * If a "-" filename is passed, the function will read from stdin.
13434
 * This feature is potentially insecure and might be removed from
13435
 * later versions.
13436
 *
13437
 * See #xmlCtxtReadFile for details.
13438
 *
13439
 * @param filename  a file or URL
13440
 * @param encoding  the document encoding (optional)
13441
 * @param options  a combination of xmlParserOption
13442
 * @returns the resulting document tree
13443
 */
13444
xmlDoc *
13445
xmlReadFile(const char *filename, const char *encoding, int options)
13446
0
{
13447
0
    xmlParserCtxtPtr ctxt;
13448
0
    xmlParserInputPtr input;
13449
0
    xmlDocPtr doc = NULL;
13450
13451
0
    ctxt = xmlNewParserCtxt();
13452
0
    if (ctxt == NULL)
13453
0
        return(NULL);
13454
13455
0
    xmlCtxtUseOptions(ctxt, options);
13456
13457
    /*
13458
     * Backward compatibility for users of command line utilities like
13459
     * xmlstarlet expecting "-" to mean stdin. This is dangerous and
13460
     * should be removed at some point.
13461
     */
13462
0
    if ((filename != NULL) && (filename[0] == '-') && (filename[1] == 0))
13463
0
        input = xmlCtxtNewInputFromFd(ctxt, filename, STDIN_FILENO,
13464
0
                                      encoding, 0);
13465
0
    else
13466
0
        input = xmlCtxtNewInputFromUrl(ctxt, filename, NULL, encoding, 0);
13467
13468
0
    if (input != NULL)
13469
0
        doc = xmlCtxtParseDocument(ctxt, input);
13470
13471
0
    xmlFreeParserCtxt(ctxt);
13472
0
    return(doc);
13473
0
}
13474
13475
/**
13476
 * Parse an XML in-memory document and build a tree. The input buffer must
13477
 * not contain a terminating null byte.
13478
 *
13479
 * See #xmlCtxtReadMemory for details.
13480
 *
13481
 * @param buffer  a pointer to a char array
13482
 * @param size  the size of the array
13483
 * @param url  base URL (optional)
13484
 * @param encoding  the document encoding (optional)
13485
 * @param options  a combination of xmlParserOption
13486
 * @returns the resulting document tree
13487
 */
13488
xmlDoc *
13489
xmlReadMemory(const char *buffer, int size, const char *url,
13490
              const char *encoding, int options)
13491
0
{
13492
0
    xmlParserCtxtPtr ctxt;
13493
0
    xmlParserInputPtr input;
13494
0
    xmlDocPtr doc = NULL;
13495
13496
0
    if (size < 0)
13497
0
  return(NULL);
13498
13499
0
    ctxt = xmlNewParserCtxt();
13500
0
    if (ctxt == NULL)
13501
0
        return(NULL);
13502
13503
0
    xmlCtxtUseOptions(ctxt, options);
13504
13505
0
    input = xmlCtxtNewInputFromMemory(ctxt, url, buffer, size, encoding,
13506
0
                                      XML_INPUT_BUF_STATIC);
13507
13508
0
    if (input != NULL)
13509
0
        doc = xmlCtxtParseDocument(ctxt, input);
13510
13511
0
    xmlFreeParserCtxt(ctxt);
13512
0
    return(doc);
13513
0
}
13514
13515
/**
13516
 * Parse an XML from a file descriptor and build a tree.
13517
 *
13518
 * See #xmlCtxtReadFd for details.
13519
 *
13520
 * NOTE that the file descriptor will not be closed when the
13521
 * context is freed or reset.
13522
 *
13523
 * @param fd  an open file descriptor
13524
 * @param URL  base URL (optional)
13525
 * @param encoding  the document encoding (optional)
13526
 * @param options  a combination of xmlParserOption
13527
 * @returns the resulting document tree
13528
 */
13529
xmlDoc *
13530
xmlReadFd(int fd, const char *URL, const char *encoding, int options)
13531
0
{
13532
0
    xmlParserCtxtPtr ctxt;
13533
0
    xmlParserInputPtr input;
13534
0
    xmlDocPtr doc = NULL;
13535
13536
0
    ctxt = xmlNewParserCtxt();
13537
0
    if (ctxt == NULL)
13538
0
        return(NULL);
13539
13540
0
    xmlCtxtUseOptions(ctxt, options);
13541
13542
0
    input = xmlCtxtNewInputFromFd(ctxt, URL, fd, encoding, 0);
13543
13544
0
    if (input != NULL)
13545
0
        doc = xmlCtxtParseDocument(ctxt, input);
13546
13547
0
    xmlFreeParserCtxt(ctxt);
13548
0
    return(doc);
13549
0
}
13550
13551
/**
13552
 * Parse an XML document from I/O functions and context and build a tree.
13553
 *
13554
 * See #xmlCtxtReadIO for details.
13555
 *
13556
 * @param ioread  an I/O read function
13557
 * @param ioclose  an I/O close function (optional)
13558
 * @param ioctx  an I/O handler
13559
 * @param URL  base URL (optional)
13560
 * @param encoding  the document encoding (optional)
13561
 * @param options  a combination of xmlParserOption
13562
 * @returns the resulting document tree
13563
 */
13564
xmlDoc *
13565
xmlReadIO(xmlInputReadCallback ioread, xmlInputCloseCallback ioclose,
13566
          void *ioctx, const char *URL, const char *encoding, int options)
13567
0
{
13568
0
    xmlParserCtxtPtr ctxt;
13569
0
    xmlParserInputPtr input;
13570
0
    xmlDocPtr doc = NULL;
13571
13572
0
    ctxt = xmlNewParserCtxt();
13573
0
    if (ctxt == NULL)
13574
0
        return(NULL);
13575
13576
0
    xmlCtxtUseOptions(ctxt, options);
13577
13578
0
    input = xmlCtxtNewInputFromIO(ctxt, URL, ioread, ioclose, ioctx,
13579
0
                                  encoding, 0);
13580
13581
0
    if (input != NULL)
13582
0
        doc = xmlCtxtParseDocument(ctxt, input);
13583
13584
0
    xmlFreeParserCtxt(ctxt);
13585
0
    return(doc);
13586
0
}
13587
13588
/**
13589
 * Parse an XML in-memory document and build a tree.
13590
 *
13591
 * `URL` is used as base to resolve external entities and for error
13592
 * reporting.
13593
 *
13594
 * @param ctxt  an XML parser context
13595
 * @param str  a pointer to a zero terminated string
13596
 * @param URL  base URL (optional)
13597
 * @param encoding  the document encoding (optional)
13598
 * @param options  a combination of xmlParserOption
13599
 * @returns the resulting document tree
13600
 */
13601
xmlDoc *
13602
xmlCtxtReadDoc(xmlParserCtxt *ctxt, const xmlChar *str,
13603
               const char *URL, const char *encoding, int options)
13604
0
{
13605
0
    xmlParserInputPtr input;
13606
13607
0
    if (ctxt == NULL)
13608
0
        return(NULL);
13609
13610
0
    xmlCtxtReset(ctxt);
13611
0
    xmlCtxtUseOptions(ctxt, options);
13612
13613
0
    input = xmlCtxtNewInputFromString(ctxt, URL, (const char *) str, encoding,
13614
0
                                      XML_INPUT_BUF_STATIC);
13615
0
    if (input == NULL)
13616
0
        return(NULL);
13617
13618
0
    return(xmlCtxtParseDocument(ctxt, input));
13619
0
}
13620
13621
/**
13622
 * Parse an XML file from the filesystem or a global, user-defined
13623
 * resource loader.
13624
 *
13625
 * @param ctxt  an XML parser context
13626
 * @param filename  a file or URL
13627
 * @param encoding  the document encoding (optional)
13628
 * @param options  a combination of xmlParserOption
13629
 * @returns the resulting document tree
13630
 */
13631
xmlDoc *
13632
xmlCtxtReadFile(xmlParserCtxt *ctxt, const char *filename,
13633
                const char *encoding, int options)
13634
0
{
13635
0
    xmlParserInputPtr input;
13636
13637
0
    if (ctxt == NULL)
13638
0
        return(NULL);
13639
13640
0
    xmlCtxtReset(ctxt);
13641
0
    xmlCtxtUseOptions(ctxt, options);
13642
13643
0
    input = xmlCtxtNewInputFromUrl(ctxt, filename, NULL, encoding, 0);
13644
0
    if (input == NULL)
13645
0
        return(NULL);
13646
13647
0
    return(xmlCtxtParseDocument(ctxt, input));
13648
0
}
13649
13650
/**
13651
 * Parse an XML in-memory document and build a tree. The input buffer must
13652
 * not contain a terminating null byte.
13653
 *
13654
 * `URL` is used as base to resolve external entities and for error
13655
 * reporting.
13656
 *
13657
 * @param ctxt  an XML parser context
13658
 * @param buffer  a pointer to a char array
13659
 * @param size  the size of the array
13660
 * @param URL  base URL (optional)
13661
 * @param encoding  the document encoding (optional)
13662
 * @param options  a combination of xmlParserOption
13663
 * @returns the resulting document tree
13664
 */
13665
xmlDoc *
13666
xmlCtxtReadMemory(xmlParserCtxt *ctxt, const char *buffer, int size,
13667
                  const char *URL, const char *encoding, int options)
13668
0
{
13669
0
    xmlParserInputPtr input;
13670
13671
0
    if ((ctxt == NULL) || (size < 0))
13672
0
        return(NULL);
13673
13674
0
    xmlCtxtReset(ctxt);
13675
0
    xmlCtxtUseOptions(ctxt, options);
13676
13677
0
    input = xmlCtxtNewInputFromMemory(ctxt, URL, buffer, size, encoding,
13678
0
                                      XML_INPUT_BUF_STATIC);
13679
0
    if (input == NULL)
13680
0
        return(NULL);
13681
13682
0
    return(xmlCtxtParseDocument(ctxt, input));
13683
0
}
13684
13685
/**
13686
 * Parse an XML document from a file descriptor and build a tree.
13687
 *
13688
 * NOTE that the file descriptor will not be closed when the
13689
 * context is freed or reset.
13690
 *
13691
 * `URL` is used as base to resolve external entities and for error
13692
 * reporting.
13693
 *
13694
 * @param ctxt  an XML parser context
13695
 * @param fd  an open file descriptor
13696
 * @param URL  base URL (optional)
13697
 * @param encoding  the document encoding (optional)
13698
 * @param options  a combination of xmlParserOption
13699
 * @returns the resulting document tree
13700
 */
13701
xmlDoc *
13702
xmlCtxtReadFd(xmlParserCtxt *ctxt, int fd,
13703
              const char *URL, const char *encoding, int options)
13704
0
{
13705
0
    xmlParserInputPtr input;
13706
13707
0
    if (ctxt == NULL)
13708
0
        return(NULL);
13709
13710
0
    xmlCtxtReset(ctxt);
13711
0
    xmlCtxtUseOptions(ctxt, options);
13712
13713
0
    input = xmlCtxtNewInputFromFd(ctxt, URL, fd, encoding, 0);
13714
0
    if (input == NULL)
13715
0
        return(NULL);
13716
13717
0
    return(xmlCtxtParseDocument(ctxt, input));
13718
0
}
13719
13720
/**
13721
 * Parse an XML document from I/O functions and source and build a tree.
13722
 * This reuses the existing `ctxt` parser context
13723
 *
13724
 * `URL` is used as base to resolve external entities and for error
13725
 * reporting.
13726
 *
13727
 * @param ctxt  an XML parser context
13728
 * @param ioread  an I/O read function
13729
 * @param ioclose  an I/O close function
13730
 * @param ioctx  an I/O handler
13731
 * @param URL  the base URL to use for the document
13732
 * @param encoding  the document encoding, or NULL
13733
 * @param options  a combination of xmlParserOption
13734
 * @returns the resulting document tree
13735
 */
13736
xmlDoc *
13737
xmlCtxtReadIO(xmlParserCtxt *ctxt, xmlInputReadCallback ioread,
13738
              xmlInputCloseCallback ioclose, void *ioctx,
13739
        const char *URL,
13740
              const char *encoding, int options)
13741
0
{
13742
0
    xmlParserInputPtr input;
13743
13744
0
    if (ctxt == NULL)
13745
0
        return(NULL);
13746
13747
0
    xmlCtxtReset(ctxt);
13748
0
    xmlCtxtUseOptions(ctxt, options);
13749
13750
0
    input = xmlCtxtNewInputFromIO(ctxt, URL, ioread, ioclose, ioctx,
13751
0
                                  encoding, 0);
13752
0
    if (input == NULL)
13753
0
        return(NULL);
13754
13755
0
    return(xmlCtxtParseDocument(ctxt, input));
13756
0
}
13757