Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/xmltodict.py: 58%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1#!/usr/bin/env python
2"Makes working with XML feel like you are working with JSON"
4from xml.parsers import expat
5from xml.sax.saxutils import XMLGenerator, escape
6from xml.sax.xmlreader import AttributesImpl
7from io import StringIO
8from inspect import isgenerator
9import codecs
11class ParsingInterrupted(Exception):
12 pass
15class _DictSAXHandler:
16 def __init__(
17 self,
18 item_depth=0,
19 item_callback=lambda *args: True,
20 xml_attribs=True,
21 attr_prefix="@",
22 cdata_key="#text",
23 force_cdata=False,
24 cdata_separator="",
25 postprocessor=None,
26 dict_constructor=dict,
27 strip_whitespace=True,
28 namespace_separator=":",
29 namespaces=None,
30 force_list=None,
31 comment_key="#comment",
32 ):
33 self.path = []
34 self.stack = []
35 self.data = []
36 self.item = None
37 self.item_depth = item_depth
38 self.xml_attribs = xml_attribs
39 self.item_callback = item_callback
40 self.attr_prefix = attr_prefix
41 self.cdata_key = cdata_key
42 self.force_cdata = force_cdata
43 self.cdata_separator = cdata_separator
44 self.postprocessor = postprocessor
45 self.dict_constructor = dict_constructor
46 self.strip_whitespace = strip_whitespace
47 self.namespace_separator = namespace_separator
48 self.namespaces = namespaces
49 self.namespace_declarations = dict_constructor()
50 self.force_list = force_list
51 self.comment_key = comment_key
53 def _build_name(self, full_name):
54 if self.namespaces is None:
55 return full_name
56 i = full_name.rfind(self.namespace_separator)
57 if i == -1:
58 return full_name
59 namespace, name = full_name[:i], full_name[i+1:]
60 try:
61 short_namespace = self.namespaces[namespace]
62 except KeyError:
63 short_namespace = namespace
64 if not short_namespace:
65 return name
66 else:
67 return self.namespace_separator.join((short_namespace, name))
69 def _attrs_to_dict(self, attrs):
70 if isinstance(attrs, dict):
71 return attrs
72 return self.dict_constructor(zip(attrs[0::2], attrs[1::2]))
74 def startNamespaceDecl(self, prefix, uri):
75 self.namespace_declarations[prefix or ''] = uri
77 def startElement(self, full_name, attrs):
78 name = self._build_name(full_name)
79 attrs = self._attrs_to_dict(attrs)
80 if self.namespace_declarations:
81 if not attrs:
82 attrs = self.dict_constructor()
83 attrs['xmlns'] = self.namespace_declarations
84 self.namespace_declarations = self.dict_constructor()
85 self.path.append((name, attrs or None))
86 if len(self.path) >= self.item_depth:
87 self.stack.append((self.item, self.data))
88 if self.xml_attribs:
89 attr_entries = []
90 for key, value in attrs.items():
91 key = self.attr_prefix+self._build_name(key)
92 if self.postprocessor:
93 entry = self.postprocessor(self.path, key, value)
94 else:
95 entry = (key, value)
96 if entry:
97 attr_entries.append(entry)
98 attrs = self.dict_constructor(attr_entries)
99 else:
100 attrs = None
101 self.item = attrs or None
102 self.data = []
104 def endElement(self, full_name):
105 name = self._build_name(full_name)
106 # If we just closed an item at the streaming depth, emit it and drop it
107 # without attaching it back to its parent. This avoids accumulating all
108 # streamed items in memory when using item_depth > 0.
109 if len(self.path) == self.item_depth:
110 data = (None if not self.data
111 else self.cdata_separator.join(self.data))
112 item = self.item
113 if self.strip_whitespace and data:
114 data = data.strip() or None
115 if data and self._should_force_cdata(name, data) and item is None:
116 item = self.dict_constructor()
117 if item is not None:
118 if data:
119 self.push_data(item, self.cdata_key, data)
120 else:
121 item = data
123 should_continue = self.item_callback(self.path, item)
124 if not should_continue:
125 raise ParsingInterrupted
126 # Reset state for the parent context without keeping a reference to
127 # the emitted item.
128 if self.stack:
129 self.item, self.data = self.stack.pop()
130 else:
131 self.item = None
132 self.data = []
133 self.path.pop()
134 return
135 if self.stack:
136 data = (None if not self.data
137 else self.cdata_separator.join(self.data))
138 item = self.item
139 self.item, self.data = self.stack.pop()
140 if self.strip_whitespace and data:
141 data = data.strip() or None
142 if data and self._should_force_cdata(name, data) and item is None:
143 item = self.dict_constructor()
144 if item is not None:
145 if data:
146 self.push_data(item, self.cdata_key, data)
147 self.item = self.push_data(self.item, name, item)
148 else:
149 self.item = self.push_data(self.item, name, data)
150 else:
151 self.item = None
152 self.data = []
153 self.path.pop()
155 def characters(self, data):
156 if not self.data:
157 self.data = [data]
158 else:
159 self.data.append(data)
161 def comments(self, data):
162 if self.strip_whitespace:
163 data = data.strip()
164 self.item = self.push_data(self.item, self.comment_key, data)
166 def push_data(self, item, key, data):
167 if self.postprocessor is not None:
168 result = self.postprocessor(self.path, key, data)
169 if result is None:
170 return item
171 key, data = result
172 if item is None:
173 item = self.dict_constructor()
174 try:
175 value = item[key]
176 if isinstance(value, list):
177 value.append(data)
178 else:
179 item[key] = [value, data]
180 except KeyError:
181 if self._should_force_list(key, data):
182 item[key] = [data]
183 else:
184 item[key] = data
185 return item
187 def _should_force_list(self, key, value):
188 if not self.force_list:
189 return False
190 if isinstance(self.force_list, bool):
191 return self.force_list
192 try:
193 return key in self.force_list
194 except TypeError:
195 return self.force_list(self.path[:-1], key, value)
197 def _should_force_cdata(self, key, value):
198 if not self.force_cdata:
199 return False
200 if isinstance(self.force_cdata, bool):
201 return self.force_cdata
202 try:
203 return key in self.force_cdata
204 except TypeError:
205 return self.force_cdata(self.path[:-1], key, value)
208def parse(xml_input, encoding=None, expat=expat, process_namespaces=False,
209 namespace_separator=':', disable_entities=True, process_comments=False, **kwargs):
210 """Parse the given XML input and convert it into a dictionary.
212 `xml_input` can either be a `string`, a file-like object, or a generator of strings.
214 If `xml_attribs` is `True`, element attributes are put in the dictionary
215 among regular child elements, using `@` as a prefix to avoid collisions. If
216 set to `False`, they are just ignored.
218 Simple example::
220 >>> import xmltodict
221 >>> doc = xmltodict.parse(\"\"\"
222 ... <a prop="x">
223 ... <b>1</b>
224 ... <b>2</b>
225 ... </a>
226 ... \"\"\")
227 >>> doc['a']['@prop']
228 'x'
229 >>> doc['a']['b']
230 ['1', '2']
232 If `item_depth` is `0`, the function returns a dictionary for the root
233 element (default behavior). Otherwise, it calls `item_callback` every time
234 an item at the specified depth is found and returns `None` in the end
235 (streaming mode).
237 The callback function receives two parameters: the `path` from the document
238 root to the item (name-attribs pairs), and the `item` (dict). If the
239 callback's return value is false-ish, parsing will be stopped with the
240 :class:`ParsingInterrupted` exception.
242 Streaming example::
244 >>> def handle(path, item):
245 ... print('path:%s item:%s' % (path, item))
246 ... return True
247 ...
248 >>> xmltodict.parse(\"\"\"
249 ... <a prop="x">
250 ... <b>1</b>
251 ... <b>2</b>
252 ... </a>\"\"\", item_depth=2, item_callback=handle)
253 path:[('a', {'prop': 'x'}), ('b', None)] item:1
254 path:[('a', {'prop': 'x'}), ('b', None)] item:2
256 The optional argument `postprocessor` is a function that takes `path`,
257 `key` and `value` as positional arguments and returns a new `(key, value)`
258 pair where both `key` and `value` may have changed. Usage example::
260 >>> def postprocessor(path, key, value):
261 ... try:
262 ... return key + ':int', int(value)
263 ... except (ValueError, TypeError):
264 ... return key, value
265 >>> xmltodict.parse('<a><b>1</b><b>2</b><b>x</b></a>',
266 ... postprocessor=postprocessor)
267 {'a': {'b:int': [1, 2], 'b': 'x'}}
269 You can pass an alternate version of `expat` (such as `defusedexpat`) by
270 using the `expat` parameter. E.g:
272 >>> import defusedexpat
273 >>> xmltodict.parse('<a>hello</a>', expat=defusedexpat.pyexpat)
274 {'a': 'hello'}
276 You can use the force_list argument to force lists to be created even
277 when there is only a single child of a given level of hierarchy. The
278 force_list argument is a tuple of keys. If the key for a given level
279 of hierarchy is in the force_list argument, that level of hierarchy
280 will have a list as a child (even if there is only one sub-element).
281 The index_keys operation takes precedence over this. This is applied
282 after any user-supplied postprocessor has already run.
284 For example, given this input:
285 <servers>
286 <server>
287 <name>host1</name>
288 <os>Linux</os>
289 <interfaces>
290 <interface>
291 <name>em0</name>
292 <ip_address>10.0.0.1</ip_address>
293 </interface>
294 </interfaces>
295 </server>
296 </servers>
298 If called with force_list=('interface',), it will produce
299 this dictionary:
300 {'servers':
301 {'server':
302 {'name': 'host1',
303 'os': 'Linux'},
304 'interfaces':
305 {'interface':
306 [ {'name': 'em0', 'ip_address': '10.0.0.1' } ] } } }
308 `force_list` can also be a callable that receives `path`, `key` and
309 `value`. This is helpful in cases where the logic that decides whether
310 a list should be forced is more complex.
313 If `process_comments` is `True`, comments will be added using `comment_key`
314 (default=`'#comment'`) to the tag that contains the comment.
316 For example, given this input:
317 <a>
318 <b>
319 <!-- b comment -->
320 <c>
321 <!-- c comment -->
322 1
323 </c>
324 <d>2</d>
325 </b>
326 </a>
328 If called with `process_comments=True`, it will produce
329 this dictionary:
330 'a': {
331 'b': {
332 '#comment': 'b comment',
333 'c': {
335 '#comment': 'c comment',
336 '#text': '1',
337 },
338 'd': '2',
339 },
340 }
341 Comment text is subject to the `strip_whitespace` flag: when it is left
342 at the default `True`, comments will have leading and trailing
343 whitespace removed. Disable `strip_whitespace` to keep comment
344 indentation or padding intact.
345 """
346 handler = _DictSAXHandler(namespace_separator=namespace_separator,
347 **kwargs)
348 if isinstance(xml_input, str):
349 encoding = encoding or 'utf-8'
350 xml_input = xml_input.encode(encoding)
351 if not process_namespaces:
352 namespace_separator = None
353 parser = expat.ParserCreate(
354 encoding,
355 namespace_separator
356 )
357 parser.ordered_attributes = True
358 parser.StartNamespaceDeclHandler = handler.startNamespaceDecl
359 parser.StartElementHandler = handler.startElement
360 parser.EndElementHandler = handler.endElement
361 parser.CharacterDataHandler = handler.characters
362 if process_comments:
363 parser.CommentHandler = handler.comments
364 parser.buffer_text = True
365 if disable_entities:
366 def _forbid_entities(*_args, **_kwargs):
367 raise ValueError("entities are disabled")
369 parser.EntityDeclHandler = _forbid_entities
370 if hasattr(xml_input, 'read'):
371 parser.ParseFile(xml_input)
372 elif isgenerator(xml_input):
373 for chunk in xml_input:
374 parser.Parse(chunk, False)
375 parser.Parse(b'', True)
376 else:
377 parser.Parse(xml_input, True)
378 return handler.item
381def _convert_value_to_string(value, encoding='utf-8', bytes_errors='replace'):
382 """Convert a value to its string representation for XML output.
384 Handles boolean values consistently by converting them to lowercase.
385 """
386 if isinstance(value, str):
387 return value
388 if isinstance(value, bool):
389 return "true" if value else "false"
390 if isinstance(value, (bytes, bytearray, memoryview)):
391 return bytes(value).decode(encoding, errors=bytes_errors)
392 return str(value)
395def _validate_name(value, kind):
396 """Validate an element/attribute name for XML safety.
398 Raises ValueError with a specific reason when invalid.
400 kind: 'element' or 'attribute' (used in error messages)
401 """
402 if not isinstance(value, str):
403 raise ValueError(f"{kind} name must be a string")
404 if value.startswith("?") or value.startswith("!"):
405 raise ValueError(f'Invalid {kind} name: cannot start with "?" or "!"')
406 if "<" in value or ">" in value:
407 raise ValueError(f'Invalid {kind} name: "<" or ">" not allowed')
408 if "/" in value:
409 raise ValueError(f'Invalid {kind} name: "/" not allowed')
410 if '"' in value or "'" in value:
411 raise ValueError(f"Invalid {kind} name: quotes not allowed")
412 if "=" in value:
413 raise ValueError(f'Invalid {kind} name: "=" not allowed')
414 if any(ch.isspace() for ch in value):
415 raise ValueError(f"Invalid {kind} name: whitespace not allowed")
418def _validate_comment(value):
419 if isinstance(value, bytes):
420 try:
421 value = value.decode("utf-8")
422 except UnicodeDecodeError as exc:
423 raise ValueError("Comment text must be valid UTF-8") from exc
424 if not isinstance(value, str):
425 raise ValueError("Comment text must be a string")
426 if "--" in value:
427 raise ValueError("Comment text cannot contain '--'")
428 if value.endswith("-"):
429 raise ValueError("Comment text cannot end with '-'")
430 return value
433def _process_namespace(name, namespaces, ns_sep=':', attr_prefix='@'):
434 if not isinstance(name, str):
435 return name
436 if not namespaces:
437 return name
438 try:
439 ns, name = name.rsplit(ns_sep, 1)
440 except ValueError:
441 pass
442 else:
443 ns_res = namespaces.get(ns.strip(attr_prefix))
444 name = '{}{}{}{}'.format(
445 attr_prefix if ns.startswith(attr_prefix) else '',
446 ns_res, ns_sep, name) if ns_res else name
447 return name
450def _emit(key, value, content_handler,
451 attr_prefix='@',
452 cdata_key='#text',
453 depth=0,
454 preprocessor=None,
455 pretty=False,
456 newl='\n',
457 indent='\t',
458 namespace_separator=':',
459 namespaces=None,
460 full_document=True,
461 expand_iter=None,
462 encoding='utf-8',
463 bytes_errors='replace',
464 comment_key='#comment'):
465 if isinstance(key, str) and key == comment_key:
466 comments_list = value if isinstance(value, list) else [value]
467 if isinstance(indent, int):
468 indent = " " * indent
469 for comment_text in comments_list:
470 if comment_text is None:
471 continue
472 comment_text = _convert_value_to_string(
473 comment_text, encoding=encoding, bytes_errors=bytes_errors
474 )
475 if not comment_text:
476 continue
477 if pretty:
478 content_handler.ignorableWhitespace(depth * indent)
479 content_handler.comment(comment_text)
480 if pretty:
481 content_handler.ignorableWhitespace(newl)
482 return
484 key = _process_namespace(key, namespaces, namespace_separator, attr_prefix)
485 if preprocessor is not None:
486 result = preprocessor(key, value)
487 if result is None:
488 return
489 key, value = result
490 # Minimal validation to avoid breaking out of tag context
491 _validate_name(key, "element")
492 if not hasattr(value, '__iter__') or isinstance(value, (str, bytes, bytearray, memoryview, dict)):
493 value = [value]
494 for index, v in enumerate(value):
495 if full_document and depth == 0 and index > 0:
496 raise ValueError('document with multiple roots')
497 if v is None:
498 v = {}
499 elif not isinstance(v, (dict, str)):
500 if expand_iter and hasattr(v, '__iter__') and not isinstance(v, (bytes, bytearray, memoryview)):
501 v = {expand_iter: v}
502 else:
503 v = _convert_value_to_string(v, encoding=encoding, bytes_errors=bytes_errors)
504 if isinstance(v, str):
505 v = {cdata_key: v}
506 cdata = None
507 attrs = {}
508 children = []
509 for ik, iv in v.items():
510 if ik == cdata_key:
511 if iv is None:
512 cdata = None
513 else:
514 cdata = _convert_value_to_string(iv, encoding=encoding, bytes_errors=bytes_errors)
515 continue
516 if isinstance(ik, str) and ik.startswith(attr_prefix):
517 ik = _process_namespace(ik, namespaces, namespace_separator,
518 attr_prefix)
519 if ik == attr_prefix + 'xmlns' and isinstance(iv, dict):
520 for k, v in iv.items():
521 _validate_name(k, "attribute")
522 attr = 'xmlns{}'.format(f':{k}' if k else '')
523 attrs[attr] = '' if v is None else _convert_value_to_string(
524 v, encoding=encoding, bytes_errors=bytes_errors
525 )
526 continue
527 if iv is None:
528 iv = ''
529 elif not isinstance(iv, str):
530 iv = _convert_value_to_string(iv, encoding=encoding, bytes_errors=bytes_errors)
531 attr_name = ik[len(attr_prefix) :]
532 _validate_name(attr_name, "attribute")
533 attrs[attr_name] = iv
534 continue
535 if isinstance(iv, list) and not iv:
536 continue # Skip empty lists to avoid creating empty child elements
537 children.append((ik, iv))
538 if isinstance(indent, int):
539 indent = ' ' * indent
540 if pretty:
541 content_handler.ignorableWhitespace(depth * indent)
542 content_handler.startElement(key, AttributesImpl(attrs))
543 if pretty and children:
544 content_handler.ignorableWhitespace(newl)
545 for child_key, child_value in children:
546 _emit(child_key, child_value, content_handler,
547 attr_prefix, cdata_key, depth+1, preprocessor,
548 pretty, newl, indent, namespaces=namespaces,
549 namespace_separator=namespace_separator,
550 expand_iter=expand_iter, encoding=encoding,
551 bytes_errors=bytes_errors, comment_key=comment_key)
552 if cdata is not None:
553 content_handler.characters(cdata)
554 if pretty and children:
555 content_handler.ignorableWhitespace(depth * indent)
556 content_handler.endElement(key)
557 if pretty and depth:
558 content_handler.ignorableWhitespace(newl)
561class _XMLGenerator(XMLGenerator):
562 def comment(self, text):
563 text = _validate_comment(text)
564 self._write(f"<!--{escape(text)}-->")
567def unparse(input_dict, output=None, encoding='utf-8', full_document=True,
568 short_empty_elements=False, comment_key='#comment',
569 **kwargs):
570 """Emit an XML document for the given `input_dict` (reverse of `parse`).
572 The resulting XML document is returned as a string, but if `output` (a
573 file-like object) is specified, it is written there instead.
575 Dictionary keys prefixed with `attr_prefix` (default=`'@'`) are interpreted
576 as XML node attributes, whereas keys equal to `cdata_key`
577 (default=`'#text'`) are treated as character data.
579 Empty lists are omitted entirely: ``{"a": []}`` produces no ``<a>`` element.
580 Provide a placeholder entry (for example ``{"a": [""]}``) when an explicit
581 empty container element must be emitted.
583 The `pretty` parameter (default=`False`) enables pretty-printing. In this
584 mode, lines are terminated with `'\n'` and indented with `'\t'`, but this
585 can be customized with the `newl` and `indent` parameters.
586 The `bytes_errors` parameter controls decoding errors for byte values and
587 defaults to `'replace'`.
589 """
590 bytes_errors = kwargs.pop('bytes_errors', 'replace')
591 try:
592 codecs.lookup_error(bytes_errors)
593 except LookupError as exc:
594 raise ValueError(f"Invalid bytes_errors handler: {bytes_errors}") from exc
596 must_return = False
597 if output is None:
598 output = StringIO()
599 must_return = True
600 if short_empty_elements:
601 content_handler = _XMLGenerator(output, encoding, True)
602 else:
603 content_handler = _XMLGenerator(output, encoding)
604 if full_document:
605 content_handler.startDocument()
606 seen_root = False
607 for key, value in input_dict.items():
608 if key != comment_key and full_document and seen_root:
609 raise ValueError("Document must have exactly one root.")
610 _emit(
611 key,
612 value,
613 content_handler,
614 full_document=full_document,
615 encoding=encoding,
616 bytes_errors=bytes_errors,
617 comment_key=comment_key,
618 **kwargs,
619 )
620 if key != comment_key:
621 seen_root = True
622 if full_document and not seen_root:
623 raise ValueError("Document must have exactly one root.")
624 if full_document:
625 content_handler.endDocument()
626 if must_return:
627 value = output.getvalue()
628 try: # pragma no cover
629 value = value.decode(encoding)
630 except AttributeError: # pragma no cover
631 pass
632 return value
635if __name__ == '__main__': # pragma: no cover
636 import marshal
637 import sys
639 stdin = sys.stdin.buffer
640 stdout = sys.stdout.buffer
642 (item_depth,) = sys.argv[1:]
643 item_depth = int(item_depth)
645 def handle_item(path, item):
646 marshal.dump((path, item), stdout)
647 return True
649 try:
650 root = parse(stdin,
651 item_depth=item_depth,
652 item_callback=handle_item,
653 dict_constructor=dict)
654 if item_depth == 0:
655 handle_item([], root)
656 except KeyboardInterrupt:
657 pass