Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/xmltodict.py: 58%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

347 statements  

1#!/usr/bin/env python 

2"Makes working with XML feel like you are working with JSON" 

3 

4from xml.parsers import expat 

5from xml.sax.saxutils import XMLGenerator, escape 

6from xml.sax.xmlreader import AttributesImpl 

7from io import StringIO 

8from inspect import isgenerator 

9import codecs 

10 

11class ParsingInterrupted(Exception): 

12 pass 

13 

14 

15class _DictSAXHandler: 

16 def __init__( 

17 self, 

18 item_depth=0, 

19 item_callback=lambda *args: True, 

20 xml_attribs=True, 

21 attr_prefix="@", 

22 cdata_key="#text", 

23 force_cdata=False, 

24 cdata_separator="", 

25 postprocessor=None, 

26 dict_constructor=dict, 

27 strip_whitespace=True, 

28 namespace_separator=":", 

29 namespaces=None, 

30 force_list=None, 

31 comment_key="#comment", 

32 ): 

33 self.path = [] 

34 self.stack = [] 

35 self.data = [] 

36 self.item = None 

37 self.item_depth = item_depth 

38 self.xml_attribs = xml_attribs 

39 self.item_callback = item_callback 

40 self.attr_prefix = attr_prefix 

41 self.cdata_key = cdata_key 

42 self.force_cdata = force_cdata 

43 self.cdata_separator = cdata_separator 

44 self.postprocessor = postprocessor 

45 self.dict_constructor = dict_constructor 

46 self.strip_whitespace = strip_whitespace 

47 self.namespace_separator = namespace_separator 

48 self.namespaces = namespaces 

49 self.namespace_declarations = dict_constructor() 

50 self.force_list = force_list 

51 self.comment_key = comment_key 

52 

53 def _build_name(self, full_name): 

54 if self.namespaces is None: 

55 return full_name 

56 i = full_name.rfind(self.namespace_separator) 

57 if i == -1: 

58 return full_name 

59 namespace, name = full_name[:i], full_name[i+1:] 

60 try: 

61 short_namespace = self.namespaces[namespace] 

62 except KeyError: 

63 short_namespace = namespace 

64 if not short_namespace: 

65 return name 

66 else: 

67 return self.namespace_separator.join((short_namespace, name)) 

68 

69 def _attrs_to_dict(self, attrs): 

70 if isinstance(attrs, dict): 

71 return attrs 

72 return self.dict_constructor(zip(attrs[0::2], attrs[1::2])) 

73 

74 def startNamespaceDecl(self, prefix, uri): 

75 self.namespace_declarations[prefix or ''] = uri 

76 

77 def startElement(self, full_name, attrs): 

78 name = self._build_name(full_name) 

79 attrs = self._attrs_to_dict(attrs) 

80 if self.namespace_declarations: 

81 if not attrs: 

82 attrs = self.dict_constructor() 

83 attrs['xmlns'] = self.namespace_declarations 

84 self.namespace_declarations = self.dict_constructor() 

85 self.path.append((name, attrs or None)) 

86 if len(self.path) >= self.item_depth: 

87 self.stack.append((self.item, self.data)) 

88 if self.xml_attribs: 

89 attr_entries = [] 

90 for key, value in attrs.items(): 

91 key = self.attr_prefix+self._build_name(key) 

92 if self.postprocessor: 

93 entry = self.postprocessor(self.path, key, value) 

94 else: 

95 entry = (key, value) 

96 if entry: 

97 attr_entries.append(entry) 

98 attrs = self.dict_constructor(attr_entries) 

99 else: 

100 attrs = None 

101 self.item = attrs or None 

102 self.data = [] 

103 

104 def endElement(self, full_name): 

105 name = self._build_name(full_name) 

106 # If we just closed an item at the streaming depth, emit it and drop it 

107 # without attaching it back to its parent. This avoids accumulating all 

108 # streamed items in memory when using item_depth > 0. 

109 if len(self.path) == self.item_depth: 

110 data = (None if not self.data 

111 else self.cdata_separator.join(self.data)) 

112 item = self.item 

113 if self.strip_whitespace and data: 

114 data = data.strip() or None 

115 if data and self._should_force_cdata(name, data) and item is None: 

116 item = self.dict_constructor() 

117 if item is not None: 

118 if data: 

119 self.push_data(item, self.cdata_key, data) 

120 else: 

121 item = data 

122 

123 should_continue = self.item_callback(self.path, item) 

124 if not should_continue: 

125 raise ParsingInterrupted 

126 # Reset state for the parent context without keeping a reference to 

127 # the emitted item. 

128 if self.stack: 

129 self.item, self.data = self.stack.pop() 

130 else: 

131 self.item = None 

132 self.data = [] 

133 self.path.pop() 

134 return 

135 if self.stack: 

136 data = (None if not self.data 

137 else self.cdata_separator.join(self.data)) 

138 item = self.item 

139 self.item, self.data = self.stack.pop() 

140 if self.strip_whitespace and data: 

141 data = data.strip() or None 

142 if data and self._should_force_cdata(name, data) and item is None: 

143 item = self.dict_constructor() 

144 if item is not None: 

145 if data: 

146 self.push_data(item, self.cdata_key, data) 

147 self.item = self.push_data(self.item, name, item) 

148 else: 

149 self.item = self.push_data(self.item, name, data) 

150 else: 

151 self.item = None 

152 self.data = [] 

153 self.path.pop() 

154 

155 def characters(self, data): 

156 if not self.data: 

157 self.data = [data] 

158 else: 

159 self.data.append(data) 

160 

161 def comments(self, data): 

162 if self.strip_whitespace: 

163 data = data.strip() 

164 self.item = self.push_data(self.item, self.comment_key, data) 

165 

166 def push_data(self, item, key, data): 

167 if self.postprocessor is not None: 

168 result = self.postprocessor(self.path, key, data) 

169 if result is None: 

170 return item 

171 key, data = result 

172 if item is None: 

173 item = self.dict_constructor() 

174 try: 

175 value = item[key] 

176 if isinstance(value, list): 

177 value.append(data) 

178 else: 

179 item[key] = [value, data] 

180 except KeyError: 

181 if self._should_force_list(key, data): 

182 item[key] = [data] 

183 else: 

184 item[key] = data 

185 return item 

186 

187 def _should_force_list(self, key, value): 

188 if not self.force_list: 

189 return False 

190 if isinstance(self.force_list, bool): 

191 return self.force_list 

192 try: 

193 return key in self.force_list 

194 except TypeError: 

195 return self.force_list(self.path[:-1], key, value) 

196 

197 def _should_force_cdata(self, key, value): 

198 if not self.force_cdata: 

199 return False 

200 if isinstance(self.force_cdata, bool): 

201 return self.force_cdata 

202 try: 

203 return key in self.force_cdata 

204 except TypeError: 

205 return self.force_cdata(self.path[:-1], key, value) 

206 

207 

208def parse(xml_input, encoding=None, expat=expat, process_namespaces=False, 

209 namespace_separator=':', disable_entities=True, process_comments=False, **kwargs): 

210 """Parse the given XML input and convert it into a dictionary. 

211 

212 `xml_input` can either be a `string`, a file-like object, or a generator of strings. 

213 

214 If `xml_attribs` is `True`, element attributes are put in the dictionary 

215 among regular child elements, using `@` as a prefix to avoid collisions. If 

216 set to `False`, they are just ignored. 

217 

218 Simple example:: 

219 

220 >>> import xmltodict 

221 >>> doc = xmltodict.parse(\"\"\" 

222 ... <a prop="x"> 

223 ... <b>1</b> 

224 ... <b>2</b> 

225 ... </a> 

226 ... \"\"\") 

227 >>> doc['a']['@prop'] 

228 'x' 

229 >>> doc['a']['b'] 

230 ['1', '2'] 

231 

232 If `item_depth` is `0`, the function returns a dictionary for the root 

233 element (default behavior). Otherwise, it calls `item_callback` every time 

234 an item at the specified depth is found and returns `None` in the end 

235 (streaming mode). 

236 

237 The callback function receives two parameters: the `path` from the document 

238 root to the item (name-attribs pairs), and the `item` (dict). If the 

239 callback's return value is false-ish, parsing will be stopped with the 

240 :class:`ParsingInterrupted` exception. 

241 

242 Streaming example:: 

243 

244 >>> def handle(path, item): 

245 ... print('path:%s item:%s' % (path, item)) 

246 ... return True 

247 ... 

248 >>> xmltodict.parse(\"\"\" 

249 ... <a prop="x"> 

250 ... <b>1</b> 

251 ... <b>2</b> 

252 ... </a>\"\"\", item_depth=2, item_callback=handle) 

253 path:[('a', {'prop': 'x'}), ('b', None)] item:1 

254 path:[('a', {'prop': 'x'}), ('b', None)] item:2 

255 

256 The optional argument `postprocessor` is a function that takes `path`, 

257 `key` and `value` as positional arguments and returns a new `(key, value)` 

258 pair where both `key` and `value` may have changed. Usage example:: 

259 

260 >>> def postprocessor(path, key, value): 

261 ... try: 

262 ... return key + ':int', int(value) 

263 ... except (ValueError, TypeError): 

264 ... return key, value 

265 >>> xmltodict.parse('<a><b>1</b><b>2</b><b>x</b></a>', 

266 ... postprocessor=postprocessor) 

267 {'a': {'b:int': [1, 2], 'b': 'x'}} 

268 

269 You can pass an alternate version of `expat` (such as `defusedexpat`) by 

270 using the `expat` parameter. E.g: 

271 

272 >>> import defusedexpat 

273 >>> xmltodict.parse('<a>hello</a>', expat=defusedexpat.pyexpat) 

274 {'a': 'hello'} 

275 

276 You can use the force_list argument to force lists to be created even 

277 when there is only a single child of a given level of hierarchy. The 

278 force_list argument is a tuple of keys. If the key for a given level 

279 of hierarchy is in the force_list argument, that level of hierarchy 

280 will have a list as a child (even if there is only one sub-element). 

281 The index_keys operation takes precedence over this. This is applied 

282 after any user-supplied postprocessor has already run. 

283 

284 For example, given this input: 

285 <servers> 

286 <server> 

287 <name>host1</name> 

288 <os>Linux</os> 

289 <interfaces> 

290 <interface> 

291 <name>em0</name> 

292 <ip_address>10.0.0.1</ip_address> 

293 </interface> 

294 </interfaces> 

295 </server> 

296 </servers> 

297 

298 If called with force_list=('interface',), it will produce 

299 this dictionary: 

300 {'servers': 

301 {'server': 

302 {'name': 'host1', 

303 'os': 'Linux'}, 

304 'interfaces': 

305 {'interface': 

306 [ {'name': 'em0', 'ip_address': '10.0.0.1' } ] } } } 

307 

308 `force_list` can also be a callable that receives `path`, `key` and 

309 `value`. This is helpful in cases where the logic that decides whether 

310 a list should be forced is more complex. 

311 

312 

313 If `process_comments` is `True`, comments will be added using `comment_key` 

314 (default=`'#comment'`) to the tag that contains the comment. 

315 

316 For example, given this input: 

317 <a> 

318 <b> 

319 <!-- b comment --> 

320 <c> 

321 <!-- c comment --> 

322 1 

323 </c> 

324 <d>2</d> 

325 </b> 

326 </a> 

327 

328 If called with `process_comments=True`, it will produce 

329 this dictionary: 

330 'a': { 

331 'b': { 

332 '#comment': 'b comment', 

333 'c': { 

334 

335 '#comment': 'c comment', 

336 '#text': '1', 

337 }, 

338 'd': '2', 

339 }, 

340 } 

341 Comment text is subject to the `strip_whitespace` flag: when it is left 

342 at the default `True`, comments will have leading and trailing 

343 whitespace removed. Disable `strip_whitespace` to keep comment 

344 indentation or padding intact. 

345 """ 

346 handler = _DictSAXHandler(namespace_separator=namespace_separator, 

347 **kwargs) 

348 if isinstance(xml_input, str): 

349 encoding = encoding or 'utf-8' 

350 xml_input = xml_input.encode(encoding) 

351 if not process_namespaces: 

352 namespace_separator = None 

353 parser = expat.ParserCreate( 

354 encoding, 

355 namespace_separator 

356 ) 

357 parser.ordered_attributes = True 

358 parser.StartNamespaceDeclHandler = handler.startNamespaceDecl 

359 parser.StartElementHandler = handler.startElement 

360 parser.EndElementHandler = handler.endElement 

361 parser.CharacterDataHandler = handler.characters 

362 if process_comments: 

363 parser.CommentHandler = handler.comments 

364 parser.buffer_text = True 

365 if disable_entities: 

366 def _forbid_entities(*_args, **_kwargs): 

367 raise ValueError("entities are disabled") 

368 

369 parser.EntityDeclHandler = _forbid_entities 

370 if hasattr(xml_input, 'read'): 

371 parser.ParseFile(xml_input) 

372 elif isgenerator(xml_input): 

373 for chunk in xml_input: 

374 parser.Parse(chunk, False) 

375 parser.Parse(b'', True) 

376 else: 

377 parser.Parse(xml_input, True) 

378 return handler.item 

379 

380 

381def _convert_value_to_string(value, encoding='utf-8', bytes_errors='replace'): 

382 """Convert a value to its string representation for XML output. 

383 

384 Handles boolean values consistently by converting them to lowercase. 

385 """ 

386 if isinstance(value, str): 

387 return value 

388 if isinstance(value, bool): 

389 return "true" if value else "false" 

390 if isinstance(value, (bytes, bytearray, memoryview)): 

391 return bytes(value).decode(encoding, errors=bytes_errors) 

392 return str(value) 

393 

394 

395def _validate_name(value, kind): 

396 """Validate an element/attribute name for XML safety. 

397 

398 Raises ValueError with a specific reason when invalid. 

399 

400 kind: 'element' or 'attribute' (used in error messages) 

401 """ 

402 if not isinstance(value, str): 

403 raise ValueError(f"{kind} name must be a string") 

404 if value.startswith("?") or value.startswith("!"): 

405 raise ValueError(f'Invalid {kind} name: cannot start with "?" or "!"') 

406 if "<" in value or ">" in value: 

407 raise ValueError(f'Invalid {kind} name: "<" or ">" not allowed') 

408 if "/" in value: 

409 raise ValueError(f'Invalid {kind} name: "/" not allowed') 

410 if '"' in value or "'" in value: 

411 raise ValueError(f"Invalid {kind} name: quotes not allowed") 

412 if "=" in value: 

413 raise ValueError(f'Invalid {kind} name: "=" not allowed') 

414 if any(ch.isspace() for ch in value): 

415 raise ValueError(f"Invalid {kind} name: whitespace not allowed") 

416 

417 

418def _validate_comment(value): 

419 if isinstance(value, bytes): 

420 try: 

421 value = value.decode("utf-8") 

422 except UnicodeDecodeError as exc: 

423 raise ValueError("Comment text must be valid UTF-8") from exc 

424 if not isinstance(value, str): 

425 raise ValueError("Comment text must be a string") 

426 if "--" in value: 

427 raise ValueError("Comment text cannot contain '--'") 

428 if value.endswith("-"): 

429 raise ValueError("Comment text cannot end with '-'") 

430 return value 

431 

432 

433def _process_namespace(name, namespaces, ns_sep=':', attr_prefix='@'): 

434 if not isinstance(name, str): 

435 return name 

436 if not namespaces: 

437 return name 

438 try: 

439 ns, name = name.rsplit(ns_sep, 1) 

440 except ValueError: 

441 pass 

442 else: 

443 ns_res = namespaces.get(ns.strip(attr_prefix)) 

444 name = '{}{}{}{}'.format( 

445 attr_prefix if ns.startswith(attr_prefix) else '', 

446 ns_res, ns_sep, name) if ns_res else name 

447 return name 

448 

449 

450def _emit(key, value, content_handler, 

451 attr_prefix='@', 

452 cdata_key='#text', 

453 depth=0, 

454 preprocessor=None, 

455 pretty=False, 

456 newl='\n', 

457 indent='\t', 

458 namespace_separator=':', 

459 namespaces=None, 

460 full_document=True, 

461 expand_iter=None, 

462 encoding='utf-8', 

463 bytes_errors='replace', 

464 comment_key='#comment'): 

465 if isinstance(key, str) and key == comment_key: 

466 comments_list = value if isinstance(value, list) else [value] 

467 if isinstance(indent, int): 

468 indent = " " * indent 

469 for comment_text in comments_list: 

470 if comment_text is None: 

471 continue 

472 comment_text = _convert_value_to_string( 

473 comment_text, encoding=encoding, bytes_errors=bytes_errors 

474 ) 

475 if not comment_text: 

476 continue 

477 if pretty: 

478 content_handler.ignorableWhitespace(depth * indent) 

479 content_handler.comment(comment_text) 

480 if pretty: 

481 content_handler.ignorableWhitespace(newl) 

482 return 

483 

484 key = _process_namespace(key, namespaces, namespace_separator, attr_prefix) 

485 if preprocessor is not None: 

486 result = preprocessor(key, value) 

487 if result is None: 

488 return 

489 key, value = result 

490 # Minimal validation to avoid breaking out of tag context 

491 _validate_name(key, "element") 

492 if not hasattr(value, '__iter__') or isinstance(value, (str, bytes, bytearray, memoryview, dict)): 

493 value = [value] 

494 for index, v in enumerate(value): 

495 if full_document and depth == 0 and index > 0: 

496 raise ValueError('document with multiple roots') 

497 if v is None: 

498 v = {} 

499 elif not isinstance(v, (dict, str)): 

500 if expand_iter and hasattr(v, '__iter__') and not isinstance(v, (bytes, bytearray, memoryview)): 

501 v = {expand_iter: v} 

502 else: 

503 v = _convert_value_to_string(v, encoding=encoding, bytes_errors=bytes_errors) 

504 if isinstance(v, str): 

505 v = {cdata_key: v} 

506 cdata = None 

507 attrs = {} 

508 children = [] 

509 for ik, iv in v.items(): 

510 if ik == cdata_key: 

511 if iv is None: 

512 cdata = None 

513 else: 

514 cdata = _convert_value_to_string(iv, encoding=encoding, bytes_errors=bytes_errors) 

515 continue 

516 if isinstance(ik, str) and ik.startswith(attr_prefix): 

517 ik = _process_namespace(ik, namespaces, namespace_separator, 

518 attr_prefix) 

519 if ik == attr_prefix + 'xmlns' and isinstance(iv, dict): 

520 for k, v in iv.items(): 

521 _validate_name(k, "attribute") 

522 attr = 'xmlns{}'.format(f':{k}' if k else '') 

523 attrs[attr] = '' if v is None else _convert_value_to_string( 

524 v, encoding=encoding, bytes_errors=bytes_errors 

525 ) 

526 continue 

527 if iv is None: 

528 iv = '' 

529 elif not isinstance(iv, str): 

530 iv = _convert_value_to_string(iv, encoding=encoding, bytes_errors=bytes_errors) 

531 attr_name = ik[len(attr_prefix) :] 

532 _validate_name(attr_name, "attribute") 

533 attrs[attr_name] = iv 

534 continue 

535 if isinstance(iv, list) and not iv: 

536 continue # Skip empty lists to avoid creating empty child elements 

537 children.append((ik, iv)) 

538 if isinstance(indent, int): 

539 indent = ' ' * indent 

540 if pretty: 

541 content_handler.ignorableWhitespace(depth * indent) 

542 content_handler.startElement(key, AttributesImpl(attrs)) 

543 if pretty and children: 

544 content_handler.ignorableWhitespace(newl) 

545 for child_key, child_value in children: 

546 _emit(child_key, child_value, content_handler, 

547 attr_prefix, cdata_key, depth+1, preprocessor, 

548 pretty, newl, indent, namespaces=namespaces, 

549 namespace_separator=namespace_separator, 

550 expand_iter=expand_iter, encoding=encoding, 

551 bytes_errors=bytes_errors, comment_key=comment_key) 

552 if cdata is not None: 

553 content_handler.characters(cdata) 

554 if pretty and children: 

555 content_handler.ignorableWhitespace(depth * indent) 

556 content_handler.endElement(key) 

557 if pretty and depth: 

558 content_handler.ignorableWhitespace(newl) 

559 

560 

561class _XMLGenerator(XMLGenerator): 

562 def comment(self, text): 

563 text = _validate_comment(text) 

564 self._write(f"<!--{escape(text)}-->") 

565 

566 

567def unparse(input_dict, output=None, encoding='utf-8', full_document=True, 

568 short_empty_elements=False, comment_key='#comment', 

569 **kwargs): 

570 """Emit an XML document for the given `input_dict` (reverse of `parse`). 

571 

572 The resulting XML document is returned as a string, but if `output` (a 

573 file-like object) is specified, it is written there instead. 

574 

575 Dictionary keys prefixed with `attr_prefix` (default=`'@'`) are interpreted 

576 as XML node attributes, whereas keys equal to `cdata_key` 

577 (default=`'#text'`) are treated as character data. 

578 

579 Empty lists are omitted entirely: ``{"a": []}`` produces no ``<a>`` element. 

580 Provide a placeholder entry (for example ``{"a": [""]}``) when an explicit 

581 empty container element must be emitted. 

582 

583 The `pretty` parameter (default=`False`) enables pretty-printing. In this 

584 mode, lines are terminated with `'\n'` and indented with `'\t'`, but this 

585 can be customized with the `newl` and `indent` parameters. 

586 The `bytes_errors` parameter controls decoding errors for byte values and 

587 defaults to `'replace'`. 

588 

589 """ 

590 bytes_errors = kwargs.pop('bytes_errors', 'replace') 

591 try: 

592 codecs.lookup_error(bytes_errors) 

593 except LookupError as exc: 

594 raise ValueError(f"Invalid bytes_errors handler: {bytes_errors}") from exc 

595 

596 must_return = False 

597 if output is None: 

598 output = StringIO() 

599 must_return = True 

600 if short_empty_elements: 

601 content_handler = _XMLGenerator(output, encoding, True) 

602 else: 

603 content_handler = _XMLGenerator(output, encoding) 

604 if full_document: 

605 content_handler.startDocument() 

606 seen_root = False 

607 for key, value in input_dict.items(): 

608 if key != comment_key and full_document and seen_root: 

609 raise ValueError("Document must have exactly one root.") 

610 _emit( 

611 key, 

612 value, 

613 content_handler, 

614 full_document=full_document, 

615 encoding=encoding, 

616 bytes_errors=bytes_errors, 

617 comment_key=comment_key, 

618 **kwargs, 

619 ) 

620 if key != comment_key: 

621 seen_root = True 

622 if full_document and not seen_root: 

623 raise ValueError("Document must have exactly one root.") 

624 if full_document: 

625 content_handler.endDocument() 

626 if must_return: 

627 value = output.getvalue() 

628 try: # pragma no cover 

629 value = value.decode(encoding) 

630 except AttributeError: # pragma no cover 

631 pass 

632 return value 

633 

634 

635if __name__ == '__main__': # pragma: no cover 

636 import marshal 

637 import sys 

638 

639 stdin = sys.stdin.buffer 

640 stdout = sys.stdout.buffer 

641 

642 (item_depth,) = sys.argv[1:] 

643 item_depth = int(item_depth) 

644 

645 def handle_item(path, item): 

646 marshal.dump((path, item), stdout) 

647 return True 

648 

649 try: 

650 root = parse(stdin, 

651 item_depth=item_depth, 

652 item_callback=handle_item, 

653 dict_constructor=dict) 

654 if item_depth == 0: 

655 handle_item([], root) 

656 except KeyboardInterrupt: 

657 pass