Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/pandas/io/xml.py: 16%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

242 statements  

1""" 

2:mod:``pandas.io.xml`` is a module for reading XML. 

3""" 

4 

5from __future__ import annotations 

6 

7import io 

8from os import PathLike 

9from typing import ( 

10 TYPE_CHECKING, 

11 Any, 

12) 

13 

14from pandas._libs import lib 

15from pandas.compat._optional import import_optional_dependency 

16from pandas.errors import ( 

17 AbstractMethodError, 

18 ParserError, 

19) 

20from pandas.util._decorators import set_module 

21from pandas.util._validators import check_dtype_backend 

22 

23from pandas.core.dtypes.common import is_list_like 

24 

25from pandas.io.common import ( 

26 get_handle, 

27 infer_compression, 

28 is_fsspec_url, 

29 is_url, 

30 stringify_path, 

31) 

32from pandas.io.parsers import TextParser 

33 

34if TYPE_CHECKING: 

35 from collections.abc import ( 

36 Callable, 

37 Sequence, 

38 ) 

39 from xml.etree.ElementTree import Element 

40 

41 from lxml import etree 

42 

43 from pandas._typing import ( 

44 CompressionOptions, 

45 ConvertersArg, 

46 DtypeArg, 

47 DtypeBackend, 

48 FilePath, 

49 ParseDatesArg, 

50 ReadBuffer, 

51 StorageOptions, 

52 XMLParsers, 

53 ) 

54 

55 from pandas import DataFrame 

56 

57 

58class _XMLFrameParser: 

59 """ 

60 Internal subclass to parse XML into DataFrames. 

61 

62 Parameters 

63 ---------- 

64 path_or_buffer : a valid JSON ``str``, path object or file-like object 

65 Any valid string path is acceptable. The string could be a URL. Valid 

66 URL schemes include http, ftp, s3, and file. 

67 

68 xpath : str or regex 

69 The ``XPath`` expression to parse required set of nodes for 

70 migration to :class:`~pandas.DataFrame`. ``etree`` supports limited ``XPath``. 

71 

72 namespaces : dict 

73 The namespaces defined in XML document (``xmlns:namespace='URI'``) 

74 as dicts with key being namespace and value the URI. 

75 

76 elems_only : bool 

77 Parse only the child elements at the specified ``xpath``. 

78 

79 attrs_only : bool 

80 Parse only the attributes at the specified ``xpath``. 

81 

82 names : list 

83 Column names for :class:`~pandas.DataFrame` of parsed XML data. 

84 

85 dtype : dict 

86 Data type for data or columns. E.g. {'a': np.float64, 

87 'b': np.int32, 'c': 'Int64'} 

88 

89 converters : dict, optional 

90 Dict of functions for converting values in certain columns. Keys can 

91 either be integers or column labels. 

92 

93 parse_dates : bool or list of int or names or list of lists or dict 

94 Converts either index or select columns to datetimes 

95 

96 encoding : str 

97 Encoding of xml object or document. 

98 

99 stylesheet : str or file-like 

100 URL, file, file-like object, or a raw string containing XSLT, 

101 ``etree`` does not support XSLT but retained for consistency. 

102 

103 iterparse : dict, optional 

104 Dict with row element as key and list of descendant elements 

105 and/or attributes as value to be retrieved in iterparsing of 

106 XML document. 

107 

108 compression : str or dict, default 'infer' 

109 For on-the-fly decompression of on-disk data. If 'infer' and 

110 'path_or_buffer' is path-like, then detect compression from the 

111 following extensions: '.gz', '.bz2', '.zip', '.xz', '.zst', '.tar', 

112 '.tar.gz', '.tar.xz' or '.tar.bz2' (otherwise no compression). 

113 If using 'zip' or 'tar', the ZIP file must contain only one data 

114 file to be read in. Set to ``None`` for no decompression. 

115 Can also be a dict with key ``'method'`` set to one of 

116 {``'zip'``, ``'gzip'``, ``'bz2'``, ``'zstd'``, ``'xz'``, ``'tar'``} 

117 and other key-value pairs are forwarded to ``zipfile.ZipFile``, 

118 ``gzip.GzipFile``, ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``, 

119 ``lzma.LZMAFile`` or ``tarfile.TarFile``, respectively. 

120 As an example, the following could be passed for Zstandard 

121 decompression using a custom compression dictionary: 

122 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``. 

123 

124 storage_options : dict, optional 

125 Extra options that make sense for a particular storage connection, 

126 e.g. host, port, username, password, etc. For HTTP(S) URLs the 

127 key-value pairs are forwarded to ``urllib.request.Request`` as header 

128 options. For other URLs (e.g. starting with "s3://", and "gcs://") 

129 the key-value pairs are forwarded to ``fsspec.open``. Please see 

130 ``fsspec`` and ``urllib`` for more details, and for more examples on 

131 storage options refer `here <https://pandas.pydata.org/docs/ 

132 user_guide/io.html?highlight=storage_options#reading-writing-remote- 

133 files>`_. 

134 

135 See also 

136 -------- 

137 pandas.io.xml._EtreeFrameParser 

138 pandas.io.xml._LxmlFrameParser 

139 

140 Notes 

141 ----- 

142 To subclass this class effectively you must override the following methods:` 

143 * :func:`parse_data` 

144 * :func:`_parse_nodes` 

145 * :func:`_iterparse_nodes` 

146 * :func:`_parse_doc` 

147 * :func:`_validate_names` 

148 * :func:`_validate_path` 

149 

150 

151 See each method's respective documentation for details on their 

152 functionality. 

153 """ 

154 

155 def __init__( 

156 self, 

157 path_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str], 

158 xpath: str, 

159 namespaces: dict[str, str] | None, 

160 elems_only: bool, 

161 attrs_only: bool, 

162 names: Sequence[str] | None, 

163 dtype: DtypeArg | None, 

164 converters: ConvertersArg | None, 

165 parse_dates: ParseDatesArg | None, 

166 encoding: str | None, 

167 stylesheet: FilePath | ReadBuffer[bytes] | ReadBuffer[str] | None, 

168 iterparse: dict[str, list[str]] | None, 

169 compression: CompressionOptions, 

170 storage_options: StorageOptions, 

171 ) -> None: 

172 self.path_or_buffer = path_or_buffer 

173 self.xpath = xpath 

174 self.namespaces = namespaces 

175 self.elems_only = elems_only 

176 self.attrs_only = attrs_only 

177 self.names = names 

178 self.dtype = dtype 

179 self.converters = converters 

180 self.parse_dates = parse_dates 

181 self.encoding = encoding 

182 self.stylesheet = stylesheet 

183 self.iterparse = iterparse 

184 self.compression: CompressionOptions = compression 

185 self.storage_options = storage_options 

186 

187 def parse_data(self) -> list[dict[str, str | None]]: 

188 """ 

189 Parse xml data. 

190 

191 This method will call the other internal methods to 

192 validate ``xpath``, names, parse and return specific nodes. 

193 """ 

194 

195 raise AbstractMethodError(self) 

196 

197 def _parse_nodes(self, elems: list[Any]) -> list[dict[str, str | None]]: 

198 """ 

199 Parse xml nodes. 

200 

201 This method will parse the children and attributes of elements 

202 in ``xpath``, conditionally for only elements, only attributes 

203 or both while optionally renaming node names. 

204 

205 Raises 

206 ------ 

207 ValueError 

208 * If only elements and only attributes are specified. 

209 

210 Notes 

211 ----- 

212 Namespace URIs will be removed from return node values. Also, 

213 elements with missing children or attributes compared to siblings 

214 will have optional keys filled with None values. 

215 """ 

216 

217 dicts: list[dict[str, str | None]] 

218 

219 if self.elems_only and self.attrs_only: 

220 raise ValueError("Either element or attributes can be parsed not both.") 

221 if self.elems_only: 

222 if self.names: 

223 dicts = [ 

224 { 

225 **( 

226 {el.tag: el.text} 

227 if el.text and not el.text.isspace() 

228 else {} 

229 ), 

230 **{ 

231 nm: ch.text if ch.text else None 

232 for nm, ch in zip(self.names, el.findall("*"), strict=True) 

233 }, 

234 } 

235 for el in elems 

236 ] 

237 else: 

238 dicts = [ 

239 {ch.tag: ch.text if ch.text else None for ch in el.findall("*")} 

240 for el in elems 

241 ] 

242 

243 elif self.attrs_only: 

244 dicts = [ 

245 {k: v if v else None for k, v in el.attrib.items()} for el in elems 

246 ] 

247 

248 elif self.names: 

249 dicts = [ 

250 { 

251 **el.attrib, 

252 **({el.tag: el.text} if el.text and not el.text.isspace() else {}), 

253 **{ 

254 nm: ch.text if ch.text else None 

255 for nm, ch in zip(self.names, el.findall("*"), strict=False) 

256 }, 

257 } 

258 for el in elems 

259 ] 

260 

261 else: 

262 dicts = [ 

263 { 

264 **el.attrib, 

265 **({el.tag: el.text} if el.text and not el.text.isspace() else {}), 

266 **{ch.tag: ch.text if ch.text else None for ch in el.findall("*")}, 

267 } 

268 for el in elems 

269 ] 

270 

271 dicts = [ 

272 {k.split("}")[1] if "}" in k else k: v for k, v in d.items()} for d in dicts 

273 ] 

274 

275 keys = list(dict.fromkeys([k for d in dicts for k in d.keys()])) 

276 dicts = [{k: d[k] if k in d.keys() else None for k in keys} for d in dicts] 

277 

278 if self.names: 

279 dicts = [dict(zip(self.names, d.values(), strict=True)) for d in dicts] 

280 

281 return dicts 

282 

283 def _iterparse_nodes(self, iterparse: Callable) -> list[dict[str, str | None]]: 

284 """ 

285 Iterparse xml nodes. 

286 

287 This method will read in local disk, decompressed XML files for elements 

288 and underlying descendants using iterparse, a method to iterate through 

289 an XML tree without holding entire XML tree in memory. 

290 

291 Raises 

292 ------ 

293 TypeError 

294 * If ``iterparse`` is not a dict or its dict value is not list-like. 

295 ParserError 

296 * If ``path_or_buffer`` is not a physical file on disk or file-like object. 

297 * If no data is returned from selected items in ``iterparse``. 

298 

299 Notes 

300 ----- 

301 Namespace URIs will be removed from return node values. Also, 

302 elements with missing children or attributes in submitted list 

303 will have optional keys filled with None values. 

304 """ 

305 

306 dicts: list[dict[str, str | None]] = [] 

307 row: dict[str, str | None] | None = None 

308 

309 if not isinstance(self.iterparse, dict): 

310 raise TypeError( 

311 f"{type(self.iterparse).__name__} is not a valid type for iterparse" 

312 ) 

313 

314 row_node = next(iter(self.iterparse.keys())) if self.iterparse else "" 

315 if not is_list_like(self.iterparse[row_node]): 

316 raise TypeError( 

317 f"{type(self.iterparse[row_node])} is not a valid type " 

318 "for value in iterparse" 

319 ) 

320 

321 if (not hasattr(self.path_or_buffer, "read")) and ( 

322 not isinstance(self.path_or_buffer, (str, PathLike)) 

323 or is_url(self.path_or_buffer) 

324 or is_fsspec_url(self.path_or_buffer) 

325 or ( 

326 isinstance(self.path_or_buffer, str) 

327 and self.path_or_buffer.startswith(("<?xml", "<")) 

328 ) 

329 or infer_compression(self.path_or_buffer, "infer") is not None 

330 ): 

331 raise ParserError( 

332 "iterparse is designed for large XML files that are fully extracted on " 

333 "local disk and not as compressed files or online sources." 

334 ) 

335 

336 iterparse_repeats = len(self.iterparse[row_node]) != len( 

337 set(self.iterparse[row_node]) 

338 ) 

339 

340 parser = iterparse(self.path_or_buffer, events=("start", "end")) 

341 try: 

342 for event, elem in parser: 

343 curr_elem = elem.tag.split("}")[1] if "}" in elem.tag else elem.tag 

344 

345 if event == "start": 

346 if curr_elem == row_node: 

347 row = {} 

348 

349 if row is not None: 

350 if self.names and iterparse_repeats: 

351 for col, nm in zip( 

352 self.iterparse[row_node], self.names, strict=True 

353 ): 

354 if curr_elem == col: 

355 elem_val = elem.text if elem.text else None 

356 if elem_val not in row.values() and nm not in row: 

357 row[nm] = elem_val 

358 

359 if col in elem.attrib: 

360 if ( 

361 elem.attrib[col] not in row.values() 

362 and nm not in row 

363 ): 

364 row[nm] = elem.attrib[col] 

365 else: 

366 for col in self.iterparse[row_node]: 

367 if curr_elem == col: 

368 row[col] = elem.text if elem.text else None 

369 if col in elem.attrib: 

370 row[col] = elem.attrib[col] 

371 

372 if event == "end": 

373 if curr_elem == row_node and row is not None: 

374 dicts.append(row) 

375 row = None 

376 

377 elem.clear() 

378 if hasattr(elem, "getprevious"): 

379 while ( 

380 elem.getprevious() is not None 

381 and elem.getparent() is not None 

382 ): 

383 del elem.getparent()[0] 

384 finally: 

385 if hasattr(parser, "close"): 

386 parser.close() 

387 

388 if dicts == []: 

389 raise ParserError("No result from selected items in iterparse.") 

390 

391 keys = list(dict.fromkeys([k for d in dicts for k in d.keys()])) 

392 dicts = [{k: d[k] if k in d.keys() else None for k in keys} for d in dicts] 

393 

394 if self.names: 

395 dicts = [dict(zip(self.names, d.values(), strict=True)) for d in dicts] 

396 

397 return dicts 

398 

399 def _validate_path(self) -> list[Any]: 

400 """ 

401 Validate ``xpath``. 

402 

403 This method checks for syntax, evaluation, or empty nodes return. 

404 

405 Raises 

406 ------ 

407 SyntaxError 

408 * If xpah is not supported or issues with namespaces. 

409 

410 ValueError 

411 * If xpah does not return any nodes. 

412 """ 

413 

414 raise AbstractMethodError(self) 

415 

416 def _validate_names(self) -> None: 

417 """ 

418 Validate names. 

419 

420 This method will check if names is a list-like and aligns 

421 with length of parse nodes. 

422 

423 Raises 

424 ------ 

425 ValueError 

426 * If value is not a list and less then length of nodes. 

427 """ 

428 raise AbstractMethodError(self) 

429 

430 def _parse_doc( 

431 self, raw_doc: FilePath | ReadBuffer[bytes] | ReadBuffer[str] 

432 ) -> Element | etree._Element: 

433 """ 

434 Build tree from path_or_buffer. 

435 

436 This method will parse XML object into tree 

437 either from string/bytes or file location. 

438 """ 

439 raise AbstractMethodError(self) 

440 

441 

442class _EtreeFrameParser(_XMLFrameParser): 

443 """ 

444 Internal class to parse XML into DataFrames with the Python 

445 standard library XML module: `xml.etree.ElementTree`. 

446 """ 

447 

448 def parse_data(self) -> list[dict[str, str | None]]: 

449 from xml.etree.ElementTree import iterparse 

450 

451 if self.stylesheet is not None: 

452 raise ValueError( 

453 "To use stylesheet, you need lxml installed and selected as parser." 

454 ) 

455 

456 if self.iterparse is None: 

457 self.xml_doc = self._parse_doc(self.path_or_buffer) 

458 elems = self._validate_path() 

459 

460 self._validate_names() 

461 

462 xml_dicts: list[dict[str, str | None]] = ( 

463 self._parse_nodes(elems) 

464 if self.iterparse is None 

465 else self._iterparse_nodes(iterparse) 

466 ) 

467 

468 return xml_dicts 

469 

470 def _validate_path(self) -> list[Any]: 

471 """ 

472 Notes 

473 ----- 

474 ``etree`` supports limited ``XPath``. If user attempts a more complex 

475 expression syntax error will raise. 

476 """ 

477 

478 msg = ( 

479 "xpath does not return any nodes or attributes. " 

480 "Be sure to specify in `xpath` the parent nodes of " 

481 "children and attributes to parse. " 

482 "If document uses namespaces denoted with " 

483 "xmlns, be sure to define namespaces and " 

484 "use them in xpath." 

485 ) 

486 try: 

487 elems = self.xml_doc.findall(self.xpath, namespaces=self.namespaces) 

488 children = [ch for el in elems for ch in el.findall("*")] 

489 attrs = {k: v for el in elems for k, v in el.attrib.items()} 

490 

491 if elems is None: 

492 raise ValueError(msg) 

493 

494 if elems is not None: 

495 if self.elems_only and children == []: 

496 raise ValueError(msg) 

497 if self.attrs_only and attrs == {}: 

498 raise ValueError(msg) 

499 if children == [] and attrs == {}: 

500 raise ValueError(msg) 

501 

502 except (KeyError, SyntaxError) as err: 

503 raise SyntaxError( 

504 "You have used an incorrect or unsupported XPath " 

505 "expression for etree library or you used an " 

506 "undeclared namespace prefix." 

507 ) from err 

508 

509 return elems 

510 

511 def _validate_names(self) -> None: 

512 children: list[Any] 

513 

514 if self.names: 

515 if self.iterparse: 

516 children = self.iterparse[next(iter(self.iterparse))] 

517 else: 

518 parent = self.xml_doc.find(self.xpath, namespaces=self.namespaces) 

519 children = parent.findall("*") if parent is not None else [] 

520 

521 if is_list_like(self.names): 

522 if len(self.names) < len(children): 

523 raise ValueError( 

524 "names does not match length of child elements in xpath." 

525 ) 

526 else: 

527 raise TypeError( 

528 f"{type(self.names).__name__} is not a valid type for names" 

529 ) 

530 

531 def _parse_doc( 

532 self, raw_doc: FilePath | ReadBuffer[bytes] | ReadBuffer[str] 

533 ) -> Element: 

534 from xml.etree.ElementTree import ( 

535 XMLParser, 

536 parse, 

537 ) 

538 

539 handle_data = get_data_from_filepath( 

540 filepath_or_buffer=raw_doc, 

541 encoding=self.encoding, 

542 compression=self.compression, 

543 storage_options=self.storage_options, 

544 ) 

545 

546 with handle_data as xml_data: 

547 curr_parser = XMLParser(encoding=self.encoding) 

548 document = parse(xml_data, parser=curr_parser) 

549 

550 return document.getroot() 

551 

552 

553class _LxmlFrameParser(_XMLFrameParser): 

554 """ 

555 Internal class to parse XML into :class:`~pandas.DataFrame` with third-party 

556 full-featured XML library, ``lxml``, that supports 

557 ``XPath`` 1.0 and XSLT 1.0. 

558 """ 

559 

560 def parse_data(self) -> list[dict[str, str | None]]: 

561 """ 

562 Parse xml data. 

563 

564 This method will call the other internal methods to 

565 validate ``xpath``, names, optionally parse and run XSLT, 

566 and parse original or transformed XML and return specific nodes. 

567 """ 

568 from lxml.etree import iterparse 

569 

570 if self.iterparse is None: 

571 self.xml_doc = self._parse_doc(self.path_or_buffer) 

572 

573 if self.stylesheet: 

574 self.xsl_doc = self._parse_doc(self.stylesheet) 

575 self.xml_doc = self._transform_doc() 

576 

577 elems = self._validate_path() 

578 

579 self._validate_names() 

580 

581 xml_dicts: list[dict[str, str | None]] = ( 

582 self._parse_nodes(elems) 

583 if self.iterparse is None 

584 else self._iterparse_nodes(iterparse) 

585 ) 

586 

587 return xml_dicts 

588 

589 def _validate_path(self) -> list[Any]: 

590 msg = ( 

591 "xpath does not return any nodes or attributes. " 

592 "Be sure to specify in `xpath` the parent nodes of " 

593 "children and attributes to parse. " 

594 "If document uses namespaces denoted with " 

595 "xmlns, be sure to define namespaces and " 

596 "use them in xpath." 

597 ) 

598 

599 elems = self.xml_doc.xpath(self.xpath, namespaces=self.namespaces) 

600 children = [ch for el in elems for ch in el.xpath("*")] 

601 attrs = {k: v for el in elems for k, v in el.attrib.items()} 

602 

603 if elems == []: 

604 raise ValueError(msg) 

605 

606 if elems != []: 

607 if self.elems_only and children == []: 

608 raise ValueError(msg) 

609 if self.attrs_only and attrs == {}: 

610 raise ValueError(msg) 

611 if children == [] and attrs == {}: 

612 raise ValueError(msg) 

613 

614 return elems 

615 

616 def _validate_names(self) -> None: 

617 children: list[Any] 

618 

619 if self.names: 

620 if self.iterparse: 

621 children = self.iterparse[next(iter(self.iterparse))] 

622 else: 

623 children = self.xml_doc.xpath( 

624 self.xpath + "[1]/*", namespaces=self.namespaces 

625 ) 

626 

627 if is_list_like(self.names): 

628 if len(self.names) < len(children): 

629 raise ValueError( 

630 "names does not match length of child elements in xpath." 

631 ) 

632 else: 

633 raise TypeError( 

634 f"{type(self.names).__name__} is not a valid type for names" 

635 ) 

636 

637 def _parse_doc( 

638 self, raw_doc: FilePath | ReadBuffer[bytes] | ReadBuffer[str] 

639 ) -> etree._Element: 

640 from lxml.etree import ( 

641 XMLParser, 

642 fromstring, 

643 parse, 

644 ) 

645 

646 handle_data = get_data_from_filepath( 

647 filepath_or_buffer=raw_doc, 

648 encoding=self.encoding, 

649 compression=self.compression, 

650 storage_options=self.storage_options, 

651 ) 

652 

653 with handle_data as xml_data: 

654 curr_parser = XMLParser(encoding=self.encoding) 

655 

656 if isinstance(xml_data, io.StringIO): 

657 if self.encoding is None: 

658 raise TypeError( 

659 "Can not pass encoding None when input is StringIO." 

660 ) 

661 

662 document = fromstring( 

663 xml_data.getvalue().encode(self.encoding), parser=curr_parser 

664 ) 

665 else: 

666 document = parse(xml_data, parser=curr_parser) 

667 

668 return document 

669 

670 def _transform_doc(self) -> etree._XSLTResultTree: 

671 """ 

672 Transform original tree using stylesheet. 

673 

674 This method will transform original xml using XSLT script into 

675 am ideally flatter xml document for easier parsing and migration 

676 to Data Frame. 

677 """ 

678 from lxml.etree import XSLT 

679 

680 transformer = XSLT(self.xsl_doc) 

681 new_doc = transformer(self.xml_doc) 

682 

683 return new_doc 

684 

685 

686def get_data_from_filepath( 

687 filepath_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str], 

688 encoding: str | None, 

689 compression: CompressionOptions, 

690 storage_options: StorageOptions, 

691): 

692 """ 

693 Extract raw XML data. 

694 

695 The method accepts two input types: 

696 1. filepath (string-like) 

697 2. file-like object (e.g. open file object, StringIO) 

698 """ 

699 filepath_or_buffer = stringify_path(filepath_or_buffer) 

700 with get_handle( 

701 filepath_or_buffer, 

702 "r", 

703 encoding=encoding, 

704 compression=compression, 

705 storage_options=storage_options, 

706 ) as handle_obj: 

707 return ( 

708 preprocess_data(handle_obj.handle.read()) 

709 if hasattr(handle_obj.handle, "read") 

710 else handle_obj.handle 

711 ) 

712 

713 

714def preprocess_data( 

715 data: str | bytes | io.StringIO | io.BytesIO, 

716) -> io.StringIO | io.BytesIO: 

717 """ 

718 Convert extracted raw data. 

719 

720 This method will return underlying data of extracted XML content. 

721 The data either has a `read` attribute (e.g. a file object or a 

722 StringIO/BytesIO) or is a string or bytes that is an XML document. 

723 """ 

724 

725 if isinstance(data, str): 

726 data = io.StringIO(data) 

727 

728 elif isinstance(data, bytes): 

729 data = io.BytesIO(data) 

730 

731 return data 

732 

733 

734def _data_to_frame(data: list[dict[str, str | None]], **kwargs) -> DataFrame: 

735 """ 

736 Convert parsed data to Data Frame. 

737 

738 This method will bind xml dictionary data of keys and values 

739 into named columns of Data Frame using the built-in TextParser 

740 class that build Data Frame and infers specific dtypes. 

741 """ 

742 

743 tags = next(iter(data)) 

744 nodes = [list(d.values()) for d in data] 

745 

746 try: 

747 with TextParser(nodes, names=tags, **kwargs) as tp: 

748 return tp.read() 

749 except ParserError as err: 

750 raise ParserError( 

751 "XML document may be too complex for import. " 

752 "Try to flatten document and use distinct " 

753 "element and attribute names." 

754 ) from err 

755 

756 

757def _parse( 

758 path_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str], 

759 xpath: str, 

760 namespaces: dict[str, str] | None, 

761 elems_only: bool, 

762 attrs_only: bool, 

763 names: Sequence[str] | None, 

764 dtype: DtypeArg | None, 

765 converters: ConvertersArg | None, 

766 parse_dates: ParseDatesArg | None, 

767 encoding: str | None, 

768 parser: XMLParsers, 

769 stylesheet: FilePath | ReadBuffer[bytes] | ReadBuffer[str] | None, 

770 iterparse: dict[str, list[str]] | None, 

771 compression: CompressionOptions, 

772 storage_options: StorageOptions, 

773 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default, 

774 **kwargs, 

775) -> DataFrame: 

776 """ 

777 Call internal parsers. 

778 

779 This method will conditionally call internal parsers: 

780 LxmlFrameParser and/or EtreeParser. 

781 

782 Raises 

783 ------ 

784 ImportError 

785 * If lxml is not installed if selected as parser. 

786 

787 ValueError 

788 * If parser is not lxml or etree. 

789 """ 

790 

791 p: _EtreeFrameParser | _LxmlFrameParser 

792 

793 if parser == "lxml": 

794 lxml = import_optional_dependency("lxml.etree", errors="ignore") 

795 

796 if lxml is not None: 

797 p = _LxmlFrameParser( 

798 path_or_buffer, 

799 xpath, 

800 namespaces, 

801 elems_only, 

802 attrs_only, 

803 names, 

804 dtype, 

805 converters, 

806 parse_dates, 

807 encoding, 

808 stylesheet, 

809 iterparse, 

810 compression, 

811 storage_options, 

812 ) 

813 else: 

814 raise ImportError("lxml not found, please install or use the etree parser.") 

815 

816 elif parser == "etree": 

817 p = _EtreeFrameParser( 

818 path_or_buffer, 

819 xpath, 

820 namespaces, 

821 elems_only, 

822 attrs_only, 

823 names, 

824 dtype, 

825 converters, 

826 parse_dates, 

827 encoding, 

828 stylesheet, 

829 iterparse, 

830 compression, 

831 storage_options, 

832 ) 

833 else: 

834 raise ValueError("Values for parser can only be lxml or etree.") 

835 

836 data_dicts = p.parse_data() 

837 

838 return _data_to_frame( 

839 data=data_dicts, 

840 dtype=dtype, 

841 converters=converters, 

842 parse_dates=parse_dates, 

843 dtype_backend=dtype_backend, 

844 **kwargs, 

845 ) 

846 

847 

848@set_module("pandas") 

849def read_xml( 

850 path_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str], 

851 *, 

852 xpath: str = "./*", 

853 namespaces: dict[str, str] | None = None, 

854 elems_only: bool = False, 

855 attrs_only: bool = False, 

856 names: Sequence[str] | None = None, 

857 dtype: DtypeArg | None = None, 

858 converters: ConvertersArg | None = None, 

859 parse_dates: ParseDatesArg | None = None, 

860 # encoding can not be None for lxml and StringIO input 

861 encoding: str | None = "utf-8", 

862 parser: XMLParsers = "lxml", 

863 stylesheet: FilePath | ReadBuffer[bytes] | ReadBuffer[str] | None = None, 

864 iterparse: dict[str, list[str]] | None = None, 

865 compression: CompressionOptions = "infer", 

866 storage_options: StorageOptions | None = None, 

867 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default, 

868) -> DataFrame: 

869 r""" 

870 Read XML document into a :class:`~pandas.DataFrame` object. 

871 

872 Parameters 

873 ---------- 

874 path_or_buffer : str, path object, or file-like object 

875 String path, path object (implementing ``os.PathLike[str]``), or file-like 

876 object implementing a ``read()`` function. The string can be a path. 

877 The string can further be a URL. Valid URL schemes 

878 include http, ftp, s3, and file. 

879 

880 xpath : str, optional, default './\*' 

881 The ``XPath`` to parse required set of nodes for migration to 

882 :class:`~pandas.DataFrame`.``XPath`` should return a collection of elements 

883 and not a single element. Note: The ``etree`` parser supports limited ``XPath`` 

884 expressions. For more complex ``XPath``, use ``lxml`` which requires 

885 installation. 

886 

887 namespaces : dict, optional 

888 The namespaces defined in XML document as dicts with key being 

889 namespace prefix and value the URI. There is no need to include all 

890 namespaces in XML, only the ones used in ``xpath`` expression. 

891 Note: if XML document uses default namespace denoted as 

892 `xmlns='<URI>'` without a prefix, you must assign any temporary 

893 namespace prefix such as 'doc' to the URI in order to parse 

894 underlying nodes and/or attributes. 

895 

896 elems_only : bool, optional, default False 

897 Parse only the child elements at the specified ``xpath``. By default, 

898 all child elements and non-empty text nodes are returned. 

899 

900 attrs_only : bool, optional, default False 

901 Parse only the attributes at the specified ``xpath``. 

902 By default, all attributes are returned. 

903 

904 names : list-like, optional 

905 Column names for DataFrame of parsed XML data. Use this parameter to 

906 rename original element names and distinguish same named elements and 

907 attributes. 

908 

909 dtype : Type name or dict of column -> type, optional 

910 Data type for data or columns. E.g. {'a': np.float64, 'b': np.int32, 

911 'c': 'Int64'} 

912 Use `str` or `object` together with suitable `na_values` settings 

913 to preserve and not interpret dtype. 

914 If converters are specified, they will be applied INSTEAD 

915 of dtype conversion. 

916 

917 converters : dict, optional 

918 Dict of functions for converting values in certain columns. Keys can either 

919 be integers or column labels. 

920 

921 parse_dates : bool or list of int or names or list of lists or dict, default False 

922 Identifiers to parse index or columns to datetime. The behavior is as follows: 

923 

924 * boolean. If True -> try parsing the index. 

925 * list of int or names. e.g. If [1, 2, 3] -> try parsing columns 1, 2, 3 

926 each as a separate date column. 

927 * list of lists. e.g. If [[1, 3]] -> combine columns 1 and 3 and parse as 

928 a single date column. 

929 * dict, e.g. {'foo' : [1, 3]} -> parse columns 1, 3 as date and call 

930 result 'foo' 

931 

932 encoding : str, optional, default 'utf-8' 

933 Encoding of XML document. 

934 

935 parser : {'lxml','etree'}, default 'lxml' 

936 Parser module to use for retrieval of data. Only 'lxml' and 

937 'etree' are supported. With 'lxml' more complex ``XPath`` searches 

938 and ability to use XSLT stylesheet are supported. 

939 

940 stylesheet : str, path object or file-like object 

941 A URL, file-like object, or a string path containing an XSLT script. 

942 This stylesheet should flatten complex, deeply nested XML documents 

943 for easier parsing. To use this feature you must have ``lxml`` module 

944 installed and specify 'lxml' as ``parser``. The ``xpath`` must 

945 reference nodes of transformed XML document generated after XSLT 

946 transformation and not the original XML document. Only XSLT 1.0 

947 scripts and not later versions is currently supported. 

948 

949 iterparse : dict, optional 

950 The nodes or attributes to retrieve in iterparsing of XML document 

951 as a dict with key being the name of repeating element and value being 

952 list of elements or attribute names that are descendants of the repeated 

953 element. Note: If this option is used, it will replace ``xpath`` parsing 

954 and unlike ``xpath``, descendants do not need to relate to each other but can 

955 exist any where in document under the repeating element. This memory- 

956 efficient method should be used for very large XML files (500MB, 1GB, or 5GB+). 

957 For example, ``{"row_element": ["child_elem", "attr", "grandchild_elem"]}``. 

958 

959 compression : str or dict, default 'infer' 

960 For on-the-fly decompression of on-disk data. If 'infer' and 

961 'path_or_buffer' is path-like, then detect compression from the 

962 following extensions: '.gz', '.bz2', '.zip', '.xz', '.zst', '.tar', 

963 '.tar.gz', '.tar.xz' or '.tar.bz2' (otherwise no compression). 

964 If using 'zip' or 'tar', the ZIP file must contain only one data 

965 file to be read in. Set to ``None`` for no decompression. 

966 Can also be a dict with key ``'method'`` set to one of 

967 {``'zip'``, ``'gzip'``, ``'bz2'``, ``'zstd'``, ``'xz'``, ``'tar'``} 

968 and other key-value pairs are forwarded to ``zipfile.ZipFile``, 

969 ``gzip.GzipFile``, ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``, 

970 ``lzma.LZMAFile`` or ``tarfile.TarFile``, respectively. 

971 As an example, the following could be passed for Zstandard 

972 decompression using a custom compression dictionary: 

973 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``. 

974 

975 storage_options : dict, optional 

976 Extra options that make sense for a particular storage connection, 

977 e.g. host, port, username, password, etc. For HTTP(S) URLs the 

978 key-value pairs are forwarded to ``urllib.request.Request`` as header 

979 options. For other URLs (e.g. starting with "s3://", and "gcs://") 

980 the key-value pairs are forwarded to ``fsspec.open``. Please see 

981 ``fsspec`` and ``urllib`` for more details, and for more examples on 

982 storage options refer `here <https://pandas.pydata.org/docs/ 

983 user_guide/io.html?highlight=storage_options#reading-writing-remote- 

984 files>`_. 

985 

986 dtype_backend : {'numpy_nullable', 'pyarrow'} 

987 Back-end data type applied to the resultant :class:`DataFrame` 

988 (still experimental). If not specified, the default behavior 

989 is to not use nullable data types. If specified, the behavior 

990 is as follows: 

991 

992 * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame` 

993 * ``"pyarrow"``: returns pyarrow-backed nullable 

994 :class:`ArrowDtype` :class:`DataFrame` 

995 

996 .. versionadded:: 2.0 

997 

998 Returns 

999 ------- 

1000 df 

1001 A DataFrame. 

1002 

1003 See Also 

1004 -------- 

1005 read_json : Convert a JSON string to pandas object. 

1006 read_html : Read HTML tables into a list of DataFrame objects. 

1007 

1008 Notes 

1009 ----- 

1010 This method is best designed to import shallow XML documents in 

1011 following format which is the ideal fit for the two-dimensions of a 

1012 ``DataFrame`` (row by column). :: 

1013 

1014 <root> 

1015 <row> 

1016 <column1>data</column1> 

1017 <column2>data</column2> 

1018 <column3>data</column3> 

1019 ... 

1020 </row> 

1021 <row> 

1022 ... 

1023 </row> 

1024 ... 

1025 </root> 

1026 

1027 As a file format, XML documents can be designed any way including 

1028 layout of elements and attributes as long as it conforms to W3C 

1029 specifications. Therefore, this method is a convenience handler for 

1030 a specific flatter design and not all possible XML structures. 

1031 

1032 However, for more complex XML documents, ``stylesheet`` allows you to 

1033 temporarily redesign original document with XSLT (a special purpose 

1034 language) for a flatter version for migration to a DataFrame. 

1035 

1036 This function will *always* return a single :class:`DataFrame` or raise 

1037 exceptions due to issues with XML document, ``xpath``, or other 

1038 parameters. 

1039 

1040 See the :ref:`read_xml documentation in the IO section of the docs 

1041 <io.read_xml>` for more information in using this method to parse XML 

1042 files to DataFrames. 

1043 

1044 Examples 

1045 -------- 

1046 >>> from io import StringIO 

1047 >>> xml = '''<?xml version='1.0' encoding='utf-8'?> 

1048 ... <data xmlns="http://example.com"> 

1049 ... <row> 

1050 ... <shape>square</shape> 

1051 ... <degrees>360</degrees> 

1052 ... <sides>4.0</sides> 

1053 ... </row> 

1054 ... <row> 

1055 ... <shape>circle</shape> 

1056 ... <degrees>360</degrees> 

1057 ... <sides/> 

1058 ... </row> 

1059 ... <row> 

1060 ... <shape>triangle</shape> 

1061 ... <degrees>180</degrees> 

1062 ... <sides>3.0</sides> 

1063 ... </row> 

1064 ... </data>''' 

1065 

1066 >>> df = pd.read_xml(StringIO(xml)) 

1067 >>> df 

1068 shape degrees sides 

1069 0 square 360 4.0 

1070 1 circle 360 NaN 

1071 2 triangle 180 3.0 

1072 

1073 >>> xml = '''<?xml version='1.0' encoding='utf-8'?> 

1074 ... <data> 

1075 ... <row shape="square" degrees="360" sides="4.0"/> 

1076 ... <row shape="circle" degrees="360"/> 

1077 ... <row shape="triangle" degrees="180" sides="3.0"/> 

1078 ... </data>''' 

1079 

1080 >>> df = pd.read_xml(StringIO(xml), xpath=".//row") 

1081 >>> df 

1082 shape degrees sides 

1083 0 square 360 4.0 

1084 1 circle 360 NaN 

1085 2 triangle 180 3.0 

1086 

1087 >>> xml = '''<?xml version='1.0' encoding='utf-8'?> 

1088 ... <doc:data xmlns:doc="https://example.com"> 

1089 ... <doc:row> 

1090 ... <doc:shape>square</doc:shape> 

1091 ... <doc:degrees>360</doc:degrees> 

1092 ... <doc:sides>4.0</doc:sides> 

1093 ... </doc:row> 

1094 ... <doc:row> 

1095 ... <doc:shape>circle</doc:shape> 

1096 ... <doc:degrees>360</doc:degrees> 

1097 ... <doc:sides/> 

1098 ... </doc:row> 

1099 ... <doc:row> 

1100 ... <doc:shape>triangle</doc:shape> 

1101 ... <doc:degrees>180</doc:degrees> 

1102 ... <doc:sides>3.0</doc:sides> 

1103 ... </doc:row> 

1104 ... </doc:data>''' 

1105 

1106 >>> df = pd.read_xml( 

1107 ... StringIO(xml), 

1108 ... xpath="//doc:row", 

1109 ... namespaces={"doc": "https://example.com"}, 

1110 ... ) 

1111 >>> df 

1112 shape degrees sides 

1113 0 square 360 4.0 

1114 1 circle 360 NaN 

1115 2 triangle 180 3.0 

1116 

1117 >>> xml_data = ''' 

1118 ... <data> 

1119 ... <row> 

1120 ... <index>0</index> 

1121 ... <a>1</a> 

1122 ... <b>2.5</b> 

1123 ... <c>True</c> 

1124 ... <d>a</d> 

1125 ... <e>2019-12-31 00:00:00</e> 

1126 ... </row> 

1127 ... <row> 

1128 ... <index>1</index> 

1129 ... <b>4.5</b> 

1130 ... <c>False</c> 

1131 ... <d>b</d> 

1132 ... <e>2019-12-31 00:00:00</e> 

1133 ... </row> 

1134 ... </data> 

1135 ... ''' 

1136 

1137 >>> df = pd.read_xml( 

1138 ... StringIO(xml_data), dtype_backend="numpy_nullable", parse_dates=["e"] 

1139 ... ) 

1140 >>> df 

1141 index a b c d e 

1142 0 0 1 2.5 True a 2019-12-31 

1143 1 1 <NA> 4.5 False b 2019-12-31 

1144 """ 

1145 check_dtype_backend(dtype_backend) 

1146 

1147 return _parse( 

1148 path_or_buffer=path_or_buffer, 

1149 xpath=xpath, 

1150 namespaces=namespaces, 

1151 elems_only=elems_only, 

1152 attrs_only=attrs_only, 

1153 names=names, 

1154 dtype=dtype, 

1155 converters=converters, 

1156 parse_dates=parse_dates, 

1157 encoding=encoding, 

1158 parser=parser, 

1159 stylesheet=stylesheet, 

1160 iterparse=iterparse, 

1161 compression=compression, 

1162 storage_options=storage_options, 

1163 dtype_backend=dtype_backend, 

1164 )