Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/pandas/io/xml.py: 16%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""
2:mod:``pandas.io.xml`` is a module for reading XML.
3"""
5from __future__ import annotations
7import io
8from os import PathLike
9from typing import (
10 TYPE_CHECKING,
11 Any,
12)
14from pandas._libs import lib
15from pandas.compat._optional import import_optional_dependency
16from pandas.errors import (
17 AbstractMethodError,
18 ParserError,
19)
20from pandas.util._decorators import set_module
21from pandas.util._validators import check_dtype_backend
23from pandas.core.dtypes.common import is_list_like
25from pandas.io.common import (
26 get_handle,
27 infer_compression,
28 is_fsspec_url,
29 is_url,
30 stringify_path,
31)
32from pandas.io.parsers import TextParser
34if TYPE_CHECKING:
35 from collections.abc import (
36 Callable,
37 Sequence,
38 )
39 from xml.etree.ElementTree import Element
41 from lxml import etree
43 from pandas._typing import (
44 CompressionOptions,
45 ConvertersArg,
46 DtypeArg,
47 DtypeBackend,
48 FilePath,
49 ParseDatesArg,
50 ReadBuffer,
51 StorageOptions,
52 XMLParsers,
53 )
55 from pandas import DataFrame
58class _XMLFrameParser:
59 """
60 Internal subclass to parse XML into DataFrames.
62 Parameters
63 ----------
64 path_or_buffer : a valid JSON ``str``, path object or file-like object
65 Any valid string path is acceptable. The string could be a URL. Valid
66 URL schemes include http, ftp, s3, and file.
68 xpath : str or regex
69 The ``XPath`` expression to parse required set of nodes for
70 migration to :class:`~pandas.DataFrame`. ``etree`` supports limited ``XPath``.
72 namespaces : dict
73 The namespaces defined in XML document (``xmlns:namespace='URI'``)
74 as dicts with key being namespace and value the URI.
76 elems_only : bool
77 Parse only the child elements at the specified ``xpath``.
79 attrs_only : bool
80 Parse only the attributes at the specified ``xpath``.
82 names : list
83 Column names for :class:`~pandas.DataFrame` of parsed XML data.
85 dtype : dict
86 Data type for data or columns. E.g. {'a': np.float64,
87 'b': np.int32, 'c': 'Int64'}
89 converters : dict, optional
90 Dict of functions for converting values in certain columns. Keys can
91 either be integers or column labels.
93 parse_dates : bool or list of int or names or list of lists or dict
94 Converts either index or select columns to datetimes
96 encoding : str
97 Encoding of xml object or document.
99 stylesheet : str or file-like
100 URL, file, file-like object, or a raw string containing XSLT,
101 ``etree`` does not support XSLT but retained for consistency.
103 iterparse : dict, optional
104 Dict with row element as key and list of descendant elements
105 and/or attributes as value to be retrieved in iterparsing of
106 XML document.
108 compression : str or dict, default 'infer'
109 For on-the-fly decompression of on-disk data. If 'infer' and
110 'path_or_buffer' is path-like, then detect compression from the
111 following extensions: '.gz', '.bz2', '.zip', '.xz', '.zst', '.tar',
112 '.tar.gz', '.tar.xz' or '.tar.bz2' (otherwise no compression).
113 If using 'zip' or 'tar', the ZIP file must contain only one data
114 file to be read in. Set to ``None`` for no decompression.
115 Can also be a dict with key ``'method'`` set to one of
116 {``'zip'``, ``'gzip'``, ``'bz2'``, ``'zstd'``, ``'xz'``, ``'tar'``}
117 and other key-value pairs are forwarded to ``zipfile.ZipFile``,
118 ``gzip.GzipFile``, ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``,
119 ``lzma.LZMAFile`` or ``tarfile.TarFile``, respectively.
120 As an example, the following could be passed for Zstandard
121 decompression using a custom compression dictionary:
122 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``.
124 storage_options : dict, optional
125 Extra options that make sense for a particular storage connection,
126 e.g. host, port, username, password, etc. For HTTP(S) URLs the
127 key-value pairs are forwarded to ``urllib.request.Request`` as header
128 options. For other URLs (e.g. starting with "s3://", and "gcs://")
129 the key-value pairs are forwarded to ``fsspec.open``. Please see
130 ``fsspec`` and ``urllib`` for more details, and for more examples on
131 storage options refer `here <https://pandas.pydata.org/docs/
132 user_guide/io.html?highlight=storage_options#reading-writing-remote-
133 files>`_.
135 See also
136 --------
137 pandas.io.xml._EtreeFrameParser
138 pandas.io.xml._LxmlFrameParser
140 Notes
141 -----
142 To subclass this class effectively you must override the following methods:`
143 * :func:`parse_data`
144 * :func:`_parse_nodes`
145 * :func:`_iterparse_nodes`
146 * :func:`_parse_doc`
147 * :func:`_validate_names`
148 * :func:`_validate_path`
151 See each method's respective documentation for details on their
152 functionality.
153 """
155 def __init__(
156 self,
157 path_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str],
158 xpath: str,
159 namespaces: dict[str, str] | None,
160 elems_only: bool,
161 attrs_only: bool,
162 names: Sequence[str] | None,
163 dtype: DtypeArg | None,
164 converters: ConvertersArg | None,
165 parse_dates: ParseDatesArg | None,
166 encoding: str | None,
167 stylesheet: FilePath | ReadBuffer[bytes] | ReadBuffer[str] | None,
168 iterparse: dict[str, list[str]] | None,
169 compression: CompressionOptions,
170 storage_options: StorageOptions,
171 ) -> None:
172 self.path_or_buffer = path_or_buffer
173 self.xpath = xpath
174 self.namespaces = namespaces
175 self.elems_only = elems_only
176 self.attrs_only = attrs_only
177 self.names = names
178 self.dtype = dtype
179 self.converters = converters
180 self.parse_dates = parse_dates
181 self.encoding = encoding
182 self.stylesheet = stylesheet
183 self.iterparse = iterparse
184 self.compression: CompressionOptions = compression
185 self.storage_options = storage_options
187 def parse_data(self) -> list[dict[str, str | None]]:
188 """
189 Parse xml data.
191 This method will call the other internal methods to
192 validate ``xpath``, names, parse and return specific nodes.
193 """
195 raise AbstractMethodError(self)
197 def _parse_nodes(self, elems: list[Any]) -> list[dict[str, str | None]]:
198 """
199 Parse xml nodes.
201 This method will parse the children and attributes of elements
202 in ``xpath``, conditionally for only elements, only attributes
203 or both while optionally renaming node names.
205 Raises
206 ------
207 ValueError
208 * If only elements and only attributes are specified.
210 Notes
211 -----
212 Namespace URIs will be removed from return node values. Also,
213 elements with missing children or attributes compared to siblings
214 will have optional keys filled with None values.
215 """
217 dicts: list[dict[str, str | None]]
219 if self.elems_only and self.attrs_only:
220 raise ValueError("Either element or attributes can be parsed not both.")
221 if self.elems_only:
222 if self.names:
223 dicts = [
224 {
225 **(
226 {el.tag: el.text}
227 if el.text and not el.text.isspace()
228 else {}
229 ),
230 **{
231 nm: ch.text if ch.text else None
232 for nm, ch in zip(self.names, el.findall("*"), strict=True)
233 },
234 }
235 for el in elems
236 ]
237 else:
238 dicts = [
239 {ch.tag: ch.text if ch.text else None for ch in el.findall("*")}
240 for el in elems
241 ]
243 elif self.attrs_only:
244 dicts = [
245 {k: v if v else None for k, v in el.attrib.items()} for el in elems
246 ]
248 elif self.names:
249 dicts = [
250 {
251 **el.attrib,
252 **({el.tag: el.text} if el.text and not el.text.isspace() else {}),
253 **{
254 nm: ch.text if ch.text else None
255 for nm, ch in zip(self.names, el.findall("*"), strict=False)
256 },
257 }
258 for el in elems
259 ]
261 else:
262 dicts = [
263 {
264 **el.attrib,
265 **({el.tag: el.text} if el.text and not el.text.isspace() else {}),
266 **{ch.tag: ch.text if ch.text else None for ch in el.findall("*")},
267 }
268 for el in elems
269 ]
271 dicts = [
272 {k.split("}")[1] if "}" in k else k: v for k, v in d.items()} for d in dicts
273 ]
275 keys = list(dict.fromkeys([k for d in dicts for k in d.keys()]))
276 dicts = [{k: d[k] if k in d.keys() else None for k in keys} for d in dicts]
278 if self.names:
279 dicts = [dict(zip(self.names, d.values(), strict=True)) for d in dicts]
281 return dicts
283 def _iterparse_nodes(self, iterparse: Callable) -> list[dict[str, str | None]]:
284 """
285 Iterparse xml nodes.
287 This method will read in local disk, decompressed XML files for elements
288 and underlying descendants using iterparse, a method to iterate through
289 an XML tree without holding entire XML tree in memory.
291 Raises
292 ------
293 TypeError
294 * If ``iterparse`` is not a dict or its dict value is not list-like.
295 ParserError
296 * If ``path_or_buffer`` is not a physical file on disk or file-like object.
297 * If no data is returned from selected items in ``iterparse``.
299 Notes
300 -----
301 Namespace URIs will be removed from return node values. Also,
302 elements with missing children or attributes in submitted list
303 will have optional keys filled with None values.
304 """
306 dicts: list[dict[str, str | None]] = []
307 row: dict[str, str | None] | None = None
309 if not isinstance(self.iterparse, dict):
310 raise TypeError(
311 f"{type(self.iterparse).__name__} is not a valid type for iterparse"
312 )
314 row_node = next(iter(self.iterparse.keys())) if self.iterparse else ""
315 if not is_list_like(self.iterparse[row_node]):
316 raise TypeError(
317 f"{type(self.iterparse[row_node])} is not a valid type "
318 "for value in iterparse"
319 )
321 if (not hasattr(self.path_or_buffer, "read")) and (
322 not isinstance(self.path_or_buffer, (str, PathLike))
323 or is_url(self.path_or_buffer)
324 or is_fsspec_url(self.path_or_buffer)
325 or (
326 isinstance(self.path_or_buffer, str)
327 and self.path_or_buffer.startswith(("<?xml", "<"))
328 )
329 or infer_compression(self.path_or_buffer, "infer") is not None
330 ):
331 raise ParserError(
332 "iterparse is designed for large XML files that are fully extracted on "
333 "local disk and not as compressed files or online sources."
334 )
336 iterparse_repeats = len(self.iterparse[row_node]) != len(
337 set(self.iterparse[row_node])
338 )
340 parser = iterparse(self.path_or_buffer, events=("start", "end"))
341 try:
342 for event, elem in parser:
343 curr_elem = elem.tag.split("}")[1] if "}" in elem.tag else elem.tag
345 if event == "start":
346 if curr_elem == row_node:
347 row = {}
349 if row is not None:
350 if self.names and iterparse_repeats:
351 for col, nm in zip(
352 self.iterparse[row_node], self.names, strict=True
353 ):
354 if curr_elem == col:
355 elem_val = elem.text if elem.text else None
356 if elem_val not in row.values() and nm not in row:
357 row[nm] = elem_val
359 if col in elem.attrib:
360 if (
361 elem.attrib[col] not in row.values()
362 and nm not in row
363 ):
364 row[nm] = elem.attrib[col]
365 else:
366 for col in self.iterparse[row_node]:
367 if curr_elem == col:
368 row[col] = elem.text if elem.text else None
369 if col in elem.attrib:
370 row[col] = elem.attrib[col]
372 if event == "end":
373 if curr_elem == row_node and row is not None:
374 dicts.append(row)
375 row = None
377 elem.clear()
378 if hasattr(elem, "getprevious"):
379 while (
380 elem.getprevious() is not None
381 and elem.getparent() is not None
382 ):
383 del elem.getparent()[0]
384 finally:
385 if hasattr(parser, "close"):
386 parser.close()
388 if dicts == []:
389 raise ParserError("No result from selected items in iterparse.")
391 keys = list(dict.fromkeys([k for d in dicts for k in d.keys()]))
392 dicts = [{k: d[k] if k in d.keys() else None for k in keys} for d in dicts]
394 if self.names:
395 dicts = [dict(zip(self.names, d.values(), strict=True)) for d in dicts]
397 return dicts
399 def _validate_path(self) -> list[Any]:
400 """
401 Validate ``xpath``.
403 This method checks for syntax, evaluation, or empty nodes return.
405 Raises
406 ------
407 SyntaxError
408 * If xpah is not supported or issues with namespaces.
410 ValueError
411 * If xpah does not return any nodes.
412 """
414 raise AbstractMethodError(self)
416 def _validate_names(self) -> None:
417 """
418 Validate names.
420 This method will check if names is a list-like and aligns
421 with length of parse nodes.
423 Raises
424 ------
425 ValueError
426 * If value is not a list and less then length of nodes.
427 """
428 raise AbstractMethodError(self)
430 def _parse_doc(
431 self, raw_doc: FilePath | ReadBuffer[bytes] | ReadBuffer[str]
432 ) -> Element | etree._Element:
433 """
434 Build tree from path_or_buffer.
436 This method will parse XML object into tree
437 either from string/bytes or file location.
438 """
439 raise AbstractMethodError(self)
442class _EtreeFrameParser(_XMLFrameParser):
443 """
444 Internal class to parse XML into DataFrames with the Python
445 standard library XML module: `xml.etree.ElementTree`.
446 """
448 def parse_data(self) -> list[dict[str, str | None]]:
449 from xml.etree.ElementTree import iterparse
451 if self.stylesheet is not None:
452 raise ValueError(
453 "To use stylesheet, you need lxml installed and selected as parser."
454 )
456 if self.iterparse is None:
457 self.xml_doc = self._parse_doc(self.path_or_buffer)
458 elems = self._validate_path()
460 self._validate_names()
462 xml_dicts: list[dict[str, str | None]] = (
463 self._parse_nodes(elems)
464 if self.iterparse is None
465 else self._iterparse_nodes(iterparse)
466 )
468 return xml_dicts
470 def _validate_path(self) -> list[Any]:
471 """
472 Notes
473 -----
474 ``etree`` supports limited ``XPath``. If user attempts a more complex
475 expression syntax error will raise.
476 """
478 msg = (
479 "xpath does not return any nodes or attributes. "
480 "Be sure to specify in `xpath` the parent nodes of "
481 "children and attributes to parse. "
482 "If document uses namespaces denoted with "
483 "xmlns, be sure to define namespaces and "
484 "use them in xpath."
485 )
486 try:
487 elems = self.xml_doc.findall(self.xpath, namespaces=self.namespaces)
488 children = [ch for el in elems for ch in el.findall("*")]
489 attrs = {k: v for el in elems for k, v in el.attrib.items()}
491 if elems is None:
492 raise ValueError(msg)
494 if elems is not None:
495 if self.elems_only and children == []:
496 raise ValueError(msg)
497 if self.attrs_only and attrs == {}:
498 raise ValueError(msg)
499 if children == [] and attrs == {}:
500 raise ValueError(msg)
502 except (KeyError, SyntaxError) as err:
503 raise SyntaxError(
504 "You have used an incorrect or unsupported XPath "
505 "expression for etree library or you used an "
506 "undeclared namespace prefix."
507 ) from err
509 return elems
511 def _validate_names(self) -> None:
512 children: list[Any]
514 if self.names:
515 if self.iterparse:
516 children = self.iterparse[next(iter(self.iterparse))]
517 else:
518 parent = self.xml_doc.find(self.xpath, namespaces=self.namespaces)
519 children = parent.findall("*") if parent is not None else []
521 if is_list_like(self.names):
522 if len(self.names) < len(children):
523 raise ValueError(
524 "names does not match length of child elements in xpath."
525 )
526 else:
527 raise TypeError(
528 f"{type(self.names).__name__} is not a valid type for names"
529 )
531 def _parse_doc(
532 self, raw_doc: FilePath | ReadBuffer[bytes] | ReadBuffer[str]
533 ) -> Element:
534 from xml.etree.ElementTree import (
535 XMLParser,
536 parse,
537 )
539 handle_data = get_data_from_filepath(
540 filepath_or_buffer=raw_doc,
541 encoding=self.encoding,
542 compression=self.compression,
543 storage_options=self.storage_options,
544 )
546 with handle_data as xml_data:
547 curr_parser = XMLParser(encoding=self.encoding)
548 document = parse(xml_data, parser=curr_parser)
550 return document.getroot()
553class _LxmlFrameParser(_XMLFrameParser):
554 """
555 Internal class to parse XML into :class:`~pandas.DataFrame` with third-party
556 full-featured XML library, ``lxml``, that supports
557 ``XPath`` 1.0 and XSLT 1.0.
558 """
560 def parse_data(self) -> list[dict[str, str | None]]:
561 """
562 Parse xml data.
564 This method will call the other internal methods to
565 validate ``xpath``, names, optionally parse and run XSLT,
566 and parse original or transformed XML and return specific nodes.
567 """
568 from lxml.etree import iterparse
570 if self.iterparse is None:
571 self.xml_doc = self._parse_doc(self.path_or_buffer)
573 if self.stylesheet:
574 self.xsl_doc = self._parse_doc(self.stylesheet)
575 self.xml_doc = self._transform_doc()
577 elems = self._validate_path()
579 self._validate_names()
581 xml_dicts: list[dict[str, str | None]] = (
582 self._parse_nodes(elems)
583 if self.iterparse is None
584 else self._iterparse_nodes(iterparse)
585 )
587 return xml_dicts
589 def _validate_path(self) -> list[Any]:
590 msg = (
591 "xpath does not return any nodes or attributes. "
592 "Be sure to specify in `xpath` the parent nodes of "
593 "children and attributes to parse. "
594 "If document uses namespaces denoted with "
595 "xmlns, be sure to define namespaces and "
596 "use them in xpath."
597 )
599 elems = self.xml_doc.xpath(self.xpath, namespaces=self.namespaces)
600 children = [ch for el in elems for ch in el.xpath("*")]
601 attrs = {k: v for el in elems for k, v in el.attrib.items()}
603 if elems == []:
604 raise ValueError(msg)
606 if elems != []:
607 if self.elems_only and children == []:
608 raise ValueError(msg)
609 if self.attrs_only and attrs == {}:
610 raise ValueError(msg)
611 if children == [] and attrs == {}:
612 raise ValueError(msg)
614 return elems
616 def _validate_names(self) -> None:
617 children: list[Any]
619 if self.names:
620 if self.iterparse:
621 children = self.iterparse[next(iter(self.iterparse))]
622 else:
623 children = self.xml_doc.xpath(
624 self.xpath + "[1]/*", namespaces=self.namespaces
625 )
627 if is_list_like(self.names):
628 if len(self.names) < len(children):
629 raise ValueError(
630 "names does not match length of child elements in xpath."
631 )
632 else:
633 raise TypeError(
634 f"{type(self.names).__name__} is not a valid type for names"
635 )
637 def _parse_doc(
638 self, raw_doc: FilePath | ReadBuffer[bytes] | ReadBuffer[str]
639 ) -> etree._Element:
640 from lxml.etree import (
641 XMLParser,
642 fromstring,
643 parse,
644 )
646 handle_data = get_data_from_filepath(
647 filepath_or_buffer=raw_doc,
648 encoding=self.encoding,
649 compression=self.compression,
650 storage_options=self.storage_options,
651 )
653 with handle_data as xml_data:
654 curr_parser = XMLParser(encoding=self.encoding)
656 if isinstance(xml_data, io.StringIO):
657 if self.encoding is None:
658 raise TypeError(
659 "Can not pass encoding None when input is StringIO."
660 )
662 document = fromstring(
663 xml_data.getvalue().encode(self.encoding), parser=curr_parser
664 )
665 else:
666 document = parse(xml_data, parser=curr_parser)
668 return document
670 def _transform_doc(self) -> etree._XSLTResultTree:
671 """
672 Transform original tree using stylesheet.
674 This method will transform original xml using XSLT script into
675 am ideally flatter xml document for easier parsing and migration
676 to Data Frame.
677 """
678 from lxml.etree import XSLT
680 transformer = XSLT(self.xsl_doc)
681 new_doc = transformer(self.xml_doc)
683 return new_doc
686def get_data_from_filepath(
687 filepath_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str],
688 encoding: str | None,
689 compression: CompressionOptions,
690 storage_options: StorageOptions,
691):
692 """
693 Extract raw XML data.
695 The method accepts two input types:
696 1. filepath (string-like)
697 2. file-like object (e.g. open file object, StringIO)
698 """
699 filepath_or_buffer = stringify_path(filepath_or_buffer)
700 with get_handle(
701 filepath_or_buffer,
702 "r",
703 encoding=encoding,
704 compression=compression,
705 storage_options=storage_options,
706 ) as handle_obj:
707 return (
708 preprocess_data(handle_obj.handle.read())
709 if hasattr(handle_obj.handle, "read")
710 else handle_obj.handle
711 )
714def preprocess_data(
715 data: str | bytes | io.StringIO | io.BytesIO,
716) -> io.StringIO | io.BytesIO:
717 """
718 Convert extracted raw data.
720 This method will return underlying data of extracted XML content.
721 The data either has a `read` attribute (e.g. a file object or a
722 StringIO/BytesIO) or is a string or bytes that is an XML document.
723 """
725 if isinstance(data, str):
726 data = io.StringIO(data)
728 elif isinstance(data, bytes):
729 data = io.BytesIO(data)
731 return data
734def _data_to_frame(data: list[dict[str, str | None]], **kwargs) -> DataFrame:
735 """
736 Convert parsed data to Data Frame.
738 This method will bind xml dictionary data of keys and values
739 into named columns of Data Frame using the built-in TextParser
740 class that build Data Frame and infers specific dtypes.
741 """
743 tags = next(iter(data))
744 nodes = [list(d.values()) for d in data]
746 try:
747 with TextParser(nodes, names=tags, **kwargs) as tp:
748 return tp.read()
749 except ParserError as err:
750 raise ParserError(
751 "XML document may be too complex for import. "
752 "Try to flatten document and use distinct "
753 "element and attribute names."
754 ) from err
757def _parse(
758 path_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str],
759 xpath: str,
760 namespaces: dict[str, str] | None,
761 elems_only: bool,
762 attrs_only: bool,
763 names: Sequence[str] | None,
764 dtype: DtypeArg | None,
765 converters: ConvertersArg | None,
766 parse_dates: ParseDatesArg | None,
767 encoding: str | None,
768 parser: XMLParsers,
769 stylesheet: FilePath | ReadBuffer[bytes] | ReadBuffer[str] | None,
770 iterparse: dict[str, list[str]] | None,
771 compression: CompressionOptions,
772 storage_options: StorageOptions,
773 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default,
774 **kwargs,
775) -> DataFrame:
776 """
777 Call internal parsers.
779 This method will conditionally call internal parsers:
780 LxmlFrameParser and/or EtreeParser.
782 Raises
783 ------
784 ImportError
785 * If lxml is not installed if selected as parser.
787 ValueError
788 * If parser is not lxml or etree.
789 """
791 p: _EtreeFrameParser | _LxmlFrameParser
793 if parser == "lxml":
794 lxml = import_optional_dependency("lxml.etree", errors="ignore")
796 if lxml is not None:
797 p = _LxmlFrameParser(
798 path_or_buffer,
799 xpath,
800 namespaces,
801 elems_only,
802 attrs_only,
803 names,
804 dtype,
805 converters,
806 parse_dates,
807 encoding,
808 stylesheet,
809 iterparse,
810 compression,
811 storage_options,
812 )
813 else:
814 raise ImportError("lxml not found, please install or use the etree parser.")
816 elif parser == "etree":
817 p = _EtreeFrameParser(
818 path_or_buffer,
819 xpath,
820 namespaces,
821 elems_only,
822 attrs_only,
823 names,
824 dtype,
825 converters,
826 parse_dates,
827 encoding,
828 stylesheet,
829 iterparse,
830 compression,
831 storage_options,
832 )
833 else:
834 raise ValueError("Values for parser can only be lxml or etree.")
836 data_dicts = p.parse_data()
838 return _data_to_frame(
839 data=data_dicts,
840 dtype=dtype,
841 converters=converters,
842 parse_dates=parse_dates,
843 dtype_backend=dtype_backend,
844 **kwargs,
845 )
848@set_module("pandas")
849def read_xml(
850 path_or_buffer: FilePath | ReadBuffer[bytes] | ReadBuffer[str],
851 *,
852 xpath: str = "./*",
853 namespaces: dict[str, str] | None = None,
854 elems_only: bool = False,
855 attrs_only: bool = False,
856 names: Sequence[str] | None = None,
857 dtype: DtypeArg | None = None,
858 converters: ConvertersArg | None = None,
859 parse_dates: ParseDatesArg | None = None,
860 # encoding can not be None for lxml and StringIO input
861 encoding: str | None = "utf-8",
862 parser: XMLParsers = "lxml",
863 stylesheet: FilePath | ReadBuffer[bytes] | ReadBuffer[str] | None = None,
864 iterparse: dict[str, list[str]] | None = None,
865 compression: CompressionOptions = "infer",
866 storage_options: StorageOptions | None = None,
867 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default,
868) -> DataFrame:
869 r"""
870 Read XML document into a :class:`~pandas.DataFrame` object.
872 Parameters
873 ----------
874 path_or_buffer : str, path object, or file-like object
875 String path, path object (implementing ``os.PathLike[str]``), or file-like
876 object implementing a ``read()`` function. The string can be a path.
877 The string can further be a URL. Valid URL schemes
878 include http, ftp, s3, and file.
880 xpath : str, optional, default './\*'
881 The ``XPath`` to parse required set of nodes for migration to
882 :class:`~pandas.DataFrame`.``XPath`` should return a collection of elements
883 and not a single element. Note: The ``etree`` parser supports limited ``XPath``
884 expressions. For more complex ``XPath``, use ``lxml`` which requires
885 installation.
887 namespaces : dict, optional
888 The namespaces defined in XML document as dicts with key being
889 namespace prefix and value the URI. There is no need to include all
890 namespaces in XML, only the ones used in ``xpath`` expression.
891 Note: if XML document uses default namespace denoted as
892 `xmlns='<URI>'` without a prefix, you must assign any temporary
893 namespace prefix such as 'doc' to the URI in order to parse
894 underlying nodes and/or attributes.
896 elems_only : bool, optional, default False
897 Parse only the child elements at the specified ``xpath``. By default,
898 all child elements and non-empty text nodes are returned.
900 attrs_only : bool, optional, default False
901 Parse only the attributes at the specified ``xpath``.
902 By default, all attributes are returned.
904 names : list-like, optional
905 Column names for DataFrame of parsed XML data. Use this parameter to
906 rename original element names and distinguish same named elements and
907 attributes.
909 dtype : Type name or dict of column -> type, optional
910 Data type for data or columns. E.g. {'a': np.float64, 'b': np.int32,
911 'c': 'Int64'}
912 Use `str` or `object` together with suitable `na_values` settings
913 to preserve and not interpret dtype.
914 If converters are specified, they will be applied INSTEAD
915 of dtype conversion.
917 converters : dict, optional
918 Dict of functions for converting values in certain columns. Keys can either
919 be integers or column labels.
921 parse_dates : bool or list of int or names or list of lists or dict, default False
922 Identifiers to parse index or columns to datetime. The behavior is as follows:
924 * boolean. If True -> try parsing the index.
925 * list of int or names. e.g. If [1, 2, 3] -> try parsing columns 1, 2, 3
926 each as a separate date column.
927 * list of lists. e.g. If [[1, 3]] -> combine columns 1 and 3 and parse as
928 a single date column.
929 * dict, e.g. {'foo' : [1, 3]} -> parse columns 1, 3 as date and call
930 result 'foo'
932 encoding : str, optional, default 'utf-8'
933 Encoding of XML document.
935 parser : {'lxml','etree'}, default 'lxml'
936 Parser module to use for retrieval of data. Only 'lxml' and
937 'etree' are supported. With 'lxml' more complex ``XPath`` searches
938 and ability to use XSLT stylesheet are supported.
940 stylesheet : str, path object or file-like object
941 A URL, file-like object, or a string path containing an XSLT script.
942 This stylesheet should flatten complex, deeply nested XML documents
943 for easier parsing. To use this feature you must have ``lxml`` module
944 installed and specify 'lxml' as ``parser``. The ``xpath`` must
945 reference nodes of transformed XML document generated after XSLT
946 transformation and not the original XML document. Only XSLT 1.0
947 scripts and not later versions is currently supported.
949 iterparse : dict, optional
950 The nodes or attributes to retrieve in iterparsing of XML document
951 as a dict with key being the name of repeating element and value being
952 list of elements or attribute names that are descendants of the repeated
953 element. Note: If this option is used, it will replace ``xpath`` parsing
954 and unlike ``xpath``, descendants do not need to relate to each other but can
955 exist any where in document under the repeating element. This memory-
956 efficient method should be used for very large XML files (500MB, 1GB, or 5GB+).
957 For example, ``{"row_element": ["child_elem", "attr", "grandchild_elem"]}``.
959 compression : str or dict, default 'infer'
960 For on-the-fly decompression of on-disk data. If 'infer' and
961 'path_or_buffer' is path-like, then detect compression from the
962 following extensions: '.gz', '.bz2', '.zip', '.xz', '.zst', '.tar',
963 '.tar.gz', '.tar.xz' or '.tar.bz2' (otherwise no compression).
964 If using 'zip' or 'tar', the ZIP file must contain only one data
965 file to be read in. Set to ``None`` for no decompression.
966 Can also be a dict with key ``'method'`` set to one of
967 {``'zip'``, ``'gzip'``, ``'bz2'``, ``'zstd'``, ``'xz'``, ``'tar'``}
968 and other key-value pairs are forwarded to ``zipfile.ZipFile``,
969 ``gzip.GzipFile``, ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``,
970 ``lzma.LZMAFile`` or ``tarfile.TarFile``, respectively.
971 As an example, the following could be passed for Zstandard
972 decompression using a custom compression dictionary:
973 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``.
975 storage_options : dict, optional
976 Extra options that make sense for a particular storage connection,
977 e.g. host, port, username, password, etc. For HTTP(S) URLs the
978 key-value pairs are forwarded to ``urllib.request.Request`` as header
979 options. For other URLs (e.g. starting with "s3://", and "gcs://")
980 the key-value pairs are forwarded to ``fsspec.open``. Please see
981 ``fsspec`` and ``urllib`` for more details, and for more examples on
982 storage options refer `here <https://pandas.pydata.org/docs/
983 user_guide/io.html?highlight=storage_options#reading-writing-remote-
984 files>`_.
986 dtype_backend : {'numpy_nullable', 'pyarrow'}
987 Back-end data type applied to the resultant :class:`DataFrame`
988 (still experimental). If not specified, the default behavior
989 is to not use nullable data types. If specified, the behavior
990 is as follows:
992 * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame`
993 * ``"pyarrow"``: returns pyarrow-backed nullable
994 :class:`ArrowDtype` :class:`DataFrame`
996 .. versionadded:: 2.0
998 Returns
999 -------
1000 df
1001 A DataFrame.
1003 See Also
1004 --------
1005 read_json : Convert a JSON string to pandas object.
1006 read_html : Read HTML tables into a list of DataFrame objects.
1008 Notes
1009 -----
1010 This method is best designed to import shallow XML documents in
1011 following format which is the ideal fit for the two-dimensions of a
1012 ``DataFrame`` (row by column). ::
1014 <root>
1015 <row>
1016 <column1>data</column1>
1017 <column2>data</column2>
1018 <column3>data</column3>
1019 ...
1020 </row>
1021 <row>
1022 ...
1023 </row>
1024 ...
1025 </root>
1027 As a file format, XML documents can be designed any way including
1028 layout of elements and attributes as long as it conforms to W3C
1029 specifications. Therefore, this method is a convenience handler for
1030 a specific flatter design and not all possible XML structures.
1032 However, for more complex XML documents, ``stylesheet`` allows you to
1033 temporarily redesign original document with XSLT (a special purpose
1034 language) for a flatter version for migration to a DataFrame.
1036 This function will *always* return a single :class:`DataFrame` or raise
1037 exceptions due to issues with XML document, ``xpath``, or other
1038 parameters.
1040 See the :ref:`read_xml documentation in the IO section of the docs
1041 <io.read_xml>` for more information in using this method to parse XML
1042 files to DataFrames.
1044 Examples
1045 --------
1046 >>> from io import StringIO
1047 >>> xml = '''<?xml version='1.0' encoding='utf-8'?>
1048 ... <data xmlns="http://example.com">
1049 ... <row>
1050 ... <shape>square</shape>
1051 ... <degrees>360</degrees>
1052 ... <sides>4.0</sides>
1053 ... </row>
1054 ... <row>
1055 ... <shape>circle</shape>
1056 ... <degrees>360</degrees>
1057 ... <sides/>
1058 ... </row>
1059 ... <row>
1060 ... <shape>triangle</shape>
1061 ... <degrees>180</degrees>
1062 ... <sides>3.0</sides>
1063 ... </row>
1064 ... </data>'''
1066 >>> df = pd.read_xml(StringIO(xml))
1067 >>> df
1068 shape degrees sides
1069 0 square 360 4.0
1070 1 circle 360 NaN
1071 2 triangle 180 3.0
1073 >>> xml = '''<?xml version='1.0' encoding='utf-8'?>
1074 ... <data>
1075 ... <row shape="square" degrees="360" sides="4.0"/>
1076 ... <row shape="circle" degrees="360"/>
1077 ... <row shape="triangle" degrees="180" sides="3.0"/>
1078 ... </data>'''
1080 >>> df = pd.read_xml(StringIO(xml), xpath=".//row")
1081 >>> df
1082 shape degrees sides
1083 0 square 360 4.0
1084 1 circle 360 NaN
1085 2 triangle 180 3.0
1087 >>> xml = '''<?xml version='1.0' encoding='utf-8'?>
1088 ... <doc:data xmlns:doc="https://example.com">
1089 ... <doc:row>
1090 ... <doc:shape>square</doc:shape>
1091 ... <doc:degrees>360</doc:degrees>
1092 ... <doc:sides>4.0</doc:sides>
1093 ... </doc:row>
1094 ... <doc:row>
1095 ... <doc:shape>circle</doc:shape>
1096 ... <doc:degrees>360</doc:degrees>
1097 ... <doc:sides/>
1098 ... </doc:row>
1099 ... <doc:row>
1100 ... <doc:shape>triangle</doc:shape>
1101 ... <doc:degrees>180</doc:degrees>
1102 ... <doc:sides>3.0</doc:sides>
1103 ... </doc:row>
1104 ... </doc:data>'''
1106 >>> df = pd.read_xml(
1107 ... StringIO(xml),
1108 ... xpath="//doc:row",
1109 ... namespaces={"doc": "https://example.com"},
1110 ... )
1111 >>> df
1112 shape degrees sides
1113 0 square 360 4.0
1114 1 circle 360 NaN
1115 2 triangle 180 3.0
1117 >>> xml_data = '''
1118 ... <data>
1119 ... <row>
1120 ... <index>0</index>
1121 ... <a>1</a>
1122 ... <b>2.5</b>
1123 ... <c>True</c>
1124 ... <d>a</d>
1125 ... <e>2019-12-31 00:00:00</e>
1126 ... </row>
1127 ... <row>
1128 ... <index>1</index>
1129 ... <b>4.5</b>
1130 ... <c>False</c>
1131 ... <d>b</d>
1132 ... <e>2019-12-31 00:00:00</e>
1133 ... </row>
1134 ... </data>
1135 ... '''
1137 >>> df = pd.read_xml(
1138 ... StringIO(xml_data), dtype_backend="numpy_nullable", parse_dates=["e"]
1139 ... )
1140 >>> df
1141 index a b c d e
1142 0 0 1 2.5 True a 2019-12-31
1143 1 1 <NA> 4.5 False b 2019-12-31
1144 """
1145 check_dtype_backend(dtype_backend)
1147 return _parse(
1148 path_or_buffer=path_or_buffer,
1149 xpath=xpath,
1150 namespaces=namespaces,
1151 elems_only=elems_only,
1152 attrs_only=attrs_only,
1153 names=names,
1154 dtype=dtype,
1155 converters=converters,
1156 parse_dates=parse_dates,
1157 encoding=encoding,
1158 parser=parser,
1159 stylesheet=stylesheet,
1160 iterparse=iterparse,
1161 compression=compression,
1162 storage_options=storage_options,
1163 dtype_backend=dtype_backend,
1164 )