1"""
2Module contains tools for processing files into DataFrames or other objects
3
4GH#48849 provides a convenient way of deprecating keyword arguments
5"""
6
7from __future__ import annotations
8
9from collections import (
10 abc,
11 defaultdict,
12)
13import csv
14import sys
15from typing import (
16 IO,
17 TYPE_CHECKING,
18 Any,
19 Generic,
20 Literal,
21 Self,
22 TypedDict,
23 Unpack,
24 cast,
25 overload,
26)
27import warnings
28
29import numpy as np
30
31from pandas._libs import lib
32from pandas._libs.parsers import STR_NA_VALUES
33from pandas.errors import (
34 AbstractMethodError,
35 ParserWarning,
36)
37from pandas.util._decorators import (
38 set_module,
39)
40from pandas.util._exceptions import find_stack_level
41from pandas.util._validators import check_dtype_backend
42
43from pandas.core.dtypes.common import (
44 is_file_like,
45 is_float,
46 is_integer,
47 is_list_like,
48 pandas_dtype,
49)
50
51from pandas import Series
52from pandas.core.frame import DataFrame
53from pandas.core.indexes.api import RangeIndex
54
55from pandas.io.common import (
56 IOHandles,
57 get_handle,
58 stringify_path,
59 validate_header_arg,
60)
61from pandas.io.parsers.arrow_parser_wrapper import ArrowParserWrapper
62from pandas.io.parsers.base_parser import (
63 ParserBase,
64 is_index_col,
65 parser_defaults,
66)
67from pandas.io.parsers.c_parser_wrapper import CParserWrapper
68from pandas.io.parsers.python_parser import (
69 FixedWidthFieldParser,
70 PythonParser,
71)
72
73if TYPE_CHECKING:
74 from collections.abc import (
75 Callable,
76 Hashable,
77 Iterable,
78 Mapping,
79 Sequence,
80 )
81 from types import TracebackType
82
83 from pandas._typing import (
84 CompressionOptions,
85 CSVEngine,
86 DtypeArg,
87 DtypeBackend,
88 FilePath,
89 HashableT,
90 IndexLabel,
91 ReadCsvBuffer,
92 StorageOptions,
93 UsecolsArgType,
94 )
95
96 class _read_shared(TypedDict, Generic[HashableT], total=False):
97 # annotations shared between read_csv/fwf/table's overloads
98 # NOTE: Keep in sync with the annotations of the implementation
99 sep: str | None | lib.NoDefault
100 delimiter: str | None | lib.NoDefault
101 header: int | Sequence[int] | None | Literal["infer"]
102 names: Sequence[Hashable] | None | lib.NoDefault
103 index_col: IndexLabel | Literal[False] | None
104 usecols: UsecolsArgType
105 dtype: DtypeArg | None
106 engine: CSVEngine | None
107 converters: Mapping[HashableT, Callable] | None
108 true_values: list | None
109 false_values: list | None
110 skipinitialspace: bool
111 skiprows: list[int] | int | Callable[[Hashable], bool] | None
112 skipfooter: int
113 nrows: int | None
114 na_values: (
115 Hashable | Iterable[Hashable] | Mapping[Hashable, Iterable[Hashable]] | None
116 )
117 keep_default_na: bool
118 na_filter: bool
119 skip_blank_lines: bool
120 parse_dates: bool | Sequence[Hashable] | None
121 date_format: str | dict[Hashable, str] | None
122 dayfirst: bool
123 cache_dates: bool
124 compression: CompressionOptions
125 thousands: str | None
126 decimal: str
127 lineterminator: str | None
128 quotechar: str
129 quoting: int
130 doublequote: bool
131 escapechar: str | None
132 comment: str | None
133 encoding: str | None
134 encoding_errors: str | None
135 dialect: str | csv.Dialect | None
136 on_bad_lines: str
137 low_memory: bool
138 memory_map: bool
139 float_precision: Literal["high", "legacy", "round_trip"] | None
140 storage_options: StorageOptions | None
141 dtype_backend: DtypeBackend | lib.NoDefault
142
143else:
144 _read_shared = dict
145
146
147class _C_Parser_Defaults(TypedDict):
148 na_filter: Literal[True]
149 low_memory: Literal[True]
150 memory_map: Literal[False]
151 float_precision: None
152
153
154_c_parser_defaults: _C_Parser_Defaults = {
155 "na_filter": True,
156 "low_memory": True,
157 "memory_map": False,
158 "float_precision": None,
159}
160
161
162class _Fwf_Defaults(TypedDict):
163 colspecs: Literal["infer"]
164 infer_nrows: Literal[100]
165 widths: None
166
167
168_fwf_defaults: _Fwf_Defaults = {"colspecs": "infer", "infer_nrows": 100, "widths": None}
169_c_unsupported = {"skipfooter"}
170_python_unsupported = {"low_memory", "float_precision"}
171_pyarrow_unsupported = {
172 "skipfooter",
173 "float_precision",
174 "chunksize",
175 "comment",
176 "nrows",
177 "thousands",
178 "memory_map",
179 "dialect",
180 "quoting",
181 "lineterminator",
182 "converters",
183 "iterator",
184 "dayfirst",
185 "skipinitialspace",
186 "low_memory",
187}
188
189
190@overload
191def validate_integer(name: str, val: None, min_val: int = ...) -> None: ...
192
193
194@overload
195def validate_integer(name: str, val: float, min_val: int = ...) -> int: ...
196
197
198@overload
199def validate_integer(name: str, val: int | None, min_val: int = ...) -> int | None: ...
200
201
202def validate_integer(
203 name: str, val: int | float | None, min_val: int = 0
204) -> int | None:
205 """
206 Checks whether the 'name' parameter for parsing is either
207 an integer OR float that can SAFELY be cast to an integer
208 without losing accuracy. Raises a ValueError if that is
209 not the case.
210
211 Parameters
212 ----------
213 name : str
214 Parameter name (used for error reporting)
215 val : int or float
216 The value to check
217 min_val : int
218 Minimum allowed value (val < min_val will result in a ValueError)
219 """
220 if val is None:
221 return val
222
223 msg = f"'{name:s}' must be an integer >={min_val:d}"
224 if is_float(val):
225 if int(val) != val:
226 raise ValueError(msg)
227 val = int(val)
228 elif not (is_integer(val) and val >= min_val):
229 raise ValueError(msg)
230
231 return int(val)
232
233
234def _validate_names(names: Sequence[Hashable] | None) -> None:
235 """
236 Raise ValueError if the `names` parameter contains duplicates or has an
237 invalid data type.
238
239 Parameters
240 ----------
241 names : array-like or None
242 An array containing a list of the names used for the output DataFrame.
243
244 Raises
245 ------
246 ValueError
247 If names are not unique or are not ordered (e.g. set).
248 """
249 if names is not None:
250 if len(names) != len(set(names)):
251 raise ValueError("Duplicate names are not allowed.")
252 if not (
253 is_list_like(names, allow_sets=False) or isinstance(names, abc.KeysView)
254 ):
255 raise ValueError("Names should be an ordered collection.")
256
257
258def _read(
259 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], kwds
260) -> DataFrame | TextFileReader:
261 """Generic reader of line files."""
262 # if we pass a date_format and parse_dates=False, we should not parse the
263 # dates GH#44366
264 if kwds.get("parse_dates", None) is None:
265 if kwds.get("date_format", None) is None:
266 kwds["parse_dates"] = False
267 else:
268 kwds["parse_dates"] = True
269
270 # Extract some of the arguments (pass chunksize on).
271 iterator = kwds.get("iterator", False)
272 chunksize = kwds.get("chunksize", None)
273
274 # Check type of encoding_errors
275 errors = kwds.get("encoding_errors", "strict")
276 if not isinstance(errors, str):
277 raise ValueError(
278 f"encoding_errors must be a string, got {type(errors).__name__}"
279 )
280
281 if kwds.get("engine") == "pyarrow":
282 if iterator:
283 raise ValueError(
284 "The 'iterator' option is not supported with the 'pyarrow' engine"
285 )
286
287 if chunksize is not None:
288 raise ValueError(
289 "The 'chunksize' option is not supported with the 'pyarrow' engine"
290 )
291 else:
292 chunksize = validate_integer("chunksize", chunksize, 1)
293
294 nrows = kwds.get("nrows", None)
295
296 # Check for duplicates in names.
297 _validate_names(kwds.get("names", None))
298
299 # Create the parser.
300 parser = TextFileReader(filepath_or_buffer, **kwds)
301
302 if chunksize or iterator:
303 return parser
304
305 with parser:
306 return parser.read(nrows)
307
308
309@overload
310def read_csv(
311 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
312 *,
313 iterator: Literal[True],
314 chunksize: int | None = ...,
315 **kwds: Unpack[_read_shared[HashableT]],
316) -> TextFileReader: ...
317
318
319@overload
320def read_csv(
321 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
322 *,
323 iterator: bool = ...,
324 chunksize: int,
325 **kwds: Unpack[_read_shared[HashableT]],
326) -> TextFileReader: ...
327
328
329@overload
330def read_csv(
331 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
332 *,
333 iterator: Literal[False] = ...,
334 chunksize: None = ...,
335 **kwds: Unpack[_read_shared[HashableT]],
336) -> DataFrame: ...
337
338
339@overload
340def read_csv(
341 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
342 *,
343 iterator: bool = ...,
344 chunksize: int | None = ...,
345 **kwds: Unpack[_read_shared[HashableT]],
346) -> DataFrame | TextFileReader: ...
347
348
349@set_module("pandas")
350def read_csv(
351 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
352 *,
353 sep: str | None | lib.NoDefault = lib.no_default,
354 delimiter: str | None | lib.NoDefault = None,
355 # Column and Index Locations and Names
356 header: int | Sequence[int] | None | Literal["infer"] = "infer",
357 names: Sequence[Hashable] | None | lib.NoDefault = lib.no_default,
358 index_col: IndexLabel | Literal[False] | None = None,
359 usecols: UsecolsArgType = None,
360 # General Parsing Configuration
361 dtype: DtypeArg | None = None,
362 engine: CSVEngine | None = None,
363 converters: Mapping[HashableT, Callable] | None = None,
364 true_values: list | None = None,
365 false_values: list | None = None,
366 skipinitialspace: bool = False,
367 skiprows: list[int] | int | Callable[[Hashable], bool] | None = None,
368 skipfooter: int = 0,
369 nrows: int | None = None,
370 # NA and Missing Data Handling
371 na_values: (
372 Hashable | Iterable[Hashable] | Mapping[Hashable, Iterable[Hashable]] | None
373 ) = None,
374 keep_default_na: bool = True,
375 na_filter: bool = True,
376 skip_blank_lines: bool = True,
377 # Datetime Handling
378 parse_dates: bool | Sequence[Hashable] | None = None,
379 date_format: str | dict[Hashable, str] | None = None,
380 dayfirst: bool = False,
381 cache_dates: bool = True,
382 # Iteration
383 iterator: bool = False,
384 chunksize: int | None = None,
385 # Quoting, Compression, and File Format
386 compression: CompressionOptions = "infer",
387 thousands: str | None = None,
388 decimal: str = ".",
389 lineterminator: str | None = None,
390 quotechar: str = '"',
391 quoting: int = csv.QUOTE_MINIMAL,
392 doublequote: bool = True,
393 escapechar: str | None = None,
394 comment: str | None = None,
395 encoding: str | None = None,
396 encoding_errors: str | None = "strict",
397 dialect: str | csv.Dialect | None = None,
398 # Error Handling
399 on_bad_lines: str = "error",
400 # Internal
401 low_memory: bool = _c_parser_defaults["low_memory"],
402 memory_map: bool = False,
403 float_precision: Literal["high", "legacy", "round_trip"] | None = None,
404 storage_options: StorageOptions | None = None,
405 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default,
406) -> DataFrame | TextFileReader:
407 """
408 Read a comma-separated values (csv) file into DataFrame.
409
410 Also supports optionally iterating or breaking of the file
411 into chunks.
412
413 Additional help can be found in the online docs for
414 `IO Tools <https://pandas.pydata.org/pandas-docs/stable/user_guide/io.html>`_.
415
416 Parameters
417 ----------
418 filepath_or_buffer : str, path object or file-like object
419 Any valid string path is acceptable. The string could be a URL. Valid
420 URL schemes include http, ftp, s3, gs, and file. For file URLs, a host is
421 expected. A local file could be: file://localhost/path/to/table.csv.
422
423 If you want to pass in a path object, pandas accepts any ``os.PathLike``.
424
425 By file-like object, we refer to objects with a ``read()`` method, such as
426 a file handle (e.g. via builtin ``open`` function) or ``StringIO``.
427 sep : str, default ','
428 Character or regex pattern to treat as the delimiter. ``sep=None`` detects
429 the separator from the first valid row of the file with Python's builtin
430 sniffer tool, ``csv.Sniffer``; it is supported only by the Python parsing
431 engine, which will be used automatically.
432 In addition, separators longer than 1 character and different from
433 ``'\\s+'`` will be interpreted as regular expressions and will also force
434 the use of the Python parsing engine. Note that regex delimiters are prone
435 to ignoring quoted data. Regex example: ``'\\r\\t'``.
436 delimiter : str, optional
437 Alias for ``sep``.
438 header : int, Sequence of int, 'infer' or None, default 'infer'
439 Row number(s) containing column labels and marking the start of the
440 data (zero-indexed). Default behavior is to infer the column names:
441 if no ``names``
442 are passed the behavior is identical to ``header=0`` and column
443 names are inferred from the first line of the file, if column
444 names are passed explicitly to ``names`` then the behavior is identical to
445 ``header=None``. Explicitly pass ``header=0`` to be able to
446 replace existing names. The header can be a list of integers that
447 specify row locations for a :class:`~pandas.MultiIndex` on the columns
448 e.g. ``[0, 1, 3]``. Intervening rows that are not specified will be
449 skipped (e.g. 2 in this example is skipped). Note that this
450 parameter ignores commented lines and empty lines if
451 ``skip_blank_lines=True``, so ``header=0`` denotes the first line of
452 data rather than the first line of the file.
453
454 When inferred from the file contents, headers are kept distinct from
455 each other by renaming duplicate names with a numeric suffix of the form
456 ``".{count}"`` starting from 1, e.g. ``"foo"`` and ``"foo.1"``.
457 Empty headers are named ``"Unnamed: {i}"`` or ``
458 "Unnamed: {i}_level_{level}"``
459 in the case of MultiIndex columns.
460 names : Sequence of Hashable, optional
461 Sequence of column labels to apply. If the file contains a header row,
462 then you should explicitly pass ``header=0`` to override the column names.
463 Duplicates in this list are not allowed.
464 index_col : Hashable, Sequence of Hashable or False, optional
465 Column(s) to use as row label(s), denoted either by column labels or column
466 indices. If a sequence of labels or indices is given,
467 :class:`~pandas.MultiIndex`
468 will be formed for the row labels.
469
470 Note: ``index_col=False`` can be used to force pandas to *not* use the first
471 column as the index, e.g., when you have a malformed file with delimiters at
472 the end of each line.
473 usecols : Sequence of Hashable or Callable, optional
474 Subset of columns to select, denoted either
475 by column labels or column indices.
476 If list-like, all elements must either
477 be positional (i.e. integer indices into the document columns) or strings
478 that correspond to column names provided either by the user in ``names`` or
479 inferred from the document header row(s).
480 If ``names`` are given, the document
481 header row(s) are not taken into account. For example, a valid list-like
482 ``usecols`` parameter would be ``[0, 1, 2]`` or ``['foo', 'bar', 'baz']``.
483 Element order is ignored, so ``usecols=[0, 1]`` is the same as ``[1, 0]``.
484 To instantiate a :class:`~pandas.DataFrame` from ``data`` with element order
485 preserved use ``pd.read_csv(data, usecols=['foo', 'bar'])[['foo', 'bar']]``
486 for columns in ``['foo', 'bar']`` order or
487 ``pd.read_csv(data, usecols=['foo', 'bar'])[['bar', 'foo']]``
488 for ``['bar', 'foo']`` order.
489
490 If callable, the callable function will be evaluated against the column
491 names, returning names where the callable function evaluates to ``True``. An
492 example of a valid callable argument would be ``lambda x: x.upper() in
493 ['AAA', 'BBB', 'DDD']``. Using this parameter results in much faster
494 parsing time and lower memory usage.
495 dtype : dtype or dict of {Hashable : dtype}, optional
496 Data type(s) to apply to either the whole dataset or individual columns.
497 E.g., ``{'a': np.float64, 'b': np.int32, 'c': 'Int64'}``
498 Use ``str`` or ``object`` together with suitable ``na_values`` settings
499 to preserve and not interpret ``dtype``.
500 If ``converters`` are specified, they will be applied INSTEAD
501 of ``dtype`` conversion. Specify a ``defaultdict`` as input where
502 the default determines the ``dtype``
503 of the columns which are not explicitly
504 listed.
505 engine : {'c', 'python', 'pyarrow'}, optional
506 Parser engine to use. The C and pyarrow engines are faster,
507 while the python engine
508 is currently more feature-complete. Multithreading
509 is currently only supported by
510 the pyarrow engine. Some features of the "pyarrow" engine
511 are unsupported or may not work correctly.
512 converters : dict of {Hashable : Callable}, optional
513 Functions for converting values in specified columns. Keys can either
514 be column labels or column indices.
515 true_values : list, optional
516 Values to consider as ``True`` in addition
517 to case-insensitive variants of 'True'.
518 false_values : list, optional
519 Values to consider as ``False`` in addition to case-insensitive
520 variants of 'False'.
521 skipinitialspace : bool, default False
522 Skip spaces after delimiter.
523 skiprows : int, list of int or Callable, optional
524 Line numbers to skip (0-indexed) or number of lines to skip (``int``)
525 at the start of the file.
526
527 If callable, the callable function will be evaluated against the row
528 indices, returning ``True`` if the row should be skipped and ``False``
529 otherwise.
530 An example of a valid callable argument would be ``lambda x: x in [0, 2]``.
531 skipfooter : int, default 0
532 Number of lines at bottom of file to skip (Unsupported with ``engine='c'``).
533 nrows : int, optional
534 Number of rows of file to read. Useful for reading pieces of large files.
535 Refers to the number of data rows in the returned DataFrame, excluding:
536
537 * The header row containing column names.
538 * Rows before the header row, if ``header=1`` or larger.
539
540 Example usage:
541
542 * To read the first 999,999 (non-header) rows:
543 ``read_csv(..., nrows=999999)``
544
545 * To read rows 1,000,000 through 1,999,999:
546 ``read_csv(..., skiprows=1000000, nrows=999999)``
547
548 na_values : Hashable, Iterable of Hashable or dict of {Hashable : Iterable},
549 optional
550 Additional strings to recognize as ``NA``/``NaN``. If ``dict``
551 passed, specific
552 per-column ``NA`` values. By default the following values
553 are interpreted as
554 ``NaN``: empty string, "NaN", "N/A", "NULL", and other common
555 representations of missing data.
556 keep_default_na : bool, default True
557 Whether or not to include the default ``NaN`` values when parsing the data.
558 Depending on whether ``na_values`` is passed in, the behavior is as follows:
559
560 * If ``keep_default_na`` is ``True``, and ``na_values``
561 are specified, ``na_values``
562 is appended to the default ``NaN`` values used for parsing.
563 * If ``keep_default_na`` is ``True``, and ``na_values`` are not specified, only
564 the default ``NaN`` values are used for parsing.
565 * If ``keep_default_na`` is ``False``, and ``na_values`` are specified, only
566 the ``NaN`` values specified ``na_values`` are used for parsing.
567 * If ``keep_default_na`` is ``False``, and ``na_values`` are not specified, no
568 strings will be parsed as ``NaN``.
569
570 Note that if ``na_filter`` is passed in as ``False``,
571 the ``keep_default_na`` and
572 ``na_values`` parameters will be ignored.
573 na_filter : bool, default True
574 Detect missing value markers (empty strings and the value of ``na_values``). In
575 data without any ``NA`` values, passing ``na_filter=False`` can improve the
576 performance of reading a large file.
577 skip_blank_lines : bool, default True
578 If ``True``, skip over blank lines rather than interpreting as ``NaN`` values.
579 parse_dates : bool, None, list of Hashable, default None
580 The behavior is as follows:
581
582 * ``bool``. If ``True`` -> try parsing the index.
583 * ``None``. Behaves like ``True`` if ``date_format`` is specified.
584 * ``list`` of ``int`` or names.
585 e.g. If ``[1, 2, 3]`` -> try parsing columns 1, 2, 3
586 each as a separate date column.
587
588 If a column or index cannot be represented as an array of ``datetime``,
589 say because of an unparsable value or a mixture of timezones, the column
590 or index will be returned unaltered as an ``object`` data type. For
591 non-standard ``datetime`` parsing, use :func:`~pandas.to_datetime` after
592 :func:`~pandas.read_csv`.
593
594 Note: A fast-path exists for iso8601-formatted dates.
595 date_format : str or dict of column -> format, optional
596 Format to use for parsing dates and/or times when
597 used in conjunction with ``parse_dates``.
598 The strftime to parse time, e.g. :const:`"%d/%m/%Y"`. See
599 `strftime documentation
600 <https://docs.python.org/3/library/datetime.html
601 #strftime-and-strptime-behavior>`_ for more information on choices, though
602 note that :const:`"%f"`` will parse all the way up to nanoseconds.
603 You can also pass:
604
605 - "ISO8601", to parse any `ISO8601 <https://en.wikipedia.org/wiki/ISO_8601>`_
606 time string (not necessarily in exactly the same format);
607 - "mixed", to infer the format for each element individually. This is risky,
608 and you should probably use it along with `dayfirst`.
609
610 .. versionadded:: 2.0.0
611 dayfirst : bool, default False
612 DD/MM format dates, international and European format.
613 cache_dates : bool, default True
614 If ``True``, use a cache of unique, converted dates to apply the ``datetime``
615 conversion. May produce significant speed-up when parsing duplicate
616 date strings, especially ones with timezone offsets.
617
618 iterator : bool, default False
619 Return ``TextFileReader`` object for iteration or getting chunks with
620 ``get_chunk()``.
621 chunksize : int, optional
622 Number of lines to read from the file per chunk. Passing a value will cause the
623 function to return a ``TextFileReader`` object for iteration.
624 See the `IO Tools docs
625 <https://pandas.pydata.org/pandas-docs/stable/io.html#io-chunking>`_
626 for more information on ``iterator`` and ``chunksize``.
627
628 compression : str or dict, default 'infer'
629 For on-the-fly decompression of on-disk data.
630 If 'infer' and 'filepath_or_buffer' is
631 path-like, then detect compression from the following extensions: '.gz',
632 '.bz2', '.zip', '.xz', '.zst', '.tar', '.tar.gz', '.tar.xz' or '.tar.bz2'
633 (otherwise no compression).
634 If using 'zip' or 'tar', the ZIP file must contain only
635 one data file to be read in.
636 Set to ``None`` for no decompression.
637 Can also be a dict with key ``'method'`` set
638 to one of {``'zip'``, ``'gzip'``, ``'bz2'``,
639 ``'zstd'``, ``'xz'``, ``'tar'``} and
640 other key-value pairs are forwarded to
641 ``zipfile.ZipFile``, ``gzip.GzipFile``,
642 ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``, ``lzma.LZMAFile`` or
643 ``tarfile.TarFile``, respectively.
644 As an example, the following could be passed for
645 Zstandard decompression using a
646 custom compression dictionary:
647 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``.
648
649 thousands : str (length 1), optional
650 Character acting as the thousands separator in numerical values.
651 decimal : str (length 1), default '.'
652 Character to recognize as decimal point (e.g., use ',' for European data).
653 lineterminator : str (length 1), optional
654 Character used to denote a line break. Only valid with C parser.
655 quotechar : str (length 1), optional
656 Character used to denote the start and end of a quoted item. Quoted
657 items can include the ``delimiter`` and it will be ignored.
658 quoting : {0 or csv.QUOTE_MINIMAL, 1 or csv.QUOTE_ALL,
659 2 or csv.QUOTE_NONNUMERIC, 3 or csv.QUOTE_NONE}, default csv.QUOTE_MINIMAL
660 Control field quoting behavior per ``csv.QUOTE_*`` constants. Default is
661 ``csv.QUOTE_MINIMAL`` (i.e., 0) which implies that
662 only fields containing special
663 characters are quoted (e.g., characters defined
664 in ``quotechar``, ``delimiter``,
665 or ``lineterminator``.
666 doublequote : bool, default True
667 When ``quotechar`` is specified and ``quoting`` is not ``QUOTE_NONE``, indicate
668 whether or not to interpret two consecutive ``quotechar`` elements INSIDE a
669 field as a single ``quotechar`` element.
670 escapechar : str (length 1), optional
671 Character used to escape other characters.
672 comment : str (length 1), optional
673 Character indicating that the remainder of line should not be parsed.
674 If found at the beginning
675 of a line, the line will be ignored altogether. This parameter must be a
676 single character. Like empty lines (as long as ``skip_blank_lines=True``),
677 fully commented lines are ignored by the parameter ``header`` but not by
678 ``skiprows``. For example, if ``comment='#'``, parsing
679 ``#empty\\na,b,c\\n1,2,3`` with ``header=0`` will result in ``'a,b,c'`` being
680 treated as the header.
681 encoding : str, optional, default 'utf-8'
682 Encoding to use for UTF when reading/writing (ex. ``'utf-8'``). `List of Python
683 standard encodings
684 <https://docs.python.org/3/library/codecs.html#standard-encodings>`_ .
685
686 encoding_errors : str, optional, default 'strict'
687 How encoding errors are treated. `List of possible values
688 <https://docs.python.org/3/library/codecs.html#error-handlers>`_ .
689
690 dialect : str or csv.Dialect, optional
691 If provided, this parameter will override values (default or not) for the
692 following parameters: ``delimiter``, ``doublequote``, ``escapechar``,
693 ``skipinitialspace``, ``quotechar``, and ``quoting``. If it is necessary to
694 override values, a ``ParserWarning`` will be issued. See ``csv.Dialect``
695 documentation for more details.
696 on_bad_lines : {'error', 'warn', 'skip'} or Callable, default 'error'
697 Specifies what to do upon encountering a bad line (a line with too many fields).
698 Allowed values are:
699
700 - ``'error'``, raise an Exception when a bad line is encountered.
701 - ``'warn'``, raise a warning when a bad line is
702 encountered and skip that line.
703 - ``'skip'``, skip bad lines without raising or warning when
704 they are encountered.
705 - Callable, function that will process a single bad line.
706 - With ``engine='python'``, function with signature
707 ``(bad_line: list[str]) -> list[str] | None``.
708 ``bad_line`` is a list of strings split by the ``sep``.
709 If the function returns ``None``, the bad line will be ignored.
710 If the function returns a new ``list`` of strings with
711 more elements than
712 expected, a ``ParserWarning`` will be emitted while
713 dropping extra elements.
714 - With ``engine='pyarrow'``, function with signature
715 as described in pyarrow documentation: `invalid_row_handler
716 <https://arrow.apache.org/docs/python
717 /generated/pyarrow.csv.ParseOptions.html
718 #pyarrow.csv.ParseOptions.invalid_row_handler>`_.
719
720 .. versionchanged:: 2.2.0
721
722 Callable for ``engine='pyarrow'``
723
724 low_memory : bool, default True
725 Internally process the file in chunks, resulting in lower memory use
726 while parsing, but possibly mixed type inference. To ensure no mixed
727 types either set ``False``, or specify the type with the ``dtype`` parameter.
728 Note that the entire file is read into a single :class:`~pandas.DataFrame`
729 regardless, use the ``chunksize`` or ``iterator``
730 parameter to return the data in
731 chunks. (Only valid with C parser).
732 memory_map : bool, default False
733 If a filepath is provided for ``filepath_or_buffer``, map the file object
734 directly onto memory and access the data directly from there. Using this
735 option can improve performance because there is no longer any I/O overhead.
736 float_precision : {'high', 'legacy', 'round_trip'}, optional
737 Specifies which converter the C engine should use for floating-point
738 values. The options are ``None`` or ``'high'`` for the ordinary converter,
739 ``'legacy'`` for the original lower precision pandas converter, and
740 ``'round_trip'`` for the round-trip converter.
741
742 storage_options : dict, optional
743 Extra options that make sense for a particular storage connection, e.g.
744 host, port, username, password, etc. For HTTP(S) URLs the key-value pairs
745 are forwarded to ``urllib.request.Request`` as header options. For other
746 URLs (e.g. starting with "s3://", and "gcs://") the key-value pairs are
747 forwarded to ``fsspec.open``. Please see ``fsspec`` and ``urllib`` for more
748 details, and for more examples on storage options refer `here
749 <https://pandas.pydata.org/docs/user_guide/io.html?
750 highlight=storage_options#reading-writing-remote-files>`_.
751
752 dtype_backend : {'numpy_nullable', 'pyarrow'}
753 Back-end data type applied to the resultant :class:`DataFrame`
754 (still experimental). If not specified, the default behavior
755 is to not use nullable data types. If specified, the behavior
756 is as follows:
757
758 * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame`
759 * ``"pyarrow"``: returns
760 pyarrow-backed nullable :class:`ArrowDtype` :class:`DataFrame`
761
762 .. versionadded:: 2.0
763
764 Returns
765 -------
766 DataFrame or TextFileReader
767 A comma-separated values (csv) file is returned as two-dimensional
768 data structure with labeled axes.
769
770 See Also
771 --------
772 DataFrame.to_csv : Write DataFrame to a comma-separated values (csv) file.
773 read_table : Read general delimited file into DataFrame.
774 read_fwf : Read a table of fixed-width formatted lines into DataFrame.
775
776 Examples
777 --------
778 >>> pd.read_csv("data.csv") # doctest: +SKIP
779 Name Value
780 0 foo 1
781 1 bar 2
782 2 #baz 3
783
784 Index and header can be specified via the `index_col` and `header` arguments.
785
786 >>> pd.read_csv("data.csv", header=None) # doctest: +SKIP
787 0 1
788 0 Name Value
789 1 foo 1
790 2 bar 2
791 3 #baz 3
792
793 >>> pd.read_csv("data.csv", index_col="Value") # doctest: +SKIP
794 Name
795 Value
796 1 foo
797 2 bar
798 3 #baz
799
800 Column types are inferred but can be explicitly specified using the dtype argument.
801
802 >>> pd.read_csv("data.csv", dtype={"Value": float}) # doctest: +SKIP
803 Name Value
804 0 foo 1.0
805 1 bar 2.0
806 2 #baz 3.0
807
808 True, False, and NA values, and thousands separators have defaults,
809 but can be explicitly specified, too. Supply the values you would like
810 as strings or lists of strings!
811
812 >>> pd.read_csv("data.csv", na_values=["foo", "bar"]) # doctest: +SKIP
813 Name Value
814 0 NaN 1
815 1 NaN 2
816 2 #baz 3
817
818 Comment lines in the input file can be skipped using the `comment` argument.
819
820 >>> pd.read_csv("data.csv", comment="#") # doctest: +SKIP
821 Name Value
822 0 foo 1
823 1 bar 2
824
825 By default, columns with dates will be read as ``object`` rather than ``datetime``.
826
827 >>> df = pd.read_csv("tmp.csv") # doctest: +SKIP
828
829 >>> df # doctest: +SKIP
830 col 1 col 2 col 3
831 0 10 10/04/2018 Sun 15 Jan 2023
832 1 20 15/04/2018 Fri 12 May 2023
833
834 >>> df.dtypes # doctest: +SKIP
835 col 1 int64
836 col 2 object
837 col 3 object
838 dtype: object
839
840 Specific columns can be parsed as dates by using the `parse_dates` and
841 `date_format` arguments.
842
843 >>> df = pd.read_csv(
844 ... "tmp.csv",
845 ... parse_dates=[1, 2],
846 ... date_format={"col 2": "%d/%m/%Y", "col 3": "%a %d %b %Y"},
847 ... ) # doctest: +SKIP
848
849 >>> df.dtypes # doctest: +SKIP
850 col 1 int64
851 col 2 datetime64[ns]
852 col 3 datetime64[ns]
853 dtype: object
854 """
855 # locals() should never be modified
856 kwds = locals().copy()
857 del kwds["filepath_or_buffer"]
858 del kwds["sep"]
859
860 kwds_defaults = _refine_defaults_read(
861 dialect,
862 delimiter,
863 engine,
864 sep,
865 on_bad_lines,
866 names,
867 defaults={"delimiter": ","},
868 dtype_backend=dtype_backend,
869 )
870 kwds.update(kwds_defaults)
871
872 return _read(filepath_or_buffer, kwds)
873
874
875@overload
876def read_table(
877 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
878 *,
879 iterator: Literal[True],
880 chunksize: int | None = ...,
881 **kwds: Unpack[_read_shared[HashableT]],
882) -> TextFileReader: ...
883
884
885@overload
886def read_table(
887 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
888 *,
889 iterator: bool = ...,
890 chunksize: int,
891 **kwds: Unpack[_read_shared[HashableT]],
892) -> TextFileReader: ...
893
894
895@overload
896def read_table(
897 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
898 *,
899 iterator: Literal[False] = ...,
900 chunksize: None = ...,
901 **kwds: Unpack[_read_shared[HashableT]],
902) -> DataFrame: ...
903
904
905@overload
906def read_table(
907 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
908 *,
909 iterator: bool = ...,
910 chunksize: int | None = ...,
911 **kwds: Unpack[_read_shared[HashableT]],
912) -> DataFrame | TextFileReader: ...
913
914
915@set_module("pandas")
916def read_table(
917 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
918 *,
919 sep: str | None | lib.NoDefault = lib.no_default,
920 delimiter: str | None | lib.NoDefault = None,
921 # Column and Index Locations and Names
922 header: int | Sequence[int] | None | Literal["infer"] = "infer",
923 names: Sequence[Hashable] | None | lib.NoDefault = lib.no_default,
924 index_col: IndexLabel | Literal[False] | None = None,
925 usecols: UsecolsArgType = None,
926 # General Parsing Configuration
927 dtype: DtypeArg | None = None,
928 engine: CSVEngine | None = None,
929 converters: Mapping[HashableT, Callable] | None = None,
930 true_values: list | None = None,
931 false_values: list | None = None,
932 skipinitialspace: bool = False,
933 skiprows: list[int] | int | Callable[[Hashable], bool] | None = None,
934 skipfooter: int = 0,
935 nrows: int | None = None,
936 # NA and Missing Data Handling
937 na_values: (
938 Hashable | Iterable[Hashable] | Mapping[Hashable, Iterable[Hashable]] | None
939 ) = None,
940 keep_default_na: bool = True,
941 na_filter: bool = True,
942 skip_blank_lines: bool = True,
943 # Datetime Handling
944 parse_dates: bool | Sequence[Hashable] | None = None,
945 date_format: str | dict[Hashable, str] | None = None,
946 dayfirst: bool = False,
947 cache_dates: bool = True,
948 # Iteration
949 iterator: bool = False,
950 chunksize: int | None = None,
951 # Quoting, Compression, and File Format
952 compression: CompressionOptions = "infer",
953 thousands: str | None = None,
954 decimal: str = ".",
955 lineterminator: str | None = None,
956 quotechar: str = '"',
957 quoting: int = csv.QUOTE_MINIMAL,
958 doublequote: bool = True,
959 escapechar: str | None = None,
960 comment: str | None = None,
961 encoding: str | None = None,
962 encoding_errors: str | None = "strict",
963 dialect: str | csv.Dialect | None = None,
964 # Error Handling
965 on_bad_lines: str = "error",
966 # Internal
967 low_memory: bool = _c_parser_defaults["low_memory"],
968 memory_map: bool = False,
969 float_precision: Literal["high", "legacy", "round_trip"] | None = None,
970 storage_options: StorageOptions | None = None,
971 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default,
972) -> DataFrame | TextFileReader:
973 """
974 Read general delimited file into DataFrame.
975
976 Also supports optionally iterating or breaking of the file
977 into chunks.
978
979 Additional help can be found in the online docs for
980 `IO Tools <https://pandas.pydata.org/pandas-docs/stable/user_guide/io.html>`_.
981
982 Parameters
983 ----------
984 filepath_or_buffer : str, path object or file-like object
985 Any valid string path is acceptable. The string could be a URL. Valid
986 URL schemes include http, ftp, s3, gs, and file. For file URLs, a host is
987 expected. A local file could be: file://localhost/path/to/table.csv.
988
989 If you want to pass in a path object, pandas accepts any ``os.PathLike``.
990
991 By file-like object, we refer to objects with a ``read()`` method, such as
992 a file handle (e.g. via builtin ``open`` function) or ``StringIO``.
993 sep : str, default '\\t' (tab-stop)
994 Character or regex pattern to treat as the delimiter. ``sep=None`` detects
995 the separator from the first valid row of the file with Python's builtin
996 sniffer tool, ``csv.Sniffer``; it is supported only by the Python parsing
997 engine, which will be used automatically.
998 In addition, separators longer than 1 character and different from
999 ``'\\s+'`` will be interpreted as regular expressions and will also force
1000 the use of the Python parsing engine. Note that regex delimiters are prone
1001 to ignoring quoted data. Regex example: ``'\\r\\t'``.
1002 delimiter : str, optional
1003 Alias for ``sep``.
1004 header : int, Sequence of int, 'infer' or None, default 'infer'
1005 Row number(s) containing column labels and marking the start of the
1006 data (zero-indexed). Default behavior
1007 is to infer the column names: if no ``names``
1008 are passed the behavior is identical to ``header=0`` and column
1009 names are inferred from the first line of the file, if column
1010 names are passed explicitly to ``names`` then the behavior is identical to
1011 ``header=None``. Explicitly pass ``header=0`` to be able to
1012 replace existing names. The header can be a list of integers that
1013 specify row locations for a :class:`~pandas.MultiIndex` on the columns
1014 e.g. ``[0, 1, 3]``. Intervening rows that are not specified will be
1015 skipped (e.g. 2 in this example is skipped). Note that this
1016 parameter ignores commented lines and empty lines if
1017 ``skip_blank_lines=True``, so ``header=0`` denotes the first line of
1018 data rather than the first line of the file.
1019
1020 When inferred from the file contents, headers are kept distinct from
1021 each other by renaming duplicate names with a numeric suffix of the form
1022 ``".{count}"`` starting from 1, e.g. ``"foo"`` and ``"foo.1"``.
1023 Empty headers are named
1024 ``"Unnamed: {i}"`` or ``"Unnamed: {i}_level_{level}"``
1025 in the case of MultiIndex columns.
1026 names : Sequence of Hashable, optional
1027 Sequence of column labels to apply. If the file contains a header row,
1028 then you should explicitly pass ``header=0`` to override the column names.
1029 Duplicates in this list are not allowed.
1030 index_col : Hashable, Sequence of Hashable or False, optional
1031 Column(s) to use as row label(s), denoted either by column labels or column
1032 indices. If a sequence of labels or indices is given,
1033 :class:`~pandas.MultiIndex`
1034 will be formed for the row labels.
1035
1036 Note: ``index_col=False`` can be used to force pandas to *not* use the first
1037 column as the index, e.g., when you have a malformed file with delimiters at
1038 the end of each line.
1039 usecols : Sequence of Hashable or Callable, optional
1040 Subset of columns to select, denoted either by column labels or column indices.
1041 If list-like, all elements must either
1042 be positional (i.e. integer indices into the document columns) or strings
1043 that correspond to column names provided either by the user in ``names`` or
1044 inferred from the document header row(s). If ``names`` are given, the document
1045 header row(s) are not taken into account. For example, a valid list-like
1046 ``usecols`` parameter would be ``[0, 1, 2]`` or ``['foo', 'bar', 'baz']``.
1047 Element order is ignored, so ``usecols=[0, 1]`` is the same as ``[1, 0]``.
1048 To instantiate a :class:`~pandas.DataFrame` from ``data`` with element order
1049 preserved use ``pd.read_csv(data, usecols=['foo', 'bar'])[['foo', 'bar']]``
1050 for columns in ``['foo', 'bar']`` order or
1051 ``pd.read_csv(data, usecols=['foo', 'bar'])[['bar', 'foo']]``
1052 for ``['bar', 'foo']`` order.
1053
1054 If callable, the callable function will be evaluated against the column
1055 names, returning names where the callable function evaluates to ``True``. An
1056 example of a valid callable argument would be ``lambda x: x.upper() in
1057 ['AAA', 'BBB', 'DDD']``. Using this parameter results in much faster
1058 parsing time and lower memory usage.
1059 dtype : dtype or dict of {Hashable : dtype}, optional
1060 Data type(s) to apply to either the whole dataset or individual columns.
1061 E.g., ``{'a': np.float64, 'b': np.int32, 'c': 'Int64'}``
1062 Use ``str`` or ``object`` together with suitable ``na_values`` settings
1063 to preserve and not interpret ``dtype``.
1064 If ``converters`` are specified, they will be applied INSTEAD
1065 of ``dtype`` conversion. Specify a ``defaultdict`` as input where
1066 the default determines the ``dtype`` of the columns which
1067 are not explicitly listed.
1068 engine : {'c', 'python', 'pyarrow'}, optional
1069 Parser engine to use. The C and pyarrow engines are faster,
1070 while the python engine
1071 is currently more feature-complete. Multithreading is
1072 currently only supported by
1073 the pyarrow engine. The 'pyarrow' engine is an *experimental* engine,
1074 and some features are unsupported, or may not work correctly, with this engine.
1075 converters : dict of {Hashable : Callable}, optional
1076 Functions for converting values in specified columns. Keys can either
1077 be column labels or column indices.
1078 true_values : list, optional
1079 Values to consider as ``True`` in addition to
1080 case-insensitive variants of 'True'.
1081 false_values : list, optional
1082 Values to consider as ``False`` in addition
1083 to case-insensitive variants of 'False'.
1084 skipinitialspace : bool, default False
1085 Skip spaces after delimiter.
1086 skiprows : int, list of int or Callable, optional
1087 Line numbers to skip (0-indexed) or number of lines to skip (``int``)
1088 at the start of the file.
1089
1090 If callable, the callable function will be evaluated against the row
1091 indices, returning ``True`` if the row
1092 should be skipped and ``False`` otherwise.
1093 An example of a valid callable argument would be ``lambda x: x in [0, 2]``.
1094 skipfooter : int, default 0
1095 Number of lines at bottom of file to skip (Unsupported with ``engine='c'``).
1096 nrows : int, optional
1097 Number of rows of file to read. Useful for reading pieces of large files.
1098 Refers to the number of data rows in the returned DataFrame, excluding:
1099
1100 * The header row containing column names.
1101 * Rows before the header row, if ``header=1`` or larger.
1102
1103 Example usage:
1104
1105 * To read the first 999,999 (non-header) rows:
1106 ``read_csv(..., nrows=999999)``
1107
1108 * To read rows 1,000,000 through 1,999,999:
1109 ``read_csv(..., skiprows=1000000, nrows=999999)``
1110
1111 na_values : Hashable, Iterable of Hashable or dict of {Hashable : Iterable},
1112 optional
1113 Additional strings to recognize as ``NA``/``NaN``.
1114 If ``dict`` passed, specific
1115 per-column ``NA`` values. By default the following values are interpreted as
1116 ``NaN``: empty string, "NaN", "N/A", "NULL", and other
1117 common representations of missing data.
1118 keep_default_na : bool, default True
1119 Whether or not to include the default ``NaN`` values when parsing the data.
1120 Depending on whether ``na_values`` is passed in, the behavior is as follows:
1121
1122 * If ``keep_default_na`` is ``True``,
1123 and ``na_values`` are specified, ``na_values``
1124 is appended to the default ``NaN`` values used for parsing.
1125 * If ``keep_default_na`` is ``True``, and ``na_values`` are not specified, only
1126 the default ``NaN`` values are used for parsing.
1127 * If ``keep_default_na`` is ``False``, and ``na_values`` are specified, only
1128 the ``NaN`` values specified ``na_values`` are used for parsing.
1129 * If ``keep_default_na`` is ``False``, and ``na_values`` are not specified, no
1130 strings will be parsed as ``NaN``.
1131
1132 Note that if ``na_filter`` is passed in as
1133 ``False``, the ``keep_default_na`` and
1134 ``na_values`` parameters will be ignored.
1135 na_filter : bool, default True
1136 Detect missing value markers (empty strings and the value of ``na_values``). In
1137 data without any ``NA`` values, passing ``na_filter=False`` can improve the
1138 performance of reading a large file.
1139 skip_blank_lines : bool, default True
1140 If ``True``, skip over blank lines rather than interpreting as ``NaN`` values.
1141 parse_dates : bool, None, list of Hashable, default None
1142 The behavior is as follows:
1143
1144 * ``bool``. If ``True`` -> try parsing the index.
1145 * ``None``. Behaves like ``True`` if ``date_format`` is specified.
1146 * ``list`` of ``int`` or names.
1147 e.g. If ``[1, 2, 3]`` -> try parsing columns 1, 2, 3
1148 each as a separate date column.
1149
1150 If a column or index cannot be represented as an array of ``datetime``,
1151 say because of an unparsable value or a mixture of timezones, the column
1152 or index will be returned unaltered as an ``object`` data type. For
1153 non-standard ``datetime`` parsing, use :func:`~pandas.to_datetime` after
1154 :func:`~pandas.read_csv`.
1155
1156 Note: A fast-path exists for iso8601-formatted dates.
1157 date_format : str or dict of column -> format, optional
1158 Format to use for parsing dates and/or times when used
1159 in conjunction with ``parse_dates``.
1160 The strftime to parse time, e.g. :const:`"%d/%m/%Y"`. See
1161 `strftime documentation
1162 <https://docs.python.org/3/library/datetime.html
1163 #strftime-and-strptime-behavior>`_ for more information on choices, though
1164 note that :const:`"%f"`` will parse all the way up to nanoseconds.
1165 You can also pass:
1166
1167 - "ISO8601", to parse any `ISO8601 <https://en.wikipedia.org/wiki/ISO_8601>`_
1168 time string (not necessarily in exactly the same format);
1169 - "mixed", to infer the format for each element individually. This is risky,
1170 and you should probably use it along with `dayfirst`.
1171
1172 .. versionadded:: 2.0.0
1173 dayfirst : bool, default False
1174 DD/MM format dates, international and European format.
1175 cache_dates : bool, default True
1176 If ``True``, use a cache of unique, converted dates to apply the ``datetime``
1177 conversion. May produce significant speed-up when parsing duplicate
1178 date strings, especially ones with timezone offsets.
1179
1180 iterator : bool, default False
1181 Return ``TextFileReader`` object for iteration or getting chunks with
1182 ``get_chunk()``.
1183 chunksize : int, optional
1184 Number of lines to read from the file per chunk. Passing a value will cause the
1185 function to return a ``TextFileReader`` object for iteration.
1186 See the `IO Tools docs
1187 <https://pandas.pydata.org/pandas-docs/stable/io.html#io-chunking>`_
1188 for more information on ``iterator`` and ``chunksize``.
1189
1190 compression : str or dict, default 'infer'
1191 For on-the-fly decompression of on-disk data. If 'infer'
1192 and 'filepath_or_buffer' is
1193 path-like, then detect compression from the following extensions: '.gz',
1194 '.bz2', '.zip', '.xz', '.zst', '.tar', '.tar.gz', '.tar.xz' or '.tar.bz2'
1195 (otherwise no compression).
1196 If using 'zip' or 'tar', the ZIP file must contain
1197 only one data file to be read in.
1198 Set to ``None`` for no decompression.
1199 Can also be a dict with key ``'method'`` set
1200 to one of {``'zip'``, ``'gzip'``, ``'bz2'``,
1201 ``'zstd'``, ``'xz'``, ``'tar'``} and
1202 other key-value pairs are forwarded to
1203 ``zipfile.ZipFile``, ``gzip.GzipFile``,
1204 ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``, ``lzma.LZMAFile`` or
1205 ``tarfile.TarFile``, respectively.
1206 As an example, the following could be passed for
1207 Zstandard decompression using a
1208 custom compression dictionary:
1209 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``.
1210
1211 thousands : str (length 1), optional
1212 Character acting as the thousands separator in numerical values.
1213 decimal : str (length 1), default '.'
1214 Character to recognize as decimal point (e.g., use ',' for European data).
1215 lineterminator : str (length 1), optional
1216 Character used to denote a line break. Only valid with C parser.
1217 quotechar : str (length 1), optional
1218 Character used to denote the start and end of a quoted item. Quoted
1219 items can include the ``delimiter`` and it will be ignored.
1220 quoting : {0 or csv.QUOTE_MINIMAL, 1 or csv.QUOTE_ALL, 2 or
1221 csv.QUOTE_NONNUMERIC, 3 or csv.QUOTE_NONE}, default csv.QUOTE_MINIMAL
1222 Control field quoting behavior per ``csv.QUOTE_*`` constants. Default is
1223 ``csv.QUOTE_MINIMAL`` (i.e., 0) which
1224 implies that only fields containing special
1225 characters are quoted (e.g., characters defined
1226 in ``quotechar``, ``delimiter``,
1227 or ``lineterminator``.
1228 doublequote : bool, default True
1229 When ``quotechar`` is specified and ``quoting`` is not ``QUOTE_NONE``, indicate
1230 whether or not to interpret two consecutive ``quotechar`` elements INSIDE a
1231 field as a single ``quotechar`` element.
1232 escapechar : str (length 1), optional
1233 Character used to escape other characters.
1234 comment : str (length 1), optional
1235 Character indicating that the remainder of line should not be parsed.
1236 If found at the beginning
1237 of a line, the line will be ignored altogether. This parameter must be a
1238 single character. Like empty lines (as long as ``skip_blank_lines=True``),
1239 fully commented lines are ignored by the parameter ``header`` but not by
1240 ``skiprows``. For example, if ``comment='#'``, parsing
1241 ``#empty\\na,b,c\\n1,2,3`` with ``header=0`` will result in ``'a,b,c'`` being
1242 treated as the header.
1243 encoding : str, optional, default 'utf-8'
1244 Encoding to use for UTF when reading/writing (ex. ``'utf-8'``). `List of Python
1245 standard encodings
1246 <https://docs.python.org/3/library/codecs.html#standard-encodings>`_ .
1247
1248 encoding_errors : str, optional, default 'strict'
1249 How encoding errors are treated. `List of possible values
1250 <https://docs.python.org/3/library/codecs.html#error-handlers>`_ .
1251
1252 dialect : str or csv.Dialect, optional
1253 If provided, this parameter will override values (default or not) for the
1254 following parameters: ``delimiter``, ``doublequote``, ``escapechar``,
1255 ``skipinitialspace``, ``quotechar``, and ``quoting``. If it is necessary to
1256 override values, a ``ParserWarning`` will be issued. See ``csv.Dialect``
1257 documentation for more details.
1258 on_bad_lines : {'error', 'warn', 'skip'} or Callable, default 'error'
1259 Specifies what to do upon encountering a bad
1260 line (a line with too many fields).
1261 Allowed values are:
1262
1263 - ``'error'``, raise an Exception when a bad line is encountered.
1264 - ``'warn'``, raise a warning when a bad line is encountered and
1265 skip that line.
1266 - ``'skip'``, skip bad lines without raising or warning when they
1267 are encountered.
1268 - Callable, function that will process a single bad line.
1269 - With ``engine='python'``, function with signature
1270 ``(bad_line: list[str]) -> list[str] | None``.
1271 ``bad_line`` is a list of strings split by the ``sep``.
1272 If the function returns ``None``, the bad line will be ignored.
1273 If the function returns a new ``list`` of strings with more elements than
1274 expected, a ``ParserWarning`` will be emitted while
1275 dropping extra elements.
1276 - With ``engine='pyarrow'``, function with signature
1277 as described in pyarrow documentation: `invalid_row_handler
1278 <https://arrow.apache.org/docs/
1279 python/generated/pyarrow.csv.ParseOptions.html
1280 #pyarrow.csv.ParseOptions.invalid_row_handler>`_.
1281
1282 .. versionadded:: 2.2.0
1283
1284 Callable for ``engine='pyarrow'``
1285
1286 low_memory : bool, default True
1287 Internally process the file in chunks, resulting in lower memory use
1288 while parsing, but possibly mixed type inference. To ensure no mixed
1289 types either set ``False``, or specify the type with the ``dtype`` parameter.
1290 Note that the entire file is read into a single :class:`~pandas.DataFrame`
1291 regardless, use the ``chunksize`` or ``iterator`` parameter
1292 to return the data in
1293 chunks. (Only valid with C parser).
1294 memory_map : bool, default False
1295 If a filepath is provided for ``filepath_or_buffer``, map the file object
1296 directly onto memory and access the data directly from there. Using this
1297 option can improve performance because there is no longer any I/O overhead.
1298 float_precision : {'high', 'legacy', 'round_trip'}, optional
1299 Specifies which converter the C engine should use for floating-point
1300 values. The options are ``None`` or ``'high'`` for the ordinary converter,
1301 ``'legacy'`` for the original lower precision pandas converter, and
1302 ``'round_trip'`` for the round-trip converter.
1303
1304 storage_options : dict, optional
1305 Extra options that make sense for a particular storage connection, e.g.
1306 host, port, username, password, etc. For HTTP(S) URLs the key-value pairs
1307 are forwarded to ``urllib.request.Request`` as header options. For other
1308 URLs (e.g. starting with "s3://", and "gcs://") the key-value pairs are
1309 forwarded to ``fsspec.open``. Please see ``fsspec`` and ``urllib`` for more
1310 details, and for more examples on storage options refer `here
1311 <https://pandas.pydata.org/docs/user_guide/io.html?
1312 highlight=storage_options#reading-writing-remote-files>`_.
1313
1314 dtype_backend : {'numpy_nullable', 'pyarrow'}
1315 Back-end data type applied to the resultant :class:`DataFrame`
1316 (still experimental). If not specified, the default behavior
1317 is to not use nullable data types. If specified, the behavior
1318 is as follows:
1319
1320 * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame`
1321 * ``"pyarrow"``: returns pyarrow-backed nullable
1322 :class:`ArrowDtype` :class:`DataFrame`
1323
1324 .. versionadded:: 2.0
1325
1326 Returns
1327 -------
1328 DataFrame or TextFileReader
1329 A comma-separated values (csv) file is returned as two-dimensional
1330 data structure with labeled axes.
1331
1332 See Also
1333 --------
1334 DataFrame.to_csv : Write DataFrame to a comma-separated values (csv) file.
1335 read_csv : Read a comma-separated values (csv) file into DataFrame.
1336 read_fwf : Read a table of fixed-width formatted lines into DataFrame.
1337
1338 Examples
1339 --------
1340 >>> pd.read_table("data.csv") # doctest: +SKIP
1341 Name Value
1342 0 foo 1
1343 1 bar 2
1344 2 #baz 3
1345
1346 Index and header can be specified via the `index_col` and `header` arguments.
1347
1348 >>> pd.read_table("data.csv", header=None) # doctest: +SKIP
1349 0 1
1350 0 Name Value
1351 1 foo 1
1352 2 bar 2
1353 3 #baz 3
1354
1355 >>> pd.read_table("data.csv", index_col="Value") # doctest: +SKIP
1356 Name
1357 Value
1358 1 foo
1359 2 bar
1360 3 #baz
1361
1362 Column types are inferred but can be explicitly specified using the dtype argument.
1363
1364 >>> pd.read_table("data.csv", dtype={"Value": float}) # doctest: +SKIP
1365 Name Value
1366 0 foo 1.0
1367 1 bar 2.0
1368 2 #baz 3.0
1369
1370 True, False, and NA values, and thousands separators have defaults,
1371 but can be explicitly specified, too. Supply the values you would like
1372 as strings or lists of strings!
1373
1374 >>> pd.read_table("data.csv", na_values=["foo", "bar"]) # doctest: +SKIP
1375 Name Value
1376 0 NaN 1
1377 1 NaN 2
1378 2 #baz 3
1379
1380 Comment lines in the input file can be skipped using the `comment` argument.
1381
1382 >>> pd.read_table("data.csv", comment="#") # doctest: +SKIP
1383 Name Value
1384 0 foo 1
1385 1 bar 2
1386
1387 By default, columns with dates will be read as ``object`` rather than ``datetime``.
1388
1389 >>> df = pd.read_table("tmp.csv") # doctest: +SKIP
1390
1391 >>> df # doctest: +SKIP
1392 col 1 col 2 col 3
1393 0 10 10/04/2018 Sun 15 Jan 2023
1394 1 20 15/04/2018 Fri 12 May 2023
1395
1396 >>> df.dtypes # doctest: +SKIP
1397 col 1 int64
1398 col 2 object
1399 col 3 object
1400 dtype: object
1401
1402 Specific columns can be parsed as dates by using the `parse_dates` and
1403 `date_format` arguments.
1404
1405 >>> df = pd.read_table(
1406 ... "tmp.csv",
1407 ... parse_dates=[1, 2],
1408 ... date_format={"col 2": "%d/%m/%Y", "col 3": "%a %d %b %Y"},
1409 ... ) # doctest: +SKIP
1410
1411 >>> df.dtypes # doctest: +SKIP
1412 col 1 int64
1413 col 2 datetime64[ns]
1414 col 3 datetime64[ns]
1415 dtype: object
1416 """
1417 # locals() should never be modified
1418 kwds = locals().copy()
1419 del kwds["filepath_or_buffer"]
1420 del kwds["sep"]
1421
1422 kwds_defaults = _refine_defaults_read(
1423 dialect,
1424 delimiter,
1425 engine,
1426 sep,
1427 on_bad_lines,
1428 names,
1429 defaults={"delimiter": "\t"},
1430 dtype_backend=dtype_backend,
1431 )
1432 kwds.update(kwds_defaults)
1433
1434 return _read(filepath_or_buffer, kwds)
1435
1436
1437@overload
1438def read_fwf(
1439 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
1440 *,
1441 colspecs: Sequence[tuple[int, int]] | str | None = ...,
1442 widths: Sequence[int] | None = ...,
1443 infer_nrows: int = ...,
1444 iterator: Literal[True],
1445 chunksize: int | None = ...,
1446 **kwds: Unpack[_read_shared[HashableT]],
1447) -> TextFileReader: ...
1448
1449
1450@overload
1451def read_fwf(
1452 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
1453 *,
1454 colspecs: Sequence[tuple[int, int]] | str | None = ...,
1455 widths: Sequence[int] | None = ...,
1456 infer_nrows: int = ...,
1457 iterator: bool = ...,
1458 chunksize: int,
1459 **kwds: Unpack[_read_shared[HashableT]],
1460) -> TextFileReader: ...
1461
1462
1463@overload
1464def read_fwf(
1465 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
1466 *,
1467 colspecs: Sequence[tuple[int, int]] | str | None = ...,
1468 widths: Sequence[int] | None = ...,
1469 infer_nrows: int = ...,
1470 iterator: Literal[False] = ...,
1471 chunksize: None = ...,
1472 **kwds: Unpack[_read_shared[HashableT]],
1473) -> DataFrame: ...
1474
1475
1476@set_module("pandas")
1477def read_fwf(
1478 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str],
1479 *,
1480 colspecs: Sequence[tuple[int, int]] | str | None = "infer",
1481 widths: Sequence[int] | None = None,
1482 infer_nrows: int = 100,
1483 iterator: bool = False,
1484 chunksize: int | None = None,
1485 **kwds: Unpack[_read_shared[HashableT]],
1486) -> DataFrame | TextFileReader:
1487 r"""
1488 Read a table of fixed-width formatted lines into DataFrame.
1489
1490 Also supports optionally iterating or breaking of the file
1491 into chunks.
1492
1493 Additional help can be found in the `online docs for IO Tools
1494 <https://pandas.pydata.org/pandas-docs/stable/user_guide/io.html>`_.
1495
1496 Parameters
1497 ----------
1498 filepath_or_buffer : str, path object, or file-like object
1499 String, path object (implementing ``os.PathLike[str]``), or file-like
1500 object implementing a text ``read()`` function.The string could be a URL.
1501 Valid URL schemes include http, ftp, s3, and file. For file URLs, a host is
1502 expected. A local file could be:
1503 ``file://localhost/path/to/table.csv``.
1504 colspecs : list of tuple (int, int) or 'infer'. optional
1505 A list of tuples giving the extents of the fixed-width
1506 fields of each line as half-open intervals (i.e., [from, to] ).
1507 String value 'infer' can be used to instruct the parser to try
1508 detecting the column specifications from the first 100 rows of
1509 the data which are not being skipped via skiprows (default='infer').
1510 widths : list of int, optional
1511 A list of field widths which can be used instead of 'colspecs' if
1512 the intervals are contiguous.
1513 infer_nrows : int, default 100
1514 The number of rows to consider when letting the parser determine the
1515 `colspecs`.
1516 iterator : bool, default False
1517 Return ``TextFileReader`` object for iteration or getting chunks with
1518 ``get_chunk()``.
1519 chunksize : int, optional
1520 Number of lines to read from the file per chunk.
1521 **kwds : optional
1522 Optional keyword arguments can be passed to ``TextFileReader``.
1523
1524 Returns
1525 -------
1526 DataFrame or TextFileReader
1527 A comma-separated values (csv) file is returned as two-dimensional
1528 data structure with labeled axes.
1529
1530 See Also
1531 --------
1532 DataFrame.to_csv : Write DataFrame to a comma-separated values (csv) file.
1533 read_csv : Read a comma-separated values (csv) file into DataFrame.
1534
1535 Examples
1536 --------
1537 >>> pd.read_fwf("data.csv") # doctest: +SKIP
1538 """
1539 # Check input arguments.
1540 if colspecs is None and widths is None:
1541 raise ValueError("Must specify either colspecs or widths")
1542 if colspecs not in (None, "infer") and widths is not None:
1543 raise ValueError("You must specify only one of 'widths' and 'colspecs'")
1544
1545 # Compute 'colspecs' from 'widths', if specified.
1546 if widths is not None:
1547 colspecs, col = [], 0
1548 for w in widths:
1549 colspecs.append((col, col + w))
1550 col += w
1551
1552 # for mypy
1553 assert colspecs is not None
1554
1555 # GH#40830
1556 # Ensure length of `colspecs` matches length of `names`
1557 names = kwds.get("names")
1558 if names is not None and names is not lib.no_default:
1559 if len(names) != len(colspecs) and colspecs != "infer":
1560 # need to check len(index_col) as it might contain
1561 # unnamed indices, in which case it's name is not required
1562 len_index = 0
1563 if kwds.get("index_col") is not None:
1564 index_col: Any = kwds.get("index_col")
1565 if index_col is not False:
1566 if not is_list_like(index_col):
1567 len_index = 1
1568 else:
1569 # for mypy: handled in the if-branch
1570 assert index_col is not lib.no_default
1571
1572 len_index = len(index_col)
1573 if kwds.get("usecols") is None and len(names) + len_index != len(colspecs):
1574 # If usecols is used colspec may be longer than names
1575 raise ValueError("Length of colspecs must match length of names")
1576
1577 check_dtype_backend(kwds.setdefault("dtype_backend", lib.no_default))
1578 return _read(
1579 filepath_or_buffer,
1580 kwds
1581 | {
1582 "colspecs": colspecs,
1583 "infer_nrows": infer_nrows,
1584 "engine": "python-fwf",
1585 "iterator": iterator,
1586 "chunksize": chunksize,
1587 },
1588 )
1589
1590
1591class TextFileReader(abc.Iterator):
1592 """
1593
1594 Passed dialect overrides any of the related parser options
1595
1596 """
1597
1598 def __init__(
1599 self,
1600 f: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str] | list,
1601 engine: CSVEngine | None = None,
1602 **kwds,
1603 ) -> None:
1604 if engine is not None:
1605 engine_specified = True
1606 else:
1607 engine = "python"
1608 engine_specified = False
1609 self.engine = engine
1610 self._engine_specified = kwds.get("engine_specified", engine_specified)
1611
1612 _validate_skipfooter(kwds)
1613
1614 dialect = _extract_dialect(kwds)
1615 if dialect is not None:
1616 if engine == "pyarrow":
1617 raise ValueError(
1618 "The 'dialect' option is not supported with the 'pyarrow' engine"
1619 )
1620 kwds = _merge_with_dialect_properties(dialect, kwds)
1621
1622 if kwds.get("header", "infer") == "infer":
1623 kwds["header"] = 0 if kwds.get("names") is None else None
1624
1625 self.orig_options = kwds
1626
1627 # miscellanea
1628 self._currow = 0
1629
1630 options = self._get_options_with_defaults(engine)
1631 options["storage_options"] = kwds.get("storage_options", None)
1632
1633 self.chunksize = options.pop("chunksize", None)
1634 self.nrows = options.pop("nrows", None)
1635
1636 self._check_file_or_buffer(f, engine)
1637 self.options, self.engine = self._clean_options(options, engine)
1638
1639 if "has_index_names" in kwds:
1640 self.options["has_index_names"] = kwds["has_index_names"]
1641
1642 self.handles: IOHandles | None = None
1643 self._engine = self._make_engine(f, self.engine)
1644
1645 def close(self) -> None:
1646 if self.handles is not None:
1647 self.handles.close()
1648 self._engine.close()
1649
1650 def _get_options_with_defaults(self, engine: CSVEngine) -> dict[str, Any]:
1651 kwds = self.orig_options
1652
1653 options = {}
1654 default: object | None
1655
1656 for argname, default in parser_defaults.items():
1657 value = kwds.get(argname, default)
1658
1659 # see gh-12935
1660 if (
1661 engine == "pyarrow"
1662 and argname in _pyarrow_unsupported
1663 and value != default
1664 and value != getattr(value, "value", default)
1665 ):
1666 raise ValueError(
1667 f"The {argname!r} option is not supported with the 'pyarrow' engine"
1668 )
1669 options[argname] = value
1670
1671 for argname, default in _c_parser_defaults.items():
1672 if argname in kwds:
1673 value = kwds[argname]
1674
1675 if engine != "c" and value != default:
1676 # TODO: Refactor this logic, its pretty convoluted
1677 if "python" in engine and argname not in _python_unsupported:
1678 pass
1679 elif "pyarrow" in engine and argname not in _pyarrow_unsupported:
1680 pass
1681 else:
1682 raise ValueError(
1683 f"The {argname!r} option is not supported with the "
1684 f"{engine!r} engine"
1685 )
1686 else:
1687 value = default
1688 options[argname] = value
1689
1690 if engine == "python-fwf":
1691 for argname, default in _fwf_defaults.items():
1692 options[argname] = kwds.get(argname, default)
1693
1694 return options
1695
1696 def _check_file_or_buffer(self, f, engine: CSVEngine) -> None:
1697 # see gh-16530
1698 if is_file_like(f) and engine != "c" and not hasattr(f, "__iter__"):
1699 # The C engine doesn't need the file-like to have the "__iter__"
1700 # attribute. However, the Python engine needs "__iter__(...)"
1701 # when iterating through such an object, meaning it
1702 # needs to have that attribute
1703 raise ValueError(
1704 "The 'python' engine cannot iterate through this file buffer."
1705 )
1706 if hasattr(f, "encoding"):
1707 file_encoding = f.encoding
1708 orig_reader_enc = self.orig_options.get("encoding", None)
1709 any_none = file_encoding is None or orig_reader_enc is None
1710 if file_encoding != orig_reader_enc and not any_none:
1711 file_path = getattr(f, "name", None)
1712 raise ValueError(
1713 f"The specified reader encoding {orig_reader_enc} is different "
1714 f"from the encoding {file_encoding} of file {file_path}."
1715 )
1716
1717 def _clean_options(
1718 self, options: dict[str, Any], engine: CSVEngine
1719 ) -> tuple[dict[str, Any], CSVEngine]:
1720 result = options.copy()
1721
1722 fallback_reason = None
1723
1724 # C engine not supported yet
1725 if engine == "c":
1726 if options["skipfooter"] > 0:
1727 fallback_reason = "the 'c' engine does not support skipfooter"
1728 engine = "python"
1729
1730 sep = options["delimiter"]
1731
1732 if sep is None:
1733 # sniffing the separator with csv.Sniffer is python-engine only
1734 if engine in ("c", "pyarrow"):
1735 fallback_reason = f"the '{engine}' engine does not support sep=None"
1736 engine = "python"
1737 elif len(sep) > 1:
1738 if engine == "c" and sep == r"\s+":
1739 # delim_whitespace passed on to pandas._libs.parsers.TextReader
1740 result["delim_whitespace"] = True
1741 del result["delimiter"]
1742 elif engine not in ("python", "python-fwf"):
1743 # wait until regex engine integrated
1744 fallback_reason = (
1745 f"the '{engine}' engine does not support "
1746 "regex separators (separators > 1 char and "
1747 r"different from '\s+' are interpreted as regex)"
1748 )
1749 engine = "python"
1750 else:
1751 encodeable = True
1752 encoding = sys.getfilesystemencoding() or "utf-8"
1753 try:
1754 if len(sep.encode(encoding)) > 1:
1755 encodeable = False
1756 except UnicodeDecodeError:
1757 encodeable = False
1758 if not encodeable and engine not in ("python", "python-fwf"):
1759 fallback_reason = (
1760 f"the separator encoded in {encoding} "
1761 f"is > 1 char long, and the '{engine}' engine "
1762 "does not support such separators"
1763 )
1764 engine = "python"
1765
1766 quotechar = options["quotechar"]
1767 if quotechar is not None and isinstance(quotechar, (str, bytes)):
1768 if (
1769 len(quotechar) == 1
1770 and ord(quotechar) > 127
1771 and engine not in ("python", "python-fwf")
1772 ):
1773 fallback_reason = (
1774 "ord(quotechar) > 127, meaning the "
1775 "quotechar is larger than one byte, "
1776 f"and the '{engine}' engine does not support such quotechars"
1777 )
1778 engine = "python"
1779
1780 if fallback_reason and self._engine_specified:
1781 raise ValueError(fallback_reason)
1782
1783 if engine == "c":
1784 for arg in _c_unsupported:
1785 del result[arg]
1786
1787 if "python" in engine:
1788 for arg in _python_unsupported:
1789 if fallback_reason and result[arg] != _c_parser_defaults.get(arg):
1790 raise ValueError(
1791 "Falling back to the 'python' engine because "
1792 f"{fallback_reason}, but this causes {arg!r} to be "
1793 "ignored as it is not supported by the 'python' engine."
1794 )
1795 del result[arg]
1796
1797 if fallback_reason:
1798 warnings.warn(
1799 (
1800 "Falling back to the 'python' engine because "
1801 f"{fallback_reason}; you can avoid this warning by specifying "
1802 "engine='python'."
1803 ),
1804 ParserWarning,
1805 stacklevel=find_stack_level(),
1806 )
1807
1808 index_col = options["index_col"]
1809 names = options["names"]
1810 converters = options["converters"]
1811 na_values = options["na_values"]
1812 skiprows = options["skiprows"]
1813
1814 validate_header_arg(options["header"])
1815
1816 if index_col is True:
1817 raise ValueError("The value of index_col couldn't be 'True'")
1818 if is_index_col(index_col):
1819 if not isinstance(index_col, (list, tuple, np.ndarray)):
1820 index_col = [index_col]
1821 result["index_col"] = index_col
1822
1823 names = list(names) if names is not None else names
1824
1825 # type conversion-related
1826 if converters is not None:
1827 if not isinstance(converters, dict):
1828 raise TypeError(
1829 "Type converters must be a dict or subclass, "
1830 f"input was a {type(converters).__name__}"
1831 )
1832 else:
1833 converters = {}
1834
1835 # Converting values to NA
1836 keep_default_na = options["keep_default_na"]
1837 floatify = engine != "pyarrow"
1838 na_values, na_fvalues = _clean_na_values(
1839 na_values, keep_default_na, floatify=floatify
1840 )
1841
1842 # handle skiprows; this is internally handled by the
1843 # c-engine, so only need for python and pyarrow parsers
1844 if engine == "pyarrow":
1845 if not is_integer(skiprows) and skiprows is not None:
1846 # pyarrow expects skiprows to be passed as an integer
1847 raise ValueError(
1848 "skiprows argument must be an integer when using engine='pyarrow'"
1849 )
1850 else:
1851 if is_integer(skiprows):
1852 skiprows = range(skiprows)
1853 if skiprows is None:
1854 skiprows = set()
1855 elif not callable(skiprows):
1856 skiprows = set(skiprows)
1857
1858 # put stuff back
1859 result["names"] = names
1860 result["converters"] = converters
1861 result["na_values"] = na_values
1862 result["na_fvalues"] = na_fvalues
1863 result["skiprows"] = skiprows
1864
1865 return result, engine
1866
1867 def __next__(self) -> DataFrame:
1868 try:
1869 return self.get_chunk()
1870 except StopIteration:
1871 self.close()
1872 raise
1873
1874 def _make_engine(
1875 self,
1876 f: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str] | list | IO,
1877 engine: CSVEngine = "c",
1878 ) -> ParserBase:
1879 mapping: dict[str, type[ParserBase]] = {
1880 "c": CParserWrapper,
1881 "python": PythonParser,
1882 "pyarrow": ArrowParserWrapper,
1883 "python-fwf": FixedWidthFieldParser,
1884 }
1885
1886 if engine not in mapping:
1887 raise ValueError(
1888 f"Unknown engine: {engine} (valid options are {mapping.keys()})"
1889 )
1890 if not isinstance(f, list):
1891 # open file here
1892 is_text = True
1893 mode = "r"
1894 if engine == "pyarrow":
1895 is_text = False
1896 mode = "rb"
1897 elif (
1898 engine == "c"
1899 and self.options.get("encoding", "utf-8") == "utf-8"
1900 and isinstance(stringify_path(f), str)
1901 ):
1902 # c engine can decode utf-8 bytes, adding TextIOWrapper makes
1903 # the c-engine especially for memory_map=True far slower
1904 is_text = False
1905 if "b" not in mode:
1906 mode += "b"
1907 self.handles = get_handle(
1908 f,
1909 mode,
1910 encoding=self.options.get("encoding", None),
1911 compression=self.options.get("compression", None),
1912 memory_map=self.options.get("memory_map", False),
1913 is_text=is_text,
1914 errors=self.options.get("encoding_errors", "strict"),
1915 storage_options=self.options.get("storage_options", None),
1916 )
1917 assert self.handles is not None
1918 f = self.handles.handle
1919
1920 elif engine != "python":
1921 msg = f"Invalid file path or buffer object type: {type(f)}"
1922 raise ValueError(msg)
1923
1924 try:
1925 return mapping[engine](f, **self.options)
1926 except Exception:
1927 if self.handles is not None:
1928 self.handles.close()
1929 raise
1930
1931 def _failover_to_python(self) -> None:
1932 raise AbstractMethodError(self)
1933
1934 def read(self, nrows: int | None = None) -> DataFrame:
1935 if self.engine == "pyarrow":
1936 try:
1937 # error: "ParserBase" has no attribute "read"
1938 df = self._engine.read() # type: ignore[attr-defined]
1939 except Exception:
1940 self.close()
1941 raise
1942 else:
1943 nrows = validate_integer("nrows", nrows)
1944 try:
1945 # error: "ParserBase" has no attribute "read"
1946 (
1947 index,
1948 columns,
1949 col_dict,
1950 ) = self._engine.read( # type: ignore[attr-defined]
1951 nrows
1952 )
1953 except Exception:
1954 self.close()
1955 raise
1956
1957 if index is None:
1958 if col_dict:
1959 # Any column is actually fine:
1960 new_rows = len(next(iter(col_dict.values())))
1961 index = RangeIndex(self._currow, self._currow + new_rows)
1962 else:
1963 new_rows = 0
1964 else:
1965 new_rows = len(index)
1966
1967 if hasattr(self, "orig_options"):
1968 dtype_arg = self.orig_options.get("dtype", None)
1969 else:
1970 dtype_arg = None
1971
1972 if isinstance(dtype_arg, dict):
1973 dtype = defaultdict(lambda: None) # type: ignore[var-annotated]
1974 dtype.update(dtype_arg)
1975 elif dtype_arg is not None and pandas_dtype(dtype_arg) in (
1976 np.str_,
1977 np.object_,
1978 ):
1979 dtype = defaultdict(lambda: dtype_arg)
1980 else:
1981 dtype = None
1982
1983 if dtype is not None:
1984 new_col_dict = {}
1985 for k, v in col_dict.items():
1986 d = (
1987 dtype[k]
1988 if pandas_dtype(dtype[k]) in (np.str_, np.object_)
1989 else None
1990 )
1991 new_col_dict[k] = Series(v, index=index, dtype=d, copy=False)
1992 else:
1993 new_col_dict = col_dict
1994
1995 df = DataFrame(
1996 new_col_dict,
1997 columns=columns,
1998 index=index,
1999 copy=False,
2000 )
2001
2002 self._currow += new_rows
2003 return df
2004
2005 def get_chunk(self, size: int | None = None) -> DataFrame:
2006 if size is None:
2007 size = self.chunksize
2008 if self.nrows is not None:
2009 if self._currow >= self.nrows:
2010 raise StopIteration
2011 if size is None:
2012 size = self.nrows - self._currow
2013 else:
2014 size = min(size, self.nrows - self._currow)
2015 return self.read(nrows=size)
2016
2017 def __enter__(self) -> Self:
2018 return self
2019
2020 def __exit__(
2021 self,
2022 exc_type: type[BaseException] | None,
2023 exc_value: BaseException | None,
2024 traceback: TracebackType | None,
2025 ) -> None:
2026 self.close()
2027
2028
2029def TextParser(*args, **kwds) -> TextFileReader:
2030 """
2031 Converts lists of lists/tuples into DataFrames with proper type inference
2032 and optional (e.g. string to datetime) conversion. Also enables iterating
2033 lazily over chunks of large files
2034
2035 Parameters
2036 ----------
2037 data : file-like object or list
2038 delimiter : separator character to use
2039 dialect : str or csv.Dialect instance, optional
2040 Ignored if delimiter is longer than 1 character
2041 names : sequence, default
2042 header : int, default 0
2043 Row to use to parse column labels. Defaults to the first row. Prior
2044 rows will be discarded
2045 index_col : int or list, optional
2046 Column or columns to use as the (possibly hierarchical) index
2047 has_index_names: bool, default False
2048 True if the cols defined in index_col have an index name and are
2049 not in the header.
2050 na_values : scalar, str, list-like, or dict, optional
2051 Additional strings to recognize as NA/NaN.
2052 keep_default_na : bool, default True
2053 thousands : str, optional
2054 Thousands separator
2055 comment : str, optional
2056 Comment out remainder of line
2057 parse_dates : bool, default False
2058 date_format : str or dict of column -> format, default ``None``
2059
2060 .. versionadded:: 2.0.0
2061 skiprows : list of integers
2062 Row numbers to skip
2063 skipfooter : int
2064 Number of line at bottom of file to skip
2065 converters : dict, optional
2066 Dict of functions for converting values in certain columns. Keys can
2067 either be integers or column labels, values are functions that take one
2068 input argument, the cell (not column) content, and return the
2069 transformed content.
2070 encoding : str, optional
2071 Encoding to use for UTF when reading/writing (ex. 'utf-8')
2072 float_precision : str, optional
2073 Specifies which converter the C engine should use for floating-point
2074 values. The options are `None` or `high` for the ordinary converter,
2075 `legacy` for the original lower precision pandas converter, and
2076 `round_trip` for the round-trip converter.
2077 """
2078 kwds["engine"] = "python"
2079 return TextFileReader(*args, **kwds)
2080
2081
2082def _clean_na_values(na_values, keep_default_na: bool = True, floatify: bool = True):
2083 na_fvalues: set | dict
2084 if na_values is None:
2085 if keep_default_na:
2086 na_values = STR_NA_VALUES
2087 else:
2088 na_values = set()
2089 na_fvalues = set()
2090 elif isinstance(na_values, dict):
2091 old_na_values = na_values.copy()
2092 na_values = {} # Prevent aliasing.
2093
2094 # Convert the values in the na_values dictionary
2095 # into array-likes for further use. This is also
2096 # where we append the default NaN values, provided
2097 # that `keep_default_na=True`.
2098 for k, v in old_na_values.items():
2099 if not is_list_like(v):
2100 v = [v]
2101
2102 if keep_default_na:
2103 v = set(v) | STR_NA_VALUES
2104
2105 na_values[k] = _stringify_na_values(v, floatify)
2106 na_fvalues = {k: _floatify_na_values(v) for k, v in na_values.items()}
2107 else:
2108 if not is_list_like(na_values):
2109 na_values = [na_values]
2110 na_values = _stringify_na_values(na_values, floatify)
2111 if keep_default_na:
2112 na_values = na_values | STR_NA_VALUES
2113
2114 na_fvalues = _floatify_na_values(na_values)
2115
2116 return na_values, na_fvalues
2117
2118
2119def _floatify_na_values(na_values) -> set[float]:
2120 # create float versions of the na_values
2121 result = set()
2122 for v in na_values:
2123 try:
2124 v = float(v)
2125 if not np.isnan(v):
2126 result.add(v)
2127 except (TypeError, ValueError, OverflowError):
2128 pass
2129 return result
2130
2131
2132def _stringify_na_values(na_values, floatify: bool) -> set[str | float]:
2133 """return a stringified and numeric for these values"""
2134 result: list[str | float] = []
2135 for x in na_values:
2136 result.append(str(x))
2137 result.append(x)
2138 try:
2139 v = float(x)
2140
2141 # we are like 999 here
2142 if v == int(v):
2143 v = int(v)
2144 result.append(f"{v}.0")
2145 result.append(str(v))
2146
2147 if floatify:
2148 result.append(v)
2149 except (TypeError, ValueError, OverflowError):
2150 pass
2151 if floatify:
2152 try:
2153 result.append(int(x))
2154 except (TypeError, ValueError, OverflowError):
2155 pass
2156 return set(result)
2157
2158
2159def _refine_defaults_read(
2160 dialect: str | csv.Dialect | None,
2161 delimiter: str | None | lib.NoDefault,
2162 engine: CSVEngine | None,
2163 sep: str | None | lib.NoDefault,
2164 on_bad_lines: str | Callable,
2165 names: Sequence[Hashable] | None | lib.NoDefault,
2166 defaults: dict[str, Any],
2167 dtype_backend: DtypeBackend | lib.NoDefault,
2168):
2169 """Validate/refine default values of input parameters of read_csv, read_table.
2170
2171 Parameters
2172 ----------
2173 dialect : str or csv.Dialect
2174 If provided, this parameter will override values (default or not) for the
2175 following parameters: `delimiter`, `doublequote`, `escapechar`,
2176 `skipinitialspace`, `quotechar`, and `quoting`. If it is necessary to
2177 override values, a ParserWarning will be issued. See csv.Dialect
2178 documentation for more details.
2179 delimiter : str or object
2180 Alias for sep.
2181 engine : {'c', 'python'}
2182 Parser engine to use. The C engine is faster while the python engine is
2183 currently more feature-complete.
2184 sep : str or object
2185 A delimiter provided by the user (str) or a sentinel value, i.e.
2186 pandas._libs.lib.no_default.
2187 on_bad_lines : str, callable
2188 An option for handling bad lines or a sentinel value(None).
2189 names : array-like, optional
2190 List of column names to use. If the file contains a header row,
2191 then you should explicitly pass ``header=0`` to override the column names.
2192 Duplicates in this list are not allowed.
2193 defaults: dict
2194 Default values of input parameters.
2195
2196 Returns
2197 -------
2198 kwds : dict
2199 Input parameters with correct values.
2200 """
2201 # fix types for sep, delimiter to Union(str, Any)
2202 delim_default = defaults["delimiter"]
2203 kwds: dict[str, Any] = {}
2204 # gh-23761
2205 #
2206 # When a dialect is passed, it overrides any of the overlapping
2207 # parameters passed in directly. We don't want to warn if the
2208 # default parameters were passed in (since it probably means
2209 # that the user didn't pass them in explicitly in the first place).
2210 #
2211 # "delimiter" is the annoying corner case because we alias it to
2212 # "sep" before doing comparison to the dialect values later on.
2213 # Thus, we need a flag to indicate that we need to "override"
2214 # the comparison to dialect values by checking if default values
2215 # for BOTH "delimiter" and "sep" were provided.
2216 if dialect is not None:
2217 kwds["sep_override"] = delimiter is None and (
2218 sep is lib.no_default or sep == delim_default
2219 )
2220
2221 if delimiter and (sep is not lib.no_default):
2222 raise ValueError("Specified a sep and a delimiter; you can only specify one.")
2223
2224 kwds["names"] = None if names is lib.no_default else names
2225
2226 # Alias sep -> delimiter.
2227 if delimiter is None:
2228 delimiter = sep
2229
2230 if delimiter == "\n":
2231 raise ValueError(
2232 r"Specified \n as separator or delimiter. This forces the python engine "
2233 "which does not accept a line terminator. Hence it is not allowed to use "
2234 "the line terminator as separator.",
2235 )
2236
2237 if delimiter is lib.no_default:
2238 # assign default separator value
2239 kwds["delimiter"] = delim_default
2240 else:
2241 kwds["delimiter"] = delimiter
2242
2243 if engine is not None:
2244 kwds["engine_specified"] = True
2245 else:
2246 kwds["engine"] = "c"
2247 kwds["engine_specified"] = False
2248
2249 if on_bad_lines == "error":
2250 kwds["on_bad_lines"] = ParserBase.BadLineHandleMethod.ERROR
2251 elif on_bad_lines == "warn":
2252 kwds["on_bad_lines"] = ParserBase.BadLineHandleMethod.WARN
2253 elif on_bad_lines == "skip":
2254 kwds["on_bad_lines"] = ParserBase.BadLineHandleMethod.SKIP
2255 elif callable(on_bad_lines):
2256 if engine not in ["python", "pyarrow"]:
2257 raise ValueError(
2258 "on_bad_line can only be a callable function "
2259 "if engine='python' or 'pyarrow'"
2260 )
2261 kwds["on_bad_lines"] = on_bad_lines
2262 else:
2263 raise ValueError(f"Argument {on_bad_lines} is invalid for on_bad_lines")
2264
2265 check_dtype_backend(dtype_backend)
2266
2267 kwds["dtype_backend"] = dtype_backend
2268
2269 return kwds
2270
2271
2272def _extract_dialect(kwds: dict[str, str | csv.Dialect]) -> csv.Dialect | None:
2273 """
2274 Extract concrete csv dialect instance.
2275
2276 Returns
2277 -------
2278 csv.Dialect or None
2279 """
2280 if kwds.get("dialect") is None:
2281 return None
2282
2283 dialect = kwds["dialect"]
2284 if isinstance(dialect, str) and dialect in csv.list_dialects():
2285 # get_dialect is typed to return a `_csv.Dialect` for some reason in typeshed
2286 tdialect = cast(csv.Dialect, csv.get_dialect(dialect))
2287 _validate_dialect(tdialect)
2288
2289 else:
2290 _validate_dialect(dialect)
2291 tdialect = cast(csv.Dialect, dialect)
2292
2293 return tdialect
2294
2295
2296MANDATORY_DIALECT_ATTRS = (
2297 "delimiter",
2298 "doublequote",
2299 "escapechar",
2300 "skipinitialspace",
2301 "quotechar",
2302 "quoting",
2303)
2304
2305
2306def _validate_dialect(dialect: csv.Dialect | str) -> None:
2307 """
2308 Validate csv dialect instance.
2309
2310 Raises
2311 ------
2312 ValueError
2313 If incorrect dialect is provided.
2314 """
2315 for param in MANDATORY_DIALECT_ATTRS:
2316 if not hasattr(dialect, param):
2317 raise ValueError(f"Invalid dialect {dialect} provided")
2318
2319
2320def _merge_with_dialect_properties(
2321 dialect: csv.Dialect,
2322 defaults: dict[str, Any],
2323) -> dict[str, Any]:
2324 """
2325 Merge default kwargs in TextFileReader with dialect parameters.
2326
2327 Parameters
2328 ----------
2329 dialect : csv.Dialect
2330 Concrete csv dialect. See csv.Dialect documentation for more details.
2331 defaults : dict
2332 Keyword arguments passed to TextFileReader.
2333
2334 Returns
2335 -------
2336 kwds : dict
2337 Updated keyword arguments, merged with dialect parameters.
2338 """
2339 kwds = defaults.copy()
2340
2341 for param in MANDATORY_DIALECT_ATTRS:
2342 dialect_val = getattr(dialect, param)
2343
2344 parser_default = parser_defaults[param]
2345 provided = kwds.get(param, parser_default)
2346
2347 # Messages for conflicting values between the dialect
2348 # instance and the actual parameters provided.
2349 conflict_msgs = []
2350
2351 # Don't warn if the default parameter was passed in,
2352 # even if it conflicts with the dialect (gh-23761).
2353 if provided not in (parser_default, dialect_val):
2354 msg = (
2355 f"Conflicting values for '{param}': '{provided}' was "
2356 f"provided, but the dialect specifies '{dialect_val}'. "
2357 "Using the dialect-specified value."
2358 )
2359
2360 # Annoying corner case for not warning about
2361 # conflicts between dialect and delimiter parameter.
2362 # Refer to the outer "_read_" function for more info.
2363 if not (param == "delimiter" and kwds.pop("sep_override", False)):
2364 conflict_msgs.append(msg)
2365
2366 if conflict_msgs:
2367 warnings.warn(
2368 "\n\n".join(conflict_msgs), ParserWarning, stacklevel=find_stack_level()
2369 )
2370 kwds[param] = dialect_val
2371 return kwds
2372
2373
2374def _validate_skipfooter(kwds: dict[str, Any]) -> None:
2375 """
2376 Check whether skipfooter is compatible with other kwargs in TextFileReader.
2377
2378 Parameters
2379 ----------
2380 kwds : dict
2381 Keyword arguments passed to TextFileReader.
2382
2383 Raises
2384 ------
2385 ValueError
2386 If skipfooter is not compatible with other parameters.
2387 """
2388 if kwds.get("skipfooter"):
2389 if kwds.get("iterator") or kwds.get("chunksize"):
2390 raise ValueError("'skipfooter' not supported for iteration")
2391 if kwds.get("nrows"):
2392 raise ValueError("'skipfooter' not supported with 'nrows'")