Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/pandas/io/parsers/readers.py: 18%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

552 statements  

1""" 

2Module contains tools for processing files into DataFrames or other objects 

3 

4GH#48849 provides a convenient way of deprecating keyword arguments 

5""" 

6 

7from __future__ import annotations 

8 

9from collections import ( 

10 abc, 

11 defaultdict, 

12) 

13import csv 

14import sys 

15from typing import ( 

16 IO, 

17 TYPE_CHECKING, 

18 Any, 

19 Generic, 

20 Literal, 

21 Self, 

22 TypedDict, 

23 Unpack, 

24 cast, 

25 overload, 

26) 

27import warnings 

28 

29import numpy as np 

30 

31from pandas._libs import lib 

32from pandas._libs.parsers import STR_NA_VALUES 

33from pandas.errors import ( 

34 AbstractMethodError, 

35 ParserWarning, 

36) 

37from pandas.util._decorators import ( 

38 set_module, 

39) 

40from pandas.util._exceptions import find_stack_level 

41from pandas.util._validators import check_dtype_backend 

42 

43from pandas.core.dtypes.common import ( 

44 is_file_like, 

45 is_float, 

46 is_integer, 

47 is_list_like, 

48 pandas_dtype, 

49) 

50 

51from pandas import Series 

52from pandas.core.frame import DataFrame 

53from pandas.core.indexes.api import RangeIndex 

54 

55from pandas.io.common import ( 

56 IOHandles, 

57 get_handle, 

58 stringify_path, 

59 validate_header_arg, 

60) 

61from pandas.io.parsers.arrow_parser_wrapper import ArrowParserWrapper 

62from pandas.io.parsers.base_parser import ( 

63 ParserBase, 

64 is_index_col, 

65 parser_defaults, 

66) 

67from pandas.io.parsers.c_parser_wrapper import CParserWrapper 

68from pandas.io.parsers.python_parser import ( 

69 FixedWidthFieldParser, 

70 PythonParser, 

71) 

72 

73if TYPE_CHECKING: 

74 from collections.abc import ( 

75 Callable, 

76 Hashable, 

77 Iterable, 

78 Mapping, 

79 Sequence, 

80 ) 

81 from types import TracebackType 

82 

83 from pandas._typing import ( 

84 CompressionOptions, 

85 CSVEngine, 

86 DtypeArg, 

87 DtypeBackend, 

88 FilePath, 

89 HashableT, 

90 IndexLabel, 

91 ReadCsvBuffer, 

92 StorageOptions, 

93 UsecolsArgType, 

94 ) 

95 

96 class _read_shared(TypedDict, Generic[HashableT], total=False): 

97 # annotations shared between read_csv/fwf/table's overloads 

98 # NOTE: Keep in sync with the annotations of the implementation 

99 sep: str | None | lib.NoDefault 

100 delimiter: str | None | lib.NoDefault 

101 header: int | Sequence[int] | None | Literal["infer"] 

102 names: Sequence[Hashable] | None | lib.NoDefault 

103 index_col: IndexLabel | Literal[False] | None 

104 usecols: UsecolsArgType 

105 dtype: DtypeArg | None 

106 engine: CSVEngine | None 

107 converters: Mapping[HashableT, Callable] | None 

108 true_values: list | None 

109 false_values: list | None 

110 skipinitialspace: bool 

111 skiprows: list[int] | int | Callable[[Hashable], bool] | None 

112 skipfooter: int 

113 nrows: int | None 

114 na_values: ( 

115 Hashable | Iterable[Hashable] | Mapping[Hashable, Iterable[Hashable]] | None 

116 ) 

117 keep_default_na: bool 

118 na_filter: bool 

119 skip_blank_lines: bool 

120 parse_dates: bool | Sequence[Hashable] | None 

121 date_format: str | dict[Hashable, str] | None 

122 dayfirst: bool 

123 cache_dates: bool 

124 compression: CompressionOptions 

125 thousands: str | None 

126 decimal: str 

127 lineterminator: str | None 

128 quotechar: str 

129 quoting: int 

130 doublequote: bool 

131 escapechar: str | None 

132 comment: str | None 

133 encoding: str | None 

134 encoding_errors: str | None 

135 dialect: str | csv.Dialect | None 

136 on_bad_lines: str 

137 low_memory: bool 

138 memory_map: bool 

139 float_precision: Literal["high", "legacy", "round_trip"] | None 

140 storage_options: StorageOptions | None 

141 dtype_backend: DtypeBackend | lib.NoDefault 

142 

143else: 

144 _read_shared = dict 

145 

146 

147class _C_Parser_Defaults(TypedDict): 

148 na_filter: Literal[True] 

149 low_memory: Literal[True] 

150 memory_map: Literal[False] 

151 float_precision: None 

152 

153 

154_c_parser_defaults: _C_Parser_Defaults = { 

155 "na_filter": True, 

156 "low_memory": True, 

157 "memory_map": False, 

158 "float_precision": None, 

159} 

160 

161 

162class _Fwf_Defaults(TypedDict): 

163 colspecs: Literal["infer"] 

164 infer_nrows: Literal[100] 

165 widths: None 

166 

167 

168_fwf_defaults: _Fwf_Defaults = {"colspecs": "infer", "infer_nrows": 100, "widths": None} 

169_c_unsupported = {"skipfooter"} 

170_python_unsupported = {"low_memory", "float_precision"} 

171_pyarrow_unsupported = { 

172 "skipfooter", 

173 "float_precision", 

174 "chunksize", 

175 "comment", 

176 "nrows", 

177 "thousands", 

178 "memory_map", 

179 "dialect", 

180 "quoting", 

181 "lineterminator", 

182 "converters", 

183 "iterator", 

184 "dayfirst", 

185 "skipinitialspace", 

186 "low_memory", 

187} 

188 

189 

190@overload 

191def validate_integer(name: str, val: None, min_val: int = ...) -> None: ... 

192 

193 

194@overload 

195def validate_integer(name: str, val: float, min_val: int = ...) -> int: ... 

196 

197 

198@overload 

199def validate_integer(name: str, val: int | None, min_val: int = ...) -> int | None: ... 

200 

201 

202def validate_integer( 

203 name: str, val: int | float | None, min_val: int = 0 

204) -> int | None: 

205 """ 

206 Checks whether the 'name' parameter for parsing is either 

207 an integer OR float that can SAFELY be cast to an integer 

208 without losing accuracy. Raises a ValueError if that is 

209 not the case. 

210 

211 Parameters 

212 ---------- 

213 name : str 

214 Parameter name (used for error reporting) 

215 val : int or float 

216 The value to check 

217 min_val : int 

218 Minimum allowed value (val < min_val will result in a ValueError) 

219 """ 

220 if val is None: 

221 return val 

222 

223 msg = f"'{name:s}' must be an integer >={min_val:d}" 

224 if is_float(val): 

225 if int(val) != val: 

226 raise ValueError(msg) 

227 val = int(val) 

228 elif not (is_integer(val) and val >= min_val): 

229 raise ValueError(msg) 

230 

231 return int(val) 

232 

233 

234def _validate_names(names: Sequence[Hashable] | None) -> None: 

235 """ 

236 Raise ValueError if the `names` parameter contains duplicates or has an 

237 invalid data type. 

238 

239 Parameters 

240 ---------- 

241 names : array-like or None 

242 An array containing a list of the names used for the output DataFrame. 

243 

244 Raises 

245 ------ 

246 ValueError 

247 If names are not unique or are not ordered (e.g. set). 

248 """ 

249 if names is not None: 

250 if len(names) != len(set(names)): 

251 raise ValueError("Duplicate names are not allowed.") 

252 if not ( 

253 is_list_like(names, allow_sets=False) or isinstance(names, abc.KeysView) 

254 ): 

255 raise ValueError("Names should be an ordered collection.") 

256 

257 

258def _read( 

259 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], kwds 

260) -> DataFrame | TextFileReader: 

261 """Generic reader of line files.""" 

262 # if we pass a date_format and parse_dates=False, we should not parse the 

263 # dates GH#44366 

264 if kwds.get("parse_dates", None) is None: 

265 if kwds.get("date_format", None) is None: 

266 kwds["parse_dates"] = False 

267 else: 

268 kwds["parse_dates"] = True 

269 

270 # Extract some of the arguments (pass chunksize on). 

271 iterator = kwds.get("iterator", False) 

272 chunksize = kwds.get("chunksize", None) 

273 

274 # Check type of encoding_errors 

275 errors = kwds.get("encoding_errors", "strict") 

276 if not isinstance(errors, str): 

277 raise ValueError( 

278 f"encoding_errors must be a string, got {type(errors).__name__}" 

279 ) 

280 

281 if kwds.get("engine") == "pyarrow": 

282 if iterator: 

283 raise ValueError( 

284 "The 'iterator' option is not supported with the 'pyarrow' engine" 

285 ) 

286 

287 if chunksize is not None: 

288 raise ValueError( 

289 "The 'chunksize' option is not supported with the 'pyarrow' engine" 

290 ) 

291 else: 

292 chunksize = validate_integer("chunksize", chunksize, 1) 

293 

294 nrows = kwds.get("nrows", None) 

295 

296 # Check for duplicates in names. 

297 _validate_names(kwds.get("names", None)) 

298 

299 # Create the parser. 

300 parser = TextFileReader(filepath_or_buffer, **kwds) 

301 

302 if chunksize or iterator: 

303 return parser 

304 

305 with parser: 

306 return parser.read(nrows) 

307 

308 

309@overload 

310def read_csv( 

311 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

312 *, 

313 iterator: Literal[True], 

314 chunksize: int | None = ..., 

315 **kwds: Unpack[_read_shared[HashableT]], 

316) -> TextFileReader: ... 

317 

318 

319@overload 

320def read_csv( 

321 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

322 *, 

323 iterator: bool = ..., 

324 chunksize: int, 

325 **kwds: Unpack[_read_shared[HashableT]], 

326) -> TextFileReader: ... 

327 

328 

329@overload 

330def read_csv( 

331 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

332 *, 

333 iterator: Literal[False] = ..., 

334 chunksize: None = ..., 

335 **kwds: Unpack[_read_shared[HashableT]], 

336) -> DataFrame: ... 

337 

338 

339@overload 

340def read_csv( 

341 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

342 *, 

343 iterator: bool = ..., 

344 chunksize: int | None = ..., 

345 **kwds: Unpack[_read_shared[HashableT]], 

346) -> DataFrame | TextFileReader: ... 

347 

348 

349@set_module("pandas") 

350def read_csv( 

351 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

352 *, 

353 sep: str | None | lib.NoDefault = lib.no_default, 

354 delimiter: str | None | lib.NoDefault = None, 

355 # Column and Index Locations and Names 

356 header: int | Sequence[int] | None | Literal["infer"] = "infer", 

357 names: Sequence[Hashable] | None | lib.NoDefault = lib.no_default, 

358 index_col: IndexLabel | Literal[False] | None = None, 

359 usecols: UsecolsArgType = None, 

360 # General Parsing Configuration 

361 dtype: DtypeArg | None = None, 

362 engine: CSVEngine | None = None, 

363 converters: Mapping[HashableT, Callable] | None = None, 

364 true_values: list | None = None, 

365 false_values: list | None = None, 

366 skipinitialspace: bool = False, 

367 skiprows: list[int] | int | Callable[[Hashable], bool] | None = None, 

368 skipfooter: int = 0, 

369 nrows: int | None = None, 

370 # NA and Missing Data Handling 

371 na_values: ( 

372 Hashable | Iterable[Hashable] | Mapping[Hashable, Iterable[Hashable]] | None 

373 ) = None, 

374 keep_default_na: bool = True, 

375 na_filter: bool = True, 

376 skip_blank_lines: bool = True, 

377 # Datetime Handling 

378 parse_dates: bool | Sequence[Hashable] | None = None, 

379 date_format: str | dict[Hashable, str] | None = None, 

380 dayfirst: bool = False, 

381 cache_dates: bool = True, 

382 # Iteration 

383 iterator: bool = False, 

384 chunksize: int | None = None, 

385 # Quoting, Compression, and File Format 

386 compression: CompressionOptions = "infer", 

387 thousands: str | None = None, 

388 decimal: str = ".", 

389 lineterminator: str | None = None, 

390 quotechar: str = '"', 

391 quoting: int = csv.QUOTE_MINIMAL, 

392 doublequote: bool = True, 

393 escapechar: str | None = None, 

394 comment: str | None = None, 

395 encoding: str | None = None, 

396 encoding_errors: str | None = "strict", 

397 dialect: str | csv.Dialect | None = None, 

398 # Error Handling 

399 on_bad_lines: str = "error", 

400 # Internal 

401 low_memory: bool = _c_parser_defaults["low_memory"], 

402 memory_map: bool = False, 

403 float_precision: Literal["high", "legacy", "round_trip"] | None = None, 

404 storage_options: StorageOptions | None = None, 

405 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default, 

406) -> DataFrame | TextFileReader: 

407 """ 

408 Read a comma-separated values (csv) file into DataFrame. 

409 

410 Also supports optionally iterating or breaking of the file 

411 into chunks. 

412 

413 Additional help can be found in the online docs for 

414 `IO Tools <https://pandas.pydata.org/pandas-docs/stable/user_guide/io.html>`_. 

415 

416 Parameters 

417 ---------- 

418 filepath_or_buffer : str, path object or file-like object 

419 Any valid string path is acceptable. The string could be a URL. Valid 

420 URL schemes include http, ftp, s3, gs, and file. For file URLs, a host is 

421 expected. A local file could be: file://localhost/path/to/table.csv. 

422 

423 If you want to pass in a path object, pandas accepts any ``os.PathLike``. 

424 

425 By file-like object, we refer to objects with a ``read()`` method, such as 

426 a file handle (e.g. via builtin ``open`` function) or ``StringIO``. 

427 sep : str, default ',' 

428 Character or regex pattern to treat as the delimiter. ``sep=None`` detects 

429 the separator from the first valid row of the file with Python's builtin 

430 sniffer tool, ``csv.Sniffer``; it is supported only by the Python parsing 

431 engine, which will be used automatically. 

432 In addition, separators longer than 1 character and different from 

433 ``'\\s+'`` will be interpreted as regular expressions and will also force 

434 the use of the Python parsing engine. Note that regex delimiters are prone 

435 to ignoring quoted data. Regex example: ``'\\r\\t'``. 

436 delimiter : str, optional 

437 Alias for ``sep``. 

438 header : int, Sequence of int, 'infer' or None, default 'infer' 

439 Row number(s) containing column labels and marking the start of the 

440 data (zero-indexed). Default behavior is to infer the column names: 

441 if no ``names`` 

442 are passed the behavior is identical to ``header=0`` and column 

443 names are inferred from the first line of the file, if column 

444 names are passed explicitly to ``names`` then the behavior is identical to 

445 ``header=None``. Explicitly pass ``header=0`` to be able to 

446 replace existing names. The header can be a list of integers that 

447 specify row locations for a :class:`~pandas.MultiIndex` on the columns 

448 e.g. ``[0, 1, 3]``. Intervening rows that are not specified will be 

449 skipped (e.g. 2 in this example is skipped). Note that this 

450 parameter ignores commented lines and empty lines if 

451 ``skip_blank_lines=True``, so ``header=0`` denotes the first line of 

452 data rather than the first line of the file. 

453 

454 When inferred from the file contents, headers are kept distinct from 

455 each other by renaming duplicate names with a numeric suffix of the form 

456 ``".{count}"`` starting from 1, e.g. ``"foo"`` and ``"foo.1"``. 

457 Empty headers are named ``"Unnamed: {i}"`` or `` 

458 "Unnamed: {i}_level_{level}"`` 

459 in the case of MultiIndex columns. 

460 names : Sequence of Hashable, optional 

461 Sequence of column labels to apply. If the file contains a header row, 

462 then you should explicitly pass ``header=0`` to override the column names. 

463 Duplicates in this list are not allowed. 

464 index_col : Hashable, Sequence of Hashable or False, optional 

465 Column(s) to use as row label(s), denoted either by column labels or column 

466 indices. If a sequence of labels or indices is given, 

467 :class:`~pandas.MultiIndex` 

468 will be formed for the row labels. 

469 

470 Note: ``index_col=False`` can be used to force pandas to *not* use the first 

471 column as the index, e.g., when you have a malformed file with delimiters at 

472 the end of each line. 

473 usecols : Sequence of Hashable or Callable, optional 

474 Subset of columns to select, denoted either 

475 by column labels or column indices. 

476 If list-like, all elements must either 

477 be positional (i.e. integer indices into the document columns) or strings 

478 that correspond to column names provided either by the user in ``names`` or 

479 inferred from the document header row(s). 

480 If ``names`` are given, the document 

481 header row(s) are not taken into account. For example, a valid list-like 

482 ``usecols`` parameter would be ``[0, 1, 2]`` or ``['foo', 'bar', 'baz']``. 

483 Element order is ignored, so ``usecols=[0, 1]`` is the same as ``[1, 0]``. 

484 To instantiate a :class:`~pandas.DataFrame` from ``data`` with element order 

485 preserved use ``pd.read_csv(data, usecols=['foo', 'bar'])[['foo', 'bar']]`` 

486 for columns in ``['foo', 'bar']`` order or 

487 ``pd.read_csv(data, usecols=['foo', 'bar'])[['bar', 'foo']]`` 

488 for ``['bar', 'foo']`` order. 

489 

490 If callable, the callable function will be evaluated against the column 

491 names, returning names where the callable function evaluates to ``True``. An 

492 example of a valid callable argument would be ``lambda x: x.upper() in 

493 ['AAA', 'BBB', 'DDD']``. Using this parameter results in much faster 

494 parsing time and lower memory usage. 

495 dtype : dtype or dict of {Hashable : dtype}, optional 

496 Data type(s) to apply to either the whole dataset or individual columns. 

497 E.g., ``{'a': np.float64, 'b': np.int32, 'c': 'Int64'}`` 

498 Use ``str`` or ``object`` together with suitable ``na_values`` settings 

499 to preserve and not interpret ``dtype``. 

500 If ``converters`` are specified, they will be applied INSTEAD 

501 of ``dtype`` conversion. Specify a ``defaultdict`` as input where 

502 the default determines the ``dtype`` 

503 of the columns which are not explicitly 

504 listed. 

505 engine : {'c', 'python', 'pyarrow'}, optional 

506 Parser engine to use. The C and pyarrow engines are faster, 

507 while the python engine 

508 is currently more feature-complete. Multithreading 

509 is currently only supported by 

510 the pyarrow engine. Some features of the "pyarrow" engine 

511 are unsupported or may not work correctly. 

512 converters : dict of {Hashable : Callable}, optional 

513 Functions for converting values in specified columns. Keys can either 

514 be column labels or column indices. 

515 true_values : list, optional 

516 Values to consider as ``True`` in addition 

517 to case-insensitive variants of 'True'. 

518 false_values : list, optional 

519 Values to consider as ``False`` in addition to case-insensitive 

520 variants of 'False'. 

521 skipinitialspace : bool, default False 

522 Skip spaces after delimiter. 

523 skiprows : int, list of int or Callable, optional 

524 Line numbers to skip (0-indexed) or number of lines to skip (``int``) 

525 at the start of the file. 

526 

527 If callable, the callable function will be evaluated against the row 

528 indices, returning ``True`` if the row should be skipped and ``False`` 

529 otherwise. 

530 An example of a valid callable argument would be ``lambda x: x in [0, 2]``. 

531 skipfooter : int, default 0 

532 Number of lines at bottom of file to skip (Unsupported with ``engine='c'``). 

533 nrows : int, optional 

534 Number of rows of file to read. Useful for reading pieces of large files. 

535 Refers to the number of data rows in the returned DataFrame, excluding: 

536 

537 * The header row containing column names. 

538 * Rows before the header row, if ``header=1`` or larger. 

539 

540 Example usage: 

541 

542 * To read the first 999,999 (non-header) rows: 

543 ``read_csv(..., nrows=999999)`` 

544 

545 * To read rows 1,000,000 through 1,999,999: 

546 ``read_csv(..., skiprows=1000000, nrows=999999)`` 

547 

548 na_values : Hashable, Iterable of Hashable or dict of {Hashable : Iterable}, 

549 optional 

550 Additional strings to recognize as ``NA``/``NaN``. If ``dict`` 

551 passed, specific 

552 per-column ``NA`` values. By default the following values 

553 are interpreted as 

554 ``NaN``: empty string, "NaN", "N/A", "NULL", and other common 

555 representations of missing data. 

556 keep_default_na : bool, default True 

557 Whether or not to include the default ``NaN`` values when parsing the data. 

558 Depending on whether ``na_values`` is passed in, the behavior is as follows: 

559 

560 * If ``keep_default_na`` is ``True``, and ``na_values`` 

561 are specified, ``na_values`` 

562 is appended to the default ``NaN`` values used for parsing. 

563 * If ``keep_default_na`` is ``True``, and ``na_values`` are not specified, only 

564 the default ``NaN`` values are used for parsing. 

565 * If ``keep_default_na`` is ``False``, and ``na_values`` are specified, only 

566 the ``NaN`` values specified ``na_values`` are used for parsing. 

567 * If ``keep_default_na`` is ``False``, and ``na_values`` are not specified, no 

568 strings will be parsed as ``NaN``. 

569 

570 Note that if ``na_filter`` is passed in as ``False``, 

571 the ``keep_default_na`` and 

572 ``na_values`` parameters will be ignored. 

573 na_filter : bool, default True 

574 Detect missing value markers (empty strings and the value of ``na_values``). In 

575 data without any ``NA`` values, passing ``na_filter=False`` can improve the 

576 performance of reading a large file. 

577 skip_blank_lines : bool, default True 

578 If ``True``, skip over blank lines rather than interpreting as ``NaN`` values. 

579 parse_dates : bool, None, list of Hashable, default None 

580 The behavior is as follows: 

581 

582 * ``bool``. If ``True`` -> try parsing the index. 

583 * ``None``. Behaves like ``True`` if ``date_format`` is specified. 

584 * ``list`` of ``int`` or names. 

585 e.g. If ``[1, 2, 3]`` -> try parsing columns 1, 2, 3 

586 each as a separate date column. 

587 

588 If a column or index cannot be represented as an array of ``datetime``, 

589 say because of an unparsable value or a mixture of timezones, the column 

590 or index will be returned unaltered as an ``object`` data type. For 

591 non-standard ``datetime`` parsing, use :func:`~pandas.to_datetime` after 

592 :func:`~pandas.read_csv`. 

593 

594 Note: A fast-path exists for iso8601-formatted dates. 

595 date_format : str or dict of column -> format, optional 

596 Format to use for parsing dates and/or times when 

597 used in conjunction with ``parse_dates``. 

598 The strftime to parse time, e.g. :const:`"%d/%m/%Y"`. See 

599 `strftime documentation 

600 <https://docs.python.org/3/library/datetime.html 

601 #strftime-and-strptime-behavior>`_ for more information on choices, though 

602 note that :const:`"%f"`` will parse all the way up to nanoseconds. 

603 You can also pass: 

604 

605 - "ISO8601", to parse any `ISO8601 <https://en.wikipedia.org/wiki/ISO_8601>`_ 

606 time string (not necessarily in exactly the same format); 

607 - "mixed", to infer the format for each element individually. This is risky, 

608 and you should probably use it along with `dayfirst`. 

609 

610 .. versionadded:: 2.0.0 

611 dayfirst : bool, default False 

612 DD/MM format dates, international and European format. 

613 cache_dates : bool, default True 

614 If ``True``, use a cache of unique, converted dates to apply the ``datetime`` 

615 conversion. May produce significant speed-up when parsing duplicate 

616 date strings, especially ones with timezone offsets. 

617 

618 iterator : bool, default False 

619 Return ``TextFileReader`` object for iteration or getting chunks with 

620 ``get_chunk()``. 

621 chunksize : int, optional 

622 Number of lines to read from the file per chunk. Passing a value will cause the 

623 function to return a ``TextFileReader`` object for iteration. 

624 See the `IO Tools docs 

625 <https://pandas.pydata.org/pandas-docs/stable/io.html#io-chunking>`_ 

626 for more information on ``iterator`` and ``chunksize``. 

627 

628 compression : str or dict, default 'infer' 

629 For on-the-fly decompression of on-disk data. 

630 If 'infer' and 'filepath_or_buffer' is 

631 path-like, then detect compression from the following extensions: '.gz', 

632 '.bz2', '.zip', '.xz', '.zst', '.tar', '.tar.gz', '.tar.xz' or '.tar.bz2' 

633 (otherwise no compression). 

634 If using 'zip' or 'tar', the ZIP file must contain only 

635 one data file to be read in. 

636 Set to ``None`` for no decompression. 

637 Can also be a dict with key ``'method'`` set 

638 to one of {``'zip'``, ``'gzip'``, ``'bz2'``, 

639 ``'zstd'``, ``'xz'``, ``'tar'``} and 

640 other key-value pairs are forwarded to 

641 ``zipfile.ZipFile``, ``gzip.GzipFile``, 

642 ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``, ``lzma.LZMAFile`` or 

643 ``tarfile.TarFile``, respectively. 

644 As an example, the following could be passed for 

645 Zstandard decompression using a 

646 custom compression dictionary: 

647 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``. 

648 

649 thousands : str (length 1), optional 

650 Character acting as the thousands separator in numerical values. 

651 decimal : str (length 1), default '.' 

652 Character to recognize as decimal point (e.g., use ',' for European data). 

653 lineterminator : str (length 1), optional 

654 Character used to denote a line break. Only valid with C parser. 

655 quotechar : str (length 1), optional 

656 Character used to denote the start and end of a quoted item. Quoted 

657 items can include the ``delimiter`` and it will be ignored. 

658 quoting : {0 or csv.QUOTE_MINIMAL, 1 or csv.QUOTE_ALL, 

659 2 or csv.QUOTE_NONNUMERIC, 3 or csv.QUOTE_NONE}, default csv.QUOTE_MINIMAL 

660 Control field quoting behavior per ``csv.QUOTE_*`` constants. Default is 

661 ``csv.QUOTE_MINIMAL`` (i.e., 0) which implies that 

662 only fields containing special 

663 characters are quoted (e.g., characters defined 

664 in ``quotechar``, ``delimiter``, 

665 or ``lineterminator``. 

666 doublequote : bool, default True 

667 When ``quotechar`` is specified and ``quoting`` is not ``QUOTE_NONE``, indicate 

668 whether or not to interpret two consecutive ``quotechar`` elements INSIDE a 

669 field as a single ``quotechar`` element. 

670 escapechar : str (length 1), optional 

671 Character used to escape other characters. 

672 comment : str (length 1), optional 

673 Character indicating that the remainder of line should not be parsed. 

674 If found at the beginning 

675 of a line, the line will be ignored altogether. This parameter must be a 

676 single character. Like empty lines (as long as ``skip_blank_lines=True``), 

677 fully commented lines are ignored by the parameter ``header`` but not by 

678 ``skiprows``. For example, if ``comment='#'``, parsing 

679 ``#empty\\na,b,c\\n1,2,3`` with ``header=0`` will result in ``'a,b,c'`` being 

680 treated as the header. 

681 encoding : str, optional, default 'utf-8' 

682 Encoding to use for UTF when reading/writing (ex. ``'utf-8'``). `List of Python 

683 standard encodings 

684 <https://docs.python.org/3/library/codecs.html#standard-encodings>`_ . 

685 

686 encoding_errors : str, optional, default 'strict' 

687 How encoding errors are treated. `List of possible values 

688 <https://docs.python.org/3/library/codecs.html#error-handlers>`_ . 

689 

690 dialect : str or csv.Dialect, optional 

691 If provided, this parameter will override values (default or not) for the 

692 following parameters: ``delimiter``, ``doublequote``, ``escapechar``, 

693 ``skipinitialspace``, ``quotechar``, and ``quoting``. If it is necessary to 

694 override values, a ``ParserWarning`` will be issued. See ``csv.Dialect`` 

695 documentation for more details. 

696 on_bad_lines : {'error', 'warn', 'skip'} or Callable, default 'error' 

697 Specifies what to do upon encountering a bad line (a line with too many fields). 

698 Allowed values are: 

699 

700 - ``'error'``, raise an Exception when a bad line is encountered. 

701 - ``'warn'``, raise a warning when a bad line is 

702 encountered and skip that line. 

703 - ``'skip'``, skip bad lines without raising or warning when 

704 they are encountered. 

705 - Callable, function that will process a single bad line. 

706 - With ``engine='python'``, function with signature 

707 ``(bad_line: list[str]) -> list[str] | None``. 

708 ``bad_line`` is a list of strings split by the ``sep``. 

709 If the function returns ``None``, the bad line will be ignored. 

710 If the function returns a new ``list`` of strings with 

711 more elements than 

712 expected, a ``ParserWarning`` will be emitted while 

713 dropping extra elements. 

714 - With ``engine='pyarrow'``, function with signature 

715 as described in pyarrow documentation: `invalid_row_handler 

716 <https://arrow.apache.org/docs/python 

717 /generated/pyarrow.csv.ParseOptions.html 

718 #pyarrow.csv.ParseOptions.invalid_row_handler>`_. 

719 

720 .. versionchanged:: 2.2.0 

721 

722 Callable for ``engine='pyarrow'`` 

723 

724 low_memory : bool, default True 

725 Internally process the file in chunks, resulting in lower memory use 

726 while parsing, but possibly mixed type inference. To ensure no mixed 

727 types either set ``False``, or specify the type with the ``dtype`` parameter. 

728 Note that the entire file is read into a single :class:`~pandas.DataFrame` 

729 regardless, use the ``chunksize`` or ``iterator`` 

730 parameter to return the data in 

731 chunks. (Only valid with C parser). 

732 memory_map : bool, default False 

733 If a filepath is provided for ``filepath_or_buffer``, map the file object 

734 directly onto memory and access the data directly from there. Using this 

735 option can improve performance because there is no longer any I/O overhead. 

736 float_precision : {'high', 'legacy', 'round_trip'}, optional 

737 Specifies which converter the C engine should use for floating-point 

738 values. The options are ``None`` or ``'high'`` for the ordinary converter, 

739 ``'legacy'`` for the original lower precision pandas converter, and 

740 ``'round_trip'`` for the round-trip converter. 

741 

742 storage_options : dict, optional 

743 Extra options that make sense for a particular storage connection, e.g. 

744 host, port, username, password, etc. For HTTP(S) URLs the key-value pairs 

745 are forwarded to ``urllib.request.Request`` as header options. For other 

746 URLs (e.g. starting with "s3://", and "gcs://") the key-value pairs are 

747 forwarded to ``fsspec.open``. Please see ``fsspec`` and ``urllib`` for more 

748 details, and for more examples on storage options refer `here 

749 <https://pandas.pydata.org/docs/user_guide/io.html? 

750 highlight=storage_options#reading-writing-remote-files>`_. 

751 

752 dtype_backend : {'numpy_nullable', 'pyarrow'} 

753 Back-end data type applied to the resultant :class:`DataFrame` 

754 (still experimental). If not specified, the default behavior 

755 is to not use nullable data types. If specified, the behavior 

756 is as follows: 

757 

758 * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame` 

759 * ``"pyarrow"``: returns 

760 pyarrow-backed nullable :class:`ArrowDtype` :class:`DataFrame` 

761 

762 .. versionadded:: 2.0 

763 

764 Returns 

765 ------- 

766 DataFrame or TextFileReader 

767 A comma-separated values (csv) file is returned as two-dimensional 

768 data structure with labeled axes. 

769 

770 See Also 

771 -------- 

772 DataFrame.to_csv : Write DataFrame to a comma-separated values (csv) file. 

773 read_table : Read general delimited file into DataFrame. 

774 read_fwf : Read a table of fixed-width formatted lines into DataFrame. 

775 

776 Examples 

777 -------- 

778 >>> pd.read_csv("data.csv") # doctest: +SKIP 

779 Name Value 

780 0 foo 1 

781 1 bar 2 

782 2 #baz 3 

783 

784 Index and header can be specified via the `index_col` and `header` arguments. 

785 

786 >>> pd.read_csv("data.csv", header=None) # doctest: +SKIP 

787 0 1 

788 0 Name Value 

789 1 foo 1 

790 2 bar 2 

791 3 #baz 3 

792 

793 >>> pd.read_csv("data.csv", index_col="Value") # doctest: +SKIP 

794 Name 

795 Value 

796 1 foo 

797 2 bar 

798 3 #baz 

799 

800 Column types are inferred but can be explicitly specified using the dtype argument. 

801 

802 >>> pd.read_csv("data.csv", dtype={"Value": float}) # doctest: +SKIP 

803 Name Value 

804 0 foo 1.0 

805 1 bar 2.0 

806 2 #baz 3.0 

807 

808 True, False, and NA values, and thousands separators have defaults, 

809 but can be explicitly specified, too. Supply the values you would like 

810 as strings or lists of strings! 

811 

812 >>> pd.read_csv("data.csv", na_values=["foo", "bar"]) # doctest: +SKIP 

813 Name Value 

814 0 NaN 1 

815 1 NaN 2 

816 2 #baz 3 

817 

818 Comment lines in the input file can be skipped using the `comment` argument. 

819 

820 >>> pd.read_csv("data.csv", comment="#") # doctest: +SKIP 

821 Name Value 

822 0 foo 1 

823 1 bar 2 

824 

825 By default, columns with dates will be read as ``object`` rather than ``datetime``. 

826 

827 >>> df = pd.read_csv("tmp.csv") # doctest: +SKIP 

828 

829 >>> df # doctest: +SKIP 

830 col 1 col 2 col 3 

831 0 10 10/04/2018 Sun 15 Jan 2023 

832 1 20 15/04/2018 Fri 12 May 2023 

833 

834 >>> df.dtypes # doctest: +SKIP 

835 col 1 int64 

836 col 2 object 

837 col 3 object 

838 dtype: object 

839 

840 Specific columns can be parsed as dates by using the `parse_dates` and 

841 `date_format` arguments. 

842 

843 >>> df = pd.read_csv( 

844 ... "tmp.csv", 

845 ... parse_dates=[1, 2], 

846 ... date_format={"col 2": "%d/%m/%Y", "col 3": "%a %d %b %Y"}, 

847 ... ) # doctest: +SKIP 

848 

849 >>> df.dtypes # doctest: +SKIP 

850 col 1 int64 

851 col 2 datetime64[ns] 

852 col 3 datetime64[ns] 

853 dtype: object 

854 """ 

855 # locals() should never be modified 

856 kwds = locals().copy() 

857 del kwds["filepath_or_buffer"] 

858 del kwds["sep"] 

859 

860 kwds_defaults = _refine_defaults_read( 

861 dialect, 

862 delimiter, 

863 engine, 

864 sep, 

865 on_bad_lines, 

866 names, 

867 defaults={"delimiter": ","}, 

868 dtype_backend=dtype_backend, 

869 ) 

870 kwds.update(kwds_defaults) 

871 

872 return _read(filepath_or_buffer, kwds) 

873 

874 

875@overload 

876def read_table( 

877 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

878 *, 

879 iterator: Literal[True], 

880 chunksize: int | None = ..., 

881 **kwds: Unpack[_read_shared[HashableT]], 

882) -> TextFileReader: ... 

883 

884 

885@overload 

886def read_table( 

887 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

888 *, 

889 iterator: bool = ..., 

890 chunksize: int, 

891 **kwds: Unpack[_read_shared[HashableT]], 

892) -> TextFileReader: ... 

893 

894 

895@overload 

896def read_table( 

897 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

898 *, 

899 iterator: Literal[False] = ..., 

900 chunksize: None = ..., 

901 **kwds: Unpack[_read_shared[HashableT]], 

902) -> DataFrame: ... 

903 

904 

905@overload 

906def read_table( 

907 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

908 *, 

909 iterator: bool = ..., 

910 chunksize: int | None = ..., 

911 **kwds: Unpack[_read_shared[HashableT]], 

912) -> DataFrame | TextFileReader: ... 

913 

914 

915@set_module("pandas") 

916def read_table( 

917 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

918 *, 

919 sep: str | None | lib.NoDefault = lib.no_default, 

920 delimiter: str | None | lib.NoDefault = None, 

921 # Column and Index Locations and Names 

922 header: int | Sequence[int] | None | Literal["infer"] = "infer", 

923 names: Sequence[Hashable] | None | lib.NoDefault = lib.no_default, 

924 index_col: IndexLabel | Literal[False] | None = None, 

925 usecols: UsecolsArgType = None, 

926 # General Parsing Configuration 

927 dtype: DtypeArg | None = None, 

928 engine: CSVEngine | None = None, 

929 converters: Mapping[HashableT, Callable] | None = None, 

930 true_values: list | None = None, 

931 false_values: list | None = None, 

932 skipinitialspace: bool = False, 

933 skiprows: list[int] | int | Callable[[Hashable], bool] | None = None, 

934 skipfooter: int = 0, 

935 nrows: int | None = None, 

936 # NA and Missing Data Handling 

937 na_values: ( 

938 Hashable | Iterable[Hashable] | Mapping[Hashable, Iterable[Hashable]] | None 

939 ) = None, 

940 keep_default_na: bool = True, 

941 na_filter: bool = True, 

942 skip_blank_lines: bool = True, 

943 # Datetime Handling 

944 parse_dates: bool | Sequence[Hashable] | None = None, 

945 date_format: str | dict[Hashable, str] | None = None, 

946 dayfirst: bool = False, 

947 cache_dates: bool = True, 

948 # Iteration 

949 iterator: bool = False, 

950 chunksize: int | None = None, 

951 # Quoting, Compression, and File Format 

952 compression: CompressionOptions = "infer", 

953 thousands: str | None = None, 

954 decimal: str = ".", 

955 lineterminator: str | None = None, 

956 quotechar: str = '"', 

957 quoting: int = csv.QUOTE_MINIMAL, 

958 doublequote: bool = True, 

959 escapechar: str | None = None, 

960 comment: str | None = None, 

961 encoding: str | None = None, 

962 encoding_errors: str | None = "strict", 

963 dialect: str | csv.Dialect | None = None, 

964 # Error Handling 

965 on_bad_lines: str = "error", 

966 # Internal 

967 low_memory: bool = _c_parser_defaults["low_memory"], 

968 memory_map: bool = False, 

969 float_precision: Literal["high", "legacy", "round_trip"] | None = None, 

970 storage_options: StorageOptions | None = None, 

971 dtype_backend: DtypeBackend | lib.NoDefault = lib.no_default, 

972) -> DataFrame | TextFileReader: 

973 """ 

974 Read general delimited file into DataFrame. 

975 

976 Also supports optionally iterating or breaking of the file 

977 into chunks. 

978 

979 Additional help can be found in the online docs for 

980 `IO Tools <https://pandas.pydata.org/pandas-docs/stable/user_guide/io.html>`_. 

981 

982 Parameters 

983 ---------- 

984 filepath_or_buffer : str, path object or file-like object 

985 Any valid string path is acceptable. The string could be a URL. Valid 

986 URL schemes include http, ftp, s3, gs, and file. For file URLs, a host is 

987 expected. A local file could be: file://localhost/path/to/table.csv. 

988 

989 If you want to pass in a path object, pandas accepts any ``os.PathLike``. 

990 

991 By file-like object, we refer to objects with a ``read()`` method, such as 

992 a file handle (e.g. via builtin ``open`` function) or ``StringIO``. 

993 sep : str, default '\\t' (tab-stop) 

994 Character or regex pattern to treat as the delimiter. ``sep=None`` detects 

995 the separator from the first valid row of the file with Python's builtin 

996 sniffer tool, ``csv.Sniffer``; it is supported only by the Python parsing 

997 engine, which will be used automatically. 

998 In addition, separators longer than 1 character and different from 

999 ``'\\s+'`` will be interpreted as regular expressions and will also force 

1000 the use of the Python parsing engine. Note that regex delimiters are prone 

1001 to ignoring quoted data. Regex example: ``'\\r\\t'``. 

1002 delimiter : str, optional 

1003 Alias for ``sep``. 

1004 header : int, Sequence of int, 'infer' or None, default 'infer' 

1005 Row number(s) containing column labels and marking the start of the 

1006 data (zero-indexed). Default behavior 

1007 is to infer the column names: if no ``names`` 

1008 are passed the behavior is identical to ``header=0`` and column 

1009 names are inferred from the first line of the file, if column 

1010 names are passed explicitly to ``names`` then the behavior is identical to 

1011 ``header=None``. Explicitly pass ``header=0`` to be able to 

1012 replace existing names. The header can be a list of integers that 

1013 specify row locations for a :class:`~pandas.MultiIndex` on the columns 

1014 e.g. ``[0, 1, 3]``. Intervening rows that are not specified will be 

1015 skipped (e.g. 2 in this example is skipped). Note that this 

1016 parameter ignores commented lines and empty lines if 

1017 ``skip_blank_lines=True``, so ``header=0`` denotes the first line of 

1018 data rather than the first line of the file. 

1019 

1020 When inferred from the file contents, headers are kept distinct from 

1021 each other by renaming duplicate names with a numeric suffix of the form 

1022 ``".{count}"`` starting from 1, e.g. ``"foo"`` and ``"foo.1"``. 

1023 Empty headers are named 

1024 ``"Unnamed: {i}"`` or ``"Unnamed: {i}_level_{level}"`` 

1025 in the case of MultiIndex columns. 

1026 names : Sequence of Hashable, optional 

1027 Sequence of column labels to apply. If the file contains a header row, 

1028 then you should explicitly pass ``header=0`` to override the column names. 

1029 Duplicates in this list are not allowed. 

1030 index_col : Hashable, Sequence of Hashable or False, optional 

1031 Column(s) to use as row label(s), denoted either by column labels or column 

1032 indices. If a sequence of labels or indices is given, 

1033 :class:`~pandas.MultiIndex` 

1034 will be formed for the row labels. 

1035 

1036 Note: ``index_col=False`` can be used to force pandas to *not* use the first 

1037 column as the index, e.g., when you have a malformed file with delimiters at 

1038 the end of each line. 

1039 usecols : Sequence of Hashable or Callable, optional 

1040 Subset of columns to select, denoted either by column labels or column indices. 

1041 If list-like, all elements must either 

1042 be positional (i.e. integer indices into the document columns) or strings 

1043 that correspond to column names provided either by the user in ``names`` or 

1044 inferred from the document header row(s). If ``names`` are given, the document 

1045 header row(s) are not taken into account. For example, a valid list-like 

1046 ``usecols`` parameter would be ``[0, 1, 2]`` or ``['foo', 'bar', 'baz']``. 

1047 Element order is ignored, so ``usecols=[0, 1]`` is the same as ``[1, 0]``. 

1048 To instantiate a :class:`~pandas.DataFrame` from ``data`` with element order 

1049 preserved use ``pd.read_csv(data, usecols=['foo', 'bar'])[['foo', 'bar']]`` 

1050 for columns in ``['foo', 'bar']`` order or 

1051 ``pd.read_csv(data, usecols=['foo', 'bar'])[['bar', 'foo']]`` 

1052 for ``['bar', 'foo']`` order. 

1053 

1054 If callable, the callable function will be evaluated against the column 

1055 names, returning names where the callable function evaluates to ``True``. An 

1056 example of a valid callable argument would be ``lambda x: x.upper() in 

1057 ['AAA', 'BBB', 'DDD']``. Using this parameter results in much faster 

1058 parsing time and lower memory usage. 

1059 dtype : dtype or dict of {Hashable : dtype}, optional 

1060 Data type(s) to apply to either the whole dataset or individual columns. 

1061 E.g., ``{'a': np.float64, 'b': np.int32, 'c': 'Int64'}`` 

1062 Use ``str`` or ``object`` together with suitable ``na_values`` settings 

1063 to preserve and not interpret ``dtype``. 

1064 If ``converters`` are specified, they will be applied INSTEAD 

1065 of ``dtype`` conversion. Specify a ``defaultdict`` as input where 

1066 the default determines the ``dtype`` of the columns which 

1067 are not explicitly listed. 

1068 engine : {'c', 'python', 'pyarrow'}, optional 

1069 Parser engine to use. The C and pyarrow engines are faster, 

1070 while the python engine 

1071 is currently more feature-complete. Multithreading is 

1072 currently only supported by 

1073 the pyarrow engine. The 'pyarrow' engine is an *experimental* engine, 

1074 and some features are unsupported, or may not work correctly, with this engine. 

1075 converters : dict of {Hashable : Callable}, optional 

1076 Functions for converting values in specified columns. Keys can either 

1077 be column labels or column indices. 

1078 true_values : list, optional 

1079 Values to consider as ``True`` in addition to 

1080 case-insensitive variants of 'True'. 

1081 false_values : list, optional 

1082 Values to consider as ``False`` in addition 

1083 to case-insensitive variants of 'False'. 

1084 skipinitialspace : bool, default False 

1085 Skip spaces after delimiter. 

1086 skiprows : int, list of int or Callable, optional 

1087 Line numbers to skip (0-indexed) or number of lines to skip (``int``) 

1088 at the start of the file. 

1089 

1090 If callable, the callable function will be evaluated against the row 

1091 indices, returning ``True`` if the row 

1092 should be skipped and ``False`` otherwise. 

1093 An example of a valid callable argument would be ``lambda x: x in [0, 2]``. 

1094 skipfooter : int, default 0 

1095 Number of lines at bottom of file to skip (Unsupported with ``engine='c'``). 

1096 nrows : int, optional 

1097 Number of rows of file to read. Useful for reading pieces of large files. 

1098 Refers to the number of data rows in the returned DataFrame, excluding: 

1099 

1100 * The header row containing column names. 

1101 * Rows before the header row, if ``header=1`` or larger. 

1102 

1103 Example usage: 

1104 

1105 * To read the first 999,999 (non-header) rows: 

1106 ``read_csv(..., nrows=999999)`` 

1107 

1108 * To read rows 1,000,000 through 1,999,999: 

1109 ``read_csv(..., skiprows=1000000, nrows=999999)`` 

1110 

1111 na_values : Hashable, Iterable of Hashable or dict of {Hashable : Iterable}, 

1112 optional 

1113 Additional strings to recognize as ``NA``/``NaN``. 

1114 If ``dict`` passed, specific 

1115 per-column ``NA`` values. By default the following values are interpreted as 

1116 ``NaN``: empty string, "NaN", "N/A", "NULL", and other 

1117 common representations of missing data. 

1118 keep_default_na : bool, default True 

1119 Whether or not to include the default ``NaN`` values when parsing the data. 

1120 Depending on whether ``na_values`` is passed in, the behavior is as follows: 

1121 

1122 * If ``keep_default_na`` is ``True``, 

1123 and ``na_values`` are specified, ``na_values`` 

1124 is appended to the default ``NaN`` values used for parsing. 

1125 * If ``keep_default_na`` is ``True``, and ``na_values`` are not specified, only 

1126 the default ``NaN`` values are used for parsing. 

1127 * If ``keep_default_na`` is ``False``, and ``na_values`` are specified, only 

1128 the ``NaN`` values specified ``na_values`` are used for parsing. 

1129 * If ``keep_default_na`` is ``False``, and ``na_values`` are not specified, no 

1130 strings will be parsed as ``NaN``. 

1131 

1132 Note that if ``na_filter`` is passed in as 

1133 ``False``, the ``keep_default_na`` and 

1134 ``na_values`` parameters will be ignored. 

1135 na_filter : bool, default True 

1136 Detect missing value markers (empty strings and the value of ``na_values``). In 

1137 data without any ``NA`` values, passing ``na_filter=False`` can improve the 

1138 performance of reading a large file. 

1139 skip_blank_lines : bool, default True 

1140 If ``True``, skip over blank lines rather than interpreting as ``NaN`` values. 

1141 parse_dates : bool, None, list of Hashable, default None 

1142 The behavior is as follows: 

1143 

1144 * ``bool``. If ``True`` -> try parsing the index. 

1145 * ``None``. Behaves like ``True`` if ``date_format`` is specified. 

1146 * ``list`` of ``int`` or names. 

1147 e.g. If ``[1, 2, 3]`` -> try parsing columns 1, 2, 3 

1148 each as a separate date column. 

1149 

1150 If a column or index cannot be represented as an array of ``datetime``, 

1151 say because of an unparsable value or a mixture of timezones, the column 

1152 or index will be returned unaltered as an ``object`` data type. For 

1153 non-standard ``datetime`` parsing, use :func:`~pandas.to_datetime` after 

1154 :func:`~pandas.read_csv`. 

1155 

1156 Note: A fast-path exists for iso8601-formatted dates. 

1157 date_format : str or dict of column -> format, optional 

1158 Format to use for parsing dates and/or times when used 

1159 in conjunction with ``parse_dates``. 

1160 The strftime to parse time, e.g. :const:`"%d/%m/%Y"`. See 

1161 `strftime documentation 

1162 <https://docs.python.org/3/library/datetime.html 

1163 #strftime-and-strptime-behavior>`_ for more information on choices, though 

1164 note that :const:`"%f"`` will parse all the way up to nanoseconds. 

1165 You can also pass: 

1166 

1167 - "ISO8601", to parse any `ISO8601 <https://en.wikipedia.org/wiki/ISO_8601>`_ 

1168 time string (not necessarily in exactly the same format); 

1169 - "mixed", to infer the format for each element individually. This is risky, 

1170 and you should probably use it along with `dayfirst`. 

1171 

1172 .. versionadded:: 2.0.0 

1173 dayfirst : bool, default False 

1174 DD/MM format dates, international and European format. 

1175 cache_dates : bool, default True 

1176 If ``True``, use a cache of unique, converted dates to apply the ``datetime`` 

1177 conversion. May produce significant speed-up when parsing duplicate 

1178 date strings, especially ones with timezone offsets. 

1179 

1180 iterator : bool, default False 

1181 Return ``TextFileReader`` object for iteration or getting chunks with 

1182 ``get_chunk()``. 

1183 chunksize : int, optional 

1184 Number of lines to read from the file per chunk. Passing a value will cause the 

1185 function to return a ``TextFileReader`` object for iteration. 

1186 See the `IO Tools docs 

1187 <https://pandas.pydata.org/pandas-docs/stable/io.html#io-chunking>`_ 

1188 for more information on ``iterator`` and ``chunksize``. 

1189 

1190 compression : str or dict, default 'infer' 

1191 For on-the-fly decompression of on-disk data. If 'infer' 

1192 and 'filepath_or_buffer' is 

1193 path-like, then detect compression from the following extensions: '.gz', 

1194 '.bz2', '.zip', '.xz', '.zst', '.tar', '.tar.gz', '.tar.xz' or '.tar.bz2' 

1195 (otherwise no compression). 

1196 If using 'zip' or 'tar', the ZIP file must contain 

1197 only one data file to be read in. 

1198 Set to ``None`` for no decompression. 

1199 Can also be a dict with key ``'method'`` set 

1200 to one of {``'zip'``, ``'gzip'``, ``'bz2'``, 

1201 ``'zstd'``, ``'xz'``, ``'tar'``} and 

1202 other key-value pairs are forwarded to 

1203 ``zipfile.ZipFile``, ``gzip.GzipFile``, 

1204 ``bz2.BZ2File``, ``zstandard.ZstdDecompressor``, ``lzma.LZMAFile`` or 

1205 ``tarfile.TarFile``, respectively. 

1206 As an example, the following could be passed for 

1207 Zstandard decompression using a 

1208 custom compression dictionary: 

1209 ``compression={'method': 'zstd', 'dict_data': my_compression_dict}``. 

1210 

1211 thousands : str (length 1), optional 

1212 Character acting as the thousands separator in numerical values. 

1213 decimal : str (length 1), default '.' 

1214 Character to recognize as decimal point (e.g., use ',' for European data). 

1215 lineterminator : str (length 1), optional 

1216 Character used to denote a line break. Only valid with C parser. 

1217 quotechar : str (length 1), optional 

1218 Character used to denote the start and end of a quoted item. Quoted 

1219 items can include the ``delimiter`` and it will be ignored. 

1220 quoting : {0 or csv.QUOTE_MINIMAL, 1 or csv.QUOTE_ALL, 2 or 

1221 csv.QUOTE_NONNUMERIC, 3 or csv.QUOTE_NONE}, default csv.QUOTE_MINIMAL 

1222 Control field quoting behavior per ``csv.QUOTE_*`` constants. Default is 

1223 ``csv.QUOTE_MINIMAL`` (i.e., 0) which 

1224 implies that only fields containing special 

1225 characters are quoted (e.g., characters defined 

1226 in ``quotechar``, ``delimiter``, 

1227 or ``lineterminator``. 

1228 doublequote : bool, default True 

1229 When ``quotechar`` is specified and ``quoting`` is not ``QUOTE_NONE``, indicate 

1230 whether or not to interpret two consecutive ``quotechar`` elements INSIDE a 

1231 field as a single ``quotechar`` element. 

1232 escapechar : str (length 1), optional 

1233 Character used to escape other characters. 

1234 comment : str (length 1), optional 

1235 Character indicating that the remainder of line should not be parsed. 

1236 If found at the beginning 

1237 of a line, the line will be ignored altogether. This parameter must be a 

1238 single character. Like empty lines (as long as ``skip_blank_lines=True``), 

1239 fully commented lines are ignored by the parameter ``header`` but not by 

1240 ``skiprows``. For example, if ``comment='#'``, parsing 

1241 ``#empty\\na,b,c\\n1,2,3`` with ``header=0`` will result in ``'a,b,c'`` being 

1242 treated as the header. 

1243 encoding : str, optional, default 'utf-8' 

1244 Encoding to use for UTF when reading/writing (ex. ``'utf-8'``). `List of Python 

1245 standard encodings 

1246 <https://docs.python.org/3/library/codecs.html#standard-encodings>`_ . 

1247 

1248 encoding_errors : str, optional, default 'strict' 

1249 How encoding errors are treated. `List of possible values 

1250 <https://docs.python.org/3/library/codecs.html#error-handlers>`_ . 

1251 

1252 dialect : str or csv.Dialect, optional 

1253 If provided, this parameter will override values (default or not) for the 

1254 following parameters: ``delimiter``, ``doublequote``, ``escapechar``, 

1255 ``skipinitialspace``, ``quotechar``, and ``quoting``. If it is necessary to 

1256 override values, a ``ParserWarning`` will be issued. See ``csv.Dialect`` 

1257 documentation for more details. 

1258 on_bad_lines : {'error', 'warn', 'skip'} or Callable, default 'error' 

1259 Specifies what to do upon encountering a bad 

1260 line (a line with too many fields). 

1261 Allowed values are: 

1262 

1263 - ``'error'``, raise an Exception when a bad line is encountered. 

1264 - ``'warn'``, raise a warning when a bad line is encountered and 

1265 skip that line. 

1266 - ``'skip'``, skip bad lines without raising or warning when they 

1267 are encountered. 

1268 - Callable, function that will process a single bad line. 

1269 - With ``engine='python'``, function with signature 

1270 ``(bad_line: list[str]) -> list[str] | None``. 

1271 ``bad_line`` is a list of strings split by the ``sep``. 

1272 If the function returns ``None``, the bad line will be ignored. 

1273 If the function returns a new ``list`` of strings with more elements than 

1274 expected, a ``ParserWarning`` will be emitted while 

1275 dropping extra elements. 

1276 - With ``engine='pyarrow'``, function with signature 

1277 as described in pyarrow documentation: `invalid_row_handler 

1278 <https://arrow.apache.org/docs/ 

1279 python/generated/pyarrow.csv.ParseOptions.html 

1280 #pyarrow.csv.ParseOptions.invalid_row_handler>`_. 

1281 

1282 .. versionadded:: 2.2.0 

1283 

1284 Callable for ``engine='pyarrow'`` 

1285 

1286 low_memory : bool, default True 

1287 Internally process the file in chunks, resulting in lower memory use 

1288 while parsing, but possibly mixed type inference. To ensure no mixed 

1289 types either set ``False``, or specify the type with the ``dtype`` parameter. 

1290 Note that the entire file is read into a single :class:`~pandas.DataFrame` 

1291 regardless, use the ``chunksize`` or ``iterator`` parameter 

1292 to return the data in 

1293 chunks. (Only valid with C parser). 

1294 memory_map : bool, default False 

1295 If a filepath is provided for ``filepath_or_buffer``, map the file object 

1296 directly onto memory and access the data directly from there. Using this 

1297 option can improve performance because there is no longer any I/O overhead. 

1298 float_precision : {'high', 'legacy', 'round_trip'}, optional 

1299 Specifies which converter the C engine should use for floating-point 

1300 values. The options are ``None`` or ``'high'`` for the ordinary converter, 

1301 ``'legacy'`` for the original lower precision pandas converter, and 

1302 ``'round_trip'`` for the round-trip converter. 

1303 

1304 storage_options : dict, optional 

1305 Extra options that make sense for a particular storage connection, e.g. 

1306 host, port, username, password, etc. For HTTP(S) URLs the key-value pairs 

1307 are forwarded to ``urllib.request.Request`` as header options. For other 

1308 URLs (e.g. starting with "s3://", and "gcs://") the key-value pairs are 

1309 forwarded to ``fsspec.open``. Please see ``fsspec`` and ``urllib`` for more 

1310 details, and for more examples on storage options refer `here 

1311 <https://pandas.pydata.org/docs/user_guide/io.html? 

1312 highlight=storage_options#reading-writing-remote-files>`_. 

1313 

1314 dtype_backend : {'numpy_nullable', 'pyarrow'} 

1315 Back-end data type applied to the resultant :class:`DataFrame` 

1316 (still experimental). If not specified, the default behavior 

1317 is to not use nullable data types. If specified, the behavior 

1318 is as follows: 

1319 

1320 * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame` 

1321 * ``"pyarrow"``: returns pyarrow-backed nullable 

1322 :class:`ArrowDtype` :class:`DataFrame` 

1323 

1324 .. versionadded:: 2.0 

1325 

1326 Returns 

1327 ------- 

1328 DataFrame or TextFileReader 

1329 A comma-separated values (csv) file is returned as two-dimensional 

1330 data structure with labeled axes. 

1331 

1332 See Also 

1333 -------- 

1334 DataFrame.to_csv : Write DataFrame to a comma-separated values (csv) file. 

1335 read_csv : Read a comma-separated values (csv) file into DataFrame. 

1336 read_fwf : Read a table of fixed-width formatted lines into DataFrame. 

1337 

1338 Examples 

1339 -------- 

1340 >>> pd.read_table("data.csv") # doctest: +SKIP 

1341 Name Value 

1342 0 foo 1 

1343 1 bar 2 

1344 2 #baz 3 

1345 

1346 Index and header can be specified via the `index_col` and `header` arguments. 

1347 

1348 >>> pd.read_table("data.csv", header=None) # doctest: +SKIP 

1349 0 1 

1350 0 Name Value 

1351 1 foo 1 

1352 2 bar 2 

1353 3 #baz 3 

1354 

1355 >>> pd.read_table("data.csv", index_col="Value") # doctest: +SKIP 

1356 Name 

1357 Value 

1358 1 foo 

1359 2 bar 

1360 3 #baz 

1361 

1362 Column types are inferred but can be explicitly specified using the dtype argument. 

1363 

1364 >>> pd.read_table("data.csv", dtype={"Value": float}) # doctest: +SKIP 

1365 Name Value 

1366 0 foo 1.0 

1367 1 bar 2.0 

1368 2 #baz 3.0 

1369 

1370 True, False, and NA values, and thousands separators have defaults, 

1371 but can be explicitly specified, too. Supply the values you would like 

1372 as strings or lists of strings! 

1373 

1374 >>> pd.read_table("data.csv", na_values=["foo", "bar"]) # doctest: +SKIP 

1375 Name Value 

1376 0 NaN 1 

1377 1 NaN 2 

1378 2 #baz 3 

1379 

1380 Comment lines in the input file can be skipped using the `comment` argument. 

1381 

1382 >>> pd.read_table("data.csv", comment="#") # doctest: +SKIP 

1383 Name Value 

1384 0 foo 1 

1385 1 bar 2 

1386 

1387 By default, columns with dates will be read as ``object`` rather than ``datetime``. 

1388 

1389 >>> df = pd.read_table("tmp.csv") # doctest: +SKIP 

1390 

1391 >>> df # doctest: +SKIP 

1392 col 1 col 2 col 3 

1393 0 10 10/04/2018 Sun 15 Jan 2023 

1394 1 20 15/04/2018 Fri 12 May 2023 

1395 

1396 >>> df.dtypes # doctest: +SKIP 

1397 col 1 int64 

1398 col 2 object 

1399 col 3 object 

1400 dtype: object 

1401 

1402 Specific columns can be parsed as dates by using the `parse_dates` and 

1403 `date_format` arguments. 

1404 

1405 >>> df = pd.read_table( 

1406 ... "tmp.csv", 

1407 ... parse_dates=[1, 2], 

1408 ... date_format={"col 2": "%d/%m/%Y", "col 3": "%a %d %b %Y"}, 

1409 ... ) # doctest: +SKIP 

1410 

1411 >>> df.dtypes # doctest: +SKIP 

1412 col 1 int64 

1413 col 2 datetime64[ns] 

1414 col 3 datetime64[ns] 

1415 dtype: object 

1416 """ 

1417 # locals() should never be modified 

1418 kwds = locals().copy() 

1419 del kwds["filepath_or_buffer"] 

1420 del kwds["sep"] 

1421 

1422 kwds_defaults = _refine_defaults_read( 

1423 dialect, 

1424 delimiter, 

1425 engine, 

1426 sep, 

1427 on_bad_lines, 

1428 names, 

1429 defaults={"delimiter": "\t"}, 

1430 dtype_backend=dtype_backend, 

1431 ) 

1432 kwds.update(kwds_defaults) 

1433 

1434 return _read(filepath_or_buffer, kwds) 

1435 

1436 

1437@overload 

1438def read_fwf( 

1439 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

1440 *, 

1441 colspecs: Sequence[tuple[int, int]] | str | None = ..., 

1442 widths: Sequence[int] | None = ..., 

1443 infer_nrows: int = ..., 

1444 iterator: Literal[True], 

1445 chunksize: int | None = ..., 

1446 **kwds: Unpack[_read_shared[HashableT]], 

1447) -> TextFileReader: ... 

1448 

1449 

1450@overload 

1451def read_fwf( 

1452 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

1453 *, 

1454 colspecs: Sequence[tuple[int, int]] | str | None = ..., 

1455 widths: Sequence[int] | None = ..., 

1456 infer_nrows: int = ..., 

1457 iterator: bool = ..., 

1458 chunksize: int, 

1459 **kwds: Unpack[_read_shared[HashableT]], 

1460) -> TextFileReader: ... 

1461 

1462 

1463@overload 

1464def read_fwf( 

1465 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

1466 *, 

1467 colspecs: Sequence[tuple[int, int]] | str | None = ..., 

1468 widths: Sequence[int] | None = ..., 

1469 infer_nrows: int = ..., 

1470 iterator: Literal[False] = ..., 

1471 chunksize: None = ..., 

1472 **kwds: Unpack[_read_shared[HashableT]], 

1473) -> DataFrame: ... 

1474 

1475 

1476@set_module("pandas") 

1477def read_fwf( 

1478 filepath_or_buffer: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str], 

1479 *, 

1480 colspecs: Sequence[tuple[int, int]] | str | None = "infer", 

1481 widths: Sequence[int] | None = None, 

1482 infer_nrows: int = 100, 

1483 iterator: bool = False, 

1484 chunksize: int | None = None, 

1485 **kwds: Unpack[_read_shared[HashableT]], 

1486) -> DataFrame | TextFileReader: 

1487 r""" 

1488 Read a table of fixed-width formatted lines into DataFrame. 

1489 

1490 Also supports optionally iterating or breaking of the file 

1491 into chunks. 

1492 

1493 Additional help can be found in the `online docs for IO Tools 

1494 <https://pandas.pydata.org/pandas-docs/stable/user_guide/io.html>`_. 

1495 

1496 Parameters 

1497 ---------- 

1498 filepath_or_buffer : str, path object, or file-like object 

1499 String, path object (implementing ``os.PathLike[str]``), or file-like 

1500 object implementing a text ``read()`` function.The string could be a URL. 

1501 Valid URL schemes include http, ftp, s3, and file. For file URLs, a host is 

1502 expected. A local file could be: 

1503 ``file://localhost/path/to/table.csv``. 

1504 colspecs : list of tuple (int, int) or 'infer'. optional 

1505 A list of tuples giving the extents of the fixed-width 

1506 fields of each line as half-open intervals (i.e., [from, to] ). 

1507 String value 'infer' can be used to instruct the parser to try 

1508 detecting the column specifications from the first 100 rows of 

1509 the data which are not being skipped via skiprows (default='infer'). 

1510 widths : list of int, optional 

1511 A list of field widths which can be used instead of 'colspecs' if 

1512 the intervals are contiguous. 

1513 infer_nrows : int, default 100 

1514 The number of rows to consider when letting the parser determine the 

1515 `colspecs`. 

1516 iterator : bool, default False 

1517 Return ``TextFileReader`` object for iteration or getting chunks with 

1518 ``get_chunk()``. 

1519 chunksize : int, optional 

1520 Number of lines to read from the file per chunk. 

1521 **kwds : optional 

1522 Optional keyword arguments can be passed to ``TextFileReader``. 

1523 

1524 Returns 

1525 ------- 

1526 DataFrame or TextFileReader 

1527 A comma-separated values (csv) file is returned as two-dimensional 

1528 data structure with labeled axes. 

1529 

1530 See Also 

1531 -------- 

1532 DataFrame.to_csv : Write DataFrame to a comma-separated values (csv) file. 

1533 read_csv : Read a comma-separated values (csv) file into DataFrame. 

1534 

1535 Examples 

1536 -------- 

1537 >>> pd.read_fwf("data.csv") # doctest: +SKIP 

1538 """ 

1539 # Check input arguments. 

1540 if colspecs is None and widths is None: 

1541 raise ValueError("Must specify either colspecs or widths") 

1542 if colspecs not in (None, "infer") and widths is not None: 

1543 raise ValueError("You must specify only one of 'widths' and 'colspecs'") 

1544 

1545 # Compute 'colspecs' from 'widths', if specified. 

1546 if widths is not None: 

1547 colspecs, col = [], 0 

1548 for w in widths: 

1549 colspecs.append((col, col + w)) 

1550 col += w 

1551 

1552 # for mypy 

1553 assert colspecs is not None 

1554 

1555 # GH#40830 

1556 # Ensure length of `colspecs` matches length of `names` 

1557 names = kwds.get("names") 

1558 if names is not None and names is not lib.no_default: 

1559 if len(names) != len(colspecs) and colspecs != "infer": 

1560 # need to check len(index_col) as it might contain 

1561 # unnamed indices, in which case it's name is not required 

1562 len_index = 0 

1563 if kwds.get("index_col") is not None: 

1564 index_col: Any = kwds.get("index_col") 

1565 if index_col is not False: 

1566 if not is_list_like(index_col): 

1567 len_index = 1 

1568 else: 

1569 # for mypy: handled in the if-branch 

1570 assert index_col is not lib.no_default 

1571 

1572 len_index = len(index_col) 

1573 if kwds.get("usecols") is None and len(names) + len_index != len(colspecs): 

1574 # If usecols is used colspec may be longer than names 

1575 raise ValueError("Length of colspecs must match length of names") 

1576 

1577 check_dtype_backend(kwds.setdefault("dtype_backend", lib.no_default)) 

1578 return _read( 

1579 filepath_or_buffer, 

1580 kwds 

1581 | { 

1582 "colspecs": colspecs, 

1583 "infer_nrows": infer_nrows, 

1584 "engine": "python-fwf", 

1585 "iterator": iterator, 

1586 "chunksize": chunksize, 

1587 }, 

1588 ) 

1589 

1590 

1591class TextFileReader(abc.Iterator): 

1592 """ 

1593 

1594 Passed dialect overrides any of the related parser options 

1595 

1596 """ 

1597 

1598 def __init__( 

1599 self, 

1600 f: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str] | list, 

1601 engine: CSVEngine | None = None, 

1602 **kwds, 

1603 ) -> None: 

1604 if engine is not None: 

1605 engine_specified = True 

1606 else: 

1607 engine = "python" 

1608 engine_specified = False 

1609 self.engine = engine 

1610 self._engine_specified = kwds.get("engine_specified", engine_specified) 

1611 

1612 _validate_skipfooter(kwds) 

1613 

1614 dialect = _extract_dialect(kwds) 

1615 if dialect is not None: 

1616 if engine == "pyarrow": 

1617 raise ValueError( 

1618 "The 'dialect' option is not supported with the 'pyarrow' engine" 

1619 ) 

1620 kwds = _merge_with_dialect_properties(dialect, kwds) 

1621 

1622 if kwds.get("header", "infer") == "infer": 

1623 kwds["header"] = 0 if kwds.get("names") is None else None 

1624 

1625 self.orig_options = kwds 

1626 

1627 # miscellanea 

1628 self._currow = 0 

1629 

1630 options = self._get_options_with_defaults(engine) 

1631 options["storage_options"] = kwds.get("storage_options", None) 

1632 

1633 self.chunksize = options.pop("chunksize", None) 

1634 self.nrows = options.pop("nrows", None) 

1635 

1636 self._check_file_or_buffer(f, engine) 

1637 self.options, self.engine = self._clean_options(options, engine) 

1638 

1639 if "has_index_names" in kwds: 

1640 self.options["has_index_names"] = kwds["has_index_names"] 

1641 

1642 self.handles: IOHandles | None = None 

1643 self._engine = self._make_engine(f, self.engine) 

1644 

1645 def close(self) -> None: 

1646 if self.handles is not None: 

1647 self.handles.close() 

1648 self._engine.close() 

1649 

1650 def _get_options_with_defaults(self, engine: CSVEngine) -> dict[str, Any]: 

1651 kwds = self.orig_options 

1652 

1653 options = {} 

1654 default: object | None 

1655 

1656 for argname, default in parser_defaults.items(): 

1657 value = kwds.get(argname, default) 

1658 

1659 # see gh-12935 

1660 if ( 

1661 engine == "pyarrow" 

1662 and argname in _pyarrow_unsupported 

1663 and value != default 

1664 and value != getattr(value, "value", default) 

1665 ): 

1666 raise ValueError( 

1667 f"The {argname!r} option is not supported with the 'pyarrow' engine" 

1668 ) 

1669 options[argname] = value 

1670 

1671 for argname, default in _c_parser_defaults.items(): 

1672 if argname in kwds: 

1673 value = kwds[argname] 

1674 

1675 if engine != "c" and value != default: 

1676 # TODO: Refactor this logic, its pretty convoluted 

1677 if "python" in engine and argname not in _python_unsupported: 

1678 pass 

1679 elif "pyarrow" in engine and argname not in _pyarrow_unsupported: 

1680 pass 

1681 else: 

1682 raise ValueError( 

1683 f"The {argname!r} option is not supported with the " 

1684 f"{engine!r} engine" 

1685 ) 

1686 else: 

1687 value = default 

1688 options[argname] = value 

1689 

1690 if engine == "python-fwf": 

1691 for argname, default in _fwf_defaults.items(): 

1692 options[argname] = kwds.get(argname, default) 

1693 

1694 return options 

1695 

1696 def _check_file_or_buffer(self, f, engine: CSVEngine) -> None: 

1697 # see gh-16530 

1698 if is_file_like(f) and engine != "c" and not hasattr(f, "__iter__"): 

1699 # The C engine doesn't need the file-like to have the "__iter__" 

1700 # attribute. However, the Python engine needs "__iter__(...)" 

1701 # when iterating through such an object, meaning it 

1702 # needs to have that attribute 

1703 raise ValueError( 

1704 "The 'python' engine cannot iterate through this file buffer." 

1705 ) 

1706 if hasattr(f, "encoding"): 

1707 file_encoding = f.encoding 

1708 orig_reader_enc = self.orig_options.get("encoding", None) 

1709 any_none = file_encoding is None or orig_reader_enc is None 

1710 if file_encoding != orig_reader_enc and not any_none: 

1711 file_path = getattr(f, "name", None) 

1712 raise ValueError( 

1713 f"The specified reader encoding {orig_reader_enc} is different " 

1714 f"from the encoding {file_encoding} of file {file_path}." 

1715 ) 

1716 

1717 def _clean_options( 

1718 self, options: dict[str, Any], engine: CSVEngine 

1719 ) -> tuple[dict[str, Any], CSVEngine]: 

1720 result = options.copy() 

1721 

1722 fallback_reason = None 

1723 

1724 # C engine not supported yet 

1725 if engine == "c": 

1726 if options["skipfooter"] > 0: 

1727 fallback_reason = "the 'c' engine does not support skipfooter" 

1728 engine = "python" 

1729 

1730 sep = options["delimiter"] 

1731 

1732 if sep is None: 

1733 # sniffing the separator with csv.Sniffer is python-engine only 

1734 if engine in ("c", "pyarrow"): 

1735 fallback_reason = f"the '{engine}' engine does not support sep=None" 

1736 engine = "python" 

1737 elif len(sep) > 1: 

1738 if engine == "c" and sep == r"\s+": 

1739 # delim_whitespace passed on to pandas._libs.parsers.TextReader 

1740 result["delim_whitespace"] = True 

1741 del result["delimiter"] 

1742 elif engine not in ("python", "python-fwf"): 

1743 # wait until regex engine integrated 

1744 fallback_reason = ( 

1745 f"the '{engine}' engine does not support " 

1746 "regex separators (separators > 1 char and " 

1747 r"different from '\s+' are interpreted as regex)" 

1748 ) 

1749 engine = "python" 

1750 else: 

1751 encodeable = True 

1752 encoding = sys.getfilesystemencoding() or "utf-8" 

1753 try: 

1754 if len(sep.encode(encoding)) > 1: 

1755 encodeable = False 

1756 except UnicodeDecodeError: 

1757 encodeable = False 

1758 if not encodeable and engine not in ("python", "python-fwf"): 

1759 fallback_reason = ( 

1760 f"the separator encoded in {encoding} " 

1761 f"is > 1 char long, and the '{engine}' engine " 

1762 "does not support such separators" 

1763 ) 

1764 engine = "python" 

1765 

1766 quotechar = options["quotechar"] 

1767 if quotechar is not None and isinstance(quotechar, (str, bytes)): 

1768 if ( 

1769 len(quotechar) == 1 

1770 and ord(quotechar) > 127 

1771 and engine not in ("python", "python-fwf") 

1772 ): 

1773 fallback_reason = ( 

1774 "ord(quotechar) > 127, meaning the " 

1775 "quotechar is larger than one byte, " 

1776 f"and the '{engine}' engine does not support such quotechars" 

1777 ) 

1778 engine = "python" 

1779 

1780 if fallback_reason and self._engine_specified: 

1781 raise ValueError(fallback_reason) 

1782 

1783 if engine == "c": 

1784 for arg in _c_unsupported: 

1785 del result[arg] 

1786 

1787 if "python" in engine: 

1788 for arg in _python_unsupported: 

1789 if fallback_reason and result[arg] != _c_parser_defaults.get(arg): 

1790 raise ValueError( 

1791 "Falling back to the 'python' engine because " 

1792 f"{fallback_reason}, but this causes {arg!r} to be " 

1793 "ignored as it is not supported by the 'python' engine." 

1794 ) 

1795 del result[arg] 

1796 

1797 if fallback_reason: 

1798 warnings.warn( 

1799 ( 

1800 "Falling back to the 'python' engine because " 

1801 f"{fallback_reason}; you can avoid this warning by specifying " 

1802 "engine='python'." 

1803 ), 

1804 ParserWarning, 

1805 stacklevel=find_stack_level(), 

1806 ) 

1807 

1808 index_col = options["index_col"] 

1809 names = options["names"] 

1810 converters = options["converters"] 

1811 na_values = options["na_values"] 

1812 skiprows = options["skiprows"] 

1813 

1814 validate_header_arg(options["header"]) 

1815 

1816 if index_col is True: 

1817 raise ValueError("The value of index_col couldn't be 'True'") 

1818 if is_index_col(index_col): 

1819 if not isinstance(index_col, (list, tuple, np.ndarray)): 

1820 index_col = [index_col] 

1821 result["index_col"] = index_col 

1822 

1823 names = list(names) if names is not None else names 

1824 

1825 # type conversion-related 

1826 if converters is not None: 

1827 if not isinstance(converters, dict): 

1828 raise TypeError( 

1829 "Type converters must be a dict or subclass, " 

1830 f"input was a {type(converters).__name__}" 

1831 ) 

1832 else: 

1833 converters = {} 

1834 

1835 # Converting values to NA 

1836 keep_default_na = options["keep_default_na"] 

1837 floatify = engine != "pyarrow" 

1838 na_values, na_fvalues = _clean_na_values( 

1839 na_values, keep_default_na, floatify=floatify 

1840 ) 

1841 

1842 # handle skiprows; this is internally handled by the 

1843 # c-engine, so only need for python and pyarrow parsers 

1844 if engine == "pyarrow": 

1845 if not is_integer(skiprows) and skiprows is not None: 

1846 # pyarrow expects skiprows to be passed as an integer 

1847 raise ValueError( 

1848 "skiprows argument must be an integer when using engine='pyarrow'" 

1849 ) 

1850 else: 

1851 if is_integer(skiprows): 

1852 skiprows = range(skiprows) 

1853 if skiprows is None: 

1854 skiprows = set() 

1855 elif not callable(skiprows): 

1856 skiprows = set(skiprows) 

1857 

1858 # put stuff back 

1859 result["names"] = names 

1860 result["converters"] = converters 

1861 result["na_values"] = na_values 

1862 result["na_fvalues"] = na_fvalues 

1863 result["skiprows"] = skiprows 

1864 

1865 return result, engine 

1866 

1867 def __next__(self) -> DataFrame: 

1868 try: 

1869 return self.get_chunk() 

1870 except StopIteration: 

1871 self.close() 

1872 raise 

1873 

1874 def _make_engine( 

1875 self, 

1876 f: FilePath | ReadCsvBuffer[bytes] | ReadCsvBuffer[str] | list | IO, 

1877 engine: CSVEngine = "c", 

1878 ) -> ParserBase: 

1879 mapping: dict[str, type[ParserBase]] = { 

1880 "c": CParserWrapper, 

1881 "python": PythonParser, 

1882 "pyarrow": ArrowParserWrapper, 

1883 "python-fwf": FixedWidthFieldParser, 

1884 } 

1885 

1886 if engine not in mapping: 

1887 raise ValueError( 

1888 f"Unknown engine: {engine} (valid options are {mapping.keys()})" 

1889 ) 

1890 if not isinstance(f, list): 

1891 # open file here 

1892 is_text = True 

1893 mode = "r" 

1894 if engine == "pyarrow": 

1895 is_text = False 

1896 mode = "rb" 

1897 elif ( 

1898 engine == "c" 

1899 and self.options.get("encoding", "utf-8") == "utf-8" 

1900 and isinstance(stringify_path(f), str) 

1901 ): 

1902 # c engine can decode utf-8 bytes, adding TextIOWrapper makes 

1903 # the c-engine especially for memory_map=True far slower 

1904 is_text = False 

1905 if "b" not in mode: 

1906 mode += "b" 

1907 self.handles = get_handle( 

1908 f, 

1909 mode, 

1910 encoding=self.options.get("encoding", None), 

1911 compression=self.options.get("compression", None), 

1912 memory_map=self.options.get("memory_map", False), 

1913 is_text=is_text, 

1914 errors=self.options.get("encoding_errors", "strict"), 

1915 storage_options=self.options.get("storage_options", None), 

1916 ) 

1917 assert self.handles is not None 

1918 f = self.handles.handle 

1919 

1920 elif engine != "python": 

1921 msg = f"Invalid file path or buffer object type: {type(f)}" 

1922 raise ValueError(msg) 

1923 

1924 try: 

1925 return mapping[engine](f, **self.options) 

1926 except Exception: 

1927 if self.handles is not None: 

1928 self.handles.close() 

1929 raise 

1930 

1931 def _failover_to_python(self) -> None: 

1932 raise AbstractMethodError(self) 

1933 

1934 def read(self, nrows: int | None = None) -> DataFrame: 

1935 if self.engine == "pyarrow": 

1936 try: 

1937 # error: "ParserBase" has no attribute "read" 

1938 df = self._engine.read() # type: ignore[attr-defined] 

1939 except Exception: 

1940 self.close() 

1941 raise 

1942 else: 

1943 nrows = validate_integer("nrows", nrows) 

1944 try: 

1945 # error: "ParserBase" has no attribute "read" 

1946 ( 

1947 index, 

1948 columns, 

1949 col_dict, 

1950 ) = self._engine.read( # type: ignore[attr-defined] 

1951 nrows 

1952 ) 

1953 except Exception: 

1954 self.close() 

1955 raise 

1956 

1957 if index is None: 

1958 if col_dict: 

1959 # Any column is actually fine: 

1960 new_rows = len(next(iter(col_dict.values()))) 

1961 index = RangeIndex(self._currow, self._currow + new_rows) 

1962 else: 

1963 new_rows = 0 

1964 else: 

1965 new_rows = len(index) 

1966 

1967 if hasattr(self, "orig_options"): 

1968 dtype_arg = self.orig_options.get("dtype", None) 

1969 else: 

1970 dtype_arg = None 

1971 

1972 if isinstance(dtype_arg, dict): 

1973 dtype = defaultdict(lambda: None) # type: ignore[var-annotated] 

1974 dtype.update(dtype_arg) 

1975 elif dtype_arg is not None and pandas_dtype(dtype_arg) in ( 

1976 np.str_, 

1977 np.object_, 

1978 ): 

1979 dtype = defaultdict(lambda: dtype_arg) 

1980 else: 

1981 dtype = None 

1982 

1983 if dtype is not None: 

1984 new_col_dict = {} 

1985 for k, v in col_dict.items(): 

1986 d = ( 

1987 dtype[k] 

1988 if pandas_dtype(dtype[k]) in (np.str_, np.object_) 

1989 else None 

1990 ) 

1991 new_col_dict[k] = Series(v, index=index, dtype=d, copy=False) 

1992 else: 

1993 new_col_dict = col_dict 

1994 

1995 df = DataFrame( 

1996 new_col_dict, 

1997 columns=columns, 

1998 index=index, 

1999 copy=False, 

2000 ) 

2001 

2002 self._currow += new_rows 

2003 return df 

2004 

2005 def get_chunk(self, size: int | None = None) -> DataFrame: 

2006 if size is None: 

2007 size = self.chunksize 

2008 if self.nrows is not None: 

2009 if self._currow >= self.nrows: 

2010 raise StopIteration 

2011 if size is None: 

2012 size = self.nrows - self._currow 

2013 else: 

2014 size = min(size, self.nrows - self._currow) 

2015 return self.read(nrows=size) 

2016 

2017 def __enter__(self) -> Self: 

2018 return self 

2019 

2020 def __exit__( 

2021 self, 

2022 exc_type: type[BaseException] | None, 

2023 exc_value: BaseException | None, 

2024 traceback: TracebackType | None, 

2025 ) -> None: 

2026 self.close() 

2027 

2028 

2029def TextParser(*args, **kwds) -> TextFileReader: 

2030 """ 

2031 Converts lists of lists/tuples into DataFrames with proper type inference 

2032 and optional (e.g. string to datetime) conversion. Also enables iterating 

2033 lazily over chunks of large files 

2034 

2035 Parameters 

2036 ---------- 

2037 data : file-like object or list 

2038 delimiter : separator character to use 

2039 dialect : str or csv.Dialect instance, optional 

2040 Ignored if delimiter is longer than 1 character 

2041 names : sequence, default 

2042 header : int, default 0 

2043 Row to use to parse column labels. Defaults to the first row. Prior 

2044 rows will be discarded 

2045 index_col : int or list, optional 

2046 Column or columns to use as the (possibly hierarchical) index 

2047 has_index_names: bool, default False 

2048 True if the cols defined in index_col have an index name and are 

2049 not in the header. 

2050 na_values : scalar, str, list-like, or dict, optional 

2051 Additional strings to recognize as NA/NaN. 

2052 keep_default_na : bool, default True 

2053 thousands : str, optional 

2054 Thousands separator 

2055 comment : str, optional 

2056 Comment out remainder of line 

2057 parse_dates : bool, default False 

2058 date_format : str or dict of column -> format, default ``None`` 

2059 

2060 .. versionadded:: 2.0.0 

2061 skiprows : list of integers 

2062 Row numbers to skip 

2063 skipfooter : int 

2064 Number of line at bottom of file to skip 

2065 converters : dict, optional 

2066 Dict of functions for converting values in certain columns. Keys can 

2067 either be integers or column labels, values are functions that take one 

2068 input argument, the cell (not column) content, and return the 

2069 transformed content. 

2070 encoding : str, optional 

2071 Encoding to use for UTF when reading/writing (ex. 'utf-8') 

2072 float_precision : str, optional 

2073 Specifies which converter the C engine should use for floating-point 

2074 values. The options are `None` or `high` for the ordinary converter, 

2075 `legacy` for the original lower precision pandas converter, and 

2076 `round_trip` for the round-trip converter. 

2077 """ 

2078 kwds["engine"] = "python" 

2079 return TextFileReader(*args, **kwds) 

2080 

2081 

2082def _clean_na_values(na_values, keep_default_na: bool = True, floatify: bool = True): 

2083 na_fvalues: set | dict 

2084 if na_values is None: 

2085 if keep_default_na: 

2086 na_values = STR_NA_VALUES 

2087 else: 

2088 na_values = set() 

2089 na_fvalues = set() 

2090 elif isinstance(na_values, dict): 

2091 old_na_values = na_values.copy() 

2092 na_values = {} # Prevent aliasing. 

2093 

2094 # Convert the values in the na_values dictionary 

2095 # into array-likes for further use. This is also 

2096 # where we append the default NaN values, provided 

2097 # that `keep_default_na=True`. 

2098 for k, v in old_na_values.items(): 

2099 if not is_list_like(v): 

2100 v = [v] 

2101 

2102 if keep_default_na: 

2103 v = set(v) | STR_NA_VALUES 

2104 

2105 na_values[k] = _stringify_na_values(v, floatify) 

2106 na_fvalues = {k: _floatify_na_values(v) for k, v in na_values.items()} 

2107 else: 

2108 if not is_list_like(na_values): 

2109 na_values = [na_values] 

2110 na_values = _stringify_na_values(na_values, floatify) 

2111 if keep_default_na: 

2112 na_values = na_values | STR_NA_VALUES 

2113 

2114 na_fvalues = _floatify_na_values(na_values) 

2115 

2116 return na_values, na_fvalues 

2117 

2118 

2119def _floatify_na_values(na_values) -> set[float]: 

2120 # create float versions of the na_values 

2121 result = set() 

2122 for v in na_values: 

2123 try: 

2124 v = float(v) 

2125 if not np.isnan(v): 

2126 result.add(v) 

2127 except (TypeError, ValueError, OverflowError): 

2128 pass 

2129 return result 

2130 

2131 

2132def _stringify_na_values(na_values, floatify: bool) -> set[str | float]: 

2133 """return a stringified and numeric for these values""" 

2134 result: list[str | float] = [] 

2135 for x in na_values: 

2136 result.append(str(x)) 

2137 result.append(x) 

2138 try: 

2139 v = float(x) 

2140 

2141 # we are like 999 here 

2142 if v == int(v): 

2143 v = int(v) 

2144 result.append(f"{v}.0") 

2145 result.append(str(v)) 

2146 

2147 if floatify: 

2148 result.append(v) 

2149 except (TypeError, ValueError, OverflowError): 

2150 pass 

2151 if floatify: 

2152 try: 

2153 result.append(int(x)) 

2154 except (TypeError, ValueError, OverflowError): 

2155 pass 

2156 return set(result) 

2157 

2158 

2159def _refine_defaults_read( 

2160 dialect: str | csv.Dialect | None, 

2161 delimiter: str | None | lib.NoDefault, 

2162 engine: CSVEngine | None, 

2163 sep: str | None | lib.NoDefault, 

2164 on_bad_lines: str | Callable, 

2165 names: Sequence[Hashable] | None | lib.NoDefault, 

2166 defaults: dict[str, Any], 

2167 dtype_backend: DtypeBackend | lib.NoDefault, 

2168): 

2169 """Validate/refine default values of input parameters of read_csv, read_table. 

2170 

2171 Parameters 

2172 ---------- 

2173 dialect : str or csv.Dialect 

2174 If provided, this parameter will override values (default or not) for the 

2175 following parameters: `delimiter`, `doublequote`, `escapechar`, 

2176 `skipinitialspace`, `quotechar`, and `quoting`. If it is necessary to 

2177 override values, a ParserWarning will be issued. See csv.Dialect 

2178 documentation for more details. 

2179 delimiter : str or object 

2180 Alias for sep. 

2181 engine : {'c', 'python'} 

2182 Parser engine to use. The C engine is faster while the python engine is 

2183 currently more feature-complete. 

2184 sep : str or object 

2185 A delimiter provided by the user (str) or a sentinel value, i.e. 

2186 pandas._libs.lib.no_default. 

2187 on_bad_lines : str, callable 

2188 An option for handling bad lines or a sentinel value(None). 

2189 names : array-like, optional 

2190 List of column names to use. If the file contains a header row, 

2191 then you should explicitly pass ``header=0`` to override the column names. 

2192 Duplicates in this list are not allowed. 

2193 defaults: dict 

2194 Default values of input parameters. 

2195 

2196 Returns 

2197 ------- 

2198 kwds : dict 

2199 Input parameters with correct values. 

2200 """ 

2201 # fix types for sep, delimiter to Union(str, Any) 

2202 delim_default = defaults["delimiter"] 

2203 kwds: dict[str, Any] = {} 

2204 # gh-23761 

2205 # 

2206 # When a dialect is passed, it overrides any of the overlapping 

2207 # parameters passed in directly. We don't want to warn if the 

2208 # default parameters were passed in (since it probably means 

2209 # that the user didn't pass them in explicitly in the first place). 

2210 # 

2211 # "delimiter" is the annoying corner case because we alias it to 

2212 # "sep" before doing comparison to the dialect values later on. 

2213 # Thus, we need a flag to indicate that we need to "override" 

2214 # the comparison to dialect values by checking if default values 

2215 # for BOTH "delimiter" and "sep" were provided. 

2216 if dialect is not None: 

2217 kwds["sep_override"] = delimiter is None and ( 

2218 sep is lib.no_default or sep == delim_default 

2219 ) 

2220 

2221 if delimiter and (sep is not lib.no_default): 

2222 raise ValueError("Specified a sep and a delimiter; you can only specify one.") 

2223 

2224 kwds["names"] = None if names is lib.no_default else names 

2225 

2226 # Alias sep -> delimiter. 

2227 if delimiter is None: 

2228 delimiter = sep 

2229 

2230 if delimiter == "\n": 

2231 raise ValueError( 

2232 r"Specified \n as separator or delimiter. This forces the python engine " 

2233 "which does not accept a line terminator. Hence it is not allowed to use " 

2234 "the line terminator as separator.", 

2235 ) 

2236 

2237 if delimiter is lib.no_default: 

2238 # assign default separator value 

2239 kwds["delimiter"] = delim_default 

2240 else: 

2241 kwds["delimiter"] = delimiter 

2242 

2243 if engine is not None: 

2244 kwds["engine_specified"] = True 

2245 else: 

2246 kwds["engine"] = "c" 

2247 kwds["engine_specified"] = False 

2248 

2249 if on_bad_lines == "error": 

2250 kwds["on_bad_lines"] = ParserBase.BadLineHandleMethod.ERROR 

2251 elif on_bad_lines == "warn": 

2252 kwds["on_bad_lines"] = ParserBase.BadLineHandleMethod.WARN 

2253 elif on_bad_lines == "skip": 

2254 kwds["on_bad_lines"] = ParserBase.BadLineHandleMethod.SKIP 

2255 elif callable(on_bad_lines): 

2256 if engine not in ["python", "pyarrow"]: 

2257 raise ValueError( 

2258 "on_bad_line can only be a callable function " 

2259 "if engine='python' or 'pyarrow'" 

2260 ) 

2261 kwds["on_bad_lines"] = on_bad_lines 

2262 else: 

2263 raise ValueError(f"Argument {on_bad_lines} is invalid for on_bad_lines") 

2264 

2265 check_dtype_backend(dtype_backend) 

2266 

2267 kwds["dtype_backend"] = dtype_backend 

2268 

2269 return kwds 

2270 

2271 

2272def _extract_dialect(kwds: dict[str, str | csv.Dialect]) -> csv.Dialect | None: 

2273 """ 

2274 Extract concrete csv dialect instance. 

2275 

2276 Returns 

2277 ------- 

2278 csv.Dialect or None 

2279 """ 

2280 if kwds.get("dialect") is None: 

2281 return None 

2282 

2283 dialect = kwds["dialect"] 

2284 if isinstance(dialect, str) and dialect in csv.list_dialects(): 

2285 # get_dialect is typed to return a `_csv.Dialect` for some reason in typeshed 

2286 tdialect = cast(csv.Dialect, csv.get_dialect(dialect)) 

2287 _validate_dialect(tdialect) 

2288 

2289 else: 

2290 _validate_dialect(dialect) 

2291 tdialect = cast(csv.Dialect, dialect) 

2292 

2293 return tdialect 

2294 

2295 

2296MANDATORY_DIALECT_ATTRS = ( 

2297 "delimiter", 

2298 "doublequote", 

2299 "escapechar", 

2300 "skipinitialspace", 

2301 "quotechar", 

2302 "quoting", 

2303) 

2304 

2305 

2306def _validate_dialect(dialect: csv.Dialect | str) -> None: 

2307 """ 

2308 Validate csv dialect instance. 

2309 

2310 Raises 

2311 ------ 

2312 ValueError 

2313 If incorrect dialect is provided. 

2314 """ 

2315 for param in MANDATORY_DIALECT_ATTRS: 

2316 if not hasattr(dialect, param): 

2317 raise ValueError(f"Invalid dialect {dialect} provided") 

2318 

2319 

2320def _merge_with_dialect_properties( 

2321 dialect: csv.Dialect, 

2322 defaults: dict[str, Any], 

2323) -> dict[str, Any]: 

2324 """ 

2325 Merge default kwargs in TextFileReader with dialect parameters. 

2326 

2327 Parameters 

2328 ---------- 

2329 dialect : csv.Dialect 

2330 Concrete csv dialect. See csv.Dialect documentation for more details. 

2331 defaults : dict 

2332 Keyword arguments passed to TextFileReader. 

2333 

2334 Returns 

2335 ------- 

2336 kwds : dict 

2337 Updated keyword arguments, merged with dialect parameters. 

2338 """ 

2339 kwds = defaults.copy() 

2340 

2341 for param in MANDATORY_DIALECT_ATTRS: 

2342 dialect_val = getattr(dialect, param) 

2343 

2344 parser_default = parser_defaults[param] 

2345 provided = kwds.get(param, parser_default) 

2346 

2347 # Messages for conflicting values between the dialect 

2348 # instance and the actual parameters provided. 

2349 conflict_msgs = [] 

2350 

2351 # Don't warn if the default parameter was passed in, 

2352 # even if it conflicts with the dialect (gh-23761). 

2353 if provided not in (parser_default, dialect_val): 

2354 msg = ( 

2355 f"Conflicting values for '{param}': '{provided}' was " 

2356 f"provided, but the dialect specifies '{dialect_val}'. " 

2357 "Using the dialect-specified value." 

2358 ) 

2359 

2360 # Annoying corner case for not warning about 

2361 # conflicts between dialect and delimiter parameter. 

2362 # Refer to the outer "_read_" function for more info. 

2363 if not (param == "delimiter" and kwds.pop("sep_override", False)): 

2364 conflict_msgs.append(msg) 

2365 

2366 if conflict_msgs: 

2367 warnings.warn( 

2368 "\n\n".join(conflict_msgs), ParserWarning, stacklevel=find_stack_level() 

2369 ) 

2370 kwds[param] = dialect_val 

2371 return kwds 

2372 

2373 

2374def _validate_skipfooter(kwds: dict[str, Any]) -> None: 

2375 """ 

2376 Check whether skipfooter is compatible with other kwargs in TextFileReader. 

2377 

2378 Parameters 

2379 ---------- 

2380 kwds : dict 

2381 Keyword arguments passed to TextFileReader. 

2382 

2383 Raises 

2384 ------ 

2385 ValueError 

2386 If skipfooter is not compatible with other parameters. 

2387 """ 

2388 if kwds.get("skipfooter"): 

2389 if kwds.get("iterator") or kwds.get("chunksize"): 

2390 raise ValueError("'skipfooter' not supported for iteration") 

2391 if kwds.get("nrows"): 

2392 raise ValueError("'skipfooter' not supported with 'nrows'")