Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/pandas/core/arrays/arrow/array.py: 16%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

1520 statements  

1from __future__ import annotations 

2 

3from datetime import ( 

4 date, 

5 datetime, 

6) 

7import functools 

8import operator 

9import re 

10import textwrap 

11from typing import ( 

12 TYPE_CHECKING, 

13 Any, 

14 Literal, 

15 Self, 

16 cast, 

17 overload, 

18) 

19import unicodedata 

20import warnings 

21 

22import numpy as np 

23 

24from pandas._config import is_nan_na 

25 

26from pandas._libs import lib 

27from pandas._libs.missing import is_pdna_or_none 

28from pandas._libs.tslibs import ( 

29 Timedelta, 

30 Timestamp, 

31 timezones, 

32) 

33from pandas.compat import ( 

34 HAS_PYARROW, 

35 PYARROW_MIN_VERSION, 

36 pa_version_under16p0, 

37 pa_version_under21p0, 

38) 

39from pandas.errors import Pandas4Warning 

40from pandas.util._decorators import ( 

41 doc, 

42 set_module, 

43) 

44from pandas.util._exceptions import find_stack_level 

45 

46from pandas.core.dtypes.cast import ( 

47 can_hold_element, 

48 construct_1d_object_array_from_listlike, 

49 infer_dtype_from_scalar, 

50) 

51from pandas.core.dtypes.common import ( 

52 is_array_like, 

53 is_bool_dtype, 

54 is_float_dtype, 

55 is_integer, 

56 is_list_like, 

57 is_numeric_dtype, 

58 is_scalar, 

59 is_string_dtype, 

60 pandas_dtype, 

61) 

62from pandas.core.dtypes.dtypes import DatetimeTZDtype 

63from pandas.core.dtypes.missing import isna 

64 

65from pandas.core import ( 

66 algorithms as algos, 

67 missing, 

68 ops, 

69 roperator, 

70) 

71from pandas.core.algorithms import map_array 

72from pandas.core.arraylike import OpsMixin 

73from pandas.core.arrays._arrow_string_mixins import ArrowStringArrayMixin 

74from pandas.core.arrays._utils import to_numpy_dtype_inference 

75from pandas.core.arrays.base import ( 

76 ExtensionArray, 

77 ExtensionArraySupportsAnyAll, 

78) 

79from pandas.core.arrays.masked import BaseMaskedArray 

80from pandas.core.arrays.string_ import StringDtype 

81import pandas.core.common as com 

82from pandas.core.construction import extract_array 

83from pandas.core.indexers import ( 

84 check_array_indexer, 

85 getitem_returns_view, 

86 unpack_tuple_and_ellipses, 

87 validate_indices, 

88) 

89from pandas.core.nanops import check_below_min_count 

90 

91from pandas.io._util import _arrow_dtype_mapping 

92from pandas.tseries.frequencies import to_offset 

93 

94if HAS_PYARROW: 

95 import pyarrow as pa 

96 import pyarrow.compute as pc 

97 

98 from pandas.compat.pyarrow import _safe_fill_null 

99 

100 from pandas.core.dtypes.dtypes import ArrowDtype 

101 

102 ARROW_CMP_FUNCS = { 

103 "eq": pc.equal, 

104 "ne": pc.not_equal, 

105 "lt": pc.less, 

106 "gt": pc.greater, 

107 "le": pc.less_equal, 

108 "ge": pc.greater_equal, 

109 } 

110 

111 ARROW_LOGICAL_FUNCS = { 

112 "and_": pc.and_kleene, 

113 "rand_": lambda x, y: pc.and_kleene(y, x), 

114 "or_": pc.or_kleene, 

115 "ror_": lambda x, y: pc.or_kleene(y, x), 

116 "xor": pc.xor, 

117 "rxor": lambda x, y: pc.xor(y, x), 

118 } 

119 

120 ARROW_BIT_WISE_FUNCS = { 

121 "and_": pc.bit_wise_and, 

122 "rand_": lambda x, y: pc.bit_wise_and(y, x), 

123 "or_": pc.bit_wise_or, 

124 "ror_": lambda x, y: pc.bit_wise_or(y, x), 

125 "xor": pc.bit_wise_xor, 

126 "rxor": lambda x, y: pc.bit_wise_xor(y, x), 

127 } 

128 

129 def cast_for_truediv( 

130 arrow_array: pa.ChunkedArray, pa_object: pa.Array | pa.Scalar 

131 ) -> tuple[pa.ChunkedArray, pa.Array | pa.Scalar]: 

132 # Ensure int / int -> float mirroring Python/Numpy behavior 

133 # as pc.divide_checked(int, int) -> int 

134 if pa.types.is_integer(arrow_array.type) and pa.types.is_integer( 

135 pa_object.type 

136 ): 

137 # GH: 56645. 

138 # https://github.com/apache/arrow/issues/35563 

139 return pc.cast(arrow_array, pa.float64(), safe=False), pc.cast( 

140 pa_object, pa.float64(), safe=False 

141 ) 

142 

143 return arrow_array, pa_object 

144 

145 def floordiv_compat( 

146 left: pa.ChunkedArray | pa.Array | pa.Scalar, 

147 right: pa.ChunkedArray | pa.Array | pa.Scalar, 

148 ) -> pa.ChunkedArray: 

149 # TODO: Replace with pyarrow floordiv kernel. 

150 # https://github.com/apache/arrow/issues/39386 

151 if pa.types.is_integer(left.type) and pa.types.is_integer(right.type): 

152 divided = pc.divide_checked(left, right) 

153 if pa.types.is_signed_integer(divided.type): 

154 # GH 56676 

155 has_remainder = pc.not_equal(pc.multiply(divided, right), left) 

156 has_one_negative_operand = pc.less( 

157 pc.bit_wise_xor(left, right), 

158 pa.scalar(0, type=divided.type), 

159 ) 

160 result = pc.if_else( 

161 pc.and_( 

162 has_remainder, 

163 has_one_negative_operand, 

164 ), 

165 # GH: 55561 

166 pc.subtract(divided, pa.scalar(1, type=divided.type)), 

167 divided, 

168 ) 

169 else: 

170 result = divided 

171 result = result.cast(left.type) 

172 else: 

173 divided = pc.divide(left, right) 

174 result = pc.floor(divided) 

175 return result 

176 

177 ARROW_ARITHMETIC_FUNCS = { 

178 "add": pc.add_checked, 

179 "radd": lambda x, y: pc.add_checked(y, x), 

180 "sub": pc.subtract_checked, 

181 "rsub": lambda x, y: pc.subtract_checked(y, x), 

182 "mul": pc.multiply_checked, 

183 "rmul": lambda x, y: pc.multiply_checked(y, x), 

184 "truediv": lambda x, y: pc.divide(*cast_for_truediv(x, y)), 

185 "rtruediv": lambda x, y: pc.divide(*cast_for_truediv(y, x)), 

186 "floordiv": lambda x, y: floordiv_compat(x, y), 

187 "rfloordiv": lambda x, y: floordiv_compat(y, x), 

188 "mod": NotImplemented, 

189 "rmod": NotImplemented, 

190 "divmod": NotImplemented, 

191 "rdivmod": NotImplemented, 

192 "pow": pc.power_checked, 

193 "rpow": lambda x, y: pc.power_checked(y, x), 

194 } 

195 

196if TYPE_CHECKING: 

197 from collections.abc import ( 

198 Callable, 

199 Sequence, 

200 ) 

201 

202 from pandas._libs.missing import NAType 

203 from pandas._typing import ( 

204 ArrayLike, 

205 AxisInt, 

206 Dtype, 

207 FillnaOptions, 

208 InterpolateOptions, 

209 Iterator, 

210 NpDtype, 

211 NumpySorter, 

212 NumpyValueArrayLike, 

213 PositionalIndexer, 

214 Scalar, 

215 SortKind, 

216 TakeIndexer, 

217 TimeAmbiguous, 

218 TimeNonexistent, 

219 npt, 

220 ) 

221 

222 from pandas.core.dtypes.dtypes import ExtensionDtype 

223 

224 from pandas import Series 

225 from pandas.core.arrays.datetimes import DatetimeArray 

226 from pandas.core.arrays.timedeltas import TimedeltaArray 

227 

228 

229def to_pyarrow_type( 

230 dtype: ArrowDtype | pa.DataType | Dtype | None, 

231) -> pa.DataType | None: 

232 """ 

233 Convert dtype to a pyarrow type instance. 

234 """ 

235 if isinstance(dtype, ArrowDtype): 

236 return dtype.pyarrow_dtype 

237 elif isinstance(dtype, pa.DataType): 

238 return dtype 

239 elif isinstance(dtype, DatetimeTZDtype): 

240 return pa.timestamp(dtype.unit, dtype.tz) 

241 elif dtype: 

242 try: 

243 # Accepts python types too 

244 # Doesn't handle all numpy types 

245 return pa.from_numpy_dtype(dtype) 

246 except pa.ArrowNotImplementedError: 

247 pass 

248 return None 

249 

250 

251def _is_varbinary_type(pa_type: pa.DataType) -> bool: 

252 """ 

253 Whether this is one of string, large_string, binary and large_binary. 

254 

255 pc.if_else misreads a non-zero offset for exactly these four, silently 

256 truncating values (GH#64320, https://github.com/apache/arrow/issues/49410). 

257 Other offset-carrying layouts such as list and map are unaffected. 

258 """ 

259 return ( 

260 pa.types.is_string(pa_type) 

261 or pa.types.is_large_string(pa_type) 

262 or pa.types.is_binary(pa_type) 

263 or pa.types.is_large_binary(pa_type) 

264 ) 

265 

266 

267def _is_string_or_binary_view(typ): 

268 return not pa_version_under16p0 and ( 

269 pa.types.is_string_view(typ) or pa.types.is_binary_view(typ) 

270 ) 

271 

272 

273def _boxing_may_borrow_memory(pa_type: pa.DataType) -> bool: 

274 """ 

275 Whether ``pa.array`` on this type can return a view on caller-owned memory. 

276 

277 Zero-copy over numpy/masked arrays for fixed-width layouts; character 

278 layouts always repack, so copying those would cost a full copy of the 

279 character data for no safety gain. Nested types may have a zero-copy child. 

280 """ 

281 return not (_is_varbinary_type(pa_type) or _is_string_or_binary_view(pa_type)) 

282 

283 

284def _copy_pyarrow_buffers( 

285 pa_array: pa.Array | pa.ChunkedArray, 

286) -> pa.Array | pa.ChunkedArray: 

287 """ 

288 Return an equal array that owns its buffers (GH#67990). 

289 

290 ``pa.concat_arrays`` reuses the ``dictionary`` child rather than copying it, 

291 so dictionary types are rebuilt from copies of both halves. 

292 """ 

293 if isinstance(pa_array, pa.ChunkedArray): 

294 return pa.chunked_array( 

295 [_copy_pyarrow_buffers(chunk) for chunk in pa_array.chunks], 

296 type=pa_array.type, 

297 ) 

298 if pa.types.is_dictionary(pa_array.type): 

299 return pa.DictionaryArray.from_arrays( 

300 _copy_pyarrow_buffers(pa_array.indices), 

301 _copy_pyarrow_buffers(pa_array.dictionary), 

302 ordered=pa_array.type.ordered, 

303 ) 

304 return pa.concat_arrays([pa_array]) 

305 

306 

307@set_module("pandas.arrays") 

308class ArrowExtensionArray( 

309 OpsMixin, 

310 ExtensionArraySupportsAnyAll, 

311 ArrowStringArrayMixin, 

312): 

313 """ 

314 Pandas ExtensionArray backed by a PyArrow ChunkedArray. 

315 

316 .. warning:: 

317 

318 ArrowExtensionArray is considered experimental. The implementation and 

319 parts of the API may change without warning. 

320 

321 Parameters 

322 ---------- 

323 values : pyarrow.Array or pyarrow.ChunkedArray 

324 The input data to initialize the ArrowExtensionArray. 

325 

326 Attributes 

327 ---------- 

328 None 

329 

330 Methods 

331 ------- 

332 None 

333 

334 Returns 

335 ------- 

336 ArrowExtensionArray 

337 

338 See Also 

339 -------- 

340 array : Create a Pandas array with a specified dtype. 

341 DataFrame.to_feather : Write a DataFrame to the binary Feather format. 

342 read_feather : Load a feather-format object from the file path. 

343 

344 Notes 

345 ----- 

346 Most methods are implemented using `pyarrow compute functions. <https://arrow.apache.org/docs/python/api/compute.html>`__ 

347 Some methods may either raise an exception or raise a ``PerformanceWarning`` if an 

348 associated compute function is not available based on the installed version of PyArrow. 

349 

350 Please install the latest version of PyArrow to enable the best functionality and avoid 

351 potential bugs in prior versions of PyArrow. 

352 

353 Examples 

354 -------- 

355 Create an ArrowExtensionArray with :func:`pandas.array`: 

356 

357 >>> pd.array([1, 1, None], dtype="int64[pyarrow]") 

358 <ArrowExtensionArray> 

359 [1, 1, <NA>] 

360 Length: 3, dtype: int64[pyarrow] 

361 """ # noqa: E501 (http link too long) 

362 

363 _pa_array: pa.ChunkedArray 

364 _dtype: ArrowDtype 

365 

366 def __init__(self, values: pa.Array | pa.ChunkedArray) -> None: 

367 if not HAS_PYARROW: 

368 msg = ( 

369 f"pyarrow>={PYARROW_MIN_VERSION} is required for PyArrow " 

370 "backed ArrowExtensionArray." 

371 ) 

372 raise ImportError(msg) 

373 if isinstance(values, pa.Array): 

374 self._pa_array = pa.chunked_array([values]) 

375 elif isinstance(values, pa.ChunkedArray): 

376 self._pa_array = values 

377 else: 

378 raise ValueError( 

379 f"Unsupported type '{type(values)}' for ArrowExtensionArray" 

380 ) 

381 self._dtype = ArrowDtype(self._pa_array.type) 

382 

383 @classmethod 

384 def _from_sequence( 

385 cls, scalars, *, dtype: Dtype | None = None, copy: bool = False 

386 ) -> Self: 

387 """ 

388 Construct a new ExtensionArray from a sequence of scalars. 

389 """ 

390 pa_type = to_pyarrow_type(dtype) 

391 pa_array = cls._box_pa_array(scalars, pa_type=pa_type, copy=copy) 

392 arr = cls(pa_array) 

393 return arr 

394 

395 @classmethod 

396 def _from_sequence_of_strings( 

397 cls, strings, *, dtype: ExtensionDtype, copy: bool = False 

398 ) -> Self: 

399 """ 

400 Construct a new ExtensionArray from a sequence of strings. 

401 """ 

402 mask = isna(strings) 

403 

404 if isinstance(strings, cls): 

405 strings = strings._pa_array 

406 

407 pa_type = to_pyarrow_type(dtype) 

408 if ( 

409 pa_type is None 

410 or pa.types.is_binary(pa_type) 

411 or pa.types.is_string(pa_type) 

412 or pa.types.is_large_string(pa_type) 

413 ): 

414 # pa_type is None: Let pa.array infer 

415 # pa_type is string/binary: scalars already correct type 

416 scalars = strings 

417 elif pa.types.is_timestamp(pa_type): 

418 from pandas.core.tools.datetimes import to_datetime 

419 

420 scalars = to_datetime(strings, errors="raise") 

421 elif pa.types.is_date(pa_type): 

422 from pandas.core.tools.datetimes import to_datetime 

423 

424 scalars = to_datetime(strings, errors="raise").date 

425 scalars = pa.array(scalars, type=pa_type, mask=mask) 

426 elif pa.types.is_duration(pa_type): 

427 from pandas.core.tools.timedeltas import to_timedelta 

428 

429 scalars = to_timedelta(strings, errors="raise") 

430 

431 if pa_type.unit != "ns": 

432 # GH51175: test_from_sequence_of_strings_pa_array 

433 # attempt to parse as int64 reflecting pyarrow's 

434 # duration to string casting behavior 

435 mask = isna(scalars) 

436 if not isinstance(strings, (pa.Array, pa.ChunkedArray)): 

437 strings = pa.array(strings, type=pa.string(), mask=mask) 

438 strings = pc.if_else(mask, None, strings) 

439 try: 

440 scalars = strings.cast(pa.int64()) 

441 except pa.ArrowInvalid: 

442 pass 

443 elif pa.types.is_time(pa_type): 

444 from pandas.core.tools.times import to_time 

445 

446 # "coerce" to allow "null times" (None) to not raise 

447 scalars = to_time(strings, errors="coerce") 

448 elif pa.types.is_boolean(pa_type): 

449 # pyarrow string->bool casting is case-insensitive: 

450 # "true" or "1" -> True 

451 # "false" or "0" -> False 

452 # Note: BooleanArray was previously used to parse these strings 

453 # and allows "1.0" and "0.0". Pyarrow casting does not support 

454 # this, but we allow it here. 

455 if isinstance(strings, (pa.Array, pa.ChunkedArray)): 

456 scalars = strings 

457 else: 

458 scalars = pa.array(strings, type=pa.string(), mask=mask) 

459 scalars = pc.if_else(pc.equal(scalars, "1.0"), "1", scalars) 

460 scalars = pc.if_else(pc.equal(scalars, "0.0"), "0", scalars) 

461 scalars = scalars.cast(pa.bool_()) 

462 elif ( 

463 pa.types.is_integer(pa_type) 

464 or pa.types.is_floating(pa_type) 

465 or pa.types.is_decimal(pa_type) 

466 ): 

467 from pandas.core.tools.numeric import to_numeric 

468 

469 scalars = to_numeric(strings, errors="raise") 

470 if isinstance(strings, (pa.Array, pa.ChunkedArray)): 

471 scalars = strings.cast(pa_type) 

472 elif mask is not None: 

473 scalars = pa.array(scalars, mask=mask, type=pa_type) 

474 

475 else: 

476 raise NotImplementedError( 

477 f"Converting strings to {pa_type} is not implemented." 

478 ) 

479 return cls._from_sequence(scalars, dtype=pa_type, copy=copy) 

480 

481 def _from_pyarrow_array(self, pa_array): 

482 """ 

483 Construct from the pyarrow array result of an operation, for 

484 compatibility with ArrowStringArray. 

485 """ 

486 return type(self)(pa_array) 

487 

488 def _cast_pointwise_result(self, values) -> ArrayLike: 

489 if len(values) == 0: 

490 # Retain our dtype 

491 return self[:0].copy() 

492 

493 try: 

494 if self.dtype.kind in "iufc" and not is_nan_na(): 

495 values = np.asarray(values, dtype=object) 

496 mask = is_pdna_or_none(values) 

497 arr = pa.array(values, mask=mask) 

498 else: 

499 arr = pa.array(values, from_pandas=True) 

500 except (ValueError, TypeError): 

501 # e.g. test_by_column_values_with_same_starting_value with nested 

502 # values, one entry of which is an ArrowStringArray 

503 # or test_agg_lambda_complex128_dtype_conversion for complex values 

504 values = np.asarray(values, dtype=object) 

505 return lib.maybe_convert_objects(values, convert_non_numeric=True) 

506 

507 if pa.types.is_null(arr.type): 

508 if lib.infer_dtype(values) == "decimal": 

509 # GH#62522; the specific decimal precision here is arbitrary 

510 arr = arr.cast(pa.decimal128(1)) 

511 if pa.types.is_duration(arr.type): 

512 # workaround for https://github.com/apache/arrow/issues/40620 

513 result = ArrowExtensionArray._from_sequence(values) 

514 if pa.types.is_duration(self._pa_array.type): 

515 result = result.astype(self.dtype) # type: ignore[assignment] 

516 elif pa.types.is_timestamp(self._pa_array.type): 

517 # Try to retain original unit 

518 new_dtype = ArrowDtype(pa.duration(self._pa_array.type.unit)) 

519 try: 

520 result = result.astype(new_dtype) # type: ignore[assignment] 

521 except ValueError: 

522 pass 

523 elif pa.types.is_date64(self._pa_array.type): 

524 # Try to match unit we get on non-pointwise op 

525 dtype = ArrowDtype(pa.duration("ms")) 

526 result = result.astype(dtype) # type: ignore[assignment] 

527 elif pa.types.is_date(self._pa_array.type): 

528 # Try to match unit we get on non-pointwise op 

529 dtype = ArrowDtype(pa.duration("s")) 

530 result = result.astype(dtype) # type: ignore[assignment] 

531 return result 

532 

533 elif pa.types.is_date(arr.type) and pa.types.is_date(self._pa_array.type): 

534 arr = arr.cast(self._pa_array.type) 

535 elif pa.types.is_time(arr.type) and pa.types.is_time(self._pa_array.type): 

536 arr = arr.cast(self._pa_array.type) 

537 elif pa.types.is_decimal(arr.type) and pa.types.is_decimal(self._pa_array.type): 

538 arr = arr.cast(self._pa_array.type) 

539 elif pa.types.is_integer(arr.type) and pa.types.is_integer(self._pa_array.type): 

540 try: 

541 arr = arr.cast(self._pa_array.type) 

542 except pa.lib.ArrowInvalid: 

543 # e.g. test_combine_add if we can't cast 

544 pass 

545 elif pa.types.is_floating(arr.type) and pa.types.is_floating( 

546 self._pa_array.type 

547 ): 

548 try: 

549 arr = arr.cast(self._pa_array.type) 

550 except pa.lib.ArrowInvalid: 

551 # e.g. test_combine_add if we can't cast 

552 pass 

553 

554 if isinstance(self.dtype, StringDtype): 

555 if pa.types.is_string(arr.type) or pa.types.is_large_string(arr.type): 

556 # ArrowStringArray preserves dtype.na_value 

557 return self._from_pyarrow_array(arr) 

558 if self.dtype.na_value is np.nan: 

559 # ArrowEA has different semantics, so we return numpy-based 

560 # result instead 

561 values = np.asarray(values, dtype=object) 

562 return lib.maybe_convert_objects(values, convert_non_numeric=True) 

563 return ArrowExtensionArray(arr) 

564 return self._from_pyarrow_array(arr) 

565 

566 @classmethod 

567 def _box_pa( 

568 cls, value, pa_type: pa.DataType | None = None 

569 ) -> pa.Array | pa.ChunkedArray | pa.Scalar: 

570 """ 

571 Box value into a pyarrow Array, ChunkedArray or Scalar. 

572 

573 Parameters 

574 ---------- 

575 value : any 

576 pa_type : pa.DataType | None 

577 

578 Returns 

579 ------- 

580 pa.Array or pa.ChunkedArray or pa.Scalar 

581 """ 

582 if isinstance(value, pa.Scalar) or not is_list_like(value): 

583 return cls._box_pa_scalar(value, pa_type) 

584 return cls._box_pa_array(value, pa_type) 

585 

586 @classmethod 

587 def _box_pa_scalar(cls, value, pa_type: pa.DataType | None = None) -> pa.Scalar: 

588 """ 

589 Box value into a pyarrow Scalar. 

590 

591 Parameters 

592 ---------- 

593 value : any 

594 pa_type : pa.DataType | None 

595 

596 Returns 

597 ------- 

598 pa.Scalar 

599 """ 

600 if isinstance(value, pa.Scalar): 

601 pa_scalar = value 

602 elif isna(value) and not (lib.is_float(value) and not is_nan_na()): 

603 pa_scalar = pa.scalar(None, type=pa_type) 

604 else: 

605 # Workaround https://github.com/apache/arrow/issues/37291 

606 if isinstance(value, Timedelta): 

607 if pa_type is None: 

608 pa_type = pa.duration(value.unit) 

609 elif value.unit != pa_type.unit: 

610 value = value.as_unit(pa_type.unit) 

611 value = value._value 

612 elif isinstance(value, Timestamp): 

613 if pa_type is None: 

614 pa_type = pa.timestamp(value.unit, tz=value.tz) 

615 elif value.unit != pa_type.unit: 

616 value = value.as_unit(pa_type.unit) 

617 value = value._value 

618 

619 pa_scalar = pa.scalar(value, type=pa_type) 

620 

621 if pa_type is not None and pa_scalar.type != pa_type: 

622 pa_scalar = pa_scalar.cast(pa_type) 

623 

624 return pa_scalar 

625 

626 @classmethod 

627 def _box_pa_array( 

628 cls, value, pa_type: pa.DataType | None = None, copy: bool = False 

629 ) -> pa.Array | pa.ChunkedArray: 

630 """ 

631 Box value into a pyarrow Array or ChunkedArray. 

632 

633 Parameters 

634 ---------- 

635 value : Sequence 

636 pa_type : pa.DataType | None 

637 

638 Returns 

639 ------- 

640 pa.Array or pa.ChunkedArray 

641 """ 

642 value = extract_array(value, extract_numpy=True) 

643 if isinstance(value, cls): 

644 pa_array = value._pa_array 

645 elif isinstance(value, (pa.Array, pa.ChunkedArray)): 

646 pa_array = value 

647 elif isinstance(value, BaseMaskedArray): 

648 # GH 52625 

649 if copy: 

650 value = value.copy() 

651 pa_array = value.__arrow_array__() 

652 

653 elif hasattr(value, "__arrow_array__"): 

654 # e.g. StringArray 

655 if copy: 

656 value = value.copy() 

657 pa_array = value.__arrow_array__() 

658 

659 else: 

660 if ( 

661 isinstance(value, np.ndarray) 

662 and pa_type is not None 

663 and ( 

664 pa.types.is_large_binary(pa_type) 

665 or pa.types.is_large_string(pa_type) 

666 ) 

667 ): 

668 # See https://github.com/apache/arrow/issues/35289 

669 value = np.asarray(value, dtype=object) 

670 elif copy and is_array_like(value): 

671 # pa array should not get updated when numpy array is updated 

672 value = value.copy() 

673 

674 if ( 

675 pa_type is not None 

676 and pa.types.is_duration(pa_type) 

677 and (not isinstance(value, np.ndarray) or value.dtype.kind not in "mi") 

678 ): 

679 # Workaround https://github.com/apache/arrow/issues/37291 

680 from pandas.core.tools.timedeltas import to_timedelta 

681 

682 value = to_timedelta(value, unit=pa_type.unit).as_unit(pa_type.unit) 

683 value = value.to_numpy() 

684 

685 if pa_type is not None and pa.types.is_timestamp(pa_type): 

686 # Use DatetimeArray to exclude Decimal(NaN) (GH#61774) and 

687 # ensure constructor treats tznaive the same as non-pyarrow 

688 # dtypes (GH#61775) 

689 from pandas.core.arrays.datetimes import ( 

690 DatetimeArray, 

691 tz_to_dtype, 

692 ) 

693 

694 pass_dtype = tz_to_dtype(tz=pa_type.tz, unit=pa_type.unit) 

695 value = extract_array(value, extract_numpy=True) 

696 if isinstance(value, DatetimeArray): 

697 dta = value 

698 else: 

699 dta = DatetimeArray._from_sequence( 

700 value, copy=copy, dtype=pass_dtype 

701 ) 

702 dta_mask = dta.isna() 

703 value_i8 = cast("npt.NDArray", dta.view("i8")) 

704 if not value_i8.flags["WRITEABLE"]: 

705 # e.g. test_setitem_frame_2d_values 

706 value_i8 = value_i8.copy() 

707 dta = DatetimeArray._from_sequence(value_i8, dtype=dta.dtype) 

708 value_i8[dta_mask] = 0 # GH#61776 avoid __sub__ overflow 

709 pa_array = pa.array(dta._ndarray, type=pa_type, mask=dta_mask) 

710 return pa_array 

711 

712 mask = None 

713 if is_nan_na(): 

714 try: 

715 arr_value = np.asarray(value) 

716 if arr_value.ndim > 1: 

717 # e.g. test_fixed_size_list we have list data. ndim > 1 

718 # means there were no scalar (NA) entries. 

719 mask = np.zeros(len(value), dtype=np.bool_) 

720 else: 

721 mask = isna(arr_value) 

722 except ValueError: 

723 # Ragged data that numpy raises on 

724 arr_value = construct_1d_object_array_from_listlike(value) 

725 mask = isna(arr_value) 

726 elif ( 

727 getattr(value, "dtype", None) is None or value.dtype.kind not in "iumMf" 

728 ): 

729 arr_value = np.asarray(value, dtype=object) 

730 # similar to isna(value) but exclude NaN, NaT, nat-like, nan-like 

731 mask = is_pdna_or_none(arr_value) 

732 

733 try: 

734 pa_array = pa.array(value, type=pa_type, mask=mask) 

735 except (pa.ArrowInvalid, pa.ArrowTypeError): 

736 # GH50430: let pyarrow infer type, then cast 

737 pa_array = pa.array(value, mask=mask) 

738 

739 if pa_type is None and pa.types.is_duration(pa_array.type): 

740 # Workaround https://github.com/apache/arrow/issues/37291 

741 from pandas.core.tools.timedeltas import to_timedelta 

742 

743 value = to_timedelta(value) 

744 value = value.to_numpy() 

745 pa_array = pa.array(value, type=pa_type) 

746 

747 if pa.types.is_duration(pa_array.type) and pa_array.null_count > 0: 

748 # GH52843: upstream bug for duration types when originally 

749 # constructed with data containing numpy NaT. 

750 # https://github.com/apache/arrow/issues/35088 

751 arr = cls(pa_array) 

752 arr = arr.fillna(arr.dtype.na_value) 

753 pa_array = arr._pa_array 

754 

755 if pa_type is not None and pa_array.type != pa_type: 

756 if pa.types.is_dictionary(pa_type): 

757 pa_array = pa_array.dictionary_encode() 

758 if pa_array.type != pa_type: 

759 pa_array = pa_array.cast(pa_type) 

760 else: 

761 try: 

762 pa_array = pa_array.cast(pa_type) 

763 except (pa.ArrowNotImplementedError, pa.ArrowTypeError): 

764 if pa.types.is_string(pa_array.type) or pa.types.is_large_string( 

765 pa_array.type 

766 ): 

767 # TODO: Move logic in _from_sequence_of_strings into 

768 # _box_pa_array 

769 dtype = ArrowDtype(pa_type) 

770 return cls._from_sequence_of_strings( 

771 value, dtype=dtype 

772 )._pa_array 

773 else: 

774 raise 

775 

776 return pa_array 

777 

778 def __getitem__(self, item: PositionalIndexer): 

779 """Select a subset of self. 

780 

781 Parameters 

782 ---------- 

783 item : int, slice, or ndarray 

784 * int: The position in 'self' to get. 

785 * slice: A slice object, where 'start', 'stop', and 'step' are 

786 integers or None 

787 * ndarray: A 1-d boolean NumPy ndarray the same length as 'self' 

788 

789 Returns 

790 ------- 

791 item : scalar or ExtensionArray 

792 

793 Notes 

794 ----- 

795 For scalar ``item``, return a scalar value suitable for the array's 

796 type. This should be an instance of ``self.dtype.type``. 

797 For slice ``key``, return an instance of ``ExtensionArray``, even 

798 if the slice is length 0 or 1. 

799 For a boolean mask, return an instance of ``ExtensionArray``, filtered 

800 to the values where ``item`` is True. 

801 """ 

802 item = check_array_indexer(self, item) 

803 

804 if isinstance(item, np.ndarray): 

805 if not len(item): 

806 # Removable once we migrate StringDtype[pyarrow] to ArrowDtype[string] 

807 if ( 

808 isinstance(self._dtype, StringDtype) 

809 and self._dtype.storage == "pyarrow" 

810 ): 

811 # TODO(infer_string) should this be large_string? 

812 pa_dtype = pa.string() 

813 else: 

814 pa_dtype = self._dtype.pyarrow_dtype 

815 result = pa.chunked_array([], type=pa_dtype) 

816 return self._from_pyarrow_array(result) 

817 

818 elif item.dtype.kind in "iu": 

819 return self.take(item) 

820 elif item.dtype.kind == "b": 

821 return self._from_pyarrow_array(self._pa_array.filter(item)) 

822 else: 

823 raise IndexError( 

824 "Only integers, slices and integer or " 

825 "boolean arrays are valid indices." 

826 ) 

827 elif isinstance(item, tuple): 

828 item = unpack_tuple_and_ellipses(item) 

829 

830 if item is Ellipsis: 

831 # TODO: should be handled by pyarrow? 

832 item = slice(None) 

833 

834 if is_scalar(item) and not is_integer(item): 

835 # e.g. "foo" or 2.5 

836 # exception message copied from numpy 

837 raise IndexError( 

838 r"only integers, slices (`:`), ellipsis (`...`), numpy.newaxis " 

839 r"(`None`) and integer or boolean arrays are valid indices" 

840 ) 

841 # We are not an array indexer, so maybe e.g. a slice or integer 

842 # indexer. We dispatch to pyarrow. 

843 if isinstance(item, slice): 

844 # Arrow bug https://github.com/apache/arrow/issues/38768 

845 if item.start == item.stop: 

846 pass 

847 elif ( 

848 item.stop is not None 

849 and item.stop < -len(self) 

850 and item.step is not None 

851 and item.step < 0 

852 ): 

853 item = slice(item.start, None, item.step) 

854 

855 value = self._pa_array[item] 

856 if isinstance(value, pa.ChunkedArray): 

857 result = self._from_pyarrow_array(value) 

858 if getitem_returns_view(self, item): 

859 result._readonly = self._readonly 

860 return result 

861 else: 

862 pa_type = self._pa_array.type 

863 scalar = value.as_py() 

864 if scalar is None: 

865 return self._dtype.na_value 

866 elif pa.types.is_timestamp(pa_type) and pa_type.unit != "ns": 

867 # GH 53326 

868 return Timestamp(scalar).as_unit(pa_type.unit) 

869 elif pa.types.is_duration(pa_type) and pa_type.unit != "ns": 

870 # GH 53326 

871 return Timedelta(scalar).as_unit(pa_type.unit) 

872 else: 

873 return scalar 

874 

875 def __iter__(self) -> Iterator[Any]: 

876 """ 

877 Iterate over elements of the array. 

878 """ 

879 na_value = self._dtype.na_value 

880 # GH 53326 

881 pa_type = self._pa_array.type 

882 box_timestamp = pa.types.is_timestamp(pa_type) and pa_type.unit != "ns" 

883 box_timedelta = pa.types.is_duration(pa_type) and pa_type.unit != "ns" 

884 for value in self._pa_array: 

885 val = value.as_py() 

886 if val is None: 

887 yield na_value 

888 elif box_timestamp: 

889 yield Timestamp(val).as_unit(pa_type.unit) 

890 elif box_timedelta: 

891 yield Timedelta(val).as_unit(pa_type.unit) 

892 else: 

893 yield val 

894 

895 def __arrow_array__(self, type=None): 

896 """Convert myself to a pyarrow ChunkedArray.""" 

897 return self._pa_array 

898 

899 def __array_ufunc__(self, ufunc: np.ufunc, method: str, *inputs, **kwargs): 

900 # Need to wrap np.array results GH#62800 

901 result = super().__array_ufunc__(ufunc, method, *inputs, **kwargs) 

902 if type(self) is ArrowExtensionArray: 

903 # Exclude ArrowStringArray 

904 return type(self)._from_sequence(result) 

905 return result 

906 

907 def __array__( 

908 self, dtype: NpDtype | None = None, copy: bool | None = None 

909 ) -> np.ndarray: 

910 """Correctly construct numpy arrays when passed to `np.asarray()`.""" 

911 if copy is False: 

912 # TODO: By using `zero_copy_only` it may be possible to implement this 

913 raise ValueError( 

914 "Unable to avoid copy while creating an array as requested." 

915 ) 

916 elif copy is None: 

917 # `to_numpy(copy=False)` has the meaning of NumPy `copy=None`. 

918 copy = False 

919 

920 return self.to_numpy(dtype=dtype, copy=copy) 

921 

922 def __invert__(self) -> Self: 

923 # This is a bit wise op for integer types 

924 if pa.types.is_integer(self._pa_array.type): 

925 return self._from_pyarrow_array(pc.bit_wise_not(self._pa_array)) 

926 elif pa.types.is_string(self._pa_array.type) or pa.types.is_large_string( 

927 self._pa_array.type 

928 ): 

929 # Raise TypeError instead of pa.ArrowNotImplementedError 

930 raise TypeError("__invert__ is not supported for string dtypes") 

931 else: 

932 return self._from_pyarrow_array(pc.invert(self._pa_array)) 

933 

934 def __neg__(self) -> Self: 

935 try: 

936 return self._from_pyarrow_array(pc.negate_checked(self._pa_array)) 

937 except pa.ArrowNotImplementedError as err: 

938 raise TypeError( 

939 f"unary '-' not supported for dtype '{self.dtype}'" 

940 ) from err 

941 

942 def __pos__(self) -> Self: 

943 return self._from_pyarrow_array(self._pa_array) 

944 

945 def __abs__(self) -> Self: 

946 return self._from_pyarrow_array(pc.abs_checked(self._pa_array)) 

947 

948 # GH 42600: __getstate__/__setstate__ not necessary once 

949 # https://issues.apache.org/jira/browse/ARROW-10739 is addressed 

950 def __getstate__(self): 

951 state = self.__dict__.copy() 

952 state["_pa_array"] = self._pa_array.combine_chunks() 

953 return state 

954 

955 def __setstate__(self, state) -> None: 

956 if "_data" in state: 

957 data = state.pop("_data") 

958 else: 

959 data = state["_pa_array"] 

960 state["_pa_array"] = pa.chunked_array(data) 

961 self.__dict__.update(state) 

962 

963 def _cmp_method(self, other, op) -> ArrowExtensionArray: 

964 pc_func = ARROW_CMP_FUNCS[op.__name__] 

965 ltype = self._pa_array.type 

966 

967 if isinstance(other, (ExtensionArray, np.ndarray, list, range)): 

968 try: 

969 boxed = self._box_pa(other) 

970 except pa.lib.ArrowInvalid: 

971 # e.g. GH#60228 [1, "b"] we have to operate pointwise 

972 res_values = [op(x, y) for x, y in zip(self, other, strict=True)] 

973 result = pa.array(res_values, type=pa.bool_(), from_pandas=True) 

974 else: 

975 rtype = boxed.type 

976 if ( 

977 (pa.types.is_timestamp(ltype) and pa.types.is_date(rtype)) 

978 or (pa.types.is_timestamp(rtype) and pa.types.is_date(ltype)) 

979 or isinstance(other, range) 

980 ): 

981 # GH#62157 match non-pyarrow behavior 

982 result = ops.invalid_comparison(self, other, op) 

983 result = pa.array(result, type=pa.bool_()) 

984 else: 

985 try: 

986 result = pc_func(self._pa_array, boxed) 

987 except pa.ArrowNotImplementedError: 

988 result = ops.invalid_comparison(self, other, op) 

989 result = pa.array(result, type=pa.bool_()) 

990 

991 elif is_scalar(other): 

992 if (isinstance(other, datetime) and pa.types.is_date(ltype)) or ( 

993 type(other) is date and pa.types.is_timestamp(ltype) 

994 ): 

995 # GH#62157 match non-pyarrow behavior 

996 result = ops.invalid_comparison(self, other, op) 

997 result = pa.array(result, type=pa.bool_()) 

998 else: 

999 try: 

1000 result = pc_func(self._pa_array, self._box_pa(other)) 

1001 except (pa.lib.ArrowNotImplementedError, pa.lib.ArrowInvalid): 

1002 mask = isna(self) | isna(other) 

1003 valid = ~mask 

1004 result = np.zeros(len(self), dtype="bool") 

1005 np_array = np.array(self) 

1006 try: 

1007 result[valid] = op(np_array[valid], other) 

1008 except TypeError: 

1009 result = ops.invalid_comparison(self, other, op) 

1010 result = pa.array(result, type=pa.bool_()) 

1011 result = pc.if_else(valid, result, None) 

1012 else: 

1013 raise NotImplementedError( 

1014 f"{op.__name__} not implemented for {type(other)}" 

1015 ) 

1016 return ArrowExtensionArray(result) 

1017 

1018 def _op_method_error_message(self, other, op) -> str: 

1019 if hasattr(other, "dtype"): 

1020 other_type = f"dtype '{other.dtype}'" 

1021 else: 

1022 other_type = f"object of type {type(other)}" 

1023 return ( 

1024 f"operation '{op.__name__}' not supported for " 

1025 f"dtype '{self.dtype}' with {other_type}" 

1026 ) 

1027 

1028 def _evaluate_op_method(self, other, op, arrow_funcs) -> Self: 

1029 pa_type = self._pa_array.type 

1030 other_original = other 

1031 other = self._box_pa(other) 

1032 

1033 if ( 

1034 pa.types.is_string(pa_type) 

1035 or pa.types.is_large_string(pa_type) 

1036 or pa.types.is_binary(pa_type) 

1037 ): 

1038 if op in [operator.add, roperator.radd]: 

1039 # binary_join_element_wise does not support mixed types, but we 

1040 # want to allow addition between string and large_string types 

1041 self_array = self._pa_array 

1042 if pa.types.is_string(pa_type) and pa.types.is_large_string(other.type): 

1043 self_array = self._pa_array.cast(pa.large_string()) 

1044 elif pa.types.is_large_string(pa_type) and pa.types.is_string( 

1045 other.type 

1046 ): 

1047 other = other.cast(pa.large_string()) 

1048 

1049 sep = pa.scalar("", type=self_array.type) 

1050 if isinstance(other, pa.Scalar) and pc.is_null(other).as_py(): 

1051 other = other.cast(self_array.type) 

1052 try: 

1053 if op is operator.add: 

1054 result = pc.binary_join_element_wise(self_array, other, sep) 

1055 elif op is roperator.radd: 

1056 result = pc.binary_join_element_wise(other, self_array, sep) 

1057 except pa.ArrowNotImplementedError as err: 

1058 raise TypeError( 

1059 self._op_method_error_message(other_original, op) 

1060 ) from err 

1061 return self._from_pyarrow_array(result) 

1062 elif op in [operator.mul, roperator.rmul]: 

1063 binary = self._pa_array 

1064 integral = other 

1065 if not pa.types.is_integer(integral.type): 

1066 raise TypeError("Can only string multiply by an integer.") 

1067 pa_integral = pc.if_else(pc.less(integral, 0), 0, integral) 

1068 result = pc.binary_repeat(binary, pa_integral) 

1069 return self._from_pyarrow_array(result) 

1070 elif ( 

1071 pa.types.is_string(other.type) 

1072 or pa.types.is_binary(other.type) 

1073 or pa.types.is_large_string(other.type) 

1074 ) and op in [operator.mul, roperator.rmul]: 

1075 binary = other 

1076 integral = self._pa_array 

1077 if not pa.types.is_integer(integral.type): 

1078 raise TypeError("Can only string multiply by an integer.") 

1079 pa_integral = pc.if_else(pc.less(integral, 0), 0, integral) 

1080 result = pc.binary_repeat(binary, pa_integral) 

1081 return self._from_pyarrow_array(result) 

1082 if ( 

1083 isinstance(other, pa.Scalar) 

1084 and pc.is_null(other).as_py() 

1085 and op.__name__ in ARROW_LOGICAL_FUNCS 

1086 ): 

1087 # pyarrow kleene ops require null to be typed 

1088 other = other.cast(pa_type) 

1089 

1090 pc_func = arrow_funcs[op.__name__] 

1091 if pc_func is NotImplemented: 

1092 if pa.types.is_string(pa_type) or pa.types.is_large_string(pa_type): 

1093 raise TypeError(self._op_method_error_message(other_original, op)) 

1094 raise NotImplementedError(f"{op.__name__} not implemented.") 

1095 

1096 try: 

1097 result = pc_func(self._pa_array, other) 

1098 except pa.ArrowNotImplementedError as err: 

1099 raise TypeError(self._op_method_error_message(other_original, op)) from err 

1100 return self._from_pyarrow_array(result) 

1101 

1102 def _logical_method(self, other, op) -> Self: 

1103 # For integer types `^`, `|`, `&` are bitwise operators and return 

1104 # integer types. Otherwise these are boolean ops. 

1105 if pa.types.is_integer(self._pa_array.type): 

1106 return self._evaluate_op_method(other, op, ARROW_BIT_WISE_FUNCS) 

1107 elif ( 

1108 ( 

1109 pa.types.is_string(self._pa_array.type) 

1110 or pa.types.is_large_string(self._pa_array.type) 

1111 ) 

1112 and op in (roperator.ror_, roperator.rand_, roperator.rxor) 

1113 and isinstance(other, np.ndarray) 

1114 and other.dtype == bool 

1115 ): 

1116 # GH#60234 backward compatibility for the move to StringDtype in 3.0 

1117 op_name = op.__name__[1:].strip("_") 

1118 warnings.warn( 

1119 f"'{op_name}' operations between boolean dtype and {self.dtype} are " 

1120 "deprecated and will raise in a future version. Explicitly " 

1121 "cast the strings to a boolean dtype before operating instead.", 

1122 Pandas4Warning, 

1123 stacklevel=find_stack_level(), 

1124 ) 

1125 return op(other, self.astype(bool)) 

1126 else: 

1127 return self._evaluate_op_method(other, op, ARROW_LOGICAL_FUNCS) 

1128 

1129 def _str_arith_method_object_fallback( 

1130 self, other, op 

1131 ) -> Self | npt.NDArray[np.object_]: 

1132 mask = isna(self) | isna(other) 

1133 valid = ~mask 

1134 

1135 if is_list_like(other): 

1136 if len(other) != len(self): 

1137 raise ValueError( 

1138 f"Lengths of operands do not match: {len(self)} != {len(other)}" 

1139 ) 

1140 if not is_array_like(other): 

1141 other = np.asarray(other) 

1142 other = other[valid] 

1143 

1144 result = np.empty(len(self), dtype=object) 

1145 result[mask] = self.dtype.na_value 

1146 result[valid] = op(np.asarray(self, dtype=object)[valid], other) 

1147 

1148 if not lib.is_string_array(result, skipna=True): 

1149 return result 

1150 return type(self)._from_sequence(result, dtype=self.dtype) 

1151 

1152 def _arith_method(self, other, op) -> Self | npt.NDArray[np.object_]: 

1153 result: Self | npt.NDArray[np.object_] 

1154 if pa.types.is_string(self._pa_array.type) or pa.types.is_large_string( 

1155 self._pa_array.type 

1156 ): 

1157 try: 

1158 result = self._evaluate_op_method(other, op, ARROW_ARITHMETIC_FUNCS) 

1159 except (pa.ArrowInvalid, pa.ArrowTypeError): 

1160 result = self._str_arith_method_object_fallback(other, op) 

1161 else: 

1162 result = self._evaluate_op_method(other, op, ARROW_ARITHMETIC_FUNCS) 

1163 if isinstance(result, np.ndarray): 

1164 return result 

1165 if is_nan_na() and result.dtype.kind == "f": 

1166 parr = result._pa_array 

1167 mask = pc.is_nan(parr).fill_null(False).to_numpy() 

1168 arr = pc.replace_with_mask(parr, mask, pa.scalar(None, type=parr.type)) 

1169 result = type(self)(arr) 

1170 return result 

1171 

1172 def equals(self, other) -> bool: 

1173 if not isinstance(other, ArrowExtensionArray): 

1174 return False 

1175 # I'm told that pyarrow makes __eq__ behave like pandas' equals; 

1176 # TODO: is this documented somewhere? 

1177 return self._pa_array == other._pa_array 

1178 

1179 @property 

1180 def dtype(self) -> ArrowDtype: 

1181 """ 

1182 An instance of 'ExtensionDtype'. 

1183 """ 

1184 return self._dtype 

1185 

1186 @property 

1187 def nbytes(self) -> int: 

1188 """ 

1189 The number of bytes needed to store this object in memory. 

1190 """ 

1191 return self._pa_array.nbytes 

1192 

1193 def __len__(self) -> int: 

1194 """ 

1195 Length of this array. 

1196 

1197 Returns 

1198 ------- 

1199 length : int 

1200 """ 

1201 return len(self._pa_array) 

1202 

1203 def __contains__(self, key) -> bool: 

1204 # https://github.com/pandas-dev/pandas/pull/51307#issuecomment-1426372604 

1205 if isna(key) and key is not self.dtype.na_value: 

1206 if lib.is_float(key) and is_nan_na(): 

1207 return self.dtype.na_value in self 

1208 elif self.dtype.kind == "f" and lib.is_float(key): 

1209 # Check specifically for NaN 

1210 return pc.any(pc.is_nan(self._pa_array)).as_py() 

1211 

1212 # e.g. date or timestamp types we do not allow None here to match pd.NA 

1213 return False 

1214 # TODO: maybe complex? object? 

1215 

1216 return bool(super().__contains__(key)) 

1217 

1218 @property 

1219 def _hasna(self) -> bool: 

1220 return self._pa_array.null_count > 0 

1221 

1222 def isna(self) -> npt.NDArray[np.bool_]: 

1223 """ 

1224 Boolean NumPy array indicating if each value is missing. 

1225 

1226 This should return a 1-D array the same length as 'self'. 

1227 """ 

1228 # GH51630: fast paths 

1229 null_count = self._pa_array.null_count 

1230 if null_count == 0: 

1231 return np.zeros(len(self), dtype=np.bool_) 

1232 elif null_count == len(self): 

1233 return np.ones(len(self), dtype=np.bool_) 

1234 

1235 return self._pa_array.is_null().to_numpy() 

1236 

1237 @overload 

1238 def any(self, *, skipna: Literal[True] = ..., **kwargs) -> bool: ... 

1239 

1240 @overload 

1241 def any(self, *, skipna: bool, **kwargs) -> bool | NAType: ... 

1242 

1243 def any(self, *, skipna: bool = True, **kwargs) -> bool | NAType: 

1244 """ 

1245 Return whether any element is truthy. 

1246 

1247 Returns False unless there is at least one element that is truthy. 

1248 By default, NAs are skipped. If ``skipna=False`` is specified and 

1249 missing values are present, similar :ref:`Kleene logic <boolean.kleene>` 

1250 is used as for logical operations. 

1251 

1252 Parameters 

1253 ---------- 

1254 skipna : bool, default True 

1255 Exclude NA values. If the entire array is NA and `skipna` is 

1256 True, then the result will be False, as for an empty array. 

1257 If `skipna` is False, the result will still be True if there is 

1258 at least one element that is truthy, otherwise NA will be returned 

1259 if there are NA's present. 

1260 

1261 Returns 

1262 ------- 

1263 bool or :attr:`pandas.NA` 

1264 

1265 See Also 

1266 -------- 

1267 ArrowExtensionArray.all : Return whether all elements are truthy. 

1268 

1269 Examples 

1270 -------- 

1271 The result indicates whether any element is truthy (and by default 

1272 skips NAs): 

1273 

1274 >>> pd.array([True, False, True], dtype="boolean[pyarrow]").any() 

1275 True 

1276 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").any() 

1277 True 

1278 >>> pd.array([False, False, pd.NA], dtype="boolean[pyarrow]").any() 

1279 False 

1280 >>> pd.array([], dtype="boolean[pyarrow]").any() 

1281 False 

1282 >>> pd.array([pd.NA], dtype="boolean[pyarrow]").any() 

1283 False 

1284 >>> pd.array([pd.NA], dtype="float64[pyarrow]").any() 

1285 False 

1286 

1287 With ``skipna=False``, the result can be NA if this is logically 

1288 required (whether ``pd.NA`` is True or False influences the result): 

1289 

1290 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").any(skipna=False) 

1291 True 

1292 >>> pd.array([1, 0, pd.NA], dtype="boolean[pyarrow]").any(skipna=False) 

1293 True 

1294 >>> pd.array([False, False, pd.NA], dtype="boolean[pyarrow]").any(skipna=False) 

1295 <NA> 

1296 >>> pd.array([0, 0, pd.NA], dtype="boolean[pyarrow]").any(skipna=False) 

1297 <NA> 

1298 """ 

1299 return self._reduce("any", skipna=skipna, **kwargs) 

1300 

1301 @overload 

1302 def all(self, *, skipna: Literal[True] = ..., **kwargs) -> bool: ... 

1303 

1304 @overload 

1305 def all(self, *, skipna: bool, **kwargs) -> bool | NAType: ... 

1306 

1307 def all(self, *, skipna: bool = True, **kwargs) -> bool | NAType: 

1308 """ 

1309 Return whether all elements are truthy. 

1310 

1311 Returns True unless there is at least one element that is falsey. 

1312 By default, NAs are skipped. If ``skipna=False`` is specified and 

1313 missing values are present, similar :ref:`Kleene logic <boolean.kleene>` 

1314 is used as for logical operations. 

1315 

1316 Parameters 

1317 ---------- 

1318 skipna : bool, default True 

1319 Exclude NA values. If the entire array is NA and `skipna` is 

1320 True, then the result will be True, as for an empty array. 

1321 If `skipna` is False, the result will still be False if there is 

1322 at least one element that is falsey, otherwise NA will be returned 

1323 if there are NA's present. 

1324 

1325 Returns 

1326 ------- 

1327 bool or :attr:`pandas.NA` 

1328 

1329 See Also 

1330 -------- 

1331 ArrowExtensionArray.any : Return whether any element is truthy. 

1332 

1333 Examples 

1334 -------- 

1335 The result indicates whether all elements are truthy (and by default 

1336 skips NAs): 

1337 

1338 >>> pd.array([True, True, pd.NA], dtype="boolean[pyarrow]").all() 

1339 True 

1340 >>> pd.array([1, 1, pd.NA], dtype="boolean[pyarrow]").all() 

1341 True 

1342 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").all() 

1343 False 

1344 >>> pd.array([], dtype="boolean[pyarrow]").all() 

1345 True 

1346 >>> pd.array([pd.NA], dtype="boolean[pyarrow]").all() 

1347 True 

1348 >>> pd.array([pd.NA], dtype="float64[pyarrow]").all() 

1349 True 

1350 

1351 With ``skipna=False``, the result can be NA if this is logically 

1352 required (whether ``pd.NA`` is True or False influences the result): 

1353 

1354 >>> pd.array([True, True, pd.NA], dtype="boolean[pyarrow]").all(skipna=False) 

1355 <NA> 

1356 >>> pd.array([1, 1, pd.NA], dtype="boolean[pyarrow]").all(skipna=False) 

1357 <NA> 

1358 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").all(skipna=False) 

1359 False 

1360 >>> pd.array([1, 0, pd.NA], dtype="boolean[pyarrow]").all(skipna=False) 

1361 False 

1362 """ 

1363 return self._reduce("all", skipna=skipna, **kwargs) 

1364 

1365 def argsort( 

1366 self, 

1367 *, 

1368 ascending: bool = True, 

1369 kind: SortKind = "quicksort", 

1370 na_position: str = "last", 

1371 **kwargs, 

1372 ) -> np.ndarray: 

1373 order = "ascending" if ascending else "descending" 

1374 null_placement = {"last": "at_end", "first": "at_start"}.get(na_position, None) 

1375 if null_placement is None: 

1376 raise ValueError(f"invalid na_position: {na_position}") 

1377 

1378 result = pc.array_sort_indices( 

1379 self._pa_array, order=order, null_placement=null_placement 

1380 ) 

1381 np_result = result.to_numpy() 

1382 return np_result.astype(np.intp, copy=False) 

1383 

1384 def _argmin_max(self, skipna: bool, method: str) -> int: 

1385 if self._pa_array.length() in (0, self._pa_array.null_count) or ( 

1386 self._hasna and not skipna 

1387 ): 

1388 # For empty or all null, pyarrow returns -1 but pandas expects TypeError 

1389 # For skipna=False and data w/ null, pandas expects NotImplementedError 

1390 # let ExtensionArray.arg{max|min} raise 

1391 return getattr(super(), f"arg{method}")(skipna=skipna) 

1392 

1393 data = self._pa_array 

1394 if pa.types.is_duration(data.type): 

1395 data = data.cast(pa.int64()) 

1396 

1397 value = getattr(pc, method)(data, skip_nulls=skipna) 

1398 return pc.index(data, value).as_py() 

1399 

1400 def argmin(self, skipna: bool = True) -> int: 

1401 return self._argmin_max(skipna, "min") 

1402 

1403 def argmax(self, skipna: bool = True) -> int: 

1404 return self._argmin_max(skipna, "max") 

1405 

1406 def copy(self) -> Self: 

1407 """ 

1408 Return a shallow copy of the array. 

1409 

1410 Underlying ChunkedArray is immutable, so a deep copy is unnecessary. 

1411 

1412 Returns 

1413 ------- 

1414 type(self) 

1415 """ 

1416 return self._from_pyarrow_array(self._pa_array) 

1417 

1418 def dropna(self) -> Self: 

1419 """ 

1420 Return ArrowExtensionArray without NA values. 

1421 

1422 Returns 

1423 ------- 

1424 ArrowExtensionArray 

1425 """ 

1426 return self._from_pyarrow_array(pc.drop_null(self._pa_array)) 

1427 

1428 def _pad_or_backfill( 

1429 self, 

1430 *, 

1431 method: FillnaOptions, 

1432 limit: int | None = None, 

1433 limit_area: Literal["inside", "outside"] | None = None, 

1434 copy: bool = True, 

1435 ) -> Self: 

1436 if not self._hasna: 

1437 return self 

1438 

1439 if limit is None and limit_area is None: 

1440 method = missing.clean_fill_method(method) 

1441 try: 

1442 if method == "pad": 

1443 return self._from_pyarrow_array( 

1444 pc.fill_null_forward(self._pa_array) 

1445 ) 

1446 elif method == "backfill": 

1447 return self._from_pyarrow_array( 

1448 pc.fill_null_backward(self._pa_array) 

1449 ) 

1450 except pa.ArrowNotImplementedError: 

1451 # ArrowNotImplementedError: Function 'coalesce' has no kernel 

1452 # matching input types (duration[ns], duration[ns]) 

1453 # TODO: remove try/except wrapper if/when pyarrow implements 

1454 # a kernel for duration types. 

1455 pass 

1456 

1457 # TODO: Why do we no longer need the above cases? 

1458 # TODO(3.0): after EA.fillna 'method' deprecation is enforced, we can remove 

1459 # this method entirely. 

1460 return super()._pad_or_backfill( 

1461 method=method, limit=limit, limit_area=limit_area, copy=copy 

1462 ) 

1463 

1464 @doc(ExtensionArray.fillna) 

1465 def fillna( 

1466 self, 

1467 value: object | ArrayLike, 

1468 limit: int | None = None, 

1469 copy: bool = True, 

1470 ) -> Self: 

1471 if not self._hasna: 

1472 return self.copy() 

1473 

1474 if limit is not None: 

1475 return super().fillna(value=value, limit=limit, copy=copy) 

1476 

1477 if isinstance(value, (np.ndarray, ExtensionArray)): 

1478 # Similar to check_value_size, but we do not mask here since we may 

1479 # end up passing it to the super() method. 

1480 if len(value) != len(self): 

1481 raise ValueError( 

1482 f"Length of 'value' does not match. Got ({len(value)}) " 

1483 f" expected {len(self)}" 

1484 ) 

1485 

1486 try: 

1487 fill_value = self._box_pa(value, pa_type=self._pa_array.type) 

1488 except pa.ArrowTypeError as err: 

1489 msg = f"Invalid value '{value!s}' for dtype '{self.dtype}'" 

1490 raise TypeError(msg) from err 

1491 

1492 try: 

1493 return self._from_pyarrow_array( 

1494 _safe_fill_null(self._pa_array, fill_value=fill_value) 

1495 ) 

1496 except pa.ArrowNotImplementedError: 

1497 # ArrowNotImplementedError: Function 'coalesce' has no kernel 

1498 # matching input types (duration[ns], duration[ns]) 

1499 # TODO: remove try/except wrapper if/when pyarrow implements 

1500 # a kernel for duration types. 

1501 pass 

1502 

1503 return super().fillna(value=value, limit=limit, copy=copy) 

1504 

1505 def isin(self, values: ArrayLike) -> npt.NDArray[np.bool_]: 

1506 # short-circuit to return all False array. 

1507 if not len(values): 

1508 return np.zeros(len(self), dtype=bool) 

1509 

1510 value_set = self._box_pa(values) 

1511 result = pc.is_in(self._pa_array, value_set=value_set) 

1512 # pyarrow 2.0.0 returned nulls, so we explicitly specify dtype to convert nulls 

1513 # to False 

1514 return np.array(result, dtype=np.bool_) 

1515 

1516 def _values_for_factorize(self) -> tuple[np.ndarray, Any]: 

1517 """ 

1518 Return an array and missing value suitable for factorization. 

1519 

1520 Returns 

1521 ------- 

1522 values : ndarray 

1523 na_value : pd.NA 

1524 

1525 Notes 

1526 ----- 

1527 The values returned by this method are also used in 

1528 :func:`pandas.util.hash_pandas_object`. 

1529 """ 

1530 values = self._pa_array.to_numpy() 

1531 return values, self.dtype.na_value 

1532 

1533 @doc(ExtensionArray.factorize) 

1534 def factorize( 

1535 self, 

1536 use_na_sentinel: bool = True, 

1537 ) -> tuple[np.ndarray, ExtensionArray]: 

1538 null_encoding = "mask" if use_na_sentinel else "encode" 

1539 

1540 data = self._pa_array 

1541 

1542 if pa.types.is_dictionary(data.type): 

1543 if null_encoding == "encode": 

1544 # dictionary encode does nothing if an already encoded array is given 

1545 data = data.cast(data.type.value_type) 

1546 encoded = data.dictionary_encode(null_encoding=null_encoding) 

1547 else: 

1548 encoded = data 

1549 else: 

1550 encoded = data.dictionary_encode(null_encoding=null_encoding) 

1551 if encoded.length() == 0: 

1552 indices = np.array([], dtype=np.intp) 

1553 uniques = self._from_pyarrow_array( 

1554 pa.chunked_array([], type=encoded.type.value_type) 

1555 ) 

1556 else: 

1557 # GH 54844 

1558 combined = encoded.combine_chunks() 

1559 pa_indices = combined.indices 

1560 if pa_indices.null_count > 0: 

1561 pa_indices = _safe_fill_null(pa_indices, -1) 

1562 indices = pa_indices.to_numpy(zero_copy_only=False, writable=True).astype( 

1563 np.intp, copy=False 

1564 ) 

1565 uniques = self._from_pyarrow_array(combined.dictionary) 

1566 

1567 return indices, uniques 

1568 

1569 def reshape(self, *args, **kwargs): 

1570 raise NotImplementedError( 

1571 f"{type(self)} does not support reshape " 

1572 f"as backed by a 1D pyarrow.ChunkedArray." 

1573 ) 

1574 

1575 def round(self, decimals: int = 0, *args, **kwargs) -> Self: 

1576 """ 

1577 Round each value in the array a to the given number of decimals. 

1578 

1579 Parameters 

1580 ---------- 

1581 decimals : int, default 0 

1582 Number of decimal places to round to. If decimals is negative, 

1583 it specifies the number of positions to the left of the decimal point. 

1584 *args, **kwargs 

1585 Additional arguments and keywords have no effect. 

1586 

1587 Returns 

1588 ------- 

1589 ArrowExtensionArray 

1590 Rounded values of the ArrowExtensionArray. 

1591 

1592 See Also 

1593 -------- 

1594 DataFrame.round : Round values of a DataFrame. 

1595 Series.round : Round values of a Series. 

1596 """ 

1597 return self._from_pyarrow_array(pc.round(self._pa_array, ndigits=decimals)) 

1598 

1599 @doc(ExtensionArray.searchsorted) 

1600 def searchsorted( 

1601 self, 

1602 value: NumpyValueArrayLike | ExtensionArray, 

1603 side: Literal["left", "right"] = "left", 

1604 sorter: NumpySorter | None = None, 

1605 ) -> npt.NDArray[np.intp] | np.intp: 

1606 if self._hasna: 

1607 raise ValueError( 

1608 "searchsorted requires array to be sorted, which is impossible " 

1609 "with NAs present." 

1610 ) 

1611 if isinstance(value, ExtensionArray): 

1612 value = value.astype(object) 

1613 # Base class searchsorted would cast to object, which is *much* slower. 

1614 dtype = None 

1615 if isinstance(self.dtype, ArrowDtype): 

1616 pa_dtype = self.dtype.pyarrow_dtype 

1617 if ( 

1618 pa.types.is_timestamp(pa_dtype) or pa.types.is_duration(pa_dtype) 

1619 ) and pa_dtype.unit == "ns": 

1620 # np.array[datetime/timedelta].searchsorted(datetime/timedelta) 

1621 # erroneously fails when numpy type resolution is nanoseconds 

1622 dtype = object 

1623 return self.to_numpy(dtype=dtype).searchsorted(value, side=side, sorter=sorter) 

1624 

1625 def take( 

1626 self, 

1627 indices: TakeIndexer, 

1628 allow_fill: bool = False, 

1629 fill_value: Any = None, 

1630 ) -> ArrowExtensionArray: 

1631 """ 

1632 Take elements from an array. 

1633 

1634 Parameters 

1635 ---------- 

1636 indices : sequence of int or one-dimensional np.ndarray of int 

1637 Indices to be taken. 

1638 allow_fill : bool, default False 

1639 How to handle negative values in `indices`. 

1640 

1641 * False: negative values in `indices` indicate positional indices 

1642 from the right (the default). This is similar to 

1643 :func:`numpy.take`. 

1644 

1645 * True: negative values in `indices` indicate 

1646 missing values. These values are set to `fill_value`. Any other 

1647 other negative values raise a ``ValueError``. 

1648 

1649 fill_value : any, optional 

1650 Fill value to use for NA-indices when `allow_fill` is True. 

1651 This may be ``None``, in which case the default NA value for 

1652 the type, ``self.dtype.na_value``, is used. 

1653 

1654 For many ExtensionArrays, there will be two representations of 

1655 `fill_value`: a user-facing "boxed" scalar, and a low-level 

1656 physical NA value. `fill_value` should be the user-facing version, 

1657 and the implementation should handle translating that to the 

1658 physical version for processing the take if necessary. 

1659 

1660 Returns 

1661 ------- 

1662 ExtensionArray 

1663 

1664 Raises 

1665 ------ 

1666 IndexError 

1667 When the indices are out of bounds for the array. 

1668 ValueError 

1669 When `indices` contains negative values other than ``-1`` 

1670 and `allow_fill` is True. 

1671 

1672 See Also 

1673 -------- 

1674 numpy.take 

1675 api.extensions.take 

1676 

1677 Notes 

1678 ----- 

1679 ExtensionArray.take is called by ``Series.__getitem__``, ``.loc``, 

1680 ``iloc``, when `indices` is a sequence of values. Additionally, 

1681 it's called by :meth:`Series.reindex`, or any other method 

1682 that causes realignment, with a `fill_value`. 

1683 """ 

1684 indices_array = np.asanyarray(indices) 

1685 

1686 if len(self._pa_array) == 0 and (indices_array >= 0).any(): 

1687 raise IndexError("cannot do a non-empty take") 

1688 if indices_array.size > 0 and indices_array.max() >= len(self._pa_array): 

1689 raise IndexError("out of bounds value in 'indices'.") 

1690 

1691 if allow_fill: 

1692 fill_mask = indices_array < 0 

1693 if fill_mask.any(): 

1694 validate_indices(indices_array, len(self._pa_array)) 

1695 # TODO(ARROW-9433): Treat negative indices as NULL 

1696 indices_array = pa.array(indices_array, mask=fill_mask) 

1697 result = self._pa_array.take(indices_array) 

1698 if isna(fill_value): 

1699 return self._from_pyarrow_array(result) 

1700 # TODO: ArrowNotImplementedError: Function fill_null has no 

1701 # kernel matching input types (array[string], scalar[string]) 

1702 result = self._from_pyarrow_array(result) 

1703 result[fill_mask] = fill_value 

1704 return result 

1705 # return type(self)(pc.fill_null(result, pa.scalar(fill_value))) 

1706 else: 

1707 # Nothing to fill 

1708 return self._from_pyarrow_array(self._pa_array.take(indices)) 

1709 else: # allow_fill=False 

1710 # TODO(ARROW-9432): Treat negative indices as indices from the right. 

1711 if (indices_array < 0).any(): 

1712 # Don't modify in-place 

1713 indices_array = np.copy(indices_array) 

1714 indices_array[indices_array < 0] += len(self._pa_array) 

1715 return self._from_pyarrow_array(self._pa_array.take(indices_array)) 

1716 

1717 def _maybe_convert_datelike_array(self): 

1718 """Maybe convert to a datelike array.""" 

1719 pa_type = self._pa_array.type 

1720 if pa.types.is_timestamp(pa_type): 

1721 return self._to_datetimearray() 

1722 elif pa.types.is_duration(pa_type): 

1723 return self._to_timedeltaarray() 

1724 return self 

1725 

1726 def _to_datetimearray(self) -> DatetimeArray: 

1727 """Convert a pyarrow timestamp typed array to a DatetimeArray.""" 

1728 from pandas.core.arrays.datetimes import ( 

1729 DatetimeArray, 

1730 tz_to_dtype, 

1731 ) 

1732 

1733 pa_type = self._pa_array.type 

1734 assert pa.types.is_timestamp(pa_type) 

1735 np_dtype = np.dtype(f"M8[{pa_type.unit}]") 

1736 dtype = tz_to_dtype(pa_type.tz, pa_type.unit) 

1737 np_array = self._pa_array.to_numpy() 

1738 np_array = np_array.astype(np_dtype, copy=False) 

1739 return DatetimeArray._simple_new(np_array, dtype=dtype) 

1740 

1741 def _to_timedeltaarray(self) -> TimedeltaArray: 

1742 """Convert a pyarrow duration typed array to a TimedeltaArray.""" 

1743 from pandas.core.arrays.timedeltas import TimedeltaArray 

1744 

1745 pa_type = self._pa_array.type 

1746 assert pa.types.is_duration(pa_type) 

1747 np_dtype = np.dtype(f"m8[{pa_type.unit}]") 

1748 np_array = self._pa_array.to_numpy() 

1749 np_array = np_array.astype(np_dtype, copy=False) 

1750 return TimedeltaArray._simple_new(np_array, dtype=np_dtype) 

1751 

1752 def _values_for_json(self) -> np.ndarray: 

1753 if is_numeric_dtype(self.dtype): 

1754 return np.asarray(self, dtype=object) 

1755 return super()._values_for_json() 

1756 

1757 @doc(ExtensionArray.to_numpy) 

1758 def to_numpy( 

1759 self, 

1760 dtype: npt.DTypeLike | None = None, 

1761 copy: bool = False, 

1762 na_value: object = lib.no_default, 

1763 ) -> np.ndarray: 

1764 original_na_value = na_value 

1765 dtype, na_value = to_numpy_dtype_inference(self, dtype, na_value, self._hasna) 

1766 pa_type = self._pa_array.type 

1767 if not self._hasna or isna(na_value) or pa.types.is_null(pa_type): 

1768 data = self 

1769 else: 

1770 data = self.fillna(na_value) 

1771 copy = False 

1772 

1773 if pa.types.is_timestamp(pa_type) or pa.types.is_duration(pa_type): 

1774 # GH 55997 

1775 if dtype != object and na_value is self.dtype.na_value: 

1776 na_value = lib.no_default 

1777 result = data._maybe_convert_datelike_array().to_numpy( 

1778 dtype=dtype, na_value=na_value 

1779 ) 

1780 elif pa.types.is_time(pa_type) or pa.types.is_date(pa_type): 

1781 # convert to list of python datetime.time objects before 

1782 # wrapping in ndarray 

1783 result = np.array(list(data), dtype=dtype) 

1784 if data._hasna: 

1785 result[data.isna()] = na_value 

1786 elif pa.types.is_null(pa_type): 

1787 if dtype is not None and isna(na_value): 

1788 na_value = None 

1789 result = np.full(len(data), fill_value=na_value, dtype=dtype) 

1790 elif not data._hasna or ( 

1791 pa.types.is_floating(pa_type) 

1792 and ( 

1793 na_value is np.nan 

1794 or ( 

1795 original_na_value is lib.no_default 

1796 and is_float_dtype(dtype) 

1797 and is_nan_na() 

1798 ) 

1799 ) 

1800 ): 

1801 result = data._pa_array.to_numpy() 

1802 if dtype is not None: 

1803 result = result.astype(dtype, copy=False) 

1804 if copy: 

1805 result = result.copy() 

1806 else: 

1807 if dtype is None: 

1808 empty = pa.array([], type=pa_type).to_numpy(zero_copy_only=False) 

1809 if can_hold_element(empty, na_value): 

1810 dtype = empty.dtype 

1811 else: 

1812 dtype = np.object_ 

1813 result = np.empty(len(data), dtype=dtype) 

1814 mask = data.isna() 

1815 result[mask] = na_value 

1816 result[~mask] = data[~mask]._pa_array.to_numpy() 

1817 return result 

1818 

1819 def map(self, mapper, na_action: Literal["ignore"] | None = None): 

1820 if is_numeric_dtype(self.dtype): 

1821 return map_array(self.to_numpy(), mapper, na_action=na_action) 

1822 else: 

1823 # For "mM" cases, the super() method passes `self` without the 

1824 # to_numpy call, which inside map_array casts to ndarray[object]. 

1825 # Without the to_numpy() call, NA is preserved instead of changed 

1826 # to None. 

1827 return super().map(mapper, na_action) 

1828 

1829 @doc(ExtensionArray.duplicated) 

1830 def duplicated( 

1831 self, keep: Literal["first", "last", False] = "first" 

1832 ) -> npt.NDArray[np.bool_]: 

1833 pa_type = self._pa_array.type 

1834 if pa.types.is_floating(pa_type) or pa.types.is_integer(pa_type): 

1835 values = self.to_numpy(na_value=0) 

1836 elif pa.types.is_boolean(pa_type): 

1837 values = self.to_numpy(na_value=False) 

1838 elif pa.types.is_temporal(pa_type): 

1839 if pa_type.bit_width == 32: 

1840 pa_type = pa.int32() 

1841 else: 

1842 pa_type = pa.int64() 

1843 arr = self.astype(ArrowDtype(pa_type)) 

1844 values = arr.to_numpy(na_value=0) 

1845 else: 

1846 # factorize the values to avoid the performance penalty of 

1847 # converting to object dtype 

1848 values = self.factorize()[0] 

1849 

1850 mask = self.isna() if self._hasna else None 

1851 return algos.duplicated(values, keep=keep, mask=mask) 

1852 

1853 def unique(self) -> Self: 

1854 """ 

1855 Compute the ArrowExtensionArray of unique values. 

1856 

1857 Returns 

1858 ------- 

1859 ArrowExtensionArray 

1860 """ 

1861 pa_result = pc.unique(self._pa_array) 

1862 return self._from_pyarrow_array(pa_result) 

1863 

1864 def value_counts(self, dropna: bool = True) -> Series: 

1865 """ 

1866 Return a Series containing counts of each unique value. 

1867 

1868 Parameters 

1869 ---------- 

1870 dropna : bool, default True 

1871 Don't include counts of missing values. 

1872 

1873 Returns 

1874 ------- 

1875 counts : Series 

1876 

1877 See Also 

1878 -------- 

1879 Series.value_counts 

1880 """ 

1881 from pandas import ( 

1882 Index, 

1883 Series, 

1884 ) 

1885 

1886 data = self._pa_array 

1887 vc = data.value_counts() 

1888 

1889 values = vc.field(0) 

1890 counts = vc.field(1) 

1891 if dropna and data.null_count > 0: 

1892 mask = values.is_valid() 

1893 values = values.filter(mask) 

1894 counts = counts.filter(mask) 

1895 

1896 counts = ArrowExtensionArray(counts) 

1897 

1898 index = Index(self._from_pyarrow_array(values), copy=False) 

1899 

1900 return Series(counts, index=index, name="count", copy=False) 

1901 

1902 @classmethod 

1903 def _concat_same_type(cls, to_concat) -> Self: 

1904 """ 

1905 Concatenate multiple ArrowExtensionArrays. 

1906 

1907 Parameters 

1908 ---------- 

1909 to_concat : sequence of ArrowExtensionArrays 

1910 

1911 Returns 

1912 ------- 

1913 ArrowExtensionArray 

1914 """ 

1915 chunks = [array for ea in to_concat for array in ea._pa_array.iterchunks()] 

1916 if to_concat[0].dtype == "string": 

1917 # StringDtype has no attribute pyarrow_dtype 

1918 pa_dtype = pa.large_string() 

1919 else: 

1920 pa_dtype = to_concat[0].dtype.pyarrow_dtype 

1921 arr = pa.chunked_array(chunks, type=pa_dtype) 

1922 return to_concat[0]._from_pyarrow_array(arr) 

1923 

1924 def _accumulate( 

1925 self, name: str, *, skipna: bool = True, **kwargs 

1926 ) -> ArrowExtensionArray | ExtensionArray: 

1927 """ 

1928 Return an ExtensionArray performing an accumulation operation. 

1929 

1930 The underlying data type might change. 

1931 

1932 Parameters 

1933 ---------- 

1934 name : str 

1935 Name of the function, supported values are: 

1936 - cummin 

1937 - cummax 

1938 - cumsum 

1939 - cumprod 

1940 skipna : bool, default True 

1941 If True, skip NA values. 

1942 **kwargs 

1943 Additional keyword arguments passed to the accumulation function. 

1944 Currently, there is no supported kwarg. 

1945 

1946 Returns 

1947 ------- 

1948 array 

1949 

1950 Raises 

1951 ------ 

1952 NotImplementedError : subclass does not define accumulations 

1953 """ 

1954 if is_string_dtype(self): 

1955 return self._str_accumulate(name=name, skipna=skipna, **kwargs) 

1956 

1957 pyarrow_name = { 

1958 "cummax": "cumulative_max", 

1959 "cummin": "cumulative_min", 

1960 "cumprod": "cumulative_prod_checked", 

1961 "cumsum": "cumulative_sum_checked", 

1962 }.get(name, name) 

1963 pyarrow_meth = getattr(pc, pyarrow_name, None) 

1964 if pyarrow_meth is None: 

1965 return super()._accumulate(name, skipna=skipna, **kwargs) 

1966 

1967 data_to_accum = self._pa_array 

1968 

1969 pa_dtype = data_to_accum.type 

1970 

1971 convert_to_int = ( 

1972 pa.types.is_temporal(pa_dtype) and name in ["cummax", "cummin"] 

1973 ) or (pa.types.is_duration(pa_dtype) and name == "cumsum") 

1974 

1975 if convert_to_int: 

1976 if pa_dtype.bit_width == 32: 

1977 data_to_accum = data_to_accum.cast(pa.int32()) 

1978 else: 

1979 data_to_accum = data_to_accum.cast(pa.int64()) 

1980 

1981 try: 

1982 result = pyarrow_meth(data_to_accum, skip_nulls=skipna, **kwargs) 

1983 except pa.ArrowNotImplementedError as err: 

1984 msg = f"operation '{name}' not supported for dtype '{self.dtype}'" 

1985 raise TypeError(msg) from err 

1986 

1987 if convert_to_int: 

1988 result = result.cast(pa_dtype) 

1989 

1990 return self._from_pyarrow_array(result) 

1991 

1992 def _str_accumulate( 

1993 self, name: str, *, skipna: bool = True, **kwargs 

1994 ) -> ArrowExtensionArray | ExtensionArray: 

1995 """ 

1996 Accumulate implementation for strings, see `_accumulate` docstring for details. 

1997 

1998 pyarrow.compute does not implement these methods for strings. 

1999 """ 

2000 if name == "cumprod": 

2001 msg = f"operation '{name}' not supported for dtype '{self.dtype}'" 

2002 raise TypeError(msg) 

2003 

2004 # We may need to strip out trailing NA values 

2005 tail: pa.array | None = None 

2006 na_mask: pa.array | None = None 

2007 pa_array = self._pa_array 

2008 np_func = { 

2009 "cumsum": np.cumsum, 

2010 "cummin": np.minimum.accumulate, 

2011 "cummax": np.maximum.accumulate, 

2012 }[name] 

2013 

2014 if self._hasna: 

2015 na_mask = pc.is_null(pa_array) 

2016 if pc.all(na_mask) == pa.scalar(True): 

2017 return self._from_pyarrow_array(pa_array) 

2018 if skipna: 

2019 if name == "cumsum": 

2020 pa_array = _safe_fill_null(pa_array, "") 

2021 else: 

2022 # We can retain the running min/max by forward/backward filling. 

2023 pa_array = pc.fill_null_forward(pa_array) 

2024 pa_array = pc.fill_null_backward(pa_array) 

2025 else: 

2026 # When not skipping NA values, the result should be null from 

2027 # the first NA value onward. 

2028 idx = pc.index(na_mask, True).as_py() 

2029 tail = pa.nulls(len(pa_array) - idx, type=pa_array.type) 

2030 pa_array = pa_array[:idx] 

2031 

2032 # error: Cannot call function of unknown type 

2033 pa_result = pa.array(np_func(pa_array), type=pa_array.type) # type: ignore[operator] 

2034 

2035 if tail is not None: 

2036 pa_result = pa.concat_arrays([pa_result, tail]) 

2037 elif na_mask is not None: 

2038 pa_result = pc.if_else(na_mask, None, pa_result) 

2039 

2040 result = self._from_pyarrow_array(pa_result) 

2041 return result 

2042 

2043 def _reduce_pyarrow(self, name: str, *, skipna: bool = True, **kwargs) -> pa.Scalar: 

2044 """ 

2045 Return a pyarrow scalar result of performing the reduction operation. 

2046 

2047 Parameters 

2048 ---------- 

2049 name : str 

2050 Name of the function, supported values are: 

2051 { any, all, min, max, sum, mean, median, prod, 

2052 std, var, sem, kurt, skew }. 

2053 skipna : bool, default True 

2054 If True, skip NaN values. 

2055 **kwargs 

2056 Additional keyword arguments passed to the reduction function. 

2057 Currently, `ddof` is the only supported kwarg. 

2058 

2059 Returns 

2060 ------- 

2061 pyarrow scalar 

2062 

2063 Raises 

2064 ------ 

2065 TypeError : subclass does not define reductions 

2066 """ 

2067 pa_type = self._pa_array.type 

2068 

2069 data_to_reduce = self._pa_array 

2070 

2071 if name in ["any", "all"] and ( 

2072 pa.types.is_integer(pa_type) 

2073 or pa.types.is_floating(pa_type) 

2074 or pa.types.is_duration(pa_type) 

2075 or pa.types.is_decimal(pa_type) 

2076 ): 

2077 # pyarrow only supports any/all for boolean dtype, we allow 

2078 # for other dtypes, matching our non-pyarrow behavior 

2079 

2080 if pa.types.is_duration(pa_type): 

2081 data_to_cmp = self._pa_array.cast(pa.int64()) 

2082 else: 

2083 data_to_cmp = self._pa_array 

2084 

2085 not_eq = pc.not_equal(data_to_cmp, 0) 

2086 data_to_reduce = not_eq 

2087 

2088 elif name in ["min", "max", "sum"] and pa.types.is_duration(pa_type): 

2089 data_to_reduce = self._pa_array.cast(pa.int64()) 

2090 

2091 elif name in ["median", "mean", "std", "sem"] and pa.types.is_temporal(pa_type): 

2092 nbits = pa_type.bit_width 

2093 if nbits == 32: 

2094 data_to_reduce = self._pa_array.cast(pa.int32()) 

2095 else: 

2096 data_to_reduce = self._pa_array.cast(pa.int64()) 

2097 

2098 if name == "sem": 

2099 

2100 def pyarrow_meth(data, skip_nulls, **kwargs): 

2101 numerator = pc.stddev(data, skip_nulls=skip_nulls, **kwargs) 

2102 denominator = pc.sqrt_checked(pc.count(self._pa_array)) 

2103 return pc.divide_checked(numerator, denominator) 

2104 

2105 elif name == "sum" and ( 

2106 pa.types.is_string(pa_type) or pa.types.is_large_string(pa_type) 

2107 ): 

2108 

2109 def pyarrow_meth(data, skip_nulls, min_count=0): # type: ignore[misc] 

2110 mask = pc.is_null(data) if data.null_count > 0 else None 

2111 if skip_nulls: 

2112 if min_count > 0 and check_below_min_count( 

2113 (len(data),), 

2114 None if mask is None else mask.to_numpy(), 

2115 min_count, 

2116 ): 

2117 return pa.scalar(None, type=data.type) 

2118 if data.null_count > 0: 

2119 # binary_join returns null if there is any null -> 

2120 # have to filter out any nulls 

2121 data = data.filter(pc.invert(mask)) 

2122 elif mask is not None or check_below_min_count( 

2123 (len(data),), None, min_count 

2124 ): 

2125 return pa.scalar(None, type=data.type) 

2126 

2127 if pa.types.is_large_string(data.type): 

2128 # binary_join only supports string, not large_string 

2129 data = data.cast(pa.string()) 

2130 data_list = pa.ListArray.from_arrays( 

2131 [0, len(data)], data.combine_chunks() 

2132 )[0] 

2133 return pc.binary_join(data_list, "") 

2134 

2135 else: 

2136 pyarrow_name = { 

2137 "median": "quantile", 

2138 "prod": "product", 

2139 "std": "stddev", 

2140 "var": "variance", 

2141 }.get(name, name) 

2142 # error: Incompatible types in assignment 

2143 # (expression has type "Optional[Any]", variable has type 

2144 # "Callable[[Any, Any, KwArg(Any)], Any]") 

2145 pyarrow_meth = getattr(pc, pyarrow_name, None) # type: ignore[assignment] 

2146 if pyarrow_meth is None: 

2147 # Let ExtensionArray._reduce raise the TypeError 

2148 return super()._reduce(name, skipna=skipna, **kwargs) 

2149 

2150 # GH51624: pyarrow defaults to min_count=1, pandas behavior is min_count=0 

2151 if name in ["any", "all"] and "min_count" not in kwargs: 

2152 kwargs["min_count"] = 0 

2153 elif name == "median": 

2154 # GH 52679: Use quantile instead of approximate_median 

2155 kwargs["q"] = 0.5 

2156 

2157 try: 

2158 result = pyarrow_meth(data_to_reduce, skip_nulls=skipna, **kwargs) 

2159 except (AttributeError, NotImplementedError, TypeError) as err: 

2160 msg = ( 

2161 f"'{type(self).__name__}' with dtype {self.dtype} " 

2162 f"does not support operation '{name}' with pyarrow " 

2163 f"version {pa.__version__}. '{name}' may be supported by " 

2164 f"upgrading pyarrow." 

2165 ) 

2166 raise TypeError(msg) from err 

2167 if name == "median": 

2168 # GH 52679: Use quantile instead of approximate_median; returns array 

2169 result = result[0] 

2170 

2171 if name in ["min", "max", "sum"] and pa.types.is_duration(pa_type): 

2172 result = result.cast(pa_type) 

2173 if name in ["median", "mean"] and pa.types.is_temporal(pa_type): 

2174 nbits = pa_type.bit_width 

2175 if nbits == 32: 

2176 result = result.cast(pa.int32(), safe=False) 

2177 else: 

2178 result = result.cast(pa.int64(), safe=False) 

2179 result = result.cast(pa_type) 

2180 if name in ["std", "sem"] and pa.types.is_temporal(pa_type): 

2181 result = result.cast(pa.int64(), safe=False) 

2182 if pa.types.is_duration(pa_type): 

2183 result = result.cast(pa_type) 

2184 elif pa.types.is_time(pa_type): 

2185 result = result.cast(pa.duration(pa_type.unit)) 

2186 elif pa.types.is_date(pa_type): 

2187 # go with closest available unit, i.e. "s" 

2188 result = result.cast(pa.duration("s")) 

2189 else: 

2190 # i.e. timestamp 

2191 result = result.cast(pa.duration(pa_type.unit)) 

2192 

2193 return result 

2194 

2195 def _reduce( 

2196 self, name: str, *, skipna: bool = True, keepdims: bool = False, **kwargs 

2197 ): 

2198 """ 

2199 Return a scalar result of performing the reduction operation. 

2200 

2201 Parameters 

2202 ---------- 

2203 name : str 

2204 Name of the function, supported values are: 

2205 { any, all, min, max, sum, mean, median, prod, 

2206 std, var, sem, kurt, skew }. 

2207 skipna : bool, default True 

2208 If True, skip NaN values. 

2209 **kwargs 

2210 Additional keyword arguments passed to the reduction function. 

2211 Currently, `ddof` is the only supported kwarg. 

2212 

2213 Returns 

2214 ------- 

2215 scalar 

2216 

2217 Raises 

2218 ------ 

2219 TypeError : subclass does not define reductions 

2220 """ 

2221 result = self._reduce_calc(name, skipna=skipna, keepdims=keepdims, **kwargs) 

2222 if isinstance(result, pa.Array): 

2223 return self._from_pyarrow_array(result) 

2224 else: 

2225 return result 

2226 

2227 def _reduce_calc( 

2228 self, name: str, *, skipna: bool = True, keepdims: bool = False, **kwargs 

2229 ): 

2230 pa_result = self._reduce_pyarrow(name, skipna=skipna, **kwargs) 

2231 

2232 if keepdims: 

2233 if isinstance(pa_result, pa.Scalar): 

2234 result = pa.array([pa_result.as_py()], type=pa_result.type) 

2235 else: 

2236 result = pa.array( 

2237 [pa_result], 

2238 type=to_pyarrow_type(infer_dtype_from_scalar(pa_result)[0]), 

2239 ) 

2240 return result 

2241 

2242 if pc.is_null(pa_result).as_py(): 

2243 return self.dtype.na_value 

2244 elif isinstance(pa_result, pa.Scalar): 

2245 result = pa_result.as_py() 

2246 pa_type = pa_result.type 

2247 if pa.types.is_duration(pa_type) and pa_type.unit != "ns": 

2248 return Timedelta(result).as_unit(pa_type.unit) 

2249 elif pa.types.is_timestamp(pa_type) and pa_type.unit != "ns": 

2250 return Timestamp(result).as_unit(pa_type.unit) 

2251 return result 

2252 else: 

2253 return pa_result 

2254 

2255 def _explode(self): 

2256 """ 

2257 See Series.explode.__doc__. 

2258 """ 

2259 # child class explode method supports only list types; return 

2260 # default implementation for non list types. 

2261 if not hasattr(self.dtype, "pyarrow_dtype") or ( 

2262 not pa.types.is_list(self.dtype.pyarrow_dtype) 

2263 and not pa.types.is_large_list(self.dtype.pyarrow_dtype) 

2264 ): 

2265 return super()._explode() 

2266 values = self 

2267 counts = pa.compute.list_value_length(values._pa_array) 

2268 counts = counts.fill_null(1).to_numpy() 

2269 fill_value = pa.scalar([None], type=self._pa_array.type) 

2270 mask = counts == 0 

2271 if mask.any(): 

2272 # pc.if_else here is similar to `values[mask] = fill_value` 

2273 # but this avoids an object-dtype round-trip. 

2274 pa_values = pc.if_else(~mask, values._pa_array, fill_value) 

2275 values = self._from_pyarrow_array(pa_values) 

2276 counts = counts.copy() 

2277 counts[mask] = 1 

2278 values = values.fillna(fill_value) 

2279 values = self._from_pyarrow_array(pa.compute.list_flatten(values._pa_array)) 

2280 return values, counts 

2281 

2282 def __setitem__(self, key, value) -> None: 

2283 """Set one or more values inplace. 

2284 

2285 Parameters 

2286 ---------- 

2287 key : int, ndarray, or slice 

2288 When called from, e.g. ``Series.__setitem__``, ``key`` will be 

2289 one of 

2290 

2291 * scalar int 

2292 * ndarray of integers. 

2293 * boolean ndarray 

2294 * slice object 

2295 

2296 value : ExtensionDtype.type, Sequence[ExtensionDtype.type], or object 

2297 value or values to be set of ``key``. 

2298 

2299 Returns 

2300 ------- 

2301 None 

2302 """ 

2303 if self._readonly: 

2304 raise ValueError("Cannot modify read-only array") 

2305 

2306 # GH50085: unwrap 1D indexers 

2307 if isinstance(key, tuple) and len(key) == 1: 

2308 key = key[0] 

2309 

2310 key = check_array_indexer(self, key) 

2311 value = self._maybe_convert_setitem_value(value) 

2312 

2313 if com.is_null_slice(key): 

2314 # fast path (GH50248) 

2315 if ( 

2316 isinstance(value, (pa.Array, pa.ChunkedArray)) 

2317 and value.type == self._pa_array.type 

2318 and len(value) == len(self) 

2319 ): 

2320 # GH#67990 this adopts ``value`` as our backing array, so copy 

2321 # first if the caller may still own and mutate its buffers. 

2322 if _boxing_may_borrow_memory(value.type): 

2323 value = _copy_pyarrow_buffers(value) 

2324 data = value 

2325 else: 

2326 data = self._if_else(True, value, self._pa_array) 

2327 

2328 elif is_integer(key): 

2329 # fast path 

2330 key = cast(int, key) 

2331 n = len(self) 

2332 if key < 0: 

2333 key += n 

2334 if not 0 <= key < n: 

2335 raise IndexError( 

2336 f"index {key} is out of bounds for axis 0 with size {n}" 

2337 ) 

2338 if isinstance(value, pa.Scalar): 

2339 value = value.as_py() 

2340 elif is_list_like(value): 

2341 raise ValueError("Length of indexer and values mismatch") 

2342 chunks = [ 

2343 *self._pa_array[:key].chunks, 

2344 pa.array([value], type=self._pa_array.type, from_pandas=is_nan_na()), 

2345 *self._pa_array[key + 1 :].chunks, 

2346 ] 

2347 data = pa.chunked_array(chunks).combine_chunks() 

2348 

2349 elif is_bool_dtype(key): 

2350 key = np.asarray(key, dtype=np.bool_) 

2351 data = self._replace_with_mask(self._pa_array, key, value) 

2352 

2353 elif is_scalar(value) or isinstance(value, pa.Scalar): 

2354 mask = np.zeros(len(self), dtype=np.bool_) 

2355 mask[key] = True 

2356 data = self._if_else(mask, value, self._pa_array) 

2357 

2358 else: 

2359 indices = np.arange(len(self))[key] 

2360 if len(indices) != len(value): 

2361 raise ValueError("Length of indexer and values mismatch") 

2362 if len(indices) == 0: 

2363 return 

2364 # GH#58530 wrong item assignment by repeated key 

2365 _, argsort = np.unique(indices, return_index=True) 

2366 indices = indices[argsort] 

2367 value = value.take(argsort) 

2368 mask = np.zeros(len(self), dtype=np.bool_) 

2369 mask[indices] = True 

2370 data = self._replace_with_mask(self._pa_array, mask, value) 

2371 

2372 if isinstance(data, pa.Array): 

2373 data = pa.chunked_array([data]) 

2374 self._pa_array = data 

2375 

2376 def _rank_calc( 

2377 self, 

2378 *, 

2379 axis: AxisInt = 0, 

2380 method: str = "average", 

2381 na_option: str = "keep", 

2382 ascending: bool = True, 

2383 pct: bool = False, 

2384 ): 

2385 if axis != 0: 

2386 ranked = super()._rank( 

2387 axis=axis, 

2388 method=method, 

2389 na_option=na_option, 

2390 ascending=ascending, 

2391 pct=pct, 

2392 ) 

2393 # keep dtypes consistent with the implementation below 

2394 if method == "average" or pct: 

2395 pa_type = pa.float64() 

2396 else: 

2397 pa_type = pa.uint64() 

2398 result = pa.array(ranked, type=pa_type, from_pandas=is_nan_na()) 

2399 return result 

2400 

2401 data = self._pa_array.combine_chunks() 

2402 sort_keys = "ascending" if ascending else "descending" 

2403 null_placement = "at_start" if na_option == "top" else "at_end" 

2404 tiebreaker = "min" if method == "average" else method 

2405 

2406 result = pc.rank( 

2407 data, 

2408 sort_keys=sort_keys, 

2409 null_placement=null_placement, 

2410 tiebreaker=tiebreaker, 

2411 ) 

2412 

2413 if na_option == "keep": 

2414 mask = pc.is_null(self._pa_array) 

2415 null = pa.scalar(None, type=result.type) 

2416 result = pc.if_else(mask, null, result) 

2417 

2418 if method == "average": 

2419 result_max = pc.rank( 

2420 data, 

2421 sort_keys=sort_keys, 

2422 null_placement=null_placement, 

2423 tiebreaker="max", 

2424 ) 

2425 result_max = result_max.cast(pa.float64()) 

2426 result_min = result.cast(pa.float64()) 

2427 result = pc.divide(pc.add(result_min, result_max), 2) 

2428 

2429 if pct: 

2430 if not pa.types.is_floating(result.type): 

2431 result = result.cast(pa.float64()) 

2432 if method == "dense": 

2433 divisor = pc.max(result) 

2434 else: 

2435 divisor = pc.count(result) 

2436 result = pc.divide(result, divisor) 

2437 

2438 return result 

2439 

2440 def _rank( 

2441 self, 

2442 *, 

2443 axis: AxisInt = 0, 

2444 method: str = "average", 

2445 na_option: str = "keep", 

2446 ascending: bool = True, 

2447 pct: bool = False, 

2448 ) -> Self: 

2449 """ 

2450 See Series.rank.__doc__. 

2451 """ 

2452 return self._convert_rank_result( 

2453 self._rank_calc( 

2454 axis=axis, 

2455 method=method, 

2456 na_option=na_option, 

2457 ascending=ascending, 

2458 pct=pct, 

2459 ) 

2460 ) 

2461 

2462 def _quantile(self, qs: npt.NDArray[np.float64], interpolation: str) -> Self: 

2463 """ 

2464 Compute the quantiles of self for each quantile in `qs`. 

2465 

2466 Parameters 

2467 ---------- 

2468 qs : np.ndarray[float64] 

2469 interpolation: str 

2470 

2471 Returns 

2472 ------- 

2473 same type as self 

2474 """ 

2475 pa_dtype = self._pa_array.type 

2476 

2477 data = self._pa_array 

2478 if pa.types.is_temporal(pa_dtype): 

2479 # https://github.com/apache/arrow/issues/33769 in these cases 

2480 # we can cast to ints and back 

2481 nbits = pa_dtype.bit_width 

2482 if nbits == 32: 

2483 data = data.cast(pa.int32()) 

2484 else: 

2485 data = data.cast(pa.int64()) 

2486 

2487 result = pc.quantile(data, q=qs, interpolation=interpolation) 

2488 

2489 if pa.types.is_temporal(pa_dtype): 

2490 if pa.types.is_floating(result.type): 

2491 result = pc.floor(result) 

2492 nbits = pa_dtype.bit_width 

2493 if nbits == 32: 

2494 result = result.cast(pa.int32()) 

2495 else: 

2496 result = result.cast(pa.int64()) 

2497 result = result.cast(pa_dtype) 

2498 

2499 return self._from_pyarrow_array(result) 

2500 

2501 def _mode(self, dropna: bool = True) -> Self: 

2502 """ 

2503 Returns the mode(s) of the ExtensionArray. 

2504 

2505 Always returns `ExtensionArray` even if only one value. 

2506 

2507 Parameters 

2508 ---------- 

2509 dropna : bool, default True 

2510 Don't consider counts of NA values. 

2511 

2512 Returns 

2513 ------- 

2514 same type as self 

2515 Sorted, if possible. 

2516 """ 

2517 pa_type = self._pa_array.type 

2518 if pa.types.is_temporal(pa_type): 

2519 nbits = pa_type.bit_width 

2520 if nbits == 32: 

2521 data = self._pa_array.cast(pa.int32()) 

2522 elif nbits == 64: 

2523 data = self._pa_array.cast(pa.int64()) 

2524 else: 

2525 raise NotImplementedError(pa_type) 

2526 else: 

2527 data = self._pa_array 

2528 

2529 if dropna: 

2530 data = data.drop_null() 

2531 

2532 res = pc.value_counts(data) 

2533 most_common = res.field("values").filter( 

2534 pc.equal(res.field("counts"), pc.max(res.field("counts"))) 

2535 ) 

2536 

2537 if pa.types.is_temporal(pa_type): 

2538 most_common = most_common.cast(pa_type) 

2539 

2540 most_common = most_common.take(pc.array_sort_indices(most_common)) 

2541 return self._from_pyarrow_array(most_common) 

2542 

2543 def _maybe_convert_setitem_value(self, value): 

2544 """Maybe convert value to be pyarrow compatible.""" 

2545 try: 

2546 value = self._box_pa(value, self._pa_array.type) 

2547 except pa.ArrowTypeError as err: 

2548 msg = f"Invalid value '{value!s}' for dtype '{self.dtype}'" 

2549 raise TypeError(msg) from err 

2550 return value 

2551 

2552 def interpolate( 

2553 self, 

2554 *, 

2555 method: InterpolateOptions, 

2556 axis: int, 

2557 index, 

2558 limit, 

2559 limit_direction, 

2560 limit_area, 

2561 copy: bool, 

2562 **kwargs, 

2563 ) -> Self: 

2564 """ 

2565 See NDFrame.interpolate.__doc__. 

2566 """ 

2567 # NB: we return type(self) even if copy=False 

2568 if not self.dtype._is_numeric: 

2569 raise TypeError(f"Cannot interpolate with {self.dtype} dtype") 

2570 

2571 # GH#65345: a pyarrow-native fast path for 

2572 # method="linear"/limit_direction="forward" was removed here because 

2573 # it only handled isolated NAs (leaving consecutive and trailing NAs 

2574 # unfilled), truncated interpolated values for integer dtypes, and 

2575 # did not upcast to float64 like the general path below. 

2576 

2577 mask = self.isna() 

2578 if self.dtype.kind == "f": 

2579 data = self._pa_array.to_numpy() 

2580 elif self.dtype.kind in "iu": 

2581 data = self.to_numpy(dtype="f8", na_value=0.0) 

2582 else: 

2583 raise NotImplementedError( 

2584 f"interpolate is not implemented for dtype={self.dtype}" 

2585 ) 

2586 

2587 missing.interpolate_2d_inplace( 

2588 data, 

2589 method=method, 

2590 axis=0, 

2591 index=index, 

2592 limit=limit, 

2593 limit_direction=limit_direction, 

2594 limit_area=limit_area, 

2595 mask=mask, 

2596 **kwargs, 

2597 ) 

2598 return self._from_pyarrow_array(self._box_pa_array(pa.array(data, mask=mask))) 

2599 

2600 @classmethod 

2601 def _if_else( 

2602 cls, 

2603 cond: npt.NDArray[np.bool_] | bool, 

2604 left: ArrayLike | Scalar, 

2605 right: ArrayLike | Scalar, 

2606 ) -> pa.Array: 

2607 """ 

2608 Choose values based on a condition. 

2609 

2610 Analogous to pyarrow.compute.if_else, with logic 

2611 to fallback to numpy for unsupported types. 

2612 

2613 Parameters 

2614 ---------- 

2615 cond : npt.NDArray[np.bool_] or bool 

2616 left : ArrayLike | Scalar 

2617 right : ArrayLike | Scalar 

2618 

2619 Returns 

2620 ------- 

2621 pa.Array 

2622 """ 

2623 

2624 # TODO: Remove this part when pa.if_else is fixed (GH#64320) 

2625 def _maybe_combine(arr): 

2626 if not isinstance(arr, pa.ChunkedArray) or not ( 

2627 pa.types.is_string(arr.type) or pa.types.is_large_string(arr.type) 

2628 ): 

2629 return arr 

2630 if not any(c.offset != 0 for c in arr.chunks): 

2631 return arr 

2632 try: 

2633 return arr.combine_chunks() 

2634 except (pa.ArrowInvalid, pa.ArrowCapacityError, MemoryError): 

2635 return None 

2636 

2637 left_c, right_c = _maybe_combine(left), _maybe_combine(right) 

2638 if left_c is not None and right_c is not None: 

2639 try: 

2640 return pc.if_else(cond, left_c, right_c) 

2641 except pa.ArrowNotImplementedError: 

2642 pass 

2643 if left_c is not None: 

2644 left = left_c 

2645 if right_c is not None: 

2646 right = right_c 

2647 

2648 def _to_numpy_and_type(value) -> tuple[np.ndarray, pa.DataType | None]: 

2649 if isinstance(value, (pa.Array, pa.ChunkedArray)): 

2650 pa_type = value.type 

2651 elif isinstance(value, pa.Scalar): 

2652 pa_type = value.type 

2653 value = value.as_py() 

2654 else: 

2655 pa_type = None 

2656 return np.array(value, dtype=object), pa_type 

2657 

2658 left, left_type = _to_numpy_and_type(left) 

2659 right, right_type = _to_numpy_and_type(right) 

2660 pa_type = left_type or right_type 

2661 result = np.where(cond, left, right) 

2662 return pa.array(result, type=pa_type, from_pandas=is_nan_na()) 

2663 

2664 @classmethod 

2665 def _replace_with_mask( 

2666 cls, 

2667 values: pa.Array | pa.ChunkedArray, 

2668 mask: npt.NDArray[np.bool_] | bool, 

2669 replacements: ArrayLike | Scalar, 

2670 ) -> pa.Array | pa.ChunkedArray: 

2671 """ 

2672 Replace items selected with a mask. 

2673 

2674 Analogous to pyarrow.compute.replace_with_mask, with logic 

2675 to fallback to numpy for unsupported types. 

2676 

2677 Parameters 

2678 ---------- 

2679 values : pa.Array or pa.ChunkedArray 

2680 mask : npt.NDArray[np.bool_] or bool 

2681 replacements : ArrayLike or Scalar 

2682 Replacement value(s) 

2683 

2684 Returns 

2685 ------- 

2686 pa.Array or pa.ChunkedArray 

2687 """ 

2688 if isinstance(replacements, pa.ChunkedArray): 

2689 # replacements must be array or scalar, not ChunkedArray 

2690 replacements = replacements.combine_chunks() 

2691 if isinstance(values, pa.ChunkedArray) and pa.types.is_boolean(values.type): 

2692 # GH#52059 replace_with_mask segfaults for chunked array 

2693 # https://github.com/apache/arrow/issues/34634 

2694 values = values.combine_chunks() 

2695 try: 

2696 return pc.replace_with_mask(values, mask, replacements) 

2697 except pa.ArrowNotImplementedError: 

2698 pass 

2699 if isinstance(replacements, pa.Array): 

2700 replacements = np.array(replacements, dtype=object) 

2701 elif isinstance(replacements, pa.Scalar): 

2702 replacements = replacements.as_py() 

2703 

2704 result = np.array(values, dtype=object) 

2705 result[mask] = replacements 

2706 return pa.array(result, type=values.type, from_pandas=is_nan_na()) 

2707 

2708 # ------------------------------------------------------------------ 

2709 # GroupBy Methods 

2710 

2711 def _to_masked(self): 

2712 pa_dtype = self._pa_array.type 

2713 

2714 if pa.types.is_floating(pa_dtype) or pa.types.is_integer(pa_dtype): 

2715 na_value = 1 

2716 elif pa.types.is_boolean(pa_dtype): 

2717 na_value = True 

2718 else: 

2719 raise NotImplementedError 

2720 

2721 dtype = _arrow_dtype_mapping()[pa_dtype] 

2722 mask = self.isna() 

2723 arr = self.to_numpy(dtype=dtype.numpy_dtype, na_value=na_value) 

2724 return dtype.construct_array_type()(arr, mask) 

2725 

2726 def _groupby_op( 

2727 self, 

2728 *, 

2729 how: str, 

2730 has_dropped_na: bool, 

2731 min_count: int, 

2732 ngroups: int, 

2733 ids: npt.NDArray[np.intp], 

2734 **kwargs, 

2735 ): 

2736 if isinstance(self.dtype, StringDtype): 

2737 if how in [ 

2738 "prod", 

2739 "mean", 

2740 "median", 

2741 "cumsum", 

2742 "cumprod", 

2743 "std", 

2744 "sem", 

2745 "var", 

2746 "skew", 

2747 ]: 

2748 raise TypeError( 

2749 f"dtype '{self.dtype}' does not support operation '{how}'" 

2750 ) 

2751 return super()._groupby_op( 

2752 how=how, 

2753 has_dropped_na=has_dropped_na, 

2754 min_count=min_count, 

2755 ngroups=ngroups, 

2756 ids=ids, 

2757 **kwargs, 

2758 ) 

2759 

2760 # maybe convert to a compatible dtype optimized for groupby 

2761 values: ExtensionArray 

2762 pa_type = self._pa_array.type 

2763 if pa.types.is_timestamp(pa_type): 

2764 values = self._to_datetimearray() 

2765 elif pa.types.is_duration(pa_type): 

2766 values = self._to_timedeltaarray() 

2767 else: 

2768 values = self._to_masked() 

2769 

2770 result = values._groupby_op( 

2771 how=how, 

2772 has_dropped_na=has_dropped_na, 

2773 min_count=min_count, 

2774 ngroups=ngroups, 

2775 ids=ids, 

2776 **kwargs, 

2777 ) 

2778 if isinstance(result, np.ndarray): 

2779 return result 

2780 elif isinstance(result, BaseMaskedArray): 

2781 pa_result = result.__arrow_array__() 

2782 return self._from_pyarrow_array(pa_result) 

2783 else: 

2784 # DatetimeArray, TimedeltaArray 

2785 pa_result = pa.array(result) 

2786 return self._from_pyarrow_array(pa_result) 

2787 

2788 def _apply_elementwise(self, func: Callable) -> list[list[Any]]: 

2789 """Apply a callable to each element while maintaining the chunking structure.""" 

2790 return [ 

2791 [ 

2792 None if val is None else func(val) 

2793 for val in chunk.to_numpy(zero_copy_only=False) 

2794 ] 

2795 for chunk in self._pa_array.iterchunks() 

2796 ] 

2797 

2798 def _convert_bool_result(self, result, na=lib.no_default, method_name=None): 

2799 if na is not lib.no_default and not isna(na): # pyright: ignore [reportGeneralTypeIssues] 

2800 result = result.fill_null(na) 

2801 return self._from_pyarrow_array(result) 

2802 

2803 def _convert_int_result(self, result): 

2804 return self._from_pyarrow_array(result) 

2805 

2806 def _convert_rank_result(self, result): 

2807 return self._from_pyarrow_array(result) 

2808 

2809 def _str_count(self, pat: str, flags: int = 0) -> Self: 

2810 if flags: 

2811 raise NotImplementedError(f"count not implemented with {flags=}") 

2812 return self._from_pyarrow_array(pc.count_substring_regex(self._pa_array, pat)) 

2813 

2814 def _str_repeat(self, repeats: int | Sequence[int]) -> Self: 

2815 if not isinstance(repeats, int): 

2816 raise NotImplementedError( 

2817 f"repeat is not implemented when repeats is {type(repeats).__name__}" 

2818 ) 

2819 return self._from_pyarrow_array(pc.binary_repeat(self._pa_array, repeats)) 

2820 

2821 def _str_join(self, sep: str) -> Self: 

2822 if pa.types.is_string(self._pa_array.type) or pa.types.is_large_string( 

2823 self._pa_array.type 

2824 ): 

2825 result = self._apply_elementwise(list) 

2826 result = pa.chunked_array(result, type=pa.list_(pa.string())) 

2827 else: 

2828 result = self._pa_array 

2829 return self._from_pyarrow_array(pc.binary_join(result, sep)) 

2830 

2831 def _str_partition(self, sep: str, expand: bool) -> Self: 

2832 predicate = lambda val: val.partition(sep) 

2833 result = self._apply_elementwise(predicate) 

2834 return self._from_pyarrow_array(pa.chunked_array(result)) 

2835 

2836 def _str_rpartition(self, sep: str, expand: bool) -> Self: 

2837 predicate = lambda val: val.rpartition(sep) 

2838 result = self._apply_elementwise(predicate) 

2839 return self._from_pyarrow_array(pa.chunked_array(result)) 

2840 

2841 def _str_casefold(self) -> Self: 

2842 predicate = lambda val: val.casefold() 

2843 result = self._apply_elementwise(predicate) 

2844 return self._from_pyarrow_array(pa.chunked_array(result)) 

2845 

2846 def _str_encode(self, encoding: str, errors: str = "strict") -> Self: 

2847 predicate = lambda val: val.encode(encoding, errors) 

2848 result = self._apply_elementwise(predicate) 

2849 return self._from_pyarrow_array(pa.chunked_array(result)) 

2850 

2851 def _str_extract(self, pat: str, flags: int = 0, expand: bool = True): 

2852 if flags: 

2853 raise NotImplementedError("Only flags=0 is implemented.") 

2854 groups = re.compile(pat).groupindex.keys() 

2855 if len(groups) == 0: 

2856 raise ValueError(f"{pat=} must contain a symbolic group name.") 

2857 result = pc.extract_regex(self._pa_array, pat) 

2858 if expand: 

2859 return { 

2860 col: self._from_pyarrow_array(pc.struct_field(result, [i])) 

2861 for col, i in zip(groups, range(result.type.num_fields), strict=True) 

2862 } 

2863 else: 

2864 return type(self)(pc.struct_field(result, [0])) 

2865 

2866 def _str_findall(self, pat: str, flags: int = 0) -> Self: 

2867 regex = re.compile(pat, flags=flags) 

2868 predicate = lambda val: regex.findall(val) 

2869 result = self._apply_elementwise(predicate) 

2870 return self._from_pyarrow_array(pa.chunked_array(result)) 

2871 

2872 def _str_get_dummies(self, sep: str = "|", dtype: NpDtype | None = None): 

2873 if dtype is None: 

2874 dtype = np.bool_ 

2875 split = pc.split_pattern(self._pa_array, sep) 

2876 flattened_values = pc.list_flatten(split) 

2877 uniques = flattened_values.unique() 

2878 uniques_sorted = uniques.take(pa.compute.array_sort_indices(uniques)) 

2879 lengths = pc.list_value_length(split).fill_null(0).to_numpy() 

2880 n_rows = len(self) 

2881 n_cols = len(uniques) 

2882 indices = pc.index_in(flattened_values, uniques_sorted).to_numpy() 

2883 indices = indices + np.arange(n_rows).repeat(lengths) * n_cols 

2884 _dtype = pandas_dtype(dtype) 

2885 dummies_dtype: NpDtype 

2886 if isinstance(_dtype, np.dtype): 

2887 dummies_dtype = _dtype 

2888 else: 

2889 dummies_dtype = np.bool_ 

2890 dummies = np.zeros(n_rows * n_cols, dtype=dummies_dtype) 

2891 dummies[indices] = True 

2892 dummies = dummies.reshape((n_rows, n_cols)) 

2893 result = self._from_pyarrow_array(pa.array(list(dummies))) 

2894 return result, uniques_sorted.to_pylist() 

2895 

2896 def _str_index(self, sub: str, start: int = 0, end: int | None = None) -> Self: 

2897 predicate = lambda val: val.index(sub, start, end) 

2898 result = self._apply_elementwise(predicate) 

2899 return self._from_pyarrow_array(pa.chunked_array(result)) 

2900 

2901 def _str_rindex(self, sub: str, start: int = 0, end: int | None = None) -> Self: 

2902 predicate = lambda val: val.rindex(sub, start, end) 

2903 result = self._apply_elementwise(predicate) 

2904 return self._from_pyarrow_array(pa.chunked_array(result)) 

2905 

2906 def _str_normalize(self, form: Literal["NFC", "NFD", "NFKC", "NFKD"]) -> Self: 

2907 predicate = lambda val: unicodedata.normalize(form, val) 

2908 result = self._apply_elementwise(predicate) 

2909 return self._from_pyarrow_array(pa.chunked_array(result)) 

2910 

2911 def _str_rfind(self, sub: str, start: int = 0, end=None) -> Self: 

2912 predicate = lambda val: val.rfind(sub, start, end) 

2913 result = self._apply_elementwise(predicate) 

2914 return self._from_pyarrow_array(pa.chunked_array(result)) 

2915 

2916 def _str_split( 

2917 self, 

2918 pat: str | None = None, 

2919 n: int | None = -1, 

2920 expand: bool = False, 

2921 regex: bool | None = None, 

2922 ) -> Self: 

2923 if n in {-1, 0}: 

2924 n = None 

2925 if pat is None: 

2926 split_func = pc.utf8_split_whitespace 

2927 elif regex: 

2928 split_func = functools.partial(pc.split_pattern_regex, pattern=pat) 

2929 else: 

2930 split_func = functools.partial(pc.split_pattern, pattern=pat) 

2931 return self._from_pyarrow_array(split_func(self._pa_array, max_splits=n)) 

2932 

2933 def _str_rsplit(self, pat: str | None = None, n: int | None = -1) -> Self: 

2934 if n in {-1, 0}: 

2935 n = None 

2936 if pat is None: 

2937 return self._from_pyarrow_array( 

2938 pc.utf8_split_whitespace(self._pa_array, max_splits=n, reverse=True) 

2939 ) 

2940 return self._from_pyarrow_array( 

2941 pc.split_pattern(self._pa_array, pat, max_splits=n, reverse=True) 

2942 ) 

2943 

2944 def _str_translate(self, table: dict[int, str]) -> Self: 

2945 predicate = lambda val: val.translate(table) 

2946 result = self._apply_elementwise(predicate) 

2947 return self._from_pyarrow_array(pa.chunked_array(result)) 

2948 

2949 def _str_wrap(self, width: int, **kwargs) -> Self: 

2950 kwargs["width"] = width 

2951 tw = textwrap.TextWrapper(**kwargs) 

2952 predicate = lambda val: "\n".join(tw.wrap(val)) 

2953 result = self._apply_elementwise(predicate) 

2954 return self._from_pyarrow_array(pa.chunked_array(result)) 

2955 

2956 def _str_zfill(self, width: int) -> Self: 

2957 if pa_version_under21p0: 

2958 predicate = lambda val: val.zfill(width) 

2959 result = self._apply_elementwise(predicate) 

2960 return type(self)(pa.chunked_array(result)) 

2961 return type(self)(pc.utf8_zfill(self._pa_array, width)) 

2962 

2963 @property 

2964 def _dt_days(self) -> Self: 

2965 return self._from_pyarrow_array( 

2966 pa.array( 

2967 self._to_timedeltaarray().components.days, 

2968 from_pandas=True, 

2969 type=pa.int32(), 

2970 ) 

2971 ) 

2972 

2973 @property 

2974 def _dt_hours(self) -> Self: 

2975 return self._from_pyarrow_array( 

2976 pa.array( 

2977 self._to_timedeltaarray().components.hours, 

2978 from_pandas=True, 

2979 type=pa.int32(), 

2980 ) 

2981 ) 

2982 

2983 @property 

2984 def _dt_minutes(self) -> Self: 

2985 return self._from_pyarrow_array( 

2986 pa.array( 

2987 self._to_timedeltaarray().components.minutes, 

2988 from_pandas=True, 

2989 type=pa.int32(), 

2990 ) 

2991 ) 

2992 

2993 @property 

2994 def _dt_seconds(self) -> Self: 

2995 return self._from_pyarrow_array( 

2996 pa.array( 

2997 self._to_timedeltaarray().components.seconds, 

2998 from_pandas=True, 

2999 type=pa.int32(), 

3000 ) 

3001 ) 

3002 

3003 @property 

3004 def _dt_milliseconds(self) -> Self: 

3005 return self._from_pyarrow_array( 

3006 pa.array( 

3007 self._to_timedeltaarray().components.milliseconds, 

3008 from_pandas=True, 

3009 type=pa.int32(), 

3010 ) 

3011 ) 

3012 

3013 @property 

3014 def _dt_microseconds(self) -> Self: 

3015 return self._from_pyarrow_array( 

3016 pa.array( 

3017 self._to_timedeltaarray().components.microseconds, 

3018 from_pandas=True, 

3019 type=pa.int32(), 

3020 ) 

3021 ) 

3022 

3023 @property 

3024 def _dt_nanoseconds(self) -> Self: 

3025 return self._from_pyarrow_array( 

3026 pa.array( 

3027 self._to_timedeltaarray().components.nanoseconds, 

3028 from_pandas=True, 

3029 type=pa.int32(), 

3030 ) 

3031 ) 

3032 

3033 def _dt_to_pytimedelta(self) -> np.ndarray: 

3034 data = self._pa_array.to_pylist() 

3035 if self._dtype.pyarrow_dtype.unit == "ns": 

3036 data = [None if ts is None else ts.to_pytimedelta() for ts in data] 

3037 return np.array(data, dtype=object) 

3038 

3039 def _dt_total_seconds(self) -> Self: 

3040 unit = self._pa_array.type.unit 

3041 unit_per_second = {"s": 1.0, "ms": 1e3, "us": 1e6, "ns": 1e9} 

3042 result = pc.divide(pc.cast(self._pa_array, pa.int64()), unit_per_second[unit]) 

3043 return self._from_pyarrow_array(result) 

3044 

3045 def _dt_as_unit(self, unit: str) -> Self: 

3046 pa_type = self._pa_array.type 

3047 if pa.types.is_timestamp(pa_type): 

3048 target_type = pa.timestamp(unit, tz=pa_type.tz) 

3049 elif pa.types.is_duration(pa_type): 

3050 target_type = pa.duration(unit) 

3051 else: 

3052 raise NotImplementedError(f"as_unit not implemented for {pa_type}") 

3053 # Use safe=False to allow truncation, matching pandas as_unit behavior 

3054 result = pc.cast(self._pa_array, target_type, safe=False) 

3055 return self._from_pyarrow_array(result) 

3056 

3057 @property 

3058 def _dt_year(self) -> Self: 

3059 result = pc.year(self._pa_array) 

3060 return self._from_pyarrow_array(result) 

3061 

3062 @property 

3063 def _dt_day(self) -> Self: 

3064 result = pc.day(self._pa_array) 

3065 return self._from_pyarrow_array(result) 

3066 

3067 @property 

3068 def _dt_day_of_week(self) -> Self: 

3069 result = pc.day_of_week(self._pa_array) 

3070 return self._from_pyarrow_array(result) 

3071 

3072 _dt_dayofweek = _dt_day_of_week 

3073 _dt_weekday = _dt_day_of_week 

3074 

3075 @property 

3076 def _dt_day_of_year(self) -> Self: 

3077 result = pc.day_of_year(self._pa_array) 

3078 return self._from_pyarrow_array(result) 

3079 

3080 _dt_dayofyear = _dt_day_of_year 

3081 

3082 @property 

3083 def _dt_hour(self) -> Self: 

3084 result = pc.hour(self._pa_array) 

3085 return self._from_pyarrow_array(result) 

3086 

3087 def _dt_isocalendar(self) -> Self: 

3088 result = pc.iso_calendar(self._pa_array) 

3089 return self._from_pyarrow_array(result) 

3090 

3091 @property 

3092 def _dt_is_leap_year(self) -> Self: 

3093 result = pc.is_leap_year(self._pa_array) 

3094 return self._from_pyarrow_array(result) 

3095 

3096 @property 

3097 def _dt_is_month_start(self) -> Self: 

3098 result = pc.equal(pc.day(self._pa_array), 1) 

3099 return self._from_pyarrow_array(result) 

3100 

3101 @property 

3102 def _dt_is_month_end(self) -> Self: 

3103 result = pc.equal( 

3104 pc.days_between( 

3105 pc.floor_temporal(self._pa_array, unit="day"), 

3106 pc.ceil_temporal(self._pa_array, unit="month"), 

3107 ), 

3108 1, 

3109 ) 

3110 return self._from_pyarrow_array(result) 

3111 

3112 @property 

3113 def _dt_is_year_start(self) -> Self: 

3114 result = pc.and_( 

3115 pc.equal(pc.month(self._pa_array), 1), 

3116 pc.equal(pc.day(self._pa_array), 1), 

3117 ) 

3118 return self._from_pyarrow_array(result) 

3119 

3120 @property 

3121 def _dt_is_year_end(self) -> Self: 

3122 result = pc.and_( 

3123 pc.equal(pc.month(self._pa_array), 12), 

3124 pc.equal(pc.day(self._pa_array), 31), 

3125 ) 

3126 return self._from_pyarrow_array(result) 

3127 

3128 @property 

3129 def _dt_is_quarter_start(self) -> Self: 

3130 result = pc.equal( 

3131 pc.floor_temporal(self._pa_array, unit="quarter"), 

3132 pc.floor_temporal(self._pa_array, unit="day"), 

3133 ) 

3134 return self._from_pyarrow_array(result) 

3135 

3136 @property 

3137 def _dt_is_quarter_end(self) -> Self: 

3138 result = pc.equal( 

3139 pc.days_between( 

3140 pc.floor_temporal(self._pa_array, unit="day"), 

3141 pc.ceil_temporal(self._pa_array, unit="quarter"), 

3142 ), 

3143 1, 

3144 ) 

3145 return self._from_pyarrow_array(result) 

3146 

3147 @property 

3148 def _dt_days_in_month(self) -> Self: 

3149 result = pc.days_between( 

3150 pc.floor_temporal(self._pa_array, unit="month"), 

3151 pc.ceil_temporal(self._pa_array, unit="month"), 

3152 ) 

3153 return self._from_pyarrow_array(result) 

3154 

3155 _dt_daysinmonth = _dt_days_in_month 

3156 

3157 @property 

3158 def _dt_microsecond(self) -> Self: 

3159 # GH 59154 

3160 us = pc.microsecond(self._pa_array) 

3161 ms_to_us = pc.multiply(pc.millisecond(self._pa_array), 1000) 

3162 result = pc.add(us, ms_to_us) 

3163 return self._from_pyarrow_array(result) 

3164 

3165 @property 

3166 def _dt_minute(self) -> Self: 

3167 result = pc.minute(self._pa_array) 

3168 return self._from_pyarrow_array(result) 

3169 

3170 @property 

3171 def _dt_month(self) -> Self: 

3172 result = pc.month(self._pa_array) 

3173 return self._from_pyarrow_array(result) 

3174 

3175 @property 

3176 def _dt_nanosecond(self) -> Self: 

3177 result = pc.nanosecond(self._pa_array) 

3178 return self._from_pyarrow_array(result) 

3179 

3180 @property 

3181 def _dt_quarter(self) -> Self: 

3182 result = pc.quarter(self._pa_array) 

3183 return self._from_pyarrow_array(result) 

3184 

3185 @property 

3186 def _dt_second(self) -> Self: 

3187 result = pc.second(self._pa_array) 

3188 return self._from_pyarrow_array(result) 

3189 

3190 @property 

3191 def _dt_date(self) -> Self: 

3192 result = self._pa_array.cast(pa.date32()) 

3193 return self._from_pyarrow_array(result) 

3194 

3195 @property 

3196 def _dt_time(self) -> Self: 

3197 unit = ( 

3198 self.dtype.pyarrow_dtype.unit 

3199 if self.dtype.pyarrow_dtype.unit in {"us", "ns"} 

3200 else "ns" 

3201 ) 

3202 result = self._pa_array.cast(pa.time64(unit)) 

3203 return self._from_pyarrow_array(result) 

3204 

3205 @property 

3206 def _dt_tz(self): 

3207 return timezones.maybe_get_tz(self.dtype.pyarrow_dtype.tz) 

3208 

3209 @property 

3210 def _dt_unit(self): 

3211 return self.dtype.pyarrow_dtype.unit 

3212 

3213 def _dt_normalize(self) -> Self: 

3214 result = pc.floor_temporal(self._pa_array, 1, "day") 

3215 return self._from_pyarrow_array(result) 

3216 

3217 def _dt_strftime(self, format: str) -> Self: 

3218 result = pc.strftime(self._pa_array, format=format) 

3219 return self._from_pyarrow_array(result) 

3220 

3221 def _round_temporally( 

3222 self, 

3223 method: Literal["ceil", "floor", "round"], 

3224 freq, 

3225 ambiguous: TimeAmbiguous = "raise", 

3226 nonexistent: TimeNonexistent = "raise", 

3227 ) -> Self: 

3228 if ambiguous != "raise": 

3229 raise NotImplementedError("ambiguous is not supported.") 

3230 if nonexistent != "raise": 

3231 raise NotImplementedError("nonexistent is not supported.") 

3232 offset = to_offset(freq) 

3233 if offset is None: 

3234 raise ValueError(f"Must specify a valid frequency: {freq}") 

3235 pa_supported_unit = { 

3236 "Y": "year", 

3237 "YS": "year", 

3238 "Q": "quarter", 

3239 "QS": "quarter", 

3240 "M": "month", 

3241 "MS": "month", 

3242 "W": "week", 

3243 "D": "day", 

3244 "h": "hour", 

3245 "min": "minute", 

3246 "s": "second", 

3247 "ms": "millisecond", 

3248 "us": "microsecond", 

3249 "ns": "nanosecond", 

3250 } 

3251 unit = pa_supported_unit.get(offset._prefix, None) 

3252 if unit is None: 

3253 raise ValueError(f"{freq=} is not supported") 

3254 multiple = offset.n 

3255 rounding_method = getattr(pc, f"{method}_temporal") 

3256 result = rounding_method(self._pa_array, multiple=multiple, unit=unit) 

3257 return self._from_pyarrow_array(result) 

3258 

3259 def _dt_ceil( 

3260 self, 

3261 freq, 

3262 ambiguous: TimeAmbiguous = "raise", 

3263 nonexistent: TimeNonexistent = "raise", 

3264 ) -> Self: 

3265 return self._round_temporally("ceil", freq, ambiguous, nonexistent) 

3266 

3267 def _dt_floor( 

3268 self, 

3269 freq, 

3270 ambiguous: TimeAmbiguous = "raise", 

3271 nonexistent: TimeNonexistent = "raise", 

3272 ) -> Self: 

3273 return self._round_temporally("floor", freq, ambiguous, nonexistent) 

3274 

3275 def _dt_round( 

3276 self, 

3277 freq, 

3278 ambiguous: TimeAmbiguous = "raise", 

3279 nonexistent: TimeNonexistent = "raise", 

3280 ) -> Self: 

3281 return self._round_temporally("round", freq, ambiguous, nonexistent) 

3282 

3283 def _dt_day_name(self, locale: str | None = None) -> Self: 

3284 if locale is None: 

3285 locale = "C" 

3286 result = pc.strftime(self._pa_array, format="%A", locale=locale) 

3287 return self._from_pyarrow_array(result) 

3288 

3289 def _dt_month_name(self, locale: str | None = None) -> Self: 

3290 if locale is None: 

3291 locale = "C" 

3292 result = pc.strftime(self._pa_array, format="%B", locale=locale) 

3293 return self._from_pyarrow_array(result) 

3294 

3295 def _dt_to_pydatetime(self) -> Series: 

3296 from pandas import Series 

3297 

3298 if pa.types.is_date(self.dtype.pyarrow_dtype): 

3299 raise ValueError( 

3300 f"to_pydatetime cannot be called with {self.dtype.pyarrow_dtype} type. " 

3301 "Convert to pyarrow timestamp type." 

3302 ) 

3303 data = self._pa_array.to_pylist() 

3304 if self._dtype.pyarrow_dtype.unit == "ns": 

3305 data = [None if ts is None else ts.to_pydatetime(warn=False) for ts in data] 

3306 return Series(data, dtype=object) 

3307 

3308 def _dt_tz_localize( 

3309 self, 

3310 tz, 

3311 ambiguous: TimeAmbiguous = "raise", 

3312 nonexistent: TimeNonexistent = "raise", 

3313 ) -> Self: 

3314 if ambiguous != "raise": 

3315 raise NotImplementedError(f"{ambiguous=} is not supported") 

3316 nonexistent_pa = { 

3317 "raise": "raise", 

3318 "shift_backward": "earliest", 

3319 "shift_forward": "latest", 

3320 }.get( 

3321 nonexistent, # type: ignore[arg-type] 

3322 None, 

3323 ) 

3324 if nonexistent_pa is None: 

3325 raise NotImplementedError(f"{nonexistent=} is not supported") 

3326 if tz is None: 

3327 result = pc.local_timestamp(self._pa_array) 

3328 else: 

3329 result = pc.assume_timezone( 

3330 self._pa_array, str(tz), ambiguous=ambiguous, nonexistent=nonexistent_pa 

3331 ) 

3332 return self._from_pyarrow_array(result) 

3333 

3334 def _dt_tz_convert(self, tz) -> Self: 

3335 if self.dtype.pyarrow_dtype.tz is None: 

3336 raise TypeError( 

3337 "Cannot convert tz-naive timestamps, use tz_localize to localize" 

3338 ) 

3339 current_unit = self.dtype.pyarrow_dtype.unit 

3340 result = self._pa_array.cast(pa.timestamp(current_unit, tz)) 

3341 return self._from_pyarrow_array(result) 

3342 

3343 

3344def transpose_homogeneous_pyarrow( 

3345 arrays: Sequence[ArrowExtensionArray], 

3346) -> list[ArrowExtensionArray]: 

3347 """Transpose arrow extension arrays in a list, but faster. 

3348 

3349 Input should be a list of arrays of equal length and all have the same 

3350 dtype. The caller is responsible for ensuring validity of input data. 

3351 """ 

3352 arrays = list(arrays) 

3353 nrows, ncols = len(arrays[0]), len(arrays) 

3354 indices = np.arange(nrows * ncols).reshape(ncols, nrows).T.reshape(-1) 

3355 arr = pa.chunked_array([chunk for arr in arrays for chunk in arr._pa_array.chunks]) 

3356 arr = arr.take(indices) 

3357 return [ArrowExtensionArray(arr.slice(i * ncols, ncols)) for i in range(nrows)]