1from __future__ import annotations
2
3from datetime import (
4 date,
5 datetime,
6)
7import functools
8import operator
9import re
10import textwrap
11from typing import (
12 TYPE_CHECKING,
13 Any,
14 Literal,
15 Self,
16 cast,
17 overload,
18)
19import unicodedata
20import warnings
21
22import numpy as np
23
24from pandas._config import is_nan_na
25
26from pandas._libs import lib
27from pandas._libs.missing import is_pdna_or_none
28from pandas._libs.tslibs import (
29 Timedelta,
30 Timestamp,
31 timezones,
32)
33from pandas.compat import (
34 HAS_PYARROW,
35 PYARROW_MIN_VERSION,
36 pa_version_under16p0,
37 pa_version_under21p0,
38)
39from pandas.errors import Pandas4Warning
40from pandas.util._decorators import (
41 doc,
42 set_module,
43)
44from pandas.util._exceptions import find_stack_level
45
46from pandas.core.dtypes.cast import (
47 can_hold_element,
48 construct_1d_object_array_from_listlike,
49 infer_dtype_from_scalar,
50)
51from pandas.core.dtypes.common import (
52 is_array_like,
53 is_bool_dtype,
54 is_float_dtype,
55 is_integer,
56 is_list_like,
57 is_numeric_dtype,
58 is_scalar,
59 is_string_dtype,
60 pandas_dtype,
61)
62from pandas.core.dtypes.dtypes import DatetimeTZDtype
63from pandas.core.dtypes.missing import isna
64
65from pandas.core import (
66 algorithms as algos,
67 missing,
68 ops,
69 roperator,
70)
71from pandas.core.algorithms import map_array
72from pandas.core.arraylike import OpsMixin
73from pandas.core.arrays._arrow_string_mixins import ArrowStringArrayMixin
74from pandas.core.arrays._utils import to_numpy_dtype_inference
75from pandas.core.arrays.base import (
76 ExtensionArray,
77 ExtensionArraySupportsAnyAll,
78)
79from pandas.core.arrays.masked import BaseMaskedArray
80from pandas.core.arrays.string_ import StringDtype
81import pandas.core.common as com
82from pandas.core.construction import extract_array
83from pandas.core.indexers import (
84 check_array_indexer,
85 getitem_returns_view,
86 unpack_tuple_and_ellipses,
87 validate_indices,
88)
89from pandas.core.nanops import check_below_min_count
90
91from pandas.io._util import _arrow_dtype_mapping
92from pandas.tseries.frequencies import to_offset
93
94if HAS_PYARROW:
95 import pyarrow as pa
96 import pyarrow.compute as pc
97
98 from pandas.compat.pyarrow import _safe_fill_null
99
100 from pandas.core.dtypes.dtypes import ArrowDtype
101
102 ARROW_CMP_FUNCS = {
103 "eq": pc.equal,
104 "ne": pc.not_equal,
105 "lt": pc.less,
106 "gt": pc.greater,
107 "le": pc.less_equal,
108 "ge": pc.greater_equal,
109 }
110
111 ARROW_LOGICAL_FUNCS = {
112 "and_": pc.and_kleene,
113 "rand_": lambda x, y: pc.and_kleene(y, x),
114 "or_": pc.or_kleene,
115 "ror_": lambda x, y: pc.or_kleene(y, x),
116 "xor": pc.xor,
117 "rxor": lambda x, y: pc.xor(y, x),
118 }
119
120 ARROW_BIT_WISE_FUNCS = {
121 "and_": pc.bit_wise_and,
122 "rand_": lambda x, y: pc.bit_wise_and(y, x),
123 "or_": pc.bit_wise_or,
124 "ror_": lambda x, y: pc.bit_wise_or(y, x),
125 "xor": pc.bit_wise_xor,
126 "rxor": lambda x, y: pc.bit_wise_xor(y, x),
127 }
128
129 def cast_for_truediv(
130 arrow_array: pa.ChunkedArray, pa_object: pa.Array | pa.Scalar
131 ) -> tuple[pa.ChunkedArray, pa.Array | pa.Scalar]:
132 # Ensure int / int -> float mirroring Python/Numpy behavior
133 # as pc.divide_checked(int, int) -> int
134 if pa.types.is_integer(arrow_array.type) and pa.types.is_integer(
135 pa_object.type
136 ):
137 # GH: 56645.
138 # https://github.com/apache/arrow/issues/35563
139 return pc.cast(arrow_array, pa.float64(), safe=False), pc.cast(
140 pa_object, pa.float64(), safe=False
141 )
142
143 return arrow_array, pa_object
144
145 def floordiv_compat(
146 left: pa.ChunkedArray | pa.Array | pa.Scalar,
147 right: pa.ChunkedArray | pa.Array | pa.Scalar,
148 ) -> pa.ChunkedArray:
149 # TODO: Replace with pyarrow floordiv kernel.
150 # https://github.com/apache/arrow/issues/39386
151 if pa.types.is_integer(left.type) and pa.types.is_integer(right.type):
152 divided = pc.divide_checked(left, right)
153 if pa.types.is_signed_integer(divided.type):
154 # GH 56676
155 has_remainder = pc.not_equal(pc.multiply(divided, right), left)
156 has_one_negative_operand = pc.less(
157 pc.bit_wise_xor(left, right),
158 pa.scalar(0, type=divided.type),
159 )
160 result = pc.if_else(
161 pc.and_(
162 has_remainder,
163 has_one_negative_operand,
164 ),
165 # GH: 55561
166 pc.subtract(divided, pa.scalar(1, type=divided.type)),
167 divided,
168 )
169 else:
170 result = divided
171 result = result.cast(left.type)
172 else:
173 divided = pc.divide(left, right)
174 result = pc.floor(divided)
175 return result
176
177 ARROW_ARITHMETIC_FUNCS = {
178 "add": pc.add_checked,
179 "radd": lambda x, y: pc.add_checked(y, x),
180 "sub": pc.subtract_checked,
181 "rsub": lambda x, y: pc.subtract_checked(y, x),
182 "mul": pc.multiply_checked,
183 "rmul": lambda x, y: pc.multiply_checked(y, x),
184 "truediv": lambda x, y: pc.divide(*cast_for_truediv(x, y)),
185 "rtruediv": lambda x, y: pc.divide(*cast_for_truediv(y, x)),
186 "floordiv": lambda x, y: floordiv_compat(x, y),
187 "rfloordiv": lambda x, y: floordiv_compat(y, x),
188 "mod": NotImplemented,
189 "rmod": NotImplemented,
190 "divmod": NotImplemented,
191 "rdivmod": NotImplemented,
192 "pow": pc.power_checked,
193 "rpow": lambda x, y: pc.power_checked(y, x),
194 }
195
196if TYPE_CHECKING:
197 from collections.abc import (
198 Callable,
199 Sequence,
200 )
201
202 from pandas._libs.missing import NAType
203 from pandas._typing import (
204 ArrayLike,
205 AxisInt,
206 Dtype,
207 FillnaOptions,
208 InterpolateOptions,
209 Iterator,
210 NpDtype,
211 NumpySorter,
212 NumpyValueArrayLike,
213 PositionalIndexer,
214 Scalar,
215 SortKind,
216 TakeIndexer,
217 TimeAmbiguous,
218 TimeNonexistent,
219 npt,
220 )
221
222 from pandas.core.dtypes.dtypes import ExtensionDtype
223
224 from pandas import Series
225 from pandas.core.arrays.datetimes import DatetimeArray
226 from pandas.core.arrays.timedeltas import TimedeltaArray
227
228
229def to_pyarrow_type(
230 dtype: ArrowDtype | pa.DataType | Dtype | None,
231) -> pa.DataType | None:
232 """
233 Convert dtype to a pyarrow type instance.
234 """
235 if isinstance(dtype, ArrowDtype):
236 return dtype.pyarrow_dtype
237 elif isinstance(dtype, pa.DataType):
238 return dtype
239 elif isinstance(dtype, DatetimeTZDtype):
240 return pa.timestamp(dtype.unit, dtype.tz)
241 elif dtype:
242 try:
243 # Accepts python types too
244 # Doesn't handle all numpy types
245 return pa.from_numpy_dtype(dtype)
246 except pa.ArrowNotImplementedError:
247 pass
248 return None
249
250
251def _is_varbinary_type(pa_type: pa.DataType) -> bool:
252 """
253 Whether this is one of string, large_string, binary and large_binary.
254
255 pc.if_else misreads a non-zero offset for exactly these four, silently
256 truncating values (GH#64320, https://github.com/apache/arrow/issues/49410).
257 Other offset-carrying layouts such as list and map are unaffected.
258 """
259 return (
260 pa.types.is_string(pa_type)
261 or pa.types.is_large_string(pa_type)
262 or pa.types.is_binary(pa_type)
263 or pa.types.is_large_binary(pa_type)
264 )
265
266
267def _is_string_or_binary_view(typ):
268 return not pa_version_under16p0 and (
269 pa.types.is_string_view(typ) or pa.types.is_binary_view(typ)
270 )
271
272
273def _boxing_may_borrow_memory(pa_type: pa.DataType) -> bool:
274 """
275 Whether ``pa.array`` on this type can return a view on caller-owned memory.
276
277 Zero-copy over numpy/masked arrays for fixed-width layouts; character
278 layouts always repack, so copying those would cost a full copy of the
279 character data for no safety gain. Nested types may have a zero-copy child.
280 """
281 return not (_is_varbinary_type(pa_type) or _is_string_or_binary_view(pa_type))
282
283
284def _copy_pyarrow_buffers(
285 pa_array: pa.Array | pa.ChunkedArray,
286) -> pa.Array | pa.ChunkedArray:
287 """
288 Return an equal array that owns its buffers (GH#67990).
289
290 ``pa.concat_arrays`` reuses the ``dictionary`` child rather than copying it,
291 so dictionary types are rebuilt from copies of both halves.
292 """
293 if isinstance(pa_array, pa.ChunkedArray):
294 return pa.chunked_array(
295 [_copy_pyarrow_buffers(chunk) for chunk in pa_array.chunks],
296 type=pa_array.type,
297 )
298 if pa.types.is_dictionary(pa_array.type):
299 return pa.DictionaryArray.from_arrays(
300 _copy_pyarrow_buffers(pa_array.indices),
301 _copy_pyarrow_buffers(pa_array.dictionary),
302 ordered=pa_array.type.ordered,
303 )
304 return pa.concat_arrays([pa_array])
305
306
307@set_module("pandas.arrays")
308class ArrowExtensionArray(
309 OpsMixin,
310 ExtensionArraySupportsAnyAll,
311 ArrowStringArrayMixin,
312):
313 """
314 Pandas ExtensionArray backed by a PyArrow ChunkedArray.
315
316 .. warning::
317
318 ArrowExtensionArray is considered experimental. The implementation and
319 parts of the API may change without warning.
320
321 Parameters
322 ----------
323 values : pyarrow.Array or pyarrow.ChunkedArray
324 The input data to initialize the ArrowExtensionArray.
325
326 Attributes
327 ----------
328 None
329
330 Methods
331 -------
332 None
333
334 Returns
335 -------
336 ArrowExtensionArray
337
338 See Also
339 --------
340 array : Create a Pandas array with a specified dtype.
341 DataFrame.to_feather : Write a DataFrame to the binary Feather format.
342 read_feather : Load a feather-format object from the file path.
343
344 Notes
345 -----
346 Most methods are implemented using `pyarrow compute functions. <https://arrow.apache.org/docs/python/api/compute.html>`__
347 Some methods may either raise an exception or raise a ``PerformanceWarning`` if an
348 associated compute function is not available based on the installed version of PyArrow.
349
350 Please install the latest version of PyArrow to enable the best functionality and avoid
351 potential bugs in prior versions of PyArrow.
352
353 Examples
354 --------
355 Create an ArrowExtensionArray with :func:`pandas.array`:
356
357 >>> pd.array([1, 1, None], dtype="int64[pyarrow]")
358 <ArrowExtensionArray>
359 [1, 1, <NA>]
360 Length: 3, dtype: int64[pyarrow]
361 """ # noqa: E501 (http link too long)
362
363 _pa_array: pa.ChunkedArray
364 _dtype: ArrowDtype
365
366 def __init__(self, values: pa.Array | pa.ChunkedArray) -> None:
367 if not HAS_PYARROW:
368 msg = (
369 f"pyarrow>={PYARROW_MIN_VERSION} is required for PyArrow "
370 "backed ArrowExtensionArray."
371 )
372 raise ImportError(msg)
373 if isinstance(values, pa.Array):
374 self._pa_array = pa.chunked_array([values])
375 elif isinstance(values, pa.ChunkedArray):
376 self._pa_array = values
377 else:
378 raise ValueError(
379 f"Unsupported type '{type(values)}' for ArrowExtensionArray"
380 )
381 self._dtype = ArrowDtype(self._pa_array.type)
382
383 @classmethod
384 def _from_sequence(
385 cls, scalars, *, dtype: Dtype | None = None, copy: bool = False
386 ) -> Self:
387 """
388 Construct a new ExtensionArray from a sequence of scalars.
389 """
390 pa_type = to_pyarrow_type(dtype)
391 pa_array = cls._box_pa_array(scalars, pa_type=pa_type, copy=copy)
392 arr = cls(pa_array)
393 return arr
394
395 @classmethod
396 def _from_sequence_of_strings(
397 cls, strings, *, dtype: ExtensionDtype, copy: bool = False
398 ) -> Self:
399 """
400 Construct a new ExtensionArray from a sequence of strings.
401 """
402 mask = isna(strings)
403
404 if isinstance(strings, cls):
405 strings = strings._pa_array
406
407 pa_type = to_pyarrow_type(dtype)
408 if (
409 pa_type is None
410 or pa.types.is_binary(pa_type)
411 or pa.types.is_string(pa_type)
412 or pa.types.is_large_string(pa_type)
413 ):
414 # pa_type is None: Let pa.array infer
415 # pa_type is string/binary: scalars already correct type
416 scalars = strings
417 elif pa.types.is_timestamp(pa_type):
418 from pandas.core.tools.datetimes import to_datetime
419
420 scalars = to_datetime(strings, errors="raise")
421 elif pa.types.is_date(pa_type):
422 from pandas.core.tools.datetimes import to_datetime
423
424 scalars = to_datetime(strings, errors="raise").date
425 scalars = pa.array(scalars, type=pa_type, mask=mask)
426 elif pa.types.is_duration(pa_type):
427 from pandas.core.tools.timedeltas import to_timedelta
428
429 scalars = to_timedelta(strings, errors="raise")
430
431 if pa_type.unit != "ns":
432 # GH51175: test_from_sequence_of_strings_pa_array
433 # attempt to parse as int64 reflecting pyarrow's
434 # duration to string casting behavior
435 mask = isna(scalars)
436 if not isinstance(strings, (pa.Array, pa.ChunkedArray)):
437 strings = pa.array(strings, type=pa.string(), mask=mask)
438 strings = pc.if_else(mask, None, strings)
439 try:
440 scalars = strings.cast(pa.int64())
441 except pa.ArrowInvalid:
442 pass
443 elif pa.types.is_time(pa_type):
444 from pandas.core.tools.times import to_time
445
446 # "coerce" to allow "null times" (None) to not raise
447 scalars = to_time(strings, errors="coerce")
448 elif pa.types.is_boolean(pa_type):
449 # pyarrow string->bool casting is case-insensitive:
450 # "true" or "1" -> True
451 # "false" or "0" -> False
452 # Note: BooleanArray was previously used to parse these strings
453 # and allows "1.0" and "0.0". Pyarrow casting does not support
454 # this, but we allow it here.
455 if isinstance(strings, (pa.Array, pa.ChunkedArray)):
456 scalars = strings
457 else:
458 scalars = pa.array(strings, type=pa.string(), mask=mask)
459 scalars = pc.if_else(pc.equal(scalars, "1.0"), "1", scalars)
460 scalars = pc.if_else(pc.equal(scalars, "0.0"), "0", scalars)
461 scalars = scalars.cast(pa.bool_())
462 elif (
463 pa.types.is_integer(pa_type)
464 or pa.types.is_floating(pa_type)
465 or pa.types.is_decimal(pa_type)
466 ):
467 from pandas.core.tools.numeric import to_numeric
468
469 scalars = to_numeric(strings, errors="raise")
470 if isinstance(strings, (pa.Array, pa.ChunkedArray)):
471 scalars = strings.cast(pa_type)
472 elif mask is not None:
473 scalars = pa.array(scalars, mask=mask, type=pa_type)
474
475 else:
476 raise NotImplementedError(
477 f"Converting strings to {pa_type} is not implemented."
478 )
479 return cls._from_sequence(scalars, dtype=pa_type, copy=copy)
480
481 def _from_pyarrow_array(self, pa_array):
482 """
483 Construct from the pyarrow array result of an operation, for
484 compatibility with ArrowStringArray.
485 """
486 return type(self)(pa_array)
487
488 def _cast_pointwise_result(self, values) -> ArrayLike:
489 if len(values) == 0:
490 # Retain our dtype
491 return self[:0].copy()
492
493 try:
494 if self.dtype.kind in "iufc" and not is_nan_na():
495 values = np.asarray(values, dtype=object)
496 mask = is_pdna_or_none(values)
497 arr = pa.array(values, mask=mask)
498 else:
499 arr = pa.array(values, from_pandas=True)
500 except (ValueError, TypeError):
501 # e.g. test_by_column_values_with_same_starting_value with nested
502 # values, one entry of which is an ArrowStringArray
503 # or test_agg_lambda_complex128_dtype_conversion for complex values
504 values = np.asarray(values, dtype=object)
505 return lib.maybe_convert_objects(values, convert_non_numeric=True)
506
507 if pa.types.is_null(arr.type):
508 if lib.infer_dtype(values) == "decimal":
509 # GH#62522; the specific decimal precision here is arbitrary
510 arr = arr.cast(pa.decimal128(1))
511 if pa.types.is_duration(arr.type):
512 # workaround for https://github.com/apache/arrow/issues/40620
513 result = ArrowExtensionArray._from_sequence(values)
514 if pa.types.is_duration(self._pa_array.type):
515 result = result.astype(self.dtype) # type: ignore[assignment]
516 elif pa.types.is_timestamp(self._pa_array.type):
517 # Try to retain original unit
518 new_dtype = ArrowDtype(pa.duration(self._pa_array.type.unit))
519 try:
520 result = result.astype(new_dtype) # type: ignore[assignment]
521 except ValueError:
522 pass
523 elif pa.types.is_date64(self._pa_array.type):
524 # Try to match unit we get on non-pointwise op
525 dtype = ArrowDtype(pa.duration("ms"))
526 result = result.astype(dtype) # type: ignore[assignment]
527 elif pa.types.is_date(self._pa_array.type):
528 # Try to match unit we get on non-pointwise op
529 dtype = ArrowDtype(pa.duration("s"))
530 result = result.astype(dtype) # type: ignore[assignment]
531 return result
532
533 elif pa.types.is_date(arr.type) and pa.types.is_date(self._pa_array.type):
534 arr = arr.cast(self._pa_array.type)
535 elif pa.types.is_time(arr.type) and pa.types.is_time(self._pa_array.type):
536 arr = arr.cast(self._pa_array.type)
537 elif pa.types.is_decimal(arr.type) and pa.types.is_decimal(self._pa_array.type):
538 arr = arr.cast(self._pa_array.type)
539 elif pa.types.is_integer(arr.type) and pa.types.is_integer(self._pa_array.type):
540 try:
541 arr = arr.cast(self._pa_array.type)
542 except pa.lib.ArrowInvalid:
543 # e.g. test_combine_add if we can't cast
544 pass
545 elif pa.types.is_floating(arr.type) and pa.types.is_floating(
546 self._pa_array.type
547 ):
548 try:
549 arr = arr.cast(self._pa_array.type)
550 except pa.lib.ArrowInvalid:
551 # e.g. test_combine_add if we can't cast
552 pass
553
554 if isinstance(self.dtype, StringDtype):
555 if pa.types.is_string(arr.type) or pa.types.is_large_string(arr.type):
556 # ArrowStringArray preserves dtype.na_value
557 return self._from_pyarrow_array(arr)
558 if self.dtype.na_value is np.nan:
559 # ArrowEA has different semantics, so we return numpy-based
560 # result instead
561 values = np.asarray(values, dtype=object)
562 return lib.maybe_convert_objects(values, convert_non_numeric=True)
563 return ArrowExtensionArray(arr)
564 return self._from_pyarrow_array(arr)
565
566 @classmethod
567 def _box_pa(
568 cls, value, pa_type: pa.DataType | None = None
569 ) -> pa.Array | pa.ChunkedArray | pa.Scalar:
570 """
571 Box value into a pyarrow Array, ChunkedArray or Scalar.
572
573 Parameters
574 ----------
575 value : any
576 pa_type : pa.DataType | None
577
578 Returns
579 -------
580 pa.Array or pa.ChunkedArray or pa.Scalar
581 """
582 if isinstance(value, pa.Scalar) or not is_list_like(value):
583 return cls._box_pa_scalar(value, pa_type)
584 return cls._box_pa_array(value, pa_type)
585
586 @classmethod
587 def _box_pa_scalar(cls, value, pa_type: pa.DataType | None = None) -> pa.Scalar:
588 """
589 Box value into a pyarrow Scalar.
590
591 Parameters
592 ----------
593 value : any
594 pa_type : pa.DataType | None
595
596 Returns
597 -------
598 pa.Scalar
599 """
600 if isinstance(value, pa.Scalar):
601 pa_scalar = value
602 elif isna(value) and not (lib.is_float(value) and not is_nan_na()):
603 pa_scalar = pa.scalar(None, type=pa_type)
604 else:
605 # Workaround https://github.com/apache/arrow/issues/37291
606 if isinstance(value, Timedelta):
607 if pa_type is None:
608 pa_type = pa.duration(value.unit)
609 elif value.unit != pa_type.unit:
610 value = value.as_unit(pa_type.unit)
611 value = value._value
612 elif isinstance(value, Timestamp):
613 if pa_type is None:
614 pa_type = pa.timestamp(value.unit, tz=value.tz)
615 elif value.unit != pa_type.unit:
616 value = value.as_unit(pa_type.unit)
617 value = value._value
618
619 pa_scalar = pa.scalar(value, type=pa_type)
620
621 if pa_type is not None and pa_scalar.type != pa_type:
622 pa_scalar = pa_scalar.cast(pa_type)
623
624 return pa_scalar
625
626 @classmethod
627 def _box_pa_array(
628 cls, value, pa_type: pa.DataType | None = None, copy: bool = False
629 ) -> pa.Array | pa.ChunkedArray:
630 """
631 Box value into a pyarrow Array or ChunkedArray.
632
633 Parameters
634 ----------
635 value : Sequence
636 pa_type : pa.DataType | None
637
638 Returns
639 -------
640 pa.Array or pa.ChunkedArray
641 """
642 value = extract_array(value, extract_numpy=True)
643 if isinstance(value, cls):
644 pa_array = value._pa_array
645 elif isinstance(value, (pa.Array, pa.ChunkedArray)):
646 pa_array = value
647 elif isinstance(value, BaseMaskedArray):
648 # GH 52625
649 if copy:
650 value = value.copy()
651 pa_array = value.__arrow_array__()
652
653 elif hasattr(value, "__arrow_array__"):
654 # e.g. StringArray
655 if copy:
656 value = value.copy()
657 pa_array = value.__arrow_array__()
658
659 else:
660 if (
661 isinstance(value, np.ndarray)
662 and pa_type is not None
663 and (
664 pa.types.is_large_binary(pa_type)
665 or pa.types.is_large_string(pa_type)
666 )
667 ):
668 # See https://github.com/apache/arrow/issues/35289
669 value = np.asarray(value, dtype=object)
670 elif copy and is_array_like(value):
671 # pa array should not get updated when numpy array is updated
672 value = value.copy()
673
674 if (
675 pa_type is not None
676 and pa.types.is_duration(pa_type)
677 and (not isinstance(value, np.ndarray) or value.dtype.kind not in "mi")
678 ):
679 # Workaround https://github.com/apache/arrow/issues/37291
680 from pandas.core.tools.timedeltas import to_timedelta
681
682 value = to_timedelta(value, unit=pa_type.unit).as_unit(pa_type.unit)
683 value = value.to_numpy()
684
685 if pa_type is not None and pa.types.is_timestamp(pa_type):
686 # Use DatetimeArray to exclude Decimal(NaN) (GH#61774) and
687 # ensure constructor treats tznaive the same as non-pyarrow
688 # dtypes (GH#61775)
689 from pandas.core.arrays.datetimes import (
690 DatetimeArray,
691 tz_to_dtype,
692 )
693
694 pass_dtype = tz_to_dtype(tz=pa_type.tz, unit=pa_type.unit)
695 value = extract_array(value, extract_numpy=True)
696 if isinstance(value, DatetimeArray):
697 dta = value
698 else:
699 dta = DatetimeArray._from_sequence(
700 value, copy=copy, dtype=pass_dtype
701 )
702 dta_mask = dta.isna()
703 value_i8 = cast("npt.NDArray", dta.view("i8"))
704 if not value_i8.flags["WRITEABLE"]:
705 # e.g. test_setitem_frame_2d_values
706 value_i8 = value_i8.copy()
707 dta = DatetimeArray._from_sequence(value_i8, dtype=dta.dtype)
708 value_i8[dta_mask] = 0 # GH#61776 avoid __sub__ overflow
709 pa_array = pa.array(dta._ndarray, type=pa_type, mask=dta_mask)
710 return pa_array
711
712 mask = None
713 if is_nan_na():
714 try:
715 arr_value = np.asarray(value)
716 if arr_value.ndim > 1:
717 # e.g. test_fixed_size_list we have list data. ndim > 1
718 # means there were no scalar (NA) entries.
719 mask = np.zeros(len(value), dtype=np.bool_)
720 else:
721 mask = isna(arr_value)
722 except ValueError:
723 # Ragged data that numpy raises on
724 arr_value = construct_1d_object_array_from_listlike(value)
725 mask = isna(arr_value)
726 elif (
727 getattr(value, "dtype", None) is None or value.dtype.kind not in "iumMf"
728 ):
729 arr_value = np.asarray(value, dtype=object)
730 # similar to isna(value) but exclude NaN, NaT, nat-like, nan-like
731 mask = is_pdna_or_none(arr_value)
732
733 try:
734 pa_array = pa.array(value, type=pa_type, mask=mask)
735 except (pa.ArrowInvalid, pa.ArrowTypeError):
736 # GH50430: let pyarrow infer type, then cast
737 pa_array = pa.array(value, mask=mask)
738
739 if pa_type is None and pa.types.is_duration(pa_array.type):
740 # Workaround https://github.com/apache/arrow/issues/37291
741 from pandas.core.tools.timedeltas import to_timedelta
742
743 value = to_timedelta(value)
744 value = value.to_numpy()
745 pa_array = pa.array(value, type=pa_type)
746
747 if pa.types.is_duration(pa_array.type) and pa_array.null_count > 0:
748 # GH52843: upstream bug for duration types when originally
749 # constructed with data containing numpy NaT.
750 # https://github.com/apache/arrow/issues/35088
751 arr = cls(pa_array)
752 arr = arr.fillna(arr.dtype.na_value)
753 pa_array = arr._pa_array
754
755 if pa_type is not None and pa_array.type != pa_type:
756 if pa.types.is_dictionary(pa_type):
757 pa_array = pa_array.dictionary_encode()
758 if pa_array.type != pa_type:
759 pa_array = pa_array.cast(pa_type)
760 else:
761 try:
762 pa_array = pa_array.cast(pa_type)
763 except (pa.ArrowNotImplementedError, pa.ArrowTypeError):
764 if pa.types.is_string(pa_array.type) or pa.types.is_large_string(
765 pa_array.type
766 ):
767 # TODO: Move logic in _from_sequence_of_strings into
768 # _box_pa_array
769 dtype = ArrowDtype(pa_type)
770 return cls._from_sequence_of_strings(
771 value, dtype=dtype
772 )._pa_array
773 else:
774 raise
775
776 return pa_array
777
778 def __getitem__(self, item: PositionalIndexer):
779 """Select a subset of self.
780
781 Parameters
782 ----------
783 item : int, slice, or ndarray
784 * int: The position in 'self' to get.
785 * slice: A slice object, where 'start', 'stop', and 'step' are
786 integers or None
787 * ndarray: A 1-d boolean NumPy ndarray the same length as 'self'
788
789 Returns
790 -------
791 item : scalar or ExtensionArray
792
793 Notes
794 -----
795 For scalar ``item``, return a scalar value suitable for the array's
796 type. This should be an instance of ``self.dtype.type``.
797 For slice ``key``, return an instance of ``ExtensionArray``, even
798 if the slice is length 0 or 1.
799 For a boolean mask, return an instance of ``ExtensionArray``, filtered
800 to the values where ``item`` is True.
801 """
802 item = check_array_indexer(self, item)
803
804 if isinstance(item, np.ndarray):
805 if not len(item):
806 # Removable once we migrate StringDtype[pyarrow] to ArrowDtype[string]
807 if (
808 isinstance(self._dtype, StringDtype)
809 and self._dtype.storage == "pyarrow"
810 ):
811 # TODO(infer_string) should this be large_string?
812 pa_dtype = pa.string()
813 else:
814 pa_dtype = self._dtype.pyarrow_dtype
815 result = pa.chunked_array([], type=pa_dtype)
816 return self._from_pyarrow_array(result)
817
818 elif item.dtype.kind in "iu":
819 return self.take(item)
820 elif item.dtype.kind == "b":
821 return self._from_pyarrow_array(self._pa_array.filter(item))
822 else:
823 raise IndexError(
824 "Only integers, slices and integer or "
825 "boolean arrays are valid indices."
826 )
827 elif isinstance(item, tuple):
828 item = unpack_tuple_and_ellipses(item)
829
830 if item is Ellipsis:
831 # TODO: should be handled by pyarrow?
832 item = slice(None)
833
834 if is_scalar(item) and not is_integer(item):
835 # e.g. "foo" or 2.5
836 # exception message copied from numpy
837 raise IndexError(
838 r"only integers, slices (`:`), ellipsis (`...`), numpy.newaxis "
839 r"(`None`) and integer or boolean arrays are valid indices"
840 )
841 # We are not an array indexer, so maybe e.g. a slice or integer
842 # indexer. We dispatch to pyarrow.
843 if isinstance(item, slice):
844 # Arrow bug https://github.com/apache/arrow/issues/38768
845 if item.start == item.stop:
846 pass
847 elif (
848 item.stop is not None
849 and item.stop < -len(self)
850 and item.step is not None
851 and item.step < 0
852 ):
853 item = slice(item.start, None, item.step)
854
855 value = self._pa_array[item]
856 if isinstance(value, pa.ChunkedArray):
857 result = self._from_pyarrow_array(value)
858 if getitem_returns_view(self, item):
859 result._readonly = self._readonly
860 return result
861 else:
862 pa_type = self._pa_array.type
863 scalar = value.as_py()
864 if scalar is None:
865 return self._dtype.na_value
866 elif pa.types.is_timestamp(pa_type) and pa_type.unit != "ns":
867 # GH 53326
868 return Timestamp(scalar).as_unit(pa_type.unit)
869 elif pa.types.is_duration(pa_type) and pa_type.unit != "ns":
870 # GH 53326
871 return Timedelta(scalar).as_unit(pa_type.unit)
872 else:
873 return scalar
874
875 def __iter__(self) -> Iterator[Any]:
876 """
877 Iterate over elements of the array.
878 """
879 na_value = self._dtype.na_value
880 # GH 53326
881 pa_type = self._pa_array.type
882 box_timestamp = pa.types.is_timestamp(pa_type) and pa_type.unit != "ns"
883 box_timedelta = pa.types.is_duration(pa_type) and pa_type.unit != "ns"
884 for value in self._pa_array:
885 val = value.as_py()
886 if val is None:
887 yield na_value
888 elif box_timestamp:
889 yield Timestamp(val).as_unit(pa_type.unit)
890 elif box_timedelta:
891 yield Timedelta(val).as_unit(pa_type.unit)
892 else:
893 yield val
894
895 def __arrow_array__(self, type=None):
896 """Convert myself to a pyarrow ChunkedArray."""
897 return self._pa_array
898
899 def __array_ufunc__(self, ufunc: np.ufunc, method: str, *inputs, **kwargs):
900 # Need to wrap np.array results GH#62800
901 result = super().__array_ufunc__(ufunc, method, *inputs, **kwargs)
902 if type(self) is ArrowExtensionArray:
903 # Exclude ArrowStringArray
904 return type(self)._from_sequence(result)
905 return result
906
907 def __array__(
908 self, dtype: NpDtype | None = None, copy: bool | None = None
909 ) -> np.ndarray:
910 """Correctly construct numpy arrays when passed to `np.asarray()`."""
911 if copy is False:
912 # TODO: By using `zero_copy_only` it may be possible to implement this
913 raise ValueError(
914 "Unable to avoid copy while creating an array as requested."
915 )
916 elif copy is None:
917 # `to_numpy(copy=False)` has the meaning of NumPy `copy=None`.
918 copy = False
919
920 return self.to_numpy(dtype=dtype, copy=copy)
921
922 def __invert__(self) -> Self:
923 # This is a bit wise op for integer types
924 if pa.types.is_integer(self._pa_array.type):
925 return self._from_pyarrow_array(pc.bit_wise_not(self._pa_array))
926 elif pa.types.is_string(self._pa_array.type) or pa.types.is_large_string(
927 self._pa_array.type
928 ):
929 # Raise TypeError instead of pa.ArrowNotImplementedError
930 raise TypeError("__invert__ is not supported for string dtypes")
931 else:
932 return self._from_pyarrow_array(pc.invert(self._pa_array))
933
934 def __neg__(self) -> Self:
935 try:
936 return self._from_pyarrow_array(pc.negate_checked(self._pa_array))
937 except pa.ArrowNotImplementedError as err:
938 raise TypeError(
939 f"unary '-' not supported for dtype '{self.dtype}'"
940 ) from err
941
942 def __pos__(self) -> Self:
943 return self._from_pyarrow_array(self._pa_array)
944
945 def __abs__(self) -> Self:
946 return self._from_pyarrow_array(pc.abs_checked(self._pa_array))
947
948 # GH 42600: __getstate__/__setstate__ not necessary once
949 # https://issues.apache.org/jira/browse/ARROW-10739 is addressed
950 def __getstate__(self):
951 state = self.__dict__.copy()
952 state["_pa_array"] = self._pa_array.combine_chunks()
953 return state
954
955 def __setstate__(self, state) -> None:
956 if "_data" in state:
957 data = state.pop("_data")
958 else:
959 data = state["_pa_array"]
960 state["_pa_array"] = pa.chunked_array(data)
961 self.__dict__.update(state)
962
963 def _cmp_method(self, other, op) -> ArrowExtensionArray:
964 pc_func = ARROW_CMP_FUNCS[op.__name__]
965 ltype = self._pa_array.type
966
967 if isinstance(other, (ExtensionArray, np.ndarray, list, range)):
968 try:
969 boxed = self._box_pa(other)
970 except pa.lib.ArrowInvalid:
971 # e.g. GH#60228 [1, "b"] we have to operate pointwise
972 res_values = [op(x, y) for x, y in zip(self, other, strict=True)]
973 result = pa.array(res_values, type=pa.bool_(), from_pandas=True)
974 else:
975 rtype = boxed.type
976 if (
977 (pa.types.is_timestamp(ltype) and pa.types.is_date(rtype))
978 or (pa.types.is_timestamp(rtype) and pa.types.is_date(ltype))
979 or isinstance(other, range)
980 ):
981 # GH#62157 match non-pyarrow behavior
982 result = ops.invalid_comparison(self, other, op)
983 result = pa.array(result, type=pa.bool_())
984 else:
985 try:
986 result = pc_func(self._pa_array, boxed)
987 except pa.ArrowNotImplementedError:
988 result = ops.invalid_comparison(self, other, op)
989 result = pa.array(result, type=pa.bool_())
990
991 elif is_scalar(other):
992 if (isinstance(other, datetime) and pa.types.is_date(ltype)) or (
993 type(other) is date and pa.types.is_timestamp(ltype)
994 ):
995 # GH#62157 match non-pyarrow behavior
996 result = ops.invalid_comparison(self, other, op)
997 result = pa.array(result, type=pa.bool_())
998 else:
999 try:
1000 result = pc_func(self._pa_array, self._box_pa(other))
1001 except (pa.lib.ArrowNotImplementedError, pa.lib.ArrowInvalid):
1002 mask = isna(self) | isna(other)
1003 valid = ~mask
1004 result = np.zeros(len(self), dtype="bool")
1005 np_array = np.array(self)
1006 try:
1007 result[valid] = op(np_array[valid], other)
1008 except TypeError:
1009 result = ops.invalid_comparison(self, other, op)
1010 result = pa.array(result, type=pa.bool_())
1011 result = pc.if_else(valid, result, None)
1012 else:
1013 raise NotImplementedError(
1014 f"{op.__name__} not implemented for {type(other)}"
1015 )
1016 return ArrowExtensionArray(result)
1017
1018 def _op_method_error_message(self, other, op) -> str:
1019 if hasattr(other, "dtype"):
1020 other_type = f"dtype '{other.dtype}'"
1021 else:
1022 other_type = f"object of type {type(other)}"
1023 return (
1024 f"operation '{op.__name__}' not supported for "
1025 f"dtype '{self.dtype}' with {other_type}"
1026 )
1027
1028 def _evaluate_op_method(self, other, op, arrow_funcs) -> Self:
1029 pa_type = self._pa_array.type
1030 other_original = other
1031 other = self._box_pa(other)
1032
1033 if (
1034 pa.types.is_string(pa_type)
1035 or pa.types.is_large_string(pa_type)
1036 or pa.types.is_binary(pa_type)
1037 ):
1038 if op in [operator.add, roperator.radd]:
1039 # binary_join_element_wise does not support mixed types, but we
1040 # want to allow addition between string and large_string types
1041 self_array = self._pa_array
1042 if pa.types.is_string(pa_type) and pa.types.is_large_string(other.type):
1043 self_array = self._pa_array.cast(pa.large_string())
1044 elif pa.types.is_large_string(pa_type) and pa.types.is_string(
1045 other.type
1046 ):
1047 other = other.cast(pa.large_string())
1048
1049 sep = pa.scalar("", type=self_array.type)
1050 if isinstance(other, pa.Scalar) and pc.is_null(other).as_py():
1051 other = other.cast(self_array.type)
1052 try:
1053 if op is operator.add:
1054 result = pc.binary_join_element_wise(self_array, other, sep)
1055 elif op is roperator.radd:
1056 result = pc.binary_join_element_wise(other, self_array, sep)
1057 except pa.ArrowNotImplementedError as err:
1058 raise TypeError(
1059 self._op_method_error_message(other_original, op)
1060 ) from err
1061 return self._from_pyarrow_array(result)
1062 elif op in [operator.mul, roperator.rmul]:
1063 binary = self._pa_array
1064 integral = other
1065 if not pa.types.is_integer(integral.type):
1066 raise TypeError("Can only string multiply by an integer.")
1067 pa_integral = pc.if_else(pc.less(integral, 0), 0, integral)
1068 result = pc.binary_repeat(binary, pa_integral)
1069 return self._from_pyarrow_array(result)
1070 elif (
1071 pa.types.is_string(other.type)
1072 or pa.types.is_binary(other.type)
1073 or pa.types.is_large_string(other.type)
1074 ) and op in [operator.mul, roperator.rmul]:
1075 binary = other
1076 integral = self._pa_array
1077 if not pa.types.is_integer(integral.type):
1078 raise TypeError("Can only string multiply by an integer.")
1079 pa_integral = pc.if_else(pc.less(integral, 0), 0, integral)
1080 result = pc.binary_repeat(binary, pa_integral)
1081 return self._from_pyarrow_array(result)
1082 if (
1083 isinstance(other, pa.Scalar)
1084 and pc.is_null(other).as_py()
1085 and op.__name__ in ARROW_LOGICAL_FUNCS
1086 ):
1087 # pyarrow kleene ops require null to be typed
1088 other = other.cast(pa_type)
1089
1090 pc_func = arrow_funcs[op.__name__]
1091 if pc_func is NotImplemented:
1092 if pa.types.is_string(pa_type) or pa.types.is_large_string(pa_type):
1093 raise TypeError(self._op_method_error_message(other_original, op))
1094 raise NotImplementedError(f"{op.__name__} not implemented.")
1095
1096 try:
1097 result = pc_func(self._pa_array, other)
1098 except pa.ArrowNotImplementedError as err:
1099 raise TypeError(self._op_method_error_message(other_original, op)) from err
1100 return self._from_pyarrow_array(result)
1101
1102 def _logical_method(self, other, op) -> Self:
1103 # For integer types `^`, `|`, `&` are bitwise operators and return
1104 # integer types. Otherwise these are boolean ops.
1105 if pa.types.is_integer(self._pa_array.type):
1106 return self._evaluate_op_method(other, op, ARROW_BIT_WISE_FUNCS)
1107 elif (
1108 (
1109 pa.types.is_string(self._pa_array.type)
1110 or pa.types.is_large_string(self._pa_array.type)
1111 )
1112 and op in (roperator.ror_, roperator.rand_, roperator.rxor)
1113 and isinstance(other, np.ndarray)
1114 and other.dtype == bool
1115 ):
1116 # GH#60234 backward compatibility for the move to StringDtype in 3.0
1117 op_name = op.__name__[1:].strip("_")
1118 warnings.warn(
1119 f"'{op_name}' operations between boolean dtype and {self.dtype} are "
1120 "deprecated and will raise in a future version. Explicitly "
1121 "cast the strings to a boolean dtype before operating instead.",
1122 Pandas4Warning,
1123 stacklevel=find_stack_level(),
1124 )
1125 return op(other, self.astype(bool))
1126 else:
1127 return self._evaluate_op_method(other, op, ARROW_LOGICAL_FUNCS)
1128
1129 def _str_arith_method_object_fallback(
1130 self, other, op
1131 ) -> Self | npt.NDArray[np.object_]:
1132 mask = isna(self) | isna(other)
1133 valid = ~mask
1134
1135 if is_list_like(other):
1136 if len(other) != len(self):
1137 raise ValueError(
1138 f"Lengths of operands do not match: {len(self)} != {len(other)}"
1139 )
1140 if not is_array_like(other):
1141 other = np.asarray(other)
1142 other = other[valid]
1143
1144 result = np.empty(len(self), dtype=object)
1145 result[mask] = self.dtype.na_value
1146 result[valid] = op(np.asarray(self, dtype=object)[valid], other)
1147
1148 if not lib.is_string_array(result, skipna=True):
1149 return result
1150 return type(self)._from_sequence(result, dtype=self.dtype)
1151
1152 def _arith_method(self, other, op) -> Self | npt.NDArray[np.object_]:
1153 result: Self | npt.NDArray[np.object_]
1154 if pa.types.is_string(self._pa_array.type) or pa.types.is_large_string(
1155 self._pa_array.type
1156 ):
1157 try:
1158 result = self._evaluate_op_method(other, op, ARROW_ARITHMETIC_FUNCS)
1159 except (pa.ArrowInvalid, pa.ArrowTypeError):
1160 result = self._str_arith_method_object_fallback(other, op)
1161 else:
1162 result = self._evaluate_op_method(other, op, ARROW_ARITHMETIC_FUNCS)
1163 if isinstance(result, np.ndarray):
1164 return result
1165 if is_nan_na() and result.dtype.kind == "f":
1166 parr = result._pa_array
1167 mask = pc.is_nan(parr).fill_null(False).to_numpy()
1168 arr = pc.replace_with_mask(parr, mask, pa.scalar(None, type=parr.type))
1169 result = type(self)(arr)
1170 return result
1171
1172 def equals(self, other) -> bool:
1173 if not isinstance(other, ArrowExtensionArray):
1174 return False
1175 # I'm told that pyarrow makes __eq__ behave like pandas' equals;
1176 # TODO: is this documented somewhere?
1177 return self._pa_array == other._pa_array
1178
1179 @property
1180 def dtype(self) -> ArrowDtype:
1181 """
1182 An instance of 'ExtensionDtype'.
1183 """
1184 return self._dtype
1185
1186 @property
1187 def nbytes(self) -> int:
1188 """
1189 The number of bytes needed to store this object in memory.
1190 """
1191 return self._pa_array.nbytes
1192
1193 def __len__(self) -> int:
1194 """
1195 Length of this array.
1196
1197 Returns
1198 -------
1199 length : int
1200 """
1201 return len(self._pa_array)
1202
1203 def __contains__(self, key) -> bool:
1204 # https://github.com/pandas-dev/pandas/pull/51307#issuecomment-1426372604
1205 if isna(key) and key is not self.dtype.na_value:
1206 if lib.is_float(key) and is_nan_na():
1207 return self.dtype.na_value in self
1208 elif self.dtype.kind == "f" and lib.is_float(key):
1209 # Check specifically for NaN
1210 return pc.any(pc.is_nan(self._pa_array)).as_py()
1211
1212 # e.g. date or timestamp types we do not allow None here to match pd.NA
1213 return False
1214 # TODO: maybe complex? object?
1215
1216 return bool(super().__contains__(key))
1217
1218 @property
1219 def _hasna(self) -> bool:
1220 return self._pa_array.null_count > 0
1221
1222 def isna(self) -> npt.NDArray[np.bool_]:
1223 """
1224 Boolean NumPy array indicating if each value is missing.
1225
1226 This should return a 1-D array the same length as 'self'.
1227 """
1228 # GH51630: fast paths
1229 null_count = self._pa_array.null_count
1230 if null_count == 0:
1231 return np.zeros(len(self), dtype=np.bool_)
1232 elif null_count == len(self):
1233 return np.ones(len(self), dtype=np.bool_)
1234
1235 return self._pa_array.is_null().to_numpy()
1236
1237 @overload
1238 def any(self, *, skipna: Literal[True] = ..., **kwargs) -> bool: ...
1239
1240 @overload
1241 def any(self, *, skipna: bool, **kwargs) -> bool | NAType: ...
1242
1243 def any(self, *, skipna: bool = True, **kwargs) -> bool | NAType:
1244 """
1245 Return whether any element is truthy.
1246
1247 Returns False unless there is at least one element that is truthy.
1248 By default, NAs are skipped. If ``skipna=False`` is specified and
1249 missing values are present, similar :ref:`Kleene logic <boolean.kleene>`
1250 is used as for logical operations.
1251
1252 Parameters
1253 ----------
1254 skipna : bool, default True
1255 Exclude NA values. If the entire array is NA and `skipna` is
1256 True, then the result will be False, as for an empty array.
1257 If `skipna` is False, the result will still be True if there is
1258 at least one element that is truthy, otherwise NA will be returned
1259 if there are NA's present.
1260
1261 Returns
1262 -------
1263 bool or :attr:`pandas.NA`
1264
1265 See Also
1266 --------
1267 ArrowExtensionArray.all : Return whether all elements are truthy.
1268
1269 Examples
1270 --------
1271 The result indicates whether any element is truthy (and by default
1272 skips NAs):
1273
1274 >>> pd.array([True, False, True], dtype="boolean[pyarrow]").any()
1275 True
1276 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").any()
1277 True
1278 >>> pd.array([False, False, pd.NA], dtype="boolean[pyarrow]").any()
1279 False
1280 >>> pd.array([], dtype="boolean[pyarrow]").any()
1281 False
1282 >>> pd.array([pd.NA], dtype="boolean[pyarrow]").any()
1283 False
1284 >>> pd.array([pd.NA], dtype="float64[pyarrow]").any()
1285 False
1286
1287 With ``skipna=False``, the result can be NA if this is logically
1288 required (whether ``pd.NA`` is True or False influences the result):
1289
1290 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").any(skipna=False)
1291 True
1292 >>> pd.array([1, 0, pd.NA], dtype="boolean[pyarrow]").any(skipna=False)
1293 True
1294 >>> pd.array([False, False, pd.NA], dtype="boolean[pyarrow]").any(skipna=False)
1295 <NA>
1296 >>> pd.array([0, 0, pd.NA], dtype="boolean[pyarrow]").any(skipna=False)
1297 <NA>
1298 """
1299 return self._reduce("any", skipna=skipna, **kwargs)
1300
1301 @overload
1302 def all(self, *, skipna: Literal[True] = ..., **kwargs) -> bool: ...
1303
1304 @overload
1305 def all(self, *, skipna: bool, **kwargs) -> bool | NAType: ...
1306
1307 def all(self, *, skipna: bool = True, **kwargs) -> bool | NAType:
1308 """
1309 Return whether all elements are truthy.
1310
1311 Returns True unless there is at least one element that is falsey.
1312 By default, NAs are skipped. If ``skipna=False`` is specified and
1313 missing values are present, similar :ref:`Kleene logic <boolean.kleene>`
1314 is used as for logical operations.
1315
1316 Parameters
1317 ----------
1318 skipna : bool, default True
1319 Exclude NA values. If the entire array is NA and `skipna` is
1320 True, then the result will be True, as for an empty array.
1321 If `skipna` is False, the result will still be False if there is
1322 at least one element that is falsey, otherwise NA will be returned
1323 if there are NA's present.
1324
1325 Returns
1326 -------
1327 bool or :attr:`pandas.NA`
1328
1329 See Also
1330 --------
1331 ArrowExtensionArray.any : Return whether any element is truthy.
1332
1333 Examples
1334 --------
1335 The result indicates whether all elements are truthy (and by default
1336 skips NAs):
1337
1338 >>> pd.array([True, True, pd.NA], dtype="boolean[pyarrow]").all()
1339 True
1340 >>> pd.array([1, 1, pd.NA], dtype="boolean[pyarrow]").all()
1341 True
1342 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").all()
1343 False
1344 >>> pd.array([], dtype="boolean[pyarrow]").all()
1345 True
1346 >>> pd.array([pd.NA], dtype="boolean[pyarrow]").all()
1347 True
1348 >>> pd.array([pd.NA], dtype="float64[pyarrow]").all()
1349 True
1350
1351 With ``skipna=False``, the result can be NA if this is logically
1352 required (whether ``pd.NA`` is True or False influences the result):
1353
1354 >>> pd.array([True, True, pd.NA], dtype="boolean[pyarrow]").all(skipna=False)
1355 <NA>
1356 >>> pd.array([1, 1, pd.NA], dtype="boolean[pyarrow]").all(skipna=False)
1357 <NA>
1358 >>> pd.array([True, False, pd.NA], dtype="boolean[pyarrow]").all(skipna=False)
1359 False
1360 >>> pd.array([1, 0, pd.NA], dtype="boolean[pyarrow]").all(skipna=False)
1361 False
1362 """
1363 return self._reduce("all", skipna=skipna, **kwargs)
1364
1365 def argsort(
1366 self,
1367 *,
1368 ascending: bool = True,
1369 kind: SortKind = "quicksort",
1370 na_position: str = "last",
1371 **kwargs,
1372 ) -> np.ndarray:
1373 order = "ascending" if ascending else "descending"
1374 null_placement = {"last": "at_end", "first": "at_start"}.get(na_position, None)
1375 if null_placement is None:
1376 raise ValueError(f"invalid na_position: {na_position}")
1377
1378 result = pc.array_sort_indices(
1379 self._pa_array, order=order, null_placement=null_placement
1380 )
1381 np_result = result.to_numpy()
1382 return np_result.astype(np.intp, copy=False)
1383
1384 def _argmin_max(self, skipna: bool, method: str) -> int:
1385 if self._pa_array.length() in (0, self._pa_array.null_count) or (
1386 self._hasna and not skipna
1387 ):
1388 # For empty or all null, pyarrow returns -1 but pandas expects TypeError
1389 # For skipna=False and data w/ null, pandas expects NotImplementedError
1390 # let ExtensionArray.arg{max|min} raise
1391 return getattr(super(), f"arg{method}")(skipna=skipna)
1392
1393 data = self._pa_array
1394 if pa.types.is_duration(data.type):
1395 data = data.cast(pa.int64())
1396
1397 value = getattr(pc, method)(data, skip_nulls=skipna)
1398 return pc.index(data, value).as_py()
1399
1400 def argmin(self, skipna: bool = True) -> int:
1401 return self._argmin_max(skipna, "min")
1402
1403 def argmax(self, skipna: bool = True) -> int:
1404 return self._argmin_max(skipna, "max")
1405
1406 def copy(self) -> Self:
1407 """
1408 Return a shallow copy of the array.
1409
1410 Underlying ChunkedArray is immutable, so a deep copy is unnecessary.
1411
1412 Returns
1413 -------
1414 type(self)
1415 """
1416 return self._from_pyarrow_array(self._pa_array)
1417
1418 def dropna(self) -> Self:
1419 """
1420 Return ArrowExtensionArray without NA values.
1421
1422 Returns
1423 -------
1424 ArrowExtensionArray
1425 """
1426 return self._from_pyarrow_array(pc.drop_null(self._pa_array))
1427
1428 def _pad_or_backfill(
1429 self,
1430 *,
1431 method: FillnaOptions,
1432 limit: int | None = None,
1433 limit_area: Literal["inside", "outside"] | None = None,
1434 copy: bool = True,
1435 ) -> Self:
1436 if not self._hasna:
1437 return self
1438
1439 if limit is None and limit_area is None:
1440 method = missing.clean_fill_method(method)
1441 try:
1442 if method == "pad":
1443 return self._from_pyarrow_array(
1444 pc.fill_null_forward(self._pa_array)
1445 )
1446 elif method == "backfill":
1447 return self._from_pyarrow_array(
1448 pc.fill_null_backward(self._pa_array)
1449 )
1450 except pa.ArrowNotImplementedError:
1451 # ArrowNotImplementedError: Function 'coalesce' has no kernel
1452 # matching input types (duration[ns], duration[ns])
1453 # TODO: remove try/except wrapper if/when pyarrow implements
1454 # a kernel for duration types.
1455 pass
1456
1457 # TODO: Why do we no longer need the above cases?
1458 # TODO(3.0): after EA.fillna 'method' deprecation is enforced, we can remove
1459 # this method entirely.
1460 return super()._pad_or_backfill(
1461 method=method, limit=limit, limit_area=limit_area, copy=copy
1462 )
1463
1464 @doc(ExtensionArray.fillna)
1465 def fillna(
1466 self,
1467 value: object | ArrayLike,
1468 limit: int | None = None,
1469 copy: bool = True,
1470 ) -> Self:
1471 if not self._hasna:
1472 return self.copy()
1473
1474 if limit is not None:
1475 return super().fillna(value=value, limit=limit, copy=copy)
1476
1477 if isinstance(value, (np.ndarray, ExtensionArray)):
1478 # Similar to check_value_size, but we do not mask here since we may
1479 # end up passing it to the super() method.
1480 if len(value) != len(self):
1481 raise ValueError(
1482 f"Length of 'value' does not match. Got ({len(value)}) "
1483 f" expected {len(self)}"
1484 )
1485
1486 try:
1487 fill_value = self._box_pa(value, pa_type=self._pa_array.type)
1488 except pa.ArrowTypeError as err:
1489 msg = f"Invalid value '{value!s}' for dtype '{self.dtype}'"
1490 raise TypeError(msg) from err
1491
1492 try:
1493 return self._from_pyarrow_array(
1494 _safe_fill_null(self._pa_array, fill_value=fill_value)
1495 )
1496 except pa.ArrowNotImplementedError:
1497 # ArrowNotImplementedError: Function 'coalesce' has no kernel
1498 # matching input types (duration[ns], duration[ns])
1499 # TODO: remove try/except wrapper if/when pyarrow implements
1500 # a kernel for duration types.
1501 pass
1502
1503 return super().fillna(value=value, limit=limit, copy=copy)
1504
1505 def isin(self, values: ArrayLike) -> npt.NDArray[np.bool_]:
1506 # short-circuit to return all False array.
1507 if not len(values):
1508 return np.zeros(len(self), dtype=bool)
1509
1510 value_set = self._box_pa(values)
1511 result = pc.is_in(self._pa_array, value_set=value_set)
1512 # pyarrow 2.0.0 returned nulls, so we explicitly specify dtype to convert nulls
1513 # to False
1514 return np.array(result, dtype=np.bool_)
1515
1516 def _values_for_factorize(self) -> tuple[np.ndarray, Any]:
1517 """
1518 Return an array and missing value suitable for factorization.
1519
1520 Returns
1521 -------
1522 values : ndarray
1523 na_value : pd.NA
1524
1525 Notes
1526 -----
1527 The values returned by this method are also used in
1528 :func:`pandas.util.hash_pandas_object`.
1529 """
1530 values = self._pa_array.to_numpy()
1531 return values, self.dtype.na_value
1532
1533 @doc(ExtensionArray.factorize)
1534 def factorize(
1535 self,
1536 use_na_sentinel: bool = True,
1537 ) -> tuple[np.ndarray, ExtensionArray]:
1538 null_encoding = "mask" if use_na_sentinel else "encode"
1539
1540 data = self._pa_array
1541
1542 if pa.types.is_dictionary(data.type):
1543 if null_encoding == "encode":
1544 # dictionary encode does nothing if an already encoded array is given
1545 data = data.cast(data.type.value_type)
1546 encoded = data.dictionary_encode(null_encoding=null_encoding)
1547 else:
1548 encoded = data
1549 else:
1550 encoded = data.dictionary_encode(null_encoding=null_encoding)
1551 if encoded.length() == 0:
1552 indices = np.array([], dtype=np.intp)
1553 uniques = self._from_pyarrow_array(
1554 pa.chunked_array([], type=encoded.type.value_type)
1555 )
1556 else:
1557 # GH 54844
1558 combined = encoded.combine_chunks()
1559 pa_indices = combined.indices
1560 if pa_indices.null_count > 0:
1561 pa_indices = _safe_fill_null(pa_indices, -1)
1562 indices = pa_indices.to_numpy(zero_copy_only=False, writable=True).astype(
1563 np.intp, copy=False
1564 )
1565 uniques = self._from_pyarrow_array(combined.dictionary)
1566
1567 return indices, uniques
1568
1569 def reshape(self, *args, **kwargs):
1570 raise NotImplementedError(
1571 f"{type(self)} does not support reshape "
1572 f"as backed by a 1D pyarrow.ChunkedArray."
1573 )
1574
1575 def round(self, decimals: int = 0, *args, **kwargs) -> Self:
1576 """
1577 Round each value in the array a to the given number of decimals.
1578
1579 Parameters
1580 ----------
1581 decimals : int, default 0
1582 Number of decimal places to round to. If decimals is negative,
1583 it specifies the number of positions to the left of the decimal point.
1584 *args, **kwargs
1585 Additional arguments and keywords have no effect.
1586
1587 Returns
1588 -------
1589 ArrowExtensionArray
1590 Rounded values of the ArrowExtensionArray.
1591
1592 See Also
1593 --------
1594 DataFrame.round : Round values of a DataFrame.
1595 Series.round : Round values of a Series.
1596 """
1597 return self._from_pyarrow_array(pc.round(self._pa_array, ndigits=decimals))
1598
1599 @doc(ExtensionArray.searchsorted)
1600 def searchsorted(
1601 self,
1602 value: NumpyValueArrayLike | ExtensionArray,
1603 side: Literal["left", "right"] = "left",
1604 sorter: NumpySorter | None = None,
1605 ) -> npt.NDArray[np.intp] | np.intp:
1606 if self._hasna:
1607 raise ValueError(
1608 "searchsorted requires array to be sorted, which is impossible "
1609 "with NAs present."
1610 )
1611 if isinstance(value, ExtensionArray):
1612 value = value.astype(object)
1613 # Base class searchsorted would cast to object, which is *much* slower.
1614 dtype = None
1615 if isinstance(self.dtype, ArrowDtype):
1616 pa_dtype = self.dtype.pyarrow_dtype
1617 if (
1618 pa.types.is_timestamp(pa_dtype) or pa.types.is_duration(pa_dtype)
1619 ) and pa_dtype.unit == "ns":
1620 # np.array[datetime/timedelta].searchsorted(datetime/timedelta)
1621 # erroneously fails when numpy type resolution is nanoseconds
1622 dtype = object
1623 return self.to_numpy(dtype=dtype).searchsorted(value, side=side, sorter=sorter)
1624
1625 def take(
1626 self,
1627 indices: TakeIndexer,
1628 allow_fill: bool = False,
1629 fill_value: Any = None,
1630 ) -> ArrowExtensionArray:
1631 """
1632 Take elements from an array.
1633
1634 Parameters
1635 ----------
1636 indices : sequence of int or one-dimensional np.ndarray of int
1637 Indices to be taken.
1638 allow_fill : bool, default False
1639 How to handle negative values in `indices`.
1640
1641 * False: negative values in `indices` indicate positional indices
1642 from the right (the default). This is similar to
1643 :func:`numpy.take`.
1644
1645 * True: negative values in `indices` indicate
1646 missing values. These values are set to `fill_value`. Any other
1647 other negative values raise a ``ValueError``.
1648
1649 fill_value : any, optional
1650 Fill value to use for NA-indices when `allow_fill` is True.
1651 This may be ``None``, in which case the default NA value for
1652 the type, ``self.dtype.na_value``, is used.
1653
1654 For many ExtensionArrays, there will be two representations of
1655 `fill_value`: a user-facing "boxed" scalar, and a low-level
1656 physical NA value. `fill_value` should be the user-facing version,
1657 and the implementation should handle translating that to the
1658 physical version for processing the take if necessary.
1659
1660 Returns
1661 -------
1662 ExtensionArray
1663
1664 Raises
1665 ------
1666 IndexError
1667 When the indices are out of bounds for the array.
1668 ValueError
1669 When `indices` contains negative values other than ``-1``
1670 and `allow_fill` is True.
1671
1672 See Also
1673 --------
1674 numpy.take
1675 api.extensions.take
1676
1677 Notes
1678 -----
1679 ExtensionArray.take is called by ``Series.__getitem__``, ``.loc``,
1680 ``iloc``, when `indices` is a sequence of values. Additionally,
1681 it's called by :meth:`Series.reindex`, or any other method
1682 that causes realignment, with a `fill_value`.
1683 """
1684 indices_array = np.asanyarray(indices)
1685
1686 if len(self._pa_array) == 0 and (indices_array >= 0).any():
1687 raise IndexError("cannot do a non-empty take")
1688 if indices_array.size > 0 and indices_array.max() >= len(self._pa_array):
1689 raise IndexError("out of bounds value in 'indices'.")
1690
1691 if allow_fill:
1692 fill_mask = indices_array < 0
1693 if fill_mask.any():
1694 validate_indices(indices_array, len(self._pa_array))
1695 # TODO(ARROW-9433): Treat negative indices as NULL
1696 indices_array = pa.array(indices_array, mask=fill_mask)
1697 result = self._pa_array.take(indices_array)
1698 if isna(fill_value):
1699 return self._from_pyarrow_array(result)
1700 # TODO: ArrowNotImplementedError: Function fill_null has no
1701 # kernel matching input types (array[string], scalar[string])
1702 result = self._from_pyarrow_array(result)
1703 result[fill_mask] = fill_value
1704 return result
1705 # return type(self)(pc.fill_null(result, pa.scalar(fill_value)))
1706 else:
1707 # Nothing to fill
1708 return self._from_pyarrow_array(self._pa_array.take(indices))
1709 else: # allow_fill=False
1710 # TODO(ARROW-9432): Treat negative indices as indices from the right.
1711 if (indices_array < 0).any():
1712 # Don't modify in-place
1713 indices_array = np.copy(indices_array)
1714 indices_array[indices_array < 0] += len(self._pa_array)
1715 return self._from_pyarrow_array(self._pa_array.take(indices_array))
1716
1717 def _maybe_convert_datelike_array(self):
1718 """Maybe convert to a datelike array."""
1719 pa_type = self._pa_array.type
1720 if pa.types.is_timestamp(pa_type):
1721 return self._to_datetimearray()
1722 elif pa.types.is_duration(pa_type):
1723 return self._to_timedeltaarray()
1724 return self
1725
1726 def _to_datetimearray(self) -> DatetimeArray:
1727 """Convert a pyarrow timestamp typed array to a DatetimeArray."""
1728 from pandas.core.arrays.datetimes import (
1729 DatetimeArray,
1730 tz_to_dtype,
1731 )
1732
1733 pa_type = self._pa_array.type
1734 assert pa.types.is_timestamp(pa_type)
1735 np_dtype = np.dtype(f"M8[{pa_type.unit}]")
1736 dtype = tz_to_dtype(pa_type.tz, pa_type.unit)
1737 np_array = self._pa_array.to_numpy()
1738 np_array = np_array.astype(np_dtype, copy=False)
1739 return DatetimeArray._simple_new(np_array, dtype=dtype)
1740
1741 def _to_timedeltaarray(self) -> TimedeltaArray:
1742 """Convert a pyarrow duration typed array to a TimedeltaArray."""
1743 from pandas.core.arrays.timedeltas import TimedeltaArray
1744
1745 pa_type = self._pa_array.type
1746 assert pa.types.is_duration(pa_type)
1747 np_dtype = np.dtype(f"m8[{pa_type.unit}]")
1748 np_array = self._pa_array.to_numpy()
1749 np_array = np_array.astype(np_dtype, copy=False)
1750 return TimedeltaArray._simple_new(np_array, dtype=np_dtype)
1751
1752 def _values_for_json(self) -> np.ndarray:
1753 if is_numeric_dtype(self.dtype):
1754 return np.asarray(self, dtype=object)
1755 return super()._values_for_json()
1756
1757 @doc(ExtensionArray.to_numpy)
1758 def to_numpy(
1759 self,
1760 dtype: npt.DTypeLike | None = None,
1761 copy: bool = False,
1762 na_value: object = lib.no_default,
1763 ) -> np.ndarray:
1764 original_na_value = na_value
1765 dtype, na_value = to_numpy_dtype_inference(self, dtype, na_value, self._hasna)
1766 pa_type = self._pa_array.type
1767 if not self._hasna or isna(na_value) or pa.types.is_null(pa_type):
1768 data = self
1769 else:
1770 data = self.fillna(na_value)
1771 copy = False
1772
1773 if pa.types.is_timestamp(pa_type) or pa.types.is_duration(pa_type):
1774 # GH 55997
1775 if dtype != object and na_value is self.dtype.na_value:
1776 na_value = lib.no_default
1777 result = data._maybe_convert_datelike_array().to_numpy(
1778 dtype=dtype, na_value=na_value
1779 )
1780 elif pa.types.is_time(pa_type) or pa.types.is_date(pa_type):
1781 # convert to list of python datetime.time objects before
1782 # wrapping in ndarray
1783 result = np.array(list(data), dtype=dtype)
1784 if data._hasna:
1785 result[data.isna()] = na_value
1786 elif pa.types.is_null(pa_type):
1787 if dtype is not None and isna(na_value):
1788 na_value = None
1789 result = np.full(len(data), fill_value=na_value, dtype=dtype)
1790 elif not data._hasna or (
1791 pa.types.is_floating(pa_type)
1792 and (
1793 na_value is np.nan
1794 or (
1795 original_na_value is lib.no_default
1796 and is_float_dtype(dtype)
1797 and is_nan_na()
1798 )
1799 )
1800 ):
1801 result = data._pa_array.to_numpy()
1802 if dtype is not None:
1803 result = result.astype(dtype, copy=False)
1804 if copy:
1805 result = result.copy()
1806 else:
1807 if dtype is None:
1808 empty = pa.array([], type=pa_type).to_numpy(zero_copy_only=False)
1809 if can_hold_element(empty, na_value):
1810 dtype = empty.dtype
1811 else:
1812 dtype = np.object_
1813 result = np.empty(len(data), dtype=dtype)
1814 mask = data.isna()
1815 result[mask] = na_value
1816 result[~mask] = data[~mask]._pa_array.to_numpy()
1817 return result
1818
1819 def map(self, mapper, na_action: Literal["ignore"] | None = None):
1820 if is_numeric_dtype(self.dtype):
1821 return map_array(self.to_numpy(), mapper, na_action=na_action)
1822 else:
1823 # For "mM" cases, the super() method passes `self` without the
1824 # to_numpy call, which inside map_array casts to ndarray[object].
1825 # Without the to_numpy() call, NA is preserved instead of changed
1826 # to None.
1827 return super().map(mapper, na_action)
1828
1829 @doc(ExtensionArray.duplicated)
1830 def duplicated(
1831 self, keep: Literal["first", "last", False] = "first"
1832 ) -> npt.NDArray[np.bool_]:
1833 pa_type = self._pa_array.type
1834 if pa.types.is_floating(pa_type) or pa.types.is_integer(pa_type):
1835 values = self.to_numpy(na_value=0)
1836 elif pa.types.is_boolean(pa_type):
1837 values = self.to_numpy(na_value=False)
1838 elif pa.types.is_temporal(pa_type):
1839 if pa_type.bit_width == 32:
1840 pa_type = pa.int32()
1841 else:
1842 pa_type = pa.int64()
1843 arr = self.astype(ArrowDtype(pa_type))
1844 values = arr.to_numpy(na_value=0)
1845 else:
1846 # factorize the values to avoid the performance penalty of
1847 # converting to object dtype
1848 values = self.factorize()[0]
1849
1850 mask = self.isna() if self._hasna else None
1851 return algos.duplicated(values, keep=keep, mask=mask)
1852
1853 def unique(self) -> Self:
1854 """
1855 Compute the ArrowExtensionArray of unique values.
1856
1857 Returns
1858 -------
1859 ArrowExtensionArray
1860 """
1861 pa_result = pc.unique(self._pa_array)
1862 return self._from_pyarrow_array(pa_result)
1863
1864 def value_counts(self, dropna: bool = True) -> Series:
1865 """
1866 Return a Series containing counts of each unique value.
1867
1868 Parameters
1869 ----------
1870 dropna : bool, default True
1871 Don't include counts of missing values.
1872
1873 Returns
1874 -------
1875 counts : Series
1876
1877 See Also
1878 --------
1879 Series.value_counts
1880 """
1881 from pandas import (
1882 Index,
1883 Series,
1884 )
1885
1886 data = self._pa_array
1887 vc = data.value_counts()
1888
1889 values = vc.field(0)
1890 counts = vc.field(1)
1891 if dropna and data.null_count > 0:
1892 mask = values.is_valid()
1893 values = values.filter(mask)
1894 counts = counts.filter(mask)
1895
1896 counts = ArrowExtensionArray(counts)
1897
1898 index = Index(self._from_pyarrow_array(values), copy=False)
1899
1900 return Series(counts, index=index, name="count", copy=False)
1901
1902 @classmethod
1903 def _concat_same_type(cls, to_concat) -> Self:
1904 """
1905 Concatenate multiple ArrowExtensionArrays.
1906
1907 Parameters
1908 ----------
1909 to_concat : sequence of ArrowExtensionArrays
1910
1911 Returns
1912 -------
1913 ArrowExtensionArray
1914 """
1915 chunks = [array for ea in to_concat for array in ea._pa_array.iterchunks()]
1916 if to_concat[0].dtype == "string":
1917 # StringDtype has no attribute pyarrow_dtype
1918 pa_dtype = pa.large_string()
1919 else:
1920 pa_dtype = to_concat[0].dtype.pyarrow_dtype
1921 arr = pa.chunked_array(chunks, type=pa_dtype)
1922 return to_concat[0]._from_pyarrow_array(arr)
1923
1924 def _accumulate(
1925 self, name: str, *, skipna: bool = True, **kwargs
1926 ) -> ArrowExtensionArray | ExtensionArray:
1927 """
1928 Return an ExtensionArray performing an accumulation operation.
1929
1930 The underlying data type might change.
1931
1932 Parameters
1933 ----------
1934 name : str
1935 Name of the function, supported values are:
1936 - cummin
1937 - cummax
1938 - cumsum
1939 - cumprod
1940 skipna : bool, default True
1941 If True, skip NA values.
1942 **kwargs
1943 Additional keyword arguments passed to the accumulation function.
1944 Currently, there is no supported kwarg.
1945
1946 Returns
1947 -------
1948 array
1949
1950 Raises
1951 ------
1952 NotImplementedError : subclass does not define accumulations
1953 """
1954 if is_string_dtype(self):
1955 return self._str_accumulate(name=name, skipna=skipna, **kwargs)
1956
1957 pyarrow_name = {
1958 "cummax": "cumulative_max",
1959 "cummin": "cumulative_min",
1960 "cumprod": "cumulative_prod_checked",
1961 "cumsum": "cumulative_sum_checked",
1962 }.get(name, name)
1963 pyarrow_meth = getattr(pc, pyarrow_name, None)
1964 if pyarrow_meth is None:
1965 return super()._accumulate(name, skipna=skipna, **kwargs)
1966
1967 data_to_accum = self._pa_array
1968
1969 pa_dtype = data_to_accum.type
1970
1971 convert_to_int = (
1972 pa.types.is_temporal(pa_dtype) and name in ["cummax", "cummin"]
1973 ) or (pa.types.is_duration(pa_dtype) and name == "cumsum")
1974
1975 if convert_to_int:
1976 if pa_dtype.bit_width == 32:
1977 data_to_accum = data_to_accum.cast(pa.int32())
1978 else:
1979 data_to_accum = data_to_accum.cast(pa.int64())
1980
1981 try:
1982 result = pyarrow_meth(data_to_accum, skip_nulls=skipna, **kwargs)
1983 except pa.ArrowNotImplementedError as err:
1984 msg = f"operation '{name}' not supported for dtype '{self.dtype}'"
1985 raise TypeError(msg) from err
1986
1987 if convert_to_int:
1988 result = result.cast(pa_dtype)
1989
1990 return self._from_pyarrow_array(result)
1991
1992 def _str_accumulate(
1993 self, name: str, *, skipna: bool = True, **kwargs
1994 ) -> ArrowExtensionArray | ExtensionArray:
1995 """
1996 Accumulate implementation for strings, see `_accumulate` docstring for details.
1997
1998 pyarrow.compute does not implement these methods for strings.
1999 """
2000 if name == "cumprod":
2001 msg = f"operation '{name}' not supported for dtype '{self.dtype}'"
2002 raise TypeError(msg)
2003
2004 # We may need to strip out trailing NA values
2005 tail: pa.array | None = None
2006 na_mask: pa.array | None = None
2007 pa_array = self._pa_array
2008 np_func = {
2009 "cumsum": np.cumsum,
2010 "cummin": np.minimum.accumulate,
2011 "cummax": np.maximum.accumulate,
2012 }[name]
2013
2014 if self._hasna:
2015 na_mask = pc.is_null(pa_array)
2016 if pc.all(na_mask) == pa.scalar(True):
2017 return self._from_pyarrow_array(pa_array)
2018 if skipna:
2019 if name == "cumsum":
2020 pa_array = _safe_fill_null(pa_array, "")
2021 else:
2022 # We can retain the running min/max by forward/backward filling.
2023 pa_array = pc.fill_null_forward(pa_array)
2024 pa_array = pc.fill_null_backward(pa_array)
2025 else:
2026 # When not skipping NA values, the result should be null from
2027 # the first NA value onward.
2028 idx = pc.index(na_mask, True).as_py()
2029 tail = pa.nulls(len(pa_array) - idx, type=pa_array.type)
2030 pa_array = pa_array[:idx]
2031
2032 # error: Cannot call function of unknown type
2033 pa_result = pa.array(np_func(pa_array), type=pa_array.type) # type: ignore[operator]
2034
2035 if tail is not None:
2036 pa_result = pa.concat_arrays([pa_result, tail])
2037 elif na_mask is not None:
2038 pa_result = pc.if_else(na_mask, None, pa_result)
2039
2040 result = self._from_pyarrow_array(pa_result)
2041 return result
2042
2043 def _reduce_pyarrow(self, name: str, *, skipna: bool = True, **kwargs) -> pa.Scalar:
2044 """
2045 Return a pyarrow scalar result of performing the reduction operation.
2046
2047 Parameters
2048 ----------
2049 name : str
2050 Name of the function, supported values are:
2051 { any, all, min, max, sum, mean, median, prod,
2052 std, var, sem, kurt, skew }.
2053 skipna : bool, default True
2054 If True, skip NaN values.
2055 **kwargs
2056 Additional keyword arguments passed to the reduction function.
2057 Currently, `ddof` is the only supported kwarg.
2058
2059 Returns
2060 -------
2061 pyarrow scalar
2062
2063 Raises
2064 ------
2065 TypeError : subclass does not define reductions
2066 """
2067 pa_type = self._pa_array.type
2068
2069 data_to_reduce = self._pa_array
2070
2071 if name in ["any", "all"] and (
2072 pa.types.is_integer(pa_type)
2073 or pa.types.is_floating(pa_type)
2074 or pa.types.is_duration(pa_type)
2075 or pa.types.is_decimal(pa_type)
2076 ):
2077 # pyarrow only supports any/all for boolean dtype, we allow
2078 # for other dtypes, matching our non-pyarrow behavior
2079
2080 if pa.types.is_duration(pa_type):
2081 data_to_cmp = self._pa_array.cast(pa.int64())
2082 else:
2083 data_to_cmp = self._pa_array
2084
2085 not_eq = pc.not_equal(data_to_cmp, 0)
2086 data_to_reduce = not_eq
2087
2088 elif name in ["min", "max", "sum"] and pa.types.is_duration(pa_type):
2089 data_to_reduce = self._pa_array.cast(pa.int64())
2090
2091 elif name in ["median", "mean", "std", "sem"] and pa.types.is_temporal(pa_type):
2092 nbits = pa_type.bit_width
2093 if nbits == 32:
2094 data_to_reduce = self._pa_array.cast(pa.int32())
2095 else:
2096 data_to_reduce = self._pa_array.cast(pa.int64())
2097
2098 if name == "sem":
2099
2100 def pyarrow_meth(data, skip_nulls, **kwargs):
2101 numerator = pc.stddev(data, skip_nulls=skip_nulls, **kwargs)
2102 denominator = pc.sqrt_checked(pc.count(self._pa_array))
2103 return pc.divide_checked(numerator, denominator)
2104
2105 elif name == "sum" and (
2106 pa.types.is_string(pa_type) or pa.types.is_large_string(pa_type)
2107 ):
2108
2109 def pyarrow_meth(data, skip_nulls, min_count=0): # type: ignore[misc]
2110 mask = pc.is_null(data) if data.null_count > 0 else None
2111 if skip_nulls:
2112 if min_count > 0 and check_below_min_count(
2113 (len(data),),
2114 None if mask is None else mask.to_numpy(),
2115 min_count,
2116 ):
2117 return pa.scalar(None, type=data.type)
2118 if data.null_count > 0:
2119 # binary_join returns null if there is any null ->
2120 # have to filter out any nulls
2121 data = data.filter(pc.invert(mask))
2122 elif mask is not None or check_below_min_count(
2123 (len(data),), None, min_count
2124 ):
2125 return pa.scalar(None, type=data.type)
2126
2127 if pa.types.is_large_string(data.type):
2128 # binary_join only supports string, not large_string
2129 data = data.cast(pa.string())
2130 data_list = pa.ListArray.from_arrays(
2131 [0, len(data)], data.combine_chunks()
2132 )[0]
2133 return pc.binary_join(data_list, "")
2134
2135 else:
2136 pyarrow_name = {
2137 "median": "quantile",
2138 "prod": "product",
2139 "std": "stddev",
2140 "var": "variance",
2141 }.get(name, name)
2142 # error: Incompatible types in assignment
2143 # (expression has type "Optional[Any]", variable has type
2144 # "Callable[[Any, Any, KwArg(Any)], Any]")
2145 pyarrow_meth = getattr(pc, pyarrow_name, None) # type: ignore[assignment]
2146 if pyarrow_meth is None:
2147 # Let ExtensionArray._reduce raise the TypeError
2148 return super()._reduce(name, skipna=skipna, **kwargs)
2149
2150 # GH51624: pyarrow defaults to min_count=1, pandas behavior is min_count=0
2151 if name in ["any", "all"] and "min_count" not in kwargs:
2152 kwargs["min_count"] = 0
2153 elif name == "median":
2154 # GH 52679: Use quantile instead of approximate_median
2155 kwargs["q"] = 0.5
2156
2157 try:
2158 result = pyarrow_meth(data_to_reduce, skip_nulls=skipna, **kwargs)
2159 except (AttributeError, NotImplementedError, TypeError) as err:
2160 msg = (
2161 f"'{type(self).__name__}' with dtype {self.dtype} "
2162 f"does not support operation '{name}' with pyarrow "
2163 f"version {pa.__version__}. '{name}' may be supported by "
2164 f"upgrading pyarrow."
2165 )
2166 raise TypeError(msg) from err
2167 if name == "median":
2168 # GH 52679: Use quantile instead of approximate_median; returns array
2169 result = result[0]
2170
2171 if name in ["min", "max", "sum"] and pa.types.is_duration(pa_type):
2172 result = result.cast(pa_type)
2173 if name in ["median", "mean"] and pa.types.is_temporal(pa_type):
2174 nbits = pa_type.bit_width
2175 if nbits == 32:
2176 result = result.cast(pa.int32(), safe=False)
2177 else:
2178 result = result.cast(pa.int64(), safe=False)
2179 result = result.cast(pa_type)
2180 if name in ["std", "sem"] and pa.types.is_temporal(pa_type):
2181 result = result.cast(pa.int64(), safe=False)
2182 if pa.types.is_duration(pa_type):
2183 result = result.cast(pa_type)
2184 elif pa.types.is_time(pa_type):
2185 result = result.cast(pa.duration(pa_type.unit))
2186 elif pa.types.is_date(pa_type):
2187 # go with closest available unit, i.e. "s"
2188 result = result.cast(pa.duration("s"))
2189 else:
2190 # i.e. timestamp
2191 result = result.cast(pa.duration(pa_type.unit))
2192
2193 return result
2194
2195 def _reduce(
2196 self, name: str, *, skipna: bool = True, keepdims: bool = False, **kwargs
2197 ):
2198 """
2199 Return a scalar result of performing the reduction operation.
2200
2201 Parameters
2202 ----------
2203 name : str
2204 Name of the function, supported values are:
2205 { any, all, min, max, sum, mean, median, prod,
2206 std, var, sem, kurt, skew }.
2207 skipna : bool, default True
2208 If True, skip NaN values.
2209 **kwargs
2210 Additional keyword arguments passed to the reduction function.
2211 Currently, `ddof` is the only supported kwarg.
2212
2213 Returns
2214 -------
2215 scalar
2216
2217 Raises
2218 ------
2219 TypeError : subclass does not define reductions
2220 """
2221 result = self._reduce_calc(name, skipna=skipna, keepdims=keepdims, **kwargs)
2222 if isinstance(result, pa.Array):
2223 return self._from_pyarrow_array(result)
2224 else:
2225 return result
2226
2227 def _reduce_calc(
2228 self, name: str, *, skipna: bool = True, keepdims: bool = False, **kwargs
2229 ):
2230 pa_result = self._reduce_pyarrow(name, skipna=skipna, **kwargs)
2231
2232 if keepdims:
2233 if isinstance(pa_result, pa.Scalar):
2234 result = pa.array([pa_result.as_py()], type=pa_result.type)
2235 else:
2236 result = pa.array(
2237 [pa_result],
2238 type=to_pyarrow_type(infer_dtype_from_scalar(pa_result)[0]),
2239 )
2240 return result
2241
2242 if pc.is_null(pa_result).as_py():
2243 return self.dtype.na_value
2244 elif isinstance(pa_result, pa.Scalar):
2245 result = pa_result.as_py()
2246 pa_type = pa_result.type
2247 if pa.types.is_duration(pa_type) and pa_type.unit != "ns":
2248 return Timedelta(result).as_unit(pa_type.unit)
2249 elif pa.types.is_timestamp(pa_type) and pa_type.unit != "ns":
2250 return Timestamp(result).as_unit(pa_type.unit)
2251 return result
2252 else:
2253 return pa_result
2254
2255 def _explode(self):
2256 """
2257 See Series.explode.__doc__.
2258 """
2259 # child class explode method supports only list types; return
2260 # default implementation for non list types.
2261 if not hasattr(self.dtype, "pyarrow_dtype") or (
2262 not pa.types.is_list(self.dtype.pyarrow_dtype)
2263 and not pa.types.is_large_list(self.dtype.pyarrow_dtype)
2264 ):
2265 return super()._explode()
2266 values = self
2267 counts = pa.compute.list_value_length(values._pa_array)
2268 counts = counts.fill_null(1).to_numpy()
2269 fill_value = pa.scalar([None], type=self._pa_array.type)
2270 mask = counts == 0
2271 if mask.any():
2272 # pc.if_else here is similar to `values[mask] = fill_value`
2273 # but this avoids an object-dtype round-trip.
2274 pa_values = pc.if_else(~mask, values._pa_array, fill_value)
2275 values = self._from_pyarrow_array(pa_values)
2276 counts = counts.copy()
2277 counts[mask] = 1
2278 values = values.fillna(fill_value)
2279 values = self._from_pyarrow_array(pa.compute.list_flatten(values._pa_array))
2280 return values, counts
2281
2282 def __setitem__(self, key, value) -> None:
2283 """Set one or more values inplace.
2284
2285 Parameters
2286 ----------
2287 key : int, ndarray, or slice
2288 When called from, e.g. ``Series.__setitem__``, ``key`` will be
2289 one of
2290
2291 * scalar int
2292 * ndarray of integers.
2293 * boolean ndarray
2294 * slice object
2295
2296 value : ExtensionDtype.type, Sequence[ExtensionDtype.type], or object
2297 value or values to be set of ``key``.
2298
2299 Returns
2300 -------
2301 None
2302 """
2303 if self._readonly:
2304 raise ValueError("Cannot modify read-only array")
2305
2306 # GH50085: unwrap 1D indexers
2307 if isinstance(key, tuple) and len(key) == 1:
2308 key = key[0]
2309
2310 key = check_array_indexer(self, key)
2311 value = self._maybe_convert_setitem_value(value)
2312
2313 if com.is_null_slice(key):
2314 # fast path (GH50248)
2315 if (
2316 isinstance(value, (pa.Array, pa.ChunkedArray))
2317 and value.type == self._pa_array.type
2318 and len(value) == len(self)
2319 ):
2320 # GH#67990 this adopts ``value`` as our backing array, so copy
2321 # first if the caller may still own and mutate its buffers.
2322 if _boxing_may_borrow_memory(value.type):
2323 value = _copy_pyarrow_buffers(value)
2324 data = value
2325 else:
2326 data = self._if_else(True, value, self._pa_array)
2327
2328 elif is_integer(key):
2329 # fast path
2330 key = cast(int, key)
2331 n = len(self)
2332 if key < 0:
2333 key += n
2334 if not 0 <= key < n:
2335 raise IndexError(
2336 f"index {key} is out of bounds for axis 0 with size {n}"
2337 )
2338 if isinstance(value, pa.Scalar):
2339 value = value.as_py()
2340 elif is_list_like(value):
2341 raise ValueError("Length of indexer and values mismatch")
2342 chunks = [
2343 *self._pa_array[:key].chunks,
2344 pa.array([value], type=self._pa_array.type, from_pandas=is_nan_na()),
2345 *self._pa_array[key + 1 :].chunks,
2346 ]
2347 data = pa.chunked_array(chunks).combine_chunks()
2348
2349 elif is_bool_dtype(key):
2350 key = np.asarray(key, dtype=np.bool_)
2351 data = self._replace_with_mask(self._pa_array, key, value)
2352
2353 elif is_scalar(value) or isinstance(value, pa.Scalar):
2354 mask = np.zeros(len(self), dtype=np.bool_)
2355 mask[key] = True
2356 data = self._if_else(mask, value, self._pa_array)
2357
2358 else:
2359 indices = np.arange(len(self))[key]
2360 if len(indices) != len(value):
2361 raise ValueError("Length of indexer and values mismatch")
2362 if len(indices) == 0:
2363 return
2364 # GH#58530 wrong item assignment by repeated key
2365 _, argsort = np.unique(indices, return_index=True)
2366 indices = indices[argsort]
2367 value = value.take(argsort)
2368 mask = np.zeros(len(self), dtype=np.bool_)
2369 mask[indices] = True
2370 data = self._replace_with_mask(self._pa_array, mask, value)
2371
2372 if isinstance(data, pa.Array):
2373 data = pa.chunked_array([data])
2374 self._pa_array = data
2375
2376 def _rank_calc(
2377 self,
2378 *,
2379 axis: AxisInt = 0,
2380 method: str = "average",
2381 na_option: str = "keep",
2382 ascending: bool = True,
2383 pct: bool = False,
2384 ):
2385 if axis != 0:
2386 ranked = super()._rank(
2387 axis=axis,
2388 method=method,
2389 na_option=na_option,
2390 ascending=ascending,
2391 pct=pct,
2392 )
2393 # keep dtypes consistent with the implementation below
2394 if method == "average" or pct:
2395 pa_type = pa.float64()
2396 else:
2397 pa_type = pa.uint64()
2398 result = pa.array(ranked, type=pa_type, from_pandas=is_nan_na())
2399 return result
2400
2401 data = self._pa_array.combine_chunks()
2402 sort_keys = "ascending" if ascending else "descending"
2403 null_placement = "at_start" if na_option == "top" else "at_end"
2404 tiebreaker = "min" if method == "average" else method
2405
2406 result = pc.rank(
2407 data,
2408 sort_keys=sort_keys,
2409 null_placement=null_placement,
2410 tiebreaker=tiebreaker,
2411 )
2412
2413 if na_option == "keep":
2414 mask = pc.is_null(self._pa_array)
2415 null = pa.scalar(None, type=result.type)
2416 result = pc.if_else(mask, null, result)
2417
2418 if method == "average":
2419 result_max = pc.rank(
2420 data,
2421 sort_keys=sort_keys,
2422 null_placement=null_placement,
2423 tiebreaker="max",
2424 )
2425 result_max = result_max.cast(pa.float64())
2426 result_min = result.cast(pa.float64())
2427 result = pc.divide(pc.add(result_min, result_max), 2)
2428
2429 if pct:
2430 if not pa.types.is_floating(result.type):
2431 result = result.cast(pa.float64())
2432 if method == "dense":
2433 divisor = pc.max(result)
2434 else:
2435 divisor = pc.count(result)
2436 result = pc.divide(result, divisor)
2437
2438 return result
2439
2440 def _rank(
2441 self,
2442 *,
2443 axis: AxisInt = 0,
2444 method: str = "average",
2445 na_option: str = "keep",
2446 ascending: bool = True,
2447 pct: bool = False,
2448 ) -> Self:
2449 """
2450 See Series.rank.__doc__.
2451 """
2452 return self._convert_rank_result(
2453 self._rank_calc(
2454 axis=axis,
2455 method=method,
2456 na_option=na_option,
2457 ascending=ascending,
2458 pct=pct,
2459 )
2460 )
2461
2462 def _quantile(self, qs: npt.NDArray[np.float64], interpolation: str) -> Self:
2463 """
2464 Compute the quantiles of self for each quantile in `qs`.
2465
2466 Parameters
2467 ----------
2468 qs : np.ndarray[float64]
2469 interpolation: str
2470
2471 Returns
2472 -------
2473 same type as self
2474 """
2475 pa_dtype = self._pa_array.type
2476
2477 data = self._pa_array
2478 if pa.types.is_temporal(pa_dtype):
2479 # https://github.com/apache/arrow/issues/33769 in these cases
2480 # we can cast to ints and back
2481 nbits = pa_dtype.bit_width
2482 if nbits == 32:
2483 data = data.cast(pa.int32())
2484 else:
2485 data = data.cast(pa.int64())
2486
2487 result = pc.quantile(data, q=qs, interpolation=interpolation)
2488
2489 if pa.types.is_temporal(pa_dtype):
2490 if pa.types.is_floating(result.type):
2491 result = pc.floor(result)
2492 nbits = pa_dtype.bit_width
2493 if nbits == 32:
2494 result = result.cast(pa.int32())
2495 else:
2496 result = result.cast(pa.int64())
2497 result = result.cast(pa_dtype)
2498
2499 return self._from_pyarrow_array(result)
2500
2501 def _mode(self, dropna: bool = True) -> Self:
2502 """
2503 Returns the mode(s) of the ExtensionArray.
2504
2505 Always returns `ExtensionArray` even if only one value.
2506
2507 Parameters
2508 ----------
2509 dropna : bool, default True
2510 Don't consider counts of NA values.
2511
2512 Returns
2513 -------
2514 same type as self
2515 Sorted, if possible.
2516 """
2517 pa_type = self._pa_array.type
2518 if pa.types.is_temporal(pa_type):
2519 nbits = pa_type.bit_width
2520 if nbits == 32:
2521 data = self._pa_array.cast(pa.int32())
2522 elif nbits == 64:
2523 data = self._pa_array.cast(pa.int64())
2524 else:
2525 raise NotImplementedError(pa_type)
2526 else:
2527 data = self._pa_array
2528
2529 if dropna:
2530 data = data.drop_null()
2531
2532 res = pc.value_counts(data)
2533 most_common = res.field("values").filter(
2534 pc.equal(res.field("counts"), pc.max(res.field("counts")))
2535 )
2536
2537 if pa.types.is_temporal(pa_type):
2538 most_common = most_common.cast(pa_type)
2539
2540 most_common = most_common.take(pc.array_sort_indices(most_common))
2541 return self._from_pyarrow_array(most_common)
2542
2543 def _maybe_convert_setitem_value(self, value):
2544 """Maybe convert value to be pyarrow compatible."""
2545 try:
2546 value = self._box_pa(value, self._pa_array.type)
2547 except pa.ArrowTypeError as err:
2548 msg = f"Invalid value '{value!s}' for dtype '{self.dtype}'"
2549 raise TypeError(msg) from err
2550 return value
2551
2552 def interpolate(
2553 self,
2554 *,
2555 method: InterpolateOptions,
2556 axis: int,
2557 index,
2558 limit,
2559 limit_direction,
2560 limit_area,
2561 copy: bool,
2562 **kwargs,
2563 ) -> Self:
2564 """
2565 See NDFrame.interpolate.__doc__.
2566 """
2567 # NB: we return type(self) even if copy=False
2568 if not self.dtype._is_numeric:
2569 raise TypeError(f"Cannot interpolate with {self.dtype} dtype")
2570
2571 # GH#65345: a pyarrow-native fast path for
2572 # method="linear"/limit_direction="forward" was removed here because
2573 # it only handled isolated NAs (leaving consecutive and trailing NAs
2574 # unfilled), truncated interpolated values for integer dtypes, and
2575 # did not upcast to float64 like the general path below.
2576
2577 mask = self.isna()
2578 if self.dtype.kind == "f":
2579 data = self._pa_array.to_numpy()
2580 elif self.dtype.kind in "iu":
2581 data = self.to_numpy(dtype="f8", na_value=0.0)
2582 else:
2583 raise NotImplementedError(
2584 f"interpolate is not implemented for dtype={self.dtype}"
2585 )
2586
2587 missing.interpolate_2d_inplace(
2588 data,
2589 method=method,
2590 axis=0,
2591 index=index,
2592 limit=limit,
2593 limit_direction=limit_direction,
2594 limit_area=limit_area,
2595 mask=mask,
2596 **kwargs,
2597 )
2598 return self._from_pyarrow_array(self._box_pa_array(pa.array(data, mask=mask)))
2599
2600 @classmethod
2601 def _if_else(
2602 cls,
2603 cond: npt.NDArray[np.bool_] | bool,
2604 left: ArrayLike | Scalar,
2605 right: ArrayLike | Scalar,
2606 ) -> pa.Array:
2607 """
2608 Choose values based on a condition.
2609
2610 Analogous to pyarrow.compute.if_else, with logic
2611 to fallback to numpy for unsupported types.
2612
2613 Parameters
2614 ----------
2615 cond : npt.NDArray[np.bool_] or bool
2616 left : ArrayLike | Scalar
2617 right : ArrayLike | Scalar
2618
2619 Returns
2620 -------
2621 pa.Array
2622 """
2623
2624 # TODO: Remove this part when pa.if_else is fixed (GH#64320)
2625 def _maybe_combine(arr):
2626 if not isinstance(arr, pa.ChunkedArray) or not (
2627 pa.types.is_string(arr.type) or pa.types.is_large_string(arr.type)
2628 ):
2629 return arr
2630 if not any(c.offset != 0 for c in arr.chunks):
2631 return arr
2632 try:
2633 return arr.combine_chunks()
2634 except (pa.ArrowInvalid, pa.ArrowCapacityError, MemoryError):
2635 return None
2636
2637 left_c, right_c = _maybe_combine(left), _maybe_combine(right)
2638 if left_c is not None and right_c is not None:
2639 try:
2640 return pc.if_else(cond, left_c, right_c)
2641 except pa.ArrowNotImplementedError:
2642 pass
2643 if left_c is not None:
2644 left = left_c
2645 if right_c is not None:
2646 right = right_c
2647
2648 def _to_numpy_and_type(value) -> tuple[np.ndarray, pa.DataType | None]:
2649 if isinstance(value, (pa.Array, pa.ChunkedArray)):
2650 pa_type = value.type
2651 elif isinstance(value, pa.Scalar):
2652 pa_type = value.type
2653 value = value.as_py()
2654 else:
2655 pa_type = None
2656 return np.array(value, dtype=object), pa_type
2657
2658 left, left_type = _to_numpy_and_type(left)
2659 right, right_type = _to_numpy_and_type(right)
2660 pa_type = left_type or right_type
2661 result = np.where(cond, left, right)
2662 return pa.array(result, type=pa_type, from_pandas=is_nan_na())
2663
2664 @classmethod
2665 def _replace_with_mask(
2666 cls,
2667 values: pa.Array | pa.ChunkedArray,
2668 mask: npt.NDArray[np.bool_] | bool,
2669 replacements: ArrayLike | Scalar,
2670 ) -> pa.Array | pa.ChunkedArray:
2671 """
2672 Replace items selected with a mask.
2673
2674 Analogous to pyarrow.compute.replace_with_mask, with logic
2675 to fallback to numpy for unsupported types.
2676
2677 Parameters
2678 ----------
2679 values : pa.Array or pa.ChunkedArray
2680 mask : npt.NDArray[np.bool_] or bool
2681 replacements : ArrayLike or Scalar
2682 Replacement value(s)
2683
2684 Returns
2685 -------
2686 pa.Array or pa.ChunkedArray
2687 """
2688 if isinstance(replacements, pa.ChunkedArray):
2689 # replacements must be array or scalar, not ChunkedArray
2690 replacements = replacements.combine_chunks()
2691 if isinstance(values, pa.ChunkedArray) and pa.types.is_boolean(values.type):
2692 # GH#52059 replace_with_mask segfaults for chunked array
2693 # https://github.com/apache/arrow/issues/34634
2694 values = values.combine_chunks()
2695 try:
2696 return pc.replace_with_mask(values, mask, replacements)
2697 except pa.ArrowNotImplementedError:
2698 pass
2699 if isinstance(replacements, pa.Array):
2700 replacements = np.array(replacements, dtype=object)
2701 elif isinstance(replacements, pa.Scalar):
2702 replacements = replacements.as_py()
2703
2704 result = np.array(values, dtype=object)
2705 result[mask] = replacements
2706 return pa.array(result, type=values.type, from_pandas=is_nan_na())
2707
2708 # ------------------------------------------------------------------
2709 # GroupBy Methods
2710
2711 def _to_masked(self):
2712 pa_dtype = self._pa_array.type
2713
2714 if pa.types.is_floating(pa_dtype) or pa.types.is_integer(pa_dtype):
2715 na_value = 1
2716 elif pa.types.is_boolean(pa_dtype):
2717 na_value = True
2718 else:
2719 raise NotImplementedError
2720
2721 dtype = _arrow_dtype_mapping()[pa_dtype]
2722 mask = self.isna()
2723 arr = self.to_numpy(dtype=dtype.numpy_dtype, na_value=na_value)
2724 return dtype.construct_array_type()(arr, mask)
2725
2726 def _groupby_op(
2727 self,
2728 *,
2729 how: str,
2730 has_dropped_na: bool,
2731 min_count: int,
2732 ngroups: int,
2733 ids: npt.NDArray[np.intp],
2734 **kwargs,
2735 ):
2736 if isinstance(self.dtype, StringDtype):
2737 if how in [
2738 "prod",
2739 "mean",
2740 "median",
2741 "cumsum",
2742 "cumprod",
2743 "std",
2744 "sem",
2745 "var",
2746 "skew",
2747 ]:
2748 raise TypeError(
2749 f"dtype '{self.dtype}' does not support operation '{how}'"
2750 )
2751 return super()._groupby_op(
2752 how=how,
2753 has_dropped_na=has_dropped_na,
2754 min_count=min_count,
2755 ngroups=ngroups,
2756 ids=ids,
2757 **kwargs,
2758 )
2759
2760 # maybe convert to a compatible dtype optimized for groupby
2761 values: ExtensionArray
2762 pa_type = self._pa_array.type
2763 if pa.types.is_timestamp(pa_type):
2764 values = self._to_datetimearray()
2765 elif pa.types.is_duration(pa_type):
2766 values = self._to_timedeltaarray()
2767 else:
2768 values = self._to_masked()
2769
2770 result = values._groupby_op(
2771 how=how,
2772 has_dropped_na=has_dropped_na,
2773 min_count=min_count,
2774 ngroups=ngroups,
2775 ids=ids,
2776 **kwargs,
2777 )
2778 if isinstance(result, np.ndarray):
2779 return result
2780 elif isinstance(result, BaseMaskedArray):
2781 pa_result = result.__arrow_array__()
2782 return self._from_pyarrow_array(pa_result)
2783 else:
2784 # DatetimeArray, TimedeltaArray
2785 pa_result = pa.array(result)
2786 return self._from_pyarrow_array(pa_result)
2787
2788 def _apply_elementwise(self, func: Callable) -> list[list[Any]]:
2789 """Apply a callable to each element while maintaining the chunking structure."""
2790 return [
2791 [
2792 None if val is None else func(val)
2793 for val in chunk.to_numpy(zero_copy_only=False)
2794 ]
2795 for chunk in self._pa_array.iterchunks()
2796 ]
2797
2798 def _convert_bool_result(self, result, na=lib.no_default, method_name=None):
2799 if na is not lib.no_default and not isna(na): # pyright: ignore [reportGeneralTypeIssues]
2800 result = result.fill_null(na)
2801 return self._from_pyarrow_array(result)
2802
2803 def _convert_int_result(self, result):
2804 return self._from_pyarrow_array(result)
2805
2806 def _convert_rank_result(self, result):
2807 return self._from_pyarrow_array(result)
2808
2809 def _str_count(self, pat: str, flags: int = 0) -> Self:
2810 if flags:
2811 raise NotImplementedError(f"count not implemented with {flags=}")
2812 return self._from_pyarrow_array(pc.count_substring_regex(self._pa_array, pat))
2813
2814 def _str_repeat(self, repeats: int | Sequence[int]) -> Self:
2815 if not isinstance(repeats, int):
2816 raise NotImplementedError(
2817 f"repeat is not implemented when repeats is {type(repeats).__name__}"
2818 )
2819 return self._from_pyarrow_array(pc.binary_repeat(self._pa_array, repeats))
2820
2821 def _str_join(self, sep: str) -> Self:
2822 if pa.types.is_string(self._pa_array.type) or pa.types.is_large_string(
2823 self._pa_array.type
2824 ):
2825 result = self._apply_elementwise(list)
2826 result = pa.chunked_array(result, type=pa.list_(pa.string()))
2827 else:
2828 result = self._pa_array
2829 return self._from_pyarrow_array(pc.binary_join(result, sep))
2830
2831 def _str_partition(self, sep: str, expand: bool) -> Self:
2832 predicate = lambda val: val.partition(sep)
2833 result = self._apply_elementwise(predicate)
2834 return self._from_pyarrow_array(pa.chunked_array(result))
2835
2836 def _str_rpartition(self, sep: str, expand: bool) -> Self:
2837 predicate = lambda val: val.rpartition(sep)
2838 result = self._apply_elementwise(predicate)
2839 return self._from_pyarrow_array(pa.chunked_array(result))
2840
2841 def _str_casefold(self) -> Self:
2842 predicate = lambda val: val.casefold()
2843 result = self._apply_elementwise(predicate)
2844 return self._from_pyarrow_array(pa.chunked_array(result))
2845
2846 def _str_encode(self, encoding: str, errors: str = "strict") -> Self:
2847 predicate = lambda val: val.encode(encoding, errors)
2848 result = self._apply_elementwise(predicate)
2849 return self._from_pyarrow_array(pa.chunked_array(result))
2850
2851 def _str_extract(self, pat: str, flags: int = 0, expand: bool = True):
2852 if flags:
2853 raise NotImplementedError("Only flags=0 is implemented.")
2854 groups = re.compile(pat).groupindex.keys()
2855 if len(groups) == 0:
2856 raise ValueError(f"{pat=} must contain a symbolic group name.")
2857 result = pc.extract_regex(self._pa_array, pat)
2858 if expand:
2859 return {
2860 col: self._from_pyarrow_array(pc.struct_field(result, [i]))
2861 for col, i in zip(groups, range(result.type.num_fields), strict=True)
2862 }
2863 else:
2864 return type(self)(pc.struct_field(result, [0]))
2865
2866 def _str_findall(self, pat: str, flags: int = 0) -> Self:
2867 regex = re.compile(pat, flags=flags)
2868 predicate = lambda val: regex.findall(val)
2869 result = self._apply_elementwise(predicate)
2870 return self._from_pyarrow_array(pa.chunked_array(result))
2871
2872 def _str_get_dummies(self, sep: str = "|", dtype: NpDtype | None = None):
2873 if dtype is None:
2874 dtype = np.bool_
2875 split = pc.split_pattern(self._pa_array, sep)
2876 flattened_values = pc.list_flatten(split)
2877 uniques = flattened_values.unique()
2878 uniques_sorted = uniques.take(pa.compute.array_sort_indices(uniques))
2879 lengths = pc.list_value_length(split).fill_null(0).to_numpy()
2880 n_rows = len(self)
2881 n_cols = len(uniques)
2882 indices = pc.index_in(flattened_values, uniques_sorted).to_numpy()
2883 indices = indices + np.arange(n_rows).repeat(lengths) * n_cols
2884 _dtype = pandas_dtype(dtype)
2885 dummies_dtype: NpDtype
2886 if isinstance(_dtype, np.dtype):
2887 dummies_dtype = _dtype
2888 else:
2889 dummies_dtype = np.bool_
2890 dummies = np.zeros(n_rows * n_cols, dtype=dummies_dtype)
2891 dummies[indices] = True
2892 dummies = dummies.reshape((n_rows, n_cols))
2893 result = self._from_pyarrow_array(pa.array(list(dummies)))
2894 return result, uniques_sorted.to_pylist()
2895
2896 def _str_index(self, sub: str, start: int = 0, end: int | None = None) -> Self:
2897 predicate = lambda val: val.index(sub, start, end)
2898 result = self._apply_elementwise(predicate)
2899 return self._from_pyarrow_array(pa.chunked_array(result))
2900
2901 def _str_rindex(self, sub: str, start: int = 0, end: int | None = None) -> Self:
2902 predicate = lambda val: val.rindex(sub, start, end)
2903 result = self._apply_elementwise(predicate)
2904 return self._from_pyarrow_array(pa.chunked_array(result))
2905
2906 def _str_normalize(self, form: Literal["NFC", "NFD", "NFKC", "NFKD"]) -> Self:
2907 predicate = lambda val: unicodedata.normalize(form, val)
2908 result = self._apply_elementwise(predicate)
2909 return self._from_pyarrow_array(pa.chunked_array(result))
2910
2911 def _str_rfind(self, sub: str, start: int = 0, end=None) -> Self:
2912 predicate = lambda val: val.rfind(sub, start, end)
2913 result = self._apply_elementwise(predicate)
2914 return self._from_pyarrow_array(pa.chunked_array(result))
2915
2916 def _str_split(
2917 self,
2918 pat: str | None = None,
2919 n: int | None = -1,
2920 expand: bool = False,
2921 regex: bool | None = None,
2922 ) -> Self:
2923 if n in {-1, 0}:
2924 n = None
2925 if pat is None:
2926 split_func = pc.utf8_split_whitespace
2927 elif regex:
2928 split_func = functools.partial(pc.split_pattern_regex, pattern=pat)
2929 else:
2930 split_func = functools.partial(pc.split_pattern, pattern=pat)
2931 return self._from_pyarrow_array(split_func(self._pa_array, max_splits=n))
2932
2933 def _str_rsplit(self, pat: str | None = None, n: int | None = -1) -> Self:
2934 if n in {-1, 0}:
2935 n = None
2936 if pat is None:
2937 return self._from_pyarrow_array(
2938 pc.utf8_split_whitespace(self._pa_array, max_splits=n, reverse=True)
2939 )
2940 return self._from_pyarrow_array(
2941 pc.split_pattern(self._pa_array, pat, max_splits=n, reverse=True)
2942 )
2943
2944 def _str_translate(self, table: dict[int, str]) -> Self:
2945 predicate = lambda val: val.translate(table)
2946 result = self._apply_elementwise(predicate)
2947 return self._from_pyarrow_array(pa.chunked_array(result))
2948
2949 def _str_wrap(self, width: int, **kwargs) -> Self:
2950 kwargs["width"] = width
2951 tw = textwrap.TextWrapper(**kwargs)
2952 predicate = lambda val: "\n".join(tw.wrap(val))
2953 result = self._apply_elementwise(predicate)
2954 return self._from_pyarrow_array(pa.chunked_array(result))
2955
2956 def _str_zfill(self, width: int) -> Self:
2957 if pa_version_under21p0:
2958 predicate = lambda val: val.zfill(width)
2959 result = self._apply_elementwise(predicate)
2960 return type(self)(pa.chunked_array(result))
2961 return type(self)(pc.utf8_zfill(self._pa_array, width))
2962
2963 @property
2964 def _dt_days(self) -> Self:
2965 return self._from_pyarrow_array(
2966 pa.array(
2967 self._to_timedeltaarray().components.days,
2968 from_pandas=True,
2969 type=pa.int32(),
2970 )
2971 )
2972
2973 @property
2974 def _dt_hours(self) -> Self:
2975 return self._from_pyarrow_array(
2976 pa.array(
2977 self._to_timedeltaarray().components.hours,
2978 from_pandas=True,
2979 type=pa.int32(),
2980 )
2981 )
2982
2983 @property
2984 def _dt_minutes(self) -> Self:
2985 return self._from_pyarrow_array(
2986 pa.array(
2987 self._to_timedeltaarray().components.minutes,
2988 from_pandas=True,
2989 type=pa.int32(),
2990 )
2991 )
2992
2993 @property
2994 def _dt_seconds(self) -> Self:
2995 return self._from_pyarrow_array(
2996 pa.array(
2997 self._to_timedeltaarray().components.seconds,
2998 from_pandas=True,
2999 type=pa.int32(),
3000 )
3001 )
3002
3003 @property
3004 def _dt_milliseconds(self) -> Self:
3005 return self._from_pyarrow_array(
3006 pa.array(
3007 self._to_timedeltaarray().components.milliseconds,
3008 from_pandas=True,
3009 type=pa.int32(),
3010 )
3011 )
3012
3013 @property
3014 def _dt_microseconds(self) -> Self:
3015 return self._from_pyarrow_array(
3016 pa.array(
3017 self._to_timedeltaarray().components.microseconds,
3018 from_pandas=True,
3019 type=pa.int32(),
3020 )
3021 )
3022
3023 @property
3024 def _dt_nanoseconds(self) -> Self:
3025 return self._from_pyarrow_array(
3026 pa.array(
3027 self._to_timedeltaarray().components.nanoseconds,
3028 from_pandas=True,
3029 type=pa.int32(),
3030 )
3031 )
3032
3033 def _dt_to_pytimedelta(self) -> np.ndarray:
3034 data = self._pa_array.to_pylist()
3035 if self._dtype.pyarrow_dtype.unit == "ns":
3036 data = [None if ts is None else ts.to_pytimedelta() for ts in data]
3037 return np.array(data, dtype=object)
3038
3039 def _dt_total_seconds(self) -> Self:
3040 unit = self._pa_array.type.unit
3041 unit_per_second = {"s": 1.0, "ms": 1e3, "us": 1e6, "ns": 1e9}
3042 result = pc.divide(pc.cast(self._pa_array, pa.int64()), unit_per_second[unit])
3043 return self._from_pyarrow_array(result)
3044
3045 def _dt_as_unit(self, unit: str) -> Self:
3046 pa_type = self._pa_array.type
3047 if pa.types.is_timestamp(pa_type):
3048 target_type = pa.timestamp(unit, tz=pa_type.tz)
3049 elif pa.types.is_duration(pa_type):
3050 target_type = pa.duration(unit)
3051 else:
3052 raise NotImplementedError(f"as_unit not implemented for {pa_type}")
3053 # Use safe=False to allow truncation, matching pandas as_unit behavior
3054 result = pc.cast(self._pa_array, target_type, safe=False)
3055 return self._from_pyarrow_array(result)
3056
3057 @property
3058 def _dt_year(self) -> Self:
3059 result = pc.year(self._pa_array)
3060 return self._from_pyarrow_array(result)
3061
3062 @property
3063 def _dt_day(self) -> Self:
3064 result = pc.day(self._pa_array)
3065 return self._from_pyarrow_array(result)
3066
3067 @property
3068 def _dt_day_of_week(self) -> Self:
3069 result = pc.day_of_week(self._pa_array)
3070 return self._from_pyarrow_array(result)
3071
3072 _dt_dayofweek = _dt_day_of_week
3073 _dt_weekday = _dt_day_of_week
3074
3075 @property
3076 def _dt_day_of_year(self) -> Self:
3077 result = pc.day_of_year(self._pa_array)
3078 return self._from_pyarrow_array(result)
3079
3080 _dt_dayofyear = _dt_day_of_year
3081
3082 @property
3083 def _dt_hour(self) -> Self:
3084 result = pc.hour(self._pa_array)
3085 return self._from_pyarrow_array(result)
3086
3087 def _dt_isocalendar(self) -> Self:
3088 result = pc.iso_calendar(self._pa_array)
3089 return self._from_pyarrow_array(result)
3090
3091 @property
3092 def _dt_is_leap_year(self) -> Self:
3093 result = pc.is_leap_year(self._pa_array)
3094 return self._from_pyarrow_array(result)
3095
3096 @property
3097 def _dt_is_month_start(self) -> Self:
3098 result = pc.equal(pc.day(self._pa_array), 1)
3099 return self._from_pyarrow_array(result)
3100
3101 @property
3102 def _dt_is_month_end(self) -> Self:
3103 result = pc.equal(
3104 pc.days_between(
3105 pc.floor_temporal(self._pa_array, unit="day"),
3106 pc.ceil_temporal(self._pa_array, unit="month"),
3107 ),
3108 1,
3109 )
3110 return self._from_pyarrow_array(result)
3111
3112 @property
3113 def _dt_is_year_start(self) -> Self:
3114 result = pc.and_(
3115 pc.equal(pc.month(self._pa_array), 1),
3116 pc.equal(pc.day(self._pa_array), 1),
3117 )
3118 return self._from_pyarrow_array(result)
3119
3120 @property
3121 def _dt_is_year_end(self) -> Self:
3122 result = pc.and_(
3123 pc.equal(pc.month(self._pa_array), 12),
3124 pc.equal(pc.day(self._pa_array), 31),
3125 )
3126 return self._from_pyarrow_array(result)
3127
3128 @property
3129 def _dt_is_quarter_start(self) -> Self:
3130 result = pc.equal(
3131 pc.floor_temporal(self._pa_array, unit="quarter"),
3132 pc.floor_temporal(self._pa_array, unit="day"),
3133 )
3134 return self._from_pyarrow_array(result)
3135
3136 @property
3137 def _dt_is_quarter_end(self) -> Self:
3138 result = pc.equal(
3139 pc.days_between(
3140 pc.floor_temporal(self._pa_array, unit="day"),
3141 pc.ceil_temporal(self._pa_array, unit="quarter"),
3142 ),
3143 1,
3144 )
3145 return self._from_pyarrow_array(result)
3146
3147 @property
3148 def _dt_days_in_month(self) -> Self:
3149 result = pc.days_between(
3150 pc.floor_temporal(self._pa_array, unit="month"),
3151 pc.ceil_temporal(self._pa_array, unit="month"),
3152 )
3153 return self._from_pyarrow_array(result)
3154
3155 _dt_daysinmonth = _dt_days_in_month
3156
3157 @property
3158 def _dt_microsecond(self) -> Self:
3159 # GH 59154
3160 us = pc.microsecond(self._pa_array)
3161 ms_to_us = pc.multiply(pc.millisecond(self._pa_array), 1000)
3162 result = pc.add(us, ms_to_us)
3163 return self._from_pyarrow_array(result)
3164
3165 @property
3166 def _dt_minute(self) -> Self:
3167 result = pc.minute(self._pa_array)
3168 return self._from_pyarrow_array(result)
3169
3170 @property
3171 def _dt_month(self) -> Self:
3172 result = pc.month(self._pa_array)
3173 return self._from_pyarrow_array(result)
3174
3175 @property
3176 def _dt_nanosecond(self) -> Self:
3177 result = pc.nanosecond(self._pa_array)
3178 return self._from_pyarrow_array(result)
3179
3180 @property
3181 def _dt_quarter(self) -> Self:
3182 result = pc.quarter(self._pa_array)
3183 return self._from_pyarrow_array(result)
3184
3185 @property
3186 def _dt_second(self) -> Self:
3187 result = pc.second(self._pa_array)
3188 return self._from_pyarrow_array(result)
3189
3190 @property
3191 def _dt_date(self) -> Self:
3192 result = self._pa_array.cast(pa.date32())
3193 return self._from_pyarrow_array(result)
3194
3195 @property
3196 def _dt_time(self) -> Self:
3197 unit = (
3198 self.dtype.pyarrow_dtype.unit
3199 if self.dtype.pyarrow_dtype.unit in {"us", "ns"}
3200 else "ns"
3201 )
3202 result = self._pa_array.cast(pa.time64(unit))
3203 return self._from_pyarrow_array(result)
3204
3205 @property
3206 def _dt_tz(self):
3207 return timezones.maybe_get_tz(self.dtype.pyarrow_dtype.tz)
3208
3209 @property
3210 def _dt_unit(self):
3211 return self.dtype.pyarrow_dtype.unit
3212
3213 def _dt_normalize(self) -> Self:
3214 result = pc.floor_temporal(self._pa_array, 1, "day")
3215 return self._from_pyarrow_array(result)
3216
3217 def _dt_strftime(self, format: str) -> Self:
3218 result = pc.strftime(self._pa_array, format=format)
3219 return self._from_pyarrow_array(result)
3220
3221 def _round_temporally(
3222 self,
3223 method: Literal["ceil", "floor", "round"],
3224 freq,
3225 ambiguous: TimeAmbiguous = "raise",
3226 nonexistent: TimeNonexistent = "raise",
3227 ) -> Self:
3228 if ambiguous != "raise":
3229 raise NotImplementedError("ambiguous is not supported.")
3230 if nonexistent != "raise":
3231 raise NotImplementedError("nonexistent is not supported.")
3232 offset = to_offset(freq)
3233 if offset is None:
3234 raise ValueError(f"Must specify a valid frequency: {freq}")
3235 pa_supported_unit = {
3236 "Y": "year",
3237 "YS": "year",
3238 "Q": "quarter",
3239 "QS": "quarter",
3240 "M": "month",
3241 "MS": "month",
3242 "W": "week",
3243 "D": "day",
3244 "h": "hour",
3245 "min": "minute",
3246 "s": "second",
3247 "ms": "millisecond",
3248 "us": "microsecond",
3249 "ns": "nanosecond",
3250 }
3251 unit = pa_supported_unit.get(offset._prefix, None)
3252 if unit is None:
3253 raise ValueError(f"{freq=} is not supported")
3254 multiple = offset.n
3255 rounding_method = getattr(pc, f"{method}_temporal")
3256 result = rounding_method(self._pa_array, multiple=multiple, unit=unit)
3257 return self._from_pyarrow_array(result)
3258
3259 def _dt_ceil(
3260 self,
3261 freq,
3262 ambiguous: TimeAmbiguous = "raise",
3263 nonexistent: TimeNonexistent = "raise",
3264 ) -> Self:
3265 return self._round_temporally("ceil", freq, ambiguous, nonexistent)
3266
3267 def _dt_floor(
3268 self,
3269 freq,
3270 ambiguous: TimeAmbiguous = "raise",
3271 nonexistent: TimeNonexistent = "raise",
3272 ) -> Self:
3273 return self._round_temporally("floor", freq, ambiguous, nonexistent)
3274
3275 def _dt_round(
3276 self,
3277 freq,
3278 ambiguous: TimeAmbiguous = "raise",
3279 nonexistent: TimeNonexistent = "raise",
3280 ) -> Self:
3281 return self._round_temporally("round", freq, ambiguous, nonexistent)
3282
3283 def _dt_day_name(self, locale: str | None = None) -> Self:
3284 if locale is None:
3285 locale = "C"
3286 result = pc.strftime(self._pa_array, format="%A", locale=locale)
3287 return self._from_pyarrow_array(result)
3288
3289 def _dt_month_name(self, locale: str | None = None) -> Self:
3290 if locale is None:
3291 locale = "C"
3292 result = pc.strftime(self._pa_array, format="%B", locale=locale)
3293 return self._from_pyarrow_array(result)
3294
3295 def _dt_to_pydatetime(self) -> Series:
3296 from pandas import Series
3297
3298 if pa.types.is_date(self.dtype.pyarrow_dtype):
3299 raise ValueError(
3300 f"to_pydatetime cannot be called with {self.dtype.pyarrow_dtype} type. "
3301 "Convert to pyarrow timestamp type."
3302 )
3303 data = self._pa_array.to_pylist()
3304 if self._dtype.pyarrow_dtype.unit == "ns":
3305 data = [None if ts is None else ts.to_pydatetime(warn=False) for ts in data]
3306 return Series(data, dtype=object)
3307
3308 def _dt_tz_localize(
3309 self,
3310 tz,
3311 ambiguous: TimeAmbiguous = "raise",
3312 nonexistent: TimeNonexistent = "raise",
3313 ) -> Self:
3314 if ambiguous != "raise":
3315 raise NotImplementedError(f"{ambiguous=} is not supported")
3316 nonexistent_pa = {
3317 "raise": "raise",
3318 "shift_backward": "earliest",
3319 "shift_forward": "latest",
3320 }.get(
3321 nonexistent, # type: ignore[arg-type]
3322 None,
3323 )
3324 if nonexistent_pa is None:
3325 raise NotImplementedError(f"{nonexistent=} is not supported")
3326 if tz is None:
3327 result = pc.local_timestamp(self._pa_array)
3328 else:
3329 result = pc.assume_timezone(
3330 self._pa_array, str(tz), ambiguous=ambiguous, nonexistent=nonexistent_pa
3331 )
3332 return self._from_pyarrow_array(result)
3333
3334 def _dt_tz_convert(self, tz) -> Self:
3335 if self.dtype.pyarrow_dtype.tz is None:
3336 raise TypeError(
3337 "Cannot convert tz-naive timestamps, use tz_localize to localize"
3338 )
3339 current_unit = self.dtype.pyarrow_dtype.unit
3340 result = self._pa_array.cast(pa.timestamp(current_unit, tz))
3341 return self._from_pyarrow_array(result)
3342
3343
3344def transpose_homogeneous_pyarrow(
3345 arrays: Sequence[ArrowExtensionArray],
3346) -> list[ArrowExtensionArray]:
3347 """Transpose arrow extension arrays in a list, but faster.
3348
3349 Input should be a list of arrays of equal length and all have the same
3350 dtype. The caller is responsible for ensuring validity of input data.
3351 """
3352 arrays = list(arrays)
3353 nrows, ncols = len(arrays[0]), len(arrays)
3354 indices = np.arange(nrows * ncols).reshape(ncols, nrows).T.reshape(-1)
3355 arr = pa.chunked_array([chunk for arr in arrays for chunk in arr._pa_array.chunks])
3356 arr = arr.take(indices)
3357 return [ArrowExtensionArray(arr.slice(i * ncols, ncols)) for i in range(nrows)]