1from __future__ import annotations
2
3import codecs
4from functools import wraps
5import re
6from typing import (
7 TYPE_CHECKING,
8 Literal,
9 cast,
10)
11import warnings
12
13import numpy as np
14
15from pandas._config import using_string_dtype
16
17from pandas._libs import lib
18from pandas._typing import (
19 AlignJoin,
20 DtypeObj,
21 F,
22 Scalar,
23 npt,
24)
25from pandas.util._exceptions import find_stack_level
26
27from pandas.core.dtypes.common import (
28 ensure_object,
29 is_bool_dtype,
30 is_extension_array_dtype,
31 is_integer,
32 is_list_like,
33 is_numeric_dtype,
34 is_object_dtype,
35 is_re,
36 is_string_dtype,
37)
38from pandas.core.dtypes.dtypes import (
39 ArrowDtype,
40 CategoricalDtype,
41)
42from pandas.core.dtypes.generic import (
43 ABCDataFrame,
44 ABCIndex,
45 ABCMultiIndex,
46 ABCSeries,
47)
48from pandas.core.dtypes.missing import isna
49
50from pandas.core.arrays import ExtensionArray
51from pandas.core.base import NoNewAttributesMixin
52from pandas.core.construction import extract_array
53
54if TYPE_CHECKING:
55 from collections.abc import (
56 Callable,
57 Hashable,
58 Iterator,
59 )
60
61 from pandas._typing import NpDtype
62
63 from pandas import (
64 DataFrame,
65 Index,
66 Series,
67 )
68
69_cpython_optimized_encoders = (
70 "utf-8",
71 "utf8",
72 "latin-1",
73 "latin1",
74 "iso-8859-1",
75 "mbcs",
76 "ascii",
77)
78_cpython_optimized_decoders = (*_cpython_optimized_encoders, "utf-16", "utf-32")
79
80
81def forbid_nonstring_types(
82 forbidden: list[str] | None, name: str | None = None
83) -> Callable[[F], F]:
84 """
85 Decorator to forbid specific types for a method of StringMethods.
86
87 For calling `.str.{method}` on a Series or Index, it is necessary to first
88 initialize the :class:`StringMethods` object, and then call the method.
89 However, different methods allow different input types, and so this can not
90 be checked during :meth:`StringMethods.__init__`, but must be done on a
91 per-method basis. This decorator exists to facilitate this process, and
92 make it explicit which (inferred) types are disallowed by the method.
93
94 :meth:`StringMethods.__init__` allows the *union* of types its different
95 methods allow (after skipping NaNs; see :meth:`StringMethods._validate`),
96 namely: ['string', 'empty', 'bytes', 'mixed', 'mixed-integer'].
97
98 The default string types ['string', 'empty'] are allowed for all methods.
99 For the additional types ['bytes', 'mixed', 'mixed-integer'], each method
100 then needs to forbid the types it is not intended for.
101
102 Parameters
103 ----------
104 forbidden : list-of-str or None
105 List of forbidden non-string types, may be one or more of
106 `['bytes', 'mixed', 'mixed-integer']`.
107 name : str, default None
108 Name of the method to use in the error message. By default, this is
109 None, in which case the name from the method being wrapped will be
110 copied. However, for working with further wrappers (like _pat_wrapper
111 and _noarg_wrapper), it is necessary to specify the name.
112
113 Returns
114 -------
115 func : wrapper
116 The method to which the decorator is applied, with an added check that
117 enforces the inferred type to not be in the list of forbidden types.
118
119 Raises
120 ------
121 TypeError
122 If the inferred type of the underlying data is in `forbidden`.
123 """
124 # deal with None
125 forbidden = [] if forbidden is None else forbidden
126
127 allowed_types = {"string", "empty", "bytes", "mixed", "mixed-integer"} - set(
128 forbidden
129 )
130
131 def _forbid_nonstring_types(func: F) -> F:
132 func_name = func.__name__ if name is None else name
133
134 @wraps(func)
135 def wrapper(self, *args, **kwargs):
136 if self._inferred_dtype not in allowed_types:
137 msg = (
138 f"Cannot use .str.{func_name} with values of "
139 f"inferred dtype '{self._inferred_dtype}'."
140 )
141 raise TypeError(msg)
142 return func(self, *args, **kwargs)
143
144 wrapper.__name__ = func_name
145 return cast(F, wrapper)
146
147 return _forbid_nonstring_types
148
149
150class StringMethods(NoNewAttributesMixin):
151 """
152 Vectorized string functions for Series and Index.
153
154 NAs stay NA unless handled otherwise by a particular method.
155 Patterned after Python's string methods, with some inspiration from
156 R's stringr package.
157
158 Parameters
159 ----------
160 data : Series or Index
161 The content of the Series or Index.
162
163 See Also
164 --------
165 Series.str : Vectorized string functions for Series.
166 Index.str : Vectorized string functions for Index.
167
168 Examples
169 --------
170 >>> s = pd.Series(["A_Str_Series"])
171 >>> s
172 0 A_Str_Series
173 dtype: str
174
175 >>> s.str.split("_")
176 0 [A, Str, Series]
177 dtype: object
178
179 >>> s.str.replace("_", "")
180 0 AStrSeries
181 dtype: str
182 """
183
184 # Note: see the docstring in pandas.core.strings.__init__
185 # for an explanation of the implementation.
186 # TODO: Dispatch all the methods
187 # Currently the following are not dispatched to the array
188 # * cat
189 # * extractall
190
191 def __init__(self, data) -> None:
192 from pandas.core.arrays.string_ import StringDtype
193
194 self._inferred_dtype = self._validate(data)
195 self._is_categorical = isinstance(data.dtype, CategoricalDtype)
196 self._is_string = isinstance(data.dtype, StringDtype)
197 self._data = data
198
199 self._index = self._name = None
200 if isinstance(data, ABCSeries):
201 self._index = data.index
202 self._name = data.name
203
204 # ._values.categories works for both Series/Index
205 self._parent = data._values.categories if self._is_categorical else data
206 # save orig to blow up categoricals to the right type
207 self._orig = data
208 self._freeze()
209
210 @staticmethod
211 def _validate(data):
212 """
213 Auxiliary function for StringMethods, infers and checks dtype of data.
214
215 This is a "first line of defence" at the creation of the StringMethods-
216 object, and just checks that the dtype is in the
217 *union* of the allowed types over all string methods below; this
218 restriction is then refined on a per-method basis using the decorator
219 @forbid_nonstring_types (more info in the corresponding docstring).
220
221 This really should exclude all series/index with any non-string values,
222 but that isn't practical for performance reasons until we have a str
223 dtype (GH 9343 / 13877)
224
225 Parameters
226 ----------
227 data : The content of the Series
228
229 Returns
230 -------
231 dtype : inferred dtype of data
232 """
233 if isinstance(data, ABCMultiIndex):
234 raise AttributeError(
235 "Can only use .str accessor with Index, not MultiIndex"
236 )
237
238 # see _libs/lib.pyx for list of inferred types
239 allowed_types = ["string", "empty", "bytes", "mixed", "mixed-integer"]
240
241 data = extract_array(data)
242
243 values = getattr(data, "categories", data) # categorical / normal
244
245 inferred_dtype = lib.infer_dtype(values, skipna=True)
246
247 if inferred_dtype not in allowed_types:
248 raise AttributeError(
249 f"Can only use .str accessor with string values, not {inferred_dtype}"
250 )
251 return inferred_dtype
252
253 def __getitem__(self, key):
254 result = self._data.array._str_getitem(key)
255 return self._wrap_result(result)
256
257 def __iter__(self) -> Iterator:
258 raise TypeError(f"'{type(self).__name__}' object is not iterable")
259
260 def _wrap_result(
261 self,
262 result,
263 name=None,
264 expand: bool | None = None,
265 fill_value=np.nan,
266 returns_string: bool = True,
267 dtype=None,
268 ):
269 from pandas import (
270 Index,
271 MultiIndex,
272 )
273
274 if not hasattr(result, "ndim") or not hasattr(result, "dtype"):
275 if isinstance(result, ABCDataFrame):
276 result = result.__finalize__(self._orig, name="str")
277 return result
278 assert result.ndim < 3
279
280 # We can be wrapping a string / object / categorical result, in which
281 # case we'll want to return the same dtype as the input.
282 # Or we can be wrapping a numeric output, in which case we don't want
283 # to return a StringArray.
284 # Ideally the array method returns the right array type.
285 if expand is None:
286 # infer from ndim if expand is not specified
287 expand = result.ndim != 1
288 elif expand is True and not isinstance(self._orig, ABCIndex):
289 # required when expand=True is explicitly specified
290 # not needed when inferred
291 if isinstance(result.dtype, ArrowDtype):
292 import pyarrow as pa
293
294 from pandas.core.arrays.arrow.array import ArrowExtensionArray
295
296 value_lengths = pa.compute.list_value_length(result._pa_array)
297 max_len = pa.compute.max(value_lengths).as_py()
298 min_len = pa.compute.min(value_lengths).as_py()
299 if result._hasna:
300 # ArrowExtensionArray.fillna doesn't work for list scalars
301 result = ArrowExtensionArray(
302 result._pa_array.fill_null([None] * max_len)
303 )
304 if min_len < max_len:
305 # append nulls to each scalar list element up to max_len
306 result = ArrowExtensionArray(
307 pa.compute.list_slice(
308 result._pa_array,
309 start=0,
310 stop=max_len,
311 return_fixed_size_list=True,
312 )
313 )
314 if name is None:
315 name = range(max_len)
316 result = (
317 pa.compute.list_flatten(result._pa_array)
318 .to_numpy()
319 .reshape(len(result), max_len)
320 )
321 result = {
322 label: ArrowExtensionArray(pa.array(res))
323 for label, res in zip(name, result.T, strict=True)
324 }
325 elif is_object_dtype(result):
326
327 def cons_row(x):
328 if is_list_like(x):
329 return x
330 else:
331 return [x]
332
333 result = [cons_row(x) for x in result]
334 if result and not self._is_string:
335 # propagate nan values to match longest sequence (GH 18450)
336 max_len = max(len(x) for x in result)
337 result = [
338 x * max_len if len(x) == 0 or x[0] is np.nan else x
339 for x in result
340 ]
341
342 if not isinstance(expand, bool):
343 raise ValueError("expand must be True or False")
344
345 if expand is False:
346 # if expand is False, result should have the same name
347 # as the original otherwise specified
348 if name is None:
349 name = getattr(result, "name", None)
350 if name is None:
351 # do not use logical or, _orig may be a DataFrame
352 # which has "name" column
353 name = self._orig.name
354
355 # Wait until we are sure result is a Series or Index before
356 # checking attributes (GH 12180)
357 if isinstance(self._orig, ABCIndex):
358 # if result is a boolean np.array, return the np.array
359 # instead of wrapping it into a boolean Index (GH 8875)
360 if is_bool_dtype(result):
361 return result
362
363 if expand:
364 result = list(result)
365 out: Index = MultiIndex.from_tuples(result, names=name)
366 if out.nlevels == 1:
367 # We had all tuples of length-one, which are
368 # better represented as a regular Index.
369 out = out.get_level_values(0)
370 return out
371 else:
372 return Index(result, name=name, dtype=dtype, copy=False)
373 else:
374 index = self._orig.index
375 # This is a mess.
376 _dtype: DtypeObj | str | None = dtype
377 vdtype = getattr(result, "dtype", None)
378 if _dtype is not None:
379 pass
380 elif self._is_string:
381 if is_bool_dtype(vdtype):
382 _dtype = result.dtype
383 elif returns_string:
384 _dtype = self._orig.dtype
385 else:
386 _dtype = vdtype
387 elif vdtype is not None:
388 _dtype = vdtype
389
390 if expand:
391 cons = self._orig._constructor_expanddim
392 result = cons(result, columns=name, index=index, dtype=_dtype)
393 else:
394 # Must be a Series
395 cons = self._orig._constructor
396 result = cons(result, name=name, index=index, dtype=_dtype)
397 result = result.__finalize__(self._orig, method="str")
398 if name is not None and result.ndim == 1:
399 # __finalize__ might copy over the original name, but we may
400 # want the new name (e.g. str.extract).
401 result.name = name
402 return result
403
404 def _get_series_list(self, others):
405 """
406 Auxiliary function for :meth:`str.cat`. Turn potentially mixed input
407 into a list of Series (elements without an index must match the length
408 of the calling Series/Index).
409
410 Parameters
411 ----------
412 others : Series, DataFrame, np.ndarray, list-like or list-like of
413 Objects that are either Series, Index or np.ndarray (1-dim).
414
415 Returns
416 -------
417 list of Series
418 Others transformed into list of Series.
419 """
420 from pandas import (
421 DataFrame,
422 Series,
423 )
424
425 # self._orig is either Series or Index
426 idx = self._orig if isinstance(self._orig, ABCIndex) else self._orig.index
427
428 # Generally speaking, all objects without an index inherit the index
429 # `idx` of the calling Series/Index - i.e. must have matching length.
430 # Objects with an index (i.e. Series/Index/DataFrame) keep their own.
431 if isinstance(others, ABCSeries):
432 return [others]
433 elif isinstance(others, ABCIndex):
434 return [Series(others, index=idx, dtype=others.dtype)]
435 elif isinstance(others, ABCDataFrame):
436 return [others[x] for x in others]
437 elif isinstance(others, np.ndarray) and others.ndim == 2:
438 others = DataFrame(others, index=idx)
439 return [others[x] for x in others]
440 elif is_list_like(others, allow_sets=False):
441 try:
442 others = list(others) # ensure iterators do not get read twice etc
443 except TypeError:
444 # e.g. ser.str, raise below
445 pass
446 else:
447 # in case of list-like `others`, all elements must be
448 # either Series/Index/np.ndarray (1-dim)...
449 if all(
450 isinstance(x, (ABCSeries, ABCIndex, ExtensionArray))
451 or (isinstance(x, np.ndarray) and x.ndim == 1)
452 for x in others
453 ):
454 los: list[Series] = []
455 while others: # iterate through list and append each element
456 los = los + self._get_series_list(others.pop(0))
457 return los
458 # ... or just strings
459 elif all(not is_list_like(x) for x in others):
460 return [Series(others, index=idx)]
461 raise TypeError(
462 "others must be Series, Index, DataFrame, np.ndarray "
463 "or list-like (either containing only strings or "
464 "containing only objects of type Series/Index/"
465 "np.ndarray[1-dim])"
466 )
467
468 @forbid_nonstring_types(["bytes", "mixed", "mixed-integer"])
469 def cat(
470 self,
471 others=None,
472 sep: str | None = None,
473 na_rep=None,
474 join: AlignJoin = "left",
475 ) -> str | Series | Index:
476 """
477 Concatenate strings in the Series/Index with given separator.
478
479 If `others` is specified, this function concatenates the Series/Index
480 and elements of `others` element-wise.
481 If `others` is not passed, then all values in the Series/Index are
482 concatenated into a single string with a given `sep`.
483
484 Parameters
485 ----------
486 others : Series, Index, DataFrame, np.ndarray or list-like
487 Series, Index, DataFrame, np.ndarray (one- or two-dimensional) and
488 other list-likes of strings must have the same length as the
489 calling Series/Index, with the exception of indexed objects (i.e.
490 Series/Index/DataFrame) if `join` is not None.
491
492 If others is a list-like that contains a combination of Series,
493 Index or np.ndarray (1-dim), then all elements will be unpacked and
494 must satisfy the above criteria individually.
495
496 If others is None, the method returns the concatenation of all
497 strings in the calling Series/Index.
498 sep : str, default ''
499 The separator between the different elements/columns. By default
500 the empty string `''` is used.
501 na_rep : str or None, default None
502 Representation that is inserted for all missing values:
503
504 - If `na_rep` is None, and `others` is None, missing values in the
505 Series/Index are omitted from the result.
506 - If `na_rep` is None, and `others` is not None, a row containing a
507 missing value in any of the columns (before concatenation) will
508 have a missing value in the result.
509 join : {'left', 'right', 'outer', 'inner'}, default 'left'
510 Determines the join-style between the calling Series/Index and any
511 Series/Index/DataFrame in `others` (objects without an index need
512 to match the length of the calling Series/Index). To disable
513 alignment, use `.values` on any Series/Index/DataFrame in `others`.
514
515 Returns
516 -------
517 str, Series or Index
518 If `others` is None, `str` is returned, otherwise a `Series/Index`
519 (same type as caller) of objects is returned.
520
521 See Also
522 --------
523 split : Split each string in the Series/Index.
524 join : Join lists contained as elements in the Series/Index.
525
526 Examples
527 --------
528 When not passing `others`, all values are concatenated into a single
529 string:
530
531 >>> s = pd.Series(["a", "b", np.nan, "d"])
532 >>> s.str.cat(sep=" ")
533 'a b d'
534
535 By default, NA values in the Series are ignored. Using `na_rep`, they
536 can be given a representation:
537
538 >>> s.str.cat(sep=" ", na_rep="?")
539 'a b ? d'
540
541 If `others` is specified, corresponding values are concatenated with
542 the separator. Result will be a Series of strings.
543
544 >>> s.str.cat(["A", "B", "C", "D"], sep=",")
545 0 a,A
546 1 b,B
547 2 NaN
548 3 d,D
549 dtype: str
550
551 Missing values will remain missing in the result, but can again be
552 represented using `na_rep`
553
554 >>> s.str.cat(["A", "B", "C", "D"], sep=",", na_rep="-")
555 0 a,A
556 1 b,B
557 2 -,C
558 3 d,D
559 dtype: str
560
561 If `sep` is not specified, the values are concatenated without
562 separation.
563
564 >>> s.str.cat(["A", "B", "C", "D"], na_rep="-")
565 0 aA
566 1 bB
567 2 -C
568 3 dD
569 dtype: str
570
571 Series with different indexes can be aligned before concatenation. The
572 `join`-keyword works as in other methods.
573
574 >>> t = pd.Series(["d", "a", "e", "c"], index=[3, 0, 4, 2])
575 >>> s.str.cat(t, join="left", na_rep="-")
576 0 aa
577 1 b-
578 2 -c
579 3 dd
580 dtype: str
581 >>>
582 >>> s.str.cat(t, join="outer", na_rep="-")
583 0 aa
584 1 b-
585 2 -c
586 3 dd
587 4 -e
588 dtype: str
589 >>>
590 >>> s.str.cat(t, join="inner", na_rep="-")
591 0 aa
592 2 -c
593 3 dd
594 dtype: str
595 >>>
596 >>> s.str.cat(t, join="right", na_rep="-")
597 3 dd
598 0 aa
599 4 -e
600 2 -c
601 dtype: str
602
603 For more examples, see :ref:`here <text.concatenate>`.
604 """
605 # TODO: dispatch
606 from pandas import (
607 Index,
608 Series,
609 concat,
610 )
611
612 if isinstance(others, str):
613 raise ValueError("Did you mean to supply a `sep` keyword?")
614 if sep is None:
615 sep = ""
616
617 if isinstance(self._orig, ABCIndex):
618 data = Series(self._orig, index=self._orig, dtype=self._orig.dtype)
619 else: # Series
620 data = self._orig
621
622 # concatenate Series/Index with itself if no "others"
623 if others is None:
624 # error: Incompatible types in assignment (expression has type
625 # "ndarray", variable has type "Series")
626 data = ensure_object(data) # type: ignore[assignment]
627 na_mask = isna(data)
628 if na_rep is None and na_mask.any():
629 return sep.join(data[~na_mask])
630 elif na_rep is not None and na_mask.any():
631 return sep.join(np.where(na_mask, na_rep, data))
632 else:
633 return sep.join(data)
634
635 try:
636 # turn anything in "others" into lists of Series
637 others = self._get_series_list(others)
638 except ValueError as err: # do not catch TypeError raised by _get_series_list
639 raise ValueError(
640 "If `others` contains arrays or lists (or other "
641 "list-likes without an index), these must all be "
642 "of the same length as the calling Series/Index."
643 ) from err
644
645 # align if required
646 if any(not data.index.equals(x.index) for x in others):
647 # Need to add keys for uniqueness in case of duplicate columns
648 others = concat(
649 others,
650 axis=1,
651 join=(join if join == "inner" else "outer"),
652 keys=range(len(others)),
653 sort=False,
654 )
655 data, others = data.align(others, join=join)
656 others = [others[x] for x in others] # again list of Series
657
658 all_cols = [ensure_object(x) for x in [data, *others]]
659 na_masks = np.array([isna(x) for x in all_cols])
660 union_mask = np.logical_or.reduce(na_masks, axis=0)
661
662 if na_rep is None and union_mask.any():
663 # no na_rep means NaNs for all rows where any column has a NaN
664 # only necessary if there are actually any NaNs
665 result = np.empty(len(data), dtype=object)
666 np.putmask(result, union_mask, np.nan)
667
668 not_masked = ~union_mask
669 result[not_masked] = cat_safe([x[not_masked] for x in all_cols], sep)
670 elif na_rep is not None and union_mask.any():
671 # fill NaNs with na_rep in case there are actually any NaNs
672 all_cols = [
673 np.where(nm, na_rep, col)
674 for nm, col in zip(na_masks, all_cols, strict=True)
675 ]
676 result = cat_safe(all_cols, sep)
677 else:
678 # no NaNs - can just concatenate
679 result = cat_safe(all_cols, sep)
680
681 out: Index | Series
682 if isinstance(self._orig.dtype, CategoricalDtype):
683 # We need to infer the new categories.
684 dtype = self._orig.dtype.categories.dtype
685 else:
686 dtype = self._orig.dtype
687 if isinstance(self._orig, ABCIndex):
688 # add dtype for case that result is all-NA
689 if isna(result).all():
690 dtype = object # type: ignore[assignment]
691
692 out = Index(result, dtype=dtype, name=self._orig.name, copy=False)
693 else: # Series
694 res_ser = Series(
695 result, dtype=dtype, index=data.index, name=self._orig.name, copy=False
696 )
697 out = res_ser.__finalize__(self._orig, method="str_cat")
698 return out
699
700 @forbid_nonstring_types(["bytes"])
701 def split(
702 self,
703 pat: str | re.Pattern | None = None,
704 *,
705 n=-1,
706 expand: bool = False,
707 regex: bool | None = None,
708 ):
709 r"""
710 Split strings around given separator/delimiter.
711
712 Splits the string in the Series/Index from the beginning,
713 at the specified delimiter string.
714
715 Parameters
716 ----------
717 pat : str or compiled regex, optional
718 String or regular expression to split on.
719 If not specified, split on whitespace.
720 n : int, default -1 (all)
721 Limit number of splits in output.
722 ``None``, 0 and -1 will be interpreted as return all splits.
723 expand : bool, default False
724 Expand the split strings into separate columns.
725
726 - If ``True``, return DataFrame/MultiIndex expanding dimensionality.
727 - If ``False``, return Series/Index, containing lists of strings.
728
729 regex : bool, default None
730 Determines if the passed-in pattern is a regular expression:
731
732 - If ``True``, assumes the passed-in pattern is a regular expression
733 - If ``False``, treats the pattern as a literal string.
734 - If ``None`` and `pat` length is 1, treats `pat` as a literal string.
735 - If ``None`` and `pat` length is not 1, treats `pat` as a regular
736 expression.
737 - Cannot be set to False if `pat` is a compiled regex
738
739 Returns
740 -------
741 Series, Index, DataFrame or MultiIndex
742 Type matches caller unless ``expand=True`` (see Notes).
743
744 Raises
745 ------
746 ValueError
747 * if `regex` is False and `pat` is a compiled regex
748
749 See Also
750 --------
751 Series.str.split : Split strings around given separator/delimiter.
752 Series.str.rsplit : Splits string around given separator/delimiter,
753 starting from the right.
754 Series.str.join : Join lists contained as elements in the Series/Index
755 with passed delimiter.
756 str.split : Standard library version for split.
757 str.rsplit : Standard library version for rsplit.
758
759 Notes
760 -----
761 The handling of the `n` keyword depends on the number of found splits:
762
763 - If found splits > `n`, make first `n` splits only
764 - If found splits <= `n`, make all splits
765 - If for a certain row the number of found splits < `n`,
766 append `None` for padding up to `n` if ``expand=True``
767
768 If using ``expand=True``, Series and Index callers return DataFrame and
769 MultiIndex objects, respectively.
770
771 Use of `regex =False` with a `pat` as a compiled regex will raise an error.
772
773 Examples
774 --------
775 >>> s = pd.Series(
776 ... [
777 ... "this is a regular sentence",
778 ... "https://docs.python.org/3/tutorial/index.html",
779 ... np.nan,
780 ... ]
781 ... )
782 >>> s
783 0 this is a regular sentence
784 1 https://docs.python.org/3/tutorial/index.html
785 2 NaN
786 dtype: str
787
788 In the default setting, the string is split by whitespace.
789
790 >>> s.str.split()
791 0 [this, is, a, regular, sentence]
792 1 [https://docs.python.org/3/tutorial/index.html]
793 2 NaN
794 dtype: object
795
796 Without the `n` parameter, the outputs of `rsplit` and `split`
797 are identical.
798
799 >>> s.str.rsplit()
800 0 [this, is, a, regular, sentence]
801 1 [https://docs.python.org/3/tutorial/index.html]
802 2 NaN
803 dtype: object
804
805 The `n` parameter can be used to limit the number of splits on the
806 delimiter. The outputs of `split` and `rsplit` are different.
807
808 >>> s.str.split(n=2)
809 0 [this, is, a regular sentence]
810 1 [https://docs.python.org/3/tutorial/index.html]
811 2 NaN
812 dtype: object
813
814 >>> s.str.rsplit(n=2)
815 0 [this is a, regular, sentence]
816 1 [https://docs.python.org/3/tutorial/index.html]
817 2 NaN
818 dtype: object
819
820 The `pat` parameter can be used to split by other characters.
821
822 >>> s.str.split(pat="/")
823 0 [this is a regular sentence]
824 1 [https:, , docs.python.org, 3, tutorial, index...
825 2 NaN
826 dtype: object
827
828 When using ``expand=True``, the split elements will expand out into
829 separate columns. If NaN is present, it is propagated throughout
830 the columns during the split.
831
832 >>> s.str.split(expand=True)
833 0 1 2 3 4
834 0 this is a regular sentence
835 1 https://docs.python.org/3/tutorial/index.html NaN NaN NaN NaN
836 2 NaN NaN NaN NaN NaN
837
838 For slightly more complex use cases like splitting the html document name
839 from a url, a combination of parameter settings can be used.
840
841 >>> s.str.rsplit("/", n=1, expand=True)
842 0 1
843 0 this is a regular sentence NaN
844 1 https://docs.python.org/3/tutorial index.html
845 2 NaN NaN
846
847 Remember to escape special characters when explicitly using regular expressions.
848
849 >>> s = pd.Series(["foo and bar plus baz"])
850 >>> s.str.split(r"and|plus", expand=True)
851 0 1 2
852 0 foo bar baz
853
854 Regular expressions can be used to handle urls or file names.
855 When `pat` is a string and ``regex=None`` (the default), the given `pat` is
856 compiled as a regex only if ``len(pat) != 1``.
857
858 >>> s = pd.Series(["foojpgbar.jpg"])
859 >>> s.str.split(r".", expand=True)
860 0 1
861 0 foojpgbar jpg
862
863 >>> s.str.split(r"\.jpg", expand=True)
864 0 1
865 0 foojpgbar
866
867 When ``regex=True``, `pat` is interpreted as a regex
868
869 >>> s.str.split(r"\.jpg", regex=True, expand=True)
870 0 1
871 0 foojpgbar
872
873 A compiled regex can be passed as `pat`
874
875 >>> import re
876 >>> s.str.split(re.compile(r"\.jpg"), expand=True)
877 0 1
878 0 foojpgbar
879
880 When ``regex=False``, `pat` is interpreted as the string itself
881
882 >>> s.str.split(r"\.jpg", regex=False, expand=True)
883 0
884 0 foojpgbar.jpg
885 """
886 if regex is False and is_re(pat):
887 raise ValueError(
888 "Cannot use a compiled regex as replacement pattern with regex=False"
889 )
890 if is_re(pat):
891 regex = True
892 result = self._data.array._str_split(pat, n, expand, regex)
893 if self._data.dtype == "category":
894 dtype = self._data.dtype.categories.dtype if expand else object
895 else:
896 dtype = object if self._data.dtype == object else None
897 return self._wrap_result(
898 result, expand=expand, returns_string=expand, dtype=dtype
899 )
900
901 @forbid_nonstring_types(["bytes"])
902 def rsplit(self, pat=None, *, n=-1, expand: bool = False):
903 """
904 Split strings around given separator/delimiter.
905
906 Splits the string in the Series/Index from the end,
907 at the specified delimiter string.
908
909 Parameters
910 ----------
911 pat : str, optional
912 String to split on.
913 If not specified, split on whitespace.
914 n : int, default -1 (all)
915 Limit number of splits in output.
916 ``None``, 0 and -1 will be interpreted as return all splits.
917 expand : bool, default False
918 Expand the split strings into separate columns.
919
920 - If ``True``, return DataFrame/MultiIndex expanding dimensionality.
921 - If ``False``, return Series/Index, containing lists of strings.
922
923 Returns
924 -------
925 Series, Index, DataFrame or MultiIndex
926 Type matches caller unless ``expand=True`` (see Notes).
927
928 See Also
929 --------
930 Series.str.split : Split strings around given separator/delimiter.
931 Series.str.rsplit : Splits string around given separator/delimiter,
932 starting from the right.
933 Series.str.join : Join lists contained as elements in the Series/Index
934 with passed delimiter.
935 str.split : Standard library version for split.
936 str.rsplit : Standard library version for rsplit.
937
938 Notes
939 -----
940 The handling of the `n` keyword depends on the number of found splits:
941
942 - If found splits > `n`, make first `n` splits only
943 - If found splits <= `n`, make all splits
944 - If for a certain row the number of found splits < `n`,
945 append `None` for padding up to `n` if ``expand=True``
946
947 If using ``expand=True``, Series and Index callers return DataFrame and
948 MultiIndex objects, respectively.
949
950 Examples
951 --------
952 >>> s = pd.Series(
953 ... [
954 ... "this is a regular sentence",
955 ... "https://docs.python.org/3/tutorial/index.html",
956 ... np.nan,
957 ... ]
958 ... )
959 >>> s
960 0 this is a regular sentence
961 1 https://docs.python.org/3/tutorial/index.html
962 2 NaN
963 dtype: str
964
965 In the default setting, the string is split by whitespace.
966
967 >>> s.str.split()
968 0 [this, is, a, regular, sentence]
969 1 [https://docs.python.org/3/tutorial/index.html]
970 2 NaN
971 dtype: object
972
973 Without the `n` parameter, the outputs of `rsplit` and `split`
974 are identical.
975
976 >>> s.str.rsplit()
977 0 [this, is, a, regular, sentence]
978 1 [https://docs.python.org/3/tutorial/index.html]
979 2 NaN
980 dtype: object
981
982 The `n` parameter can be used to limit the number of splits on the
983 delimiter. The outputs of `split` and `rsplit` are different.
984
985 >>> s.str.split(n=2)
986 0 [this, is, a regular sentence]
987 1 [https://docs.python.org/3/tutorial/index.html]
988 2 NaN
989 dtype: object
990
991 >>> s.str.rsplit(n=2)
992 0 [this is a, regular, sentence]
993 1 [https://docs.python.org/3/tutorial/index.html]
994 2 NaN
995 dtype: object
996
997 The `pat` parameter can be used to split by other characters.
998
999 >>> s.str.split(pat="/")
1000 0 [this is a regular sentence]
1001 1 [https:, , docs.python.org, 3, tutorial, index...
1002 2 NaN
1003 dtype: object
1004
1005 When using ``expand=True``, the split elements will expand out into
1006 separate columns. If NaN is present, it is propagated throughout
1007 the columns during the split.
1008
1009 >>> s.str.split(expand=True)
1010 0 1 2 3 4
1011 0 this is a regular sentence
1012 1 https://docs.python.org/3/tutorial/index.html NaN NaN NaN NaN
1013 2 NaN NaN NaN NaN NaN
1014
1015 For slightly more complex use cases like splitting the html document name
1016 from a url, a combination of parameter settings can be used.
1017
1018 >>> s.str.rsplit("/", n=1, expand=True)
1019 0 1
1020 0 this is a regular sentence NaN
1021 1 https://docs.python.org/3/tutorial index.html
1022 2 NaN NaN
1023 """
1024 result = self._data.array._str_rsplit(pat, n=n)
1025 if self._data.dtype == "category":
1026 dtype = self._data.dtype.categories.dtype if expand else object
1027 else:
1028 dtype = object if self._data.dtype == object else None
1029 return self._wrap_result(
1030 result, expand=expand, returns_string=expand, dtype=dtype
1031 )
1032
1033 @forbid_nonstring_types(["bytes"])
1034 def partition(self, sep: str = " ", expand: bool = True):
1035 """
1036 Split the string at the first occurrence of `sep`.
1037
1038 This method splits the string at the first occurrence of `sep`,
1039 and returns 3 elements containing the part before the separator,
1040 the separator itself, and the part after the separator.
1041 If the separator is not found, return 3 elements containing the string itself,
1042 followed by two empty strings.
1043
1044 Parameters
1045 ----------
1046 sep : str, default whitespace
1047 String to split on.
1048 expand : bool, default True
1049 If True, return DataFrame/MultiIndex expanding dimensionality.
1050 If False, return Series/Index.
1051
1052 Returns
1053 -------
1054 DataFrame/MultiIndex or Series/Index of objects
1055 Returns appropriate type based on `expand` parameter with strings
1056 split based on the `sep` parameter.
1057
1058 See Also
1059 --------
1060 rpartition : Split the string at the last occurrence of `sep`.
1061 Series.str.split : Split strings around given separators.
1062 str.partition : Standard library version.
1063
1064 Examples
1065 --------
1066 >>> s = pd.Series(["Linda van der Berg", "George Pitt-Rivers"])
1067 >>> s
1068 0 Linda van der Berg
1069 1 George Pitt-Rivers
1070 dtype: str
1071
1072 >>> s.str.partition()
1073 0 1 2
1074 0 Linda van der Berg
1075 1 George Pitt-Rivers
1076
1077 To partition by the last space instead of the first one:
1078
1079 >>> s.str.rpartition()
1080 0 1 2
1081 0 Linda van der Berg
1082 1 George Pitt-Rivers
1083
1084 To partition by something different than a space:
1085
1086 >>> s.str.partition("-")
1087 0 1 2
1088 0 Linda van der Berg
1089 1 George Pitt - Rivers
1090
1091 To return a Series containing tuples instead of a DataFrame:
1092
1093 >>> s.str.partition("-", expand=False)
1094 0 (Linda van der Berg, , )
1095 1 (George Pitt, -, Rivers)
1096 dtype: object
1097
1098 Also available on indices:
1099
1100 >>> idx = pd.Index(["X 123", "Y 999"])
1101 >>> idx
1102 Index(['X 123', 'Y 999'], dtype='str')
1103
1104 Which will create a MultiIndex:
1105
1106 >>> idx.str.partition()
1107 MultiIndex([('X', ' ', '123'),
1108 ('Y', ' ', '999')],
1109 )
1110
1111 Or an index with tuples with ``expand=False``:
1112
1113 >>> idx.str.partition(expand=False)
1114 Index([('X', ' ', '123'), ('Y', ' ', '999')], dtype='object')
1115 """
1116 result = self._data.array._str_partition(sep, expand)
1117 if self._data.dtype == "category":
1118 dtype = self._data.dtype.categories.dtype if expand else object
1119 else:
1120 dtype = object if self._data.dtype == object else None
1121 return self._wrap_result(
1122 result, expand=expand, returns_string=expand, dtype=dtype
1123 )
1124
1125 @forbid_nonstring_types(["bytes"])
1126 def rpartition(self, sep: str = " ", expand: bool = True):
1127 """
1128 Split the string at the last occurrence of `sep`.
1129
1130 This method splits the string at the last occurrence of `sep`,
1131 and returns 3 elements containing the part before the separator,
1132 the separator itself, and the part after the separator.
1133 If the separator is not found, return 3 elements containing two empty strings,
1134 followed by the string itself.
1135
1136 Parameters
1137 ----------
1138 sep : str, default " "
1139 String to split on.
1140 expand : bool, default True
1141 If True, return DataFrame/MultiIndex expanding dimensionality.
1142 If False, return Series/Index.
1143
1144 Returns
1145 -------
1146 DataFrame/MultiIndex or Series/Index of objects
1147 Returns appropriate type based on `expand` parameter with strings
1148 split based on the `sep` parameter.
1149
1150 See Also
1151 --------
1152 partition : Split the string at the first occurrence of `sep`.
1153 Series.str.split : Split strings around given separators.
1154 str.partition : Standard library version.
1155
1156 Examples
1157 --------
1158 >>> s = pd.Series(["Linda van der Berg", "George Pitt-Rivers"])
1159 >>> s
1160 0 Linda van der Berg
1161 1 George Pitt-Rivers
1162 dtype: str
1163
1164 >>> s.str.partition()
1165 0 1 2
1166 0 Linda van der Berg
1167 1 George Pitt-Rivers
1168
1169 To partition by the last space instead of the first one:
1170
1171 >>> s.str.rpartition()
1172 0 1 2
1173 0 Linda van der Berg
1174 1 George Pitt-Rivers
1175
1176 To partition by something different than a space:
1177
1178 >>> s.str.partition("-")
1179 0 1 2
1180 0 Linda van der Berg
1181 1 George Pitt - Rivers
1182
1183 To return a Series containing tuples instead of a DataFrame:
1184
1185 >>> s.str.partition("-", expand=False)
1186 0 (Linda van der Berg, , )
1187 1 (George Pitt, -, Rivers)
1188 dtype: object
1189
1190 Also available on indices:
1191
1192 >>> idx = pd.Index(["X 123", "Y 999"])
1193 >>> idx
1194 Index(['X 123', 'Y 999'], dtype='str')
1195
1196 Which will create a MultiIndex:
1197
1198 >>> idx.str.partition()
1199 MultiIndex([('X', ' ', '123'),
1200 ('Y', ' ', '999')],
1201 )
1202
1203 Or an index with tuples with ``expand=False``:
1204
1205 >>> idx.str.partition(expand=False)
1206 Index([('X', ' ', '123'), ('Y', ' ', '999')], dtype='object')
1207 """
1208 result = self._data.array._str_rpartition(sep, expand)
1209 if self._data.dtype == "category":
1210 dtype = self._data.dtype.categories.dtype if expand else object
1211 else:
1212 dtype = object if self._data.dtype == object else None
1213 return self._wrap_result(
1214 result, expand=expand, returns_string=expand, dtype=dtype
1215 )
1216
1217 def get(self, i):
1218 """
1219 Extract element from each component at specified position or with specified key.
1220
1221 Extract element from lists, tuples, dict, or strings in each element in the
1222 Series/Index.
1223
1224 Parameters
1225 ----------
1226 i : int or hashable dict label
1227 Position or key of element to extract.
1228
1229 Returns
1230 -------
1231 Series or Index
1232 Series or Index where each value is the extracted element from
1233 the corresponding input component.
1234
1235 See Also
1236 --------
1237 Series.str.extract : Extract capture groups in the regex as columns
1238 in a DataFrame.
1239
1240 Examples
1241 --------
1242 >>> s = pd.Series(
1243 ... [
1244 ... "String",
1245 ... (1, 2, 3),
1246 ... ["a", "b", "c"],
1247 ... 123,
1248 ... -456,
1249 ... {1: "Hello", "2": "World"},
1250 ... ]
1251 ... )
1252 >>> s
1253 0 String
1254 1 (1, 2, 3)
1255 2 [a, b, c]
1256 3 123
1257 4 -456
1258 5 {1: 'Hello', '2': 'World'}
1259 dtype: object
1260
1261 >>> s.str.get(1)
1262 0 t
1263 1 2
1264 2 b
1265 3 NaN
1266 4 NaN
1267 5 Hello
1268 dtype: object
1269
1270 >>> s.str.get(-1)
1271 0 g
1272 1 3
1273 2 c
1274 3 NaN
1275 4 NaN
1276 5 None
1277 dtype: object
1278
1279 Return element with given key
1280
1281 >>> s = pd.Series(
1282 ... [
1283 ... {"name": "Hello", "value": "World"},
1284 ... {"name": "Goodbye", "value": "Planet"},
1285 ... ]
1286 ... )
1287 >>> s.str.get("name")
1288 0 Hello
1289 1 Goodbye
1290 dtype: object
1291 """
1292 result = self._data.array._str_get(i)
1293 return self._wrap_result(result)
1294
1295 @forbid_nonstring_types(["bytes"])
1296 def join(self, sep: str):
1297 """
1298 Join lists contained as elements in the Series/Index with passed delimiter.
1299
1300 If the elements of a Series are lists themselves, join the content of these
1301 lists using the delimiter passed to the function.
1302 This function is an equivalent to :meth:`str.join`.
1303
1304 Parameters
1305 ----------
1306 sep : str
1307 Delimiter to use between list entries.
1308
1309 Returns
1310 -------
1311 Series/Index: object
1312 The list entries concatenated by intervening occurrences of the
1313 delimiter.
1314
1315 Raises
1316 ------
1317 AttributeError
1318 If the supplied Series contains neither strings nor lists.
1319
1320 See Also
1321 --------
1322 str.join : Standard library version of this method.
1323 Series.str.split : Split strings around given separator/delimiter.
1324
1325 Notes
1326 -----
1327 If any of the list items is not a string object, the result of the join
1328 will be `NaN`.
1329
1330 Examples
1331 --------
1332 Example with a list that contains non-string elements.
1333
1334 >>> s = pd.Series(
1335 ... [
1336 ... ["lion", "elephant", "zebra"],
1337 ... [1.1, 2.2, 3.3],
1338 ... ["cat", np.nan, "dog"],
1339 ... ["cow", 4.5, "goat"],
1340 ... ["duck", ["swan", "fish"], "guppy"],
1341 ... ]
1342 ... )
1343 >>> s
1344 0 [lion, elephant, zebra]
1345 1 [1.1, 2.2, 3.3]
1346 2 [cat, nan, dog]
1347 3 [cow, 4.5, goat]
1348 4 [duck, [swan, fish], guppy]
1349 dtype: object
1350
1351 Join all lists using a '-'. The lists containing object(s) of types other
1352 than str will produce a NaN.
1353
1354 >>> s.str.join("-")
1355 0 lion-elephant-zebra
1356 1 NaN
1357 2 NaN
1358 3 NaN
1359 4 NaN
1360 dtype: object
1361 """
1362 result = self._data.array._str_join(sep)
1363 return self._wrap_result(result)
1364
1365 @forbid_nonstring_types(["bytes"])
1366 def contains(
1367 self,
1368 pat,
1369 case: bool = True,
1370 flags: int = 0,
1371 na=lib.no_default,
1372 regex: bool = True,
1373 ):
1374 r"""
1375 Test if pattern or regex is contained within a string of a Series or Index.
1376
1377 Return boolean Series or Index based on whether a given pattern or regex is
1378 contained within a string of a Series or Index.
1379
1380 Parameters
1381 ----------
1382 pat : str
1383 Character sequence or regular expression.
1384 case : bool, default True
1385 If True, case sensitive.
1386 flags : int, default 0 (no flags)
1387 Flags to pass through to the re module, e.g. re.IGNORECASE.
1388 na : scalar, optional
1389 Fill value for missing values. The default depends on dtype of the
1390 array. For the ``"str"`` dtype, ``False`` is used. For object
1391 dtype, ``numpy.nan`` is used. For the nullable ``StringDtype``,
1392 ``pandas.NA`` is used.
1393 regex : bool, default True
1394 If True, assumes the pat is a regular expression.
1395
1396 If False, treats the pat as a literal string.
1397
1398 Returns
1399 -------
1400 Series or Index of boolean values
1401 A Series or Index of boolean values indicating whether the
1402 given pattern is contained within the string of each element
1403 of the Series or Index.
1404
1405 See Also
1406 --------
1407 match : Analogous, but stricter, relying on re.match instead of re.search.
1408 Series.str.startswith : Test if the start of each string element matches a
1409 pattern.
1410 Series.str.endswith : Same as startswith, but tests the end of string.
1411
1412 Examples
1413 --------
1414 Returning a Series of booleans using only a literal pattern.
1415
1416 >>> s1 = pd.Series(["Mouse", "dog", "house and parrot", "23", np.nan])
1417 >>> s1.str.contains("og", regex=False)
1418 0 False
1419 1 True
1420 2 False
1421 3 False
1422 4 False
1423 dtype: bool
1424
1425 Returning an Index of booleans using only a literal pattern.
1426
1427 >>> ind = pd.Index(["Mouse", "dog", "house and parrot", "23.0", np.nan])
1428 >>> ind.str.contains("23", regex=False)
1429 array([False, False, False, True, False])
1430
1431 Specifying case sensitivity using `case`.
1432
1433 >>> s1.str.contains("oG", case=True, regex=True)
1434 0 False
1435 1 False
1436 2 False
1437 3 False
1438 4 False
1439 dtype: bool
1440
1441 Returning 'house' or 'dog' when either expression occurs in a string.
1442
1443 >>> s1.str.contains("house|dog", regex=True)
1444 0 False
1445 1 True
1446 2 True
1447 3 False
1448 4 False
1449 dtype: bool
1450
1451 Ignoring case sensitivity using `flags` with regex.
1452
1453 >>> import re
1454 >>> s1.str.contains("PARROT", flags=re.IGNORECASE, regex=True)
1455 0 False
1456 1 False
1457 2 True
1458 3 False
1459 4 False
1460 dtype: bool
1461
1462 Returning any digit using regular expression.
1463
1464 >>> s1.str.contains("\\d", regex=True)
1465 0 False
1466 1 False
1467 2 False
1468 3 True
1469 4 False
1470 dtype: bool
1471
1472 Ensure `pat` is a not a literal pattern when `regex` is set to True.
1473 Note in the following example one might expect only `s2[1]` and `s2[3]` to
1474 return `True`. However, '.0' as a regex matches any character
1475 followed by a 0.
1476
1477 >>> s2 = pd.Series(["40", "40.0", "41", "41.0", "35"])
1478 >>> s2.str.contains(".0", regex=True)
1479 0 True
1480 1 True
1481 2 False
1482 3 True
1483 4 False
1484 dtype: bool
1485 """
1486 if regex:
1487 try:
1488 has_groups = re.compile(pat).groups
1489 except re.error:
1490 has_groups = False
1491 if has_groups:
1492 warnings.warn(
1493 "This pattern is interpreted as a regular expression, and has "
1494 "match groups. To actually get the groups, use str.extract.",
1495 UserWarning,
1496 stacklevel=find_stack_level(),
1497 )
1498
1499 result = self._data.array._str_contains(pat, case, flags, na, regex)
1500 return self._wrap_result(result, fill_value=na, returns_string=False)
1501
1502 @forbid_nonstring_types(["bytes"])
1503 def match(
1504 self,
1505 pat: str | re.Pattern,
1506 case: bool | lib.NoDefault = lib.no_default,
1507 flags: int | lib.NoDefault = lib.no_default,
1508 na=lib.no_default,
1509 ):
1510 """
1511 Determine if each string starts with a match of a regular expression.
1512
1513 Determines whether each string in the Series or Index starts with a
1514 match to a specified regular expression. This function is especially
1515 useful for validating prefixes, such as ensuring that codes, tags, or
1516 identifiers begin with a specific pattern.
1517
1518 Parameters
1519 ----------
1520 pat : str or compiled regex
1521 Character sequence or regular expression.
1522 case : bool, default True
1523 If True, case sensitive.
1524 flags : int, default 0 (no flags)
1525 Regex module flags, e.g. re.IGNORECASE.
1526 na : scalar, optional
1527 Fill value for missing values. The default depends on dtype of the
1528 array. For the ``"str"`` dtype, ``False`` is used. For object
1529 dtype, ``numpy.nan`` is used. For the nullable ``StringDtype``,
1530 ``pandas.NA`` is used.
1531
1532 Returns
1533 -------
1534 Series/Index/array of boolean values
1535 A Series, Index, or array of boolean values indicating whether the start
1536 of each string matches the pattern. The result will be of the same type
1537 as the input.
1538
1539 See Also
1540 --------
1541 fullmatch : Stricter matching that requires the entire string to match.
1542 contains : Analogous, but less strict, relying on re.search instead of
1543 re.match.
1544 extract : Extract matched groups.
1545
1546 Examples
1547 --------
1548 >>> ser = pd.Series(["horse", "eagle", "donkey"])
1549 >>> ser.str.match("e")
1550 0 False
1551 1 True
1552 2 False
1553 dtype: bool
1554 """
1555 if flags is not lib.no_default:
1556 # pat.flags will have re.U regardless, so we need to add it here
1557 # before checking for a match
1558 flags = flags | re.U
1559 if is_re(pat):
1560 if pat.flags != flags:
1561 raise ValueError(
1562 "Cannot both specify 'flags' and pass a compiled regexp "
1563 "object with conflicting flags"
1564 )
1565 else:
1566 pat = re.compile(pat, flags=flags)
1567 # set flags=0 to ensure that when we call
1568 # re.compile(pat, flags=flags) the constructor does not raise.
1569 flags = 0
1570 else:
1571 flags = 0
1572
1573 if case is lib.no_default:
1574 if is_re(pat):
1575 case = not bool(pat.flags & re.IGNORECASE)
1576 else:
1577 # Case-sensitive default
1578 case = True
1579 elif is_re(pat):
1580 implicit_case = not bool(pat.flags & re.IGNORECASE)
1581 if implicit_case != case:
1582 # GH#62240
1583 raise ValueError(
1584 "Cannot both specify 'case' and pass a compiled regexp "
1585 "object with conflicting case-sensitivity"
1586 )
1587
1588 result = self._data.array._str_match(pat, case=case, flags=flags, na=na)
1589 return self._wrap_result(result, fill_value=na, returns_string=False)
1590
1591 @forbid_nonstring_types(["bytes"])
1592 def fullmatch(self, pat, case: bool = True, flags: int = 0, na=lib.no_default):
1593 """
1594 Determine if each string entirely matches a regular expression.
1595
1596 Checks if each string in the Series or Index fully matches the
1597 specified regular expression pattern. This function is useful when the
1598 requirement is for an entire string to conform to a pattern, such as
1599 validating formats like phone numbers or email addresses.
1600
1601 Parameters
1602 ----------
1603 pat : str
1604 Character sequence or regular expression.
1605 case : bool, default True
1606 If True, case sensitive.
1607 flags : int, default 0 (no flags)
1608 Regex module flags, e.g. re.IGNORECASE.
1609 na : scalar, optional
1610 Fill value for missing values. The default depends on dtype of the
1611 array. For the ``"str"`` dtype, ``False`` is used. For object
1612 dtype, ``numpy.nan`` is used. For the nullable ``StringDtype``,
1613 ``pandas.NA`` is used.
1614
1615 Returns
1616 -------
1617 Series/Index/array of boolean values
1618 The function returns a Series, Index, or array of boolean values,
1619 where True indicates that the entire string matches the regular
1620 expression pattern and False indicates that it does not.
1621
1622 See Also
1623 --------
1624 match : Similar, but also returns `True` when only a *prefix* of the string
1625 matches the regular expression.
1626 extract : Extract matched groups.
1627
1628 Examples
1629 --------
1630 >>> ser = pd.Series(["cat", "duck", "dove"])
1631 >>> ser.str.fullmatch(r"d.+")
1632 0 False
1633 1 True
1634 2 True
1635 dtype: bool
1636 """
1637 result = self._data.array._str_fullmatch(pat, case=case, flags=flags, na=na)
1638 return self._wrap_result(result, fill_value=na, returns_string=False)
1639
1640 @forbid_nonstring_types(["bytes"])
1641 def replace(
1642 self,
1643 pat: str | re.Pattern | dict,
1644 repl: str | Callable | None = None,
1645 n: int = -1,
1646 case: bool | None = None,
1647 flags: int = 0,
1648 regex: bool = False,
1649 ):
1650 r"""
1651 Replace each occurrence of pattern/regex in the Series/Index.
1652
1653 Equivalent to :meth:`str.replace` or :func:`re.sub`, depending on
1654 the regex value.
1655
1656 Parameters
1657 ----------
1658 pat : str, compiled regex, or a dict
1659 String can be a character sequence or regular expression.
1660 Dictionary contains <key : value> pairs of strings to be replaced
1661 along with the updated value.
1662 repl : str or callable
1663 Replacement string or a callable. The callable is passed the regex
1664 match object and must return a replacement string to be used.
1665 Must have a value of None if `pat` is a dict
1666 See :func:`re.sub`.
1667 n : int, default -1 (all)
1668 Number of replacements to make from start.
1669 case : bool, default None
1670 Determines if replace is case sensitive:
1671
1672 - If True, case sensitive (the default if `pat` is a string)
1673 - Set to False for case insensitive
1674 - Cannot be set if `pat` is a compiled regex.
1675
1676 flags : int, default 0 (no flags)
1677 Regex module flags, e.g. re.IGNORECASE. Cannot be set if `pat` is a compiled
1678 regex.
1679 regex : bool, default False
1680 Determines if the passed-in pattern is a regular expression:
1681
1682 - If True, assumes the passed-in pattern is a regular expression.
1683 - If False, treats the pattern as a literal string
1684 - Cannot be set to False if `pat` is a compiled regex or `repl` is
1685 a callable.
1686
1687 Returns
1688 -------
1689 Series or Index of object
1690 A copy of the object with all matching occurrences of `pat` replaced by
1691 `repl`.
1692
1693 Raises
1694 ------
1695 ValueError
1696 * if `regex` is False and `repl` is a callable or `pat` is a compiled
1697 regex
1698 * if `pat` is a compiled regex and `case` or `flags` is set
1699 * if `pat` is a dictionary and `repl` is not None.
1700
1701 See Also
1702 --------
1703 Series.str.replace : Method to replace occurrences of a substring with another
1704 substring.
1705 Series.str.extract : Extract substrings using a regular expression.
1706 Series.str.findall : Find all occurrences of a pattern or regex in each string.
1707 Series.str.split : Split each string by a specified delimiter or pattern.
1708
1709 Notes
1710 -----
1711 When `pat` is a compiled regex, all flags should be included in the
1712 compiled regex. Use of `case`, `flags`, or `regex=False` with a compiled
1713 regex will raise an error.
1714
1715 Examples
1716 --------
1717 When `pat` is a dictionary, every key in `pat` is replaced
1718 with its corresponding value:
1719
1720 >>> pd.Series(["A", "B", np.nan]).str.replace(pat={"A": "a", "B": "b"})
1721 0 a
1722 1 b
1723 2 NaN
1724 dtype: str
1725
1726 When `pat` is a string and `regex` is True, the given `pat`
1727 is compiled as a regex. When `repl` is a string, it replaces matching
1728 regex patterns as with :meth:`re.sub`. NaN value(s) in the Series are
1729 left as is:
1730
1731 >>> pd.Series(["foo", "fuz", np.nan]).str.replace("f.", "ba", regex=True)
1732 0 bao
1733 1 baz
1734 2 NaN
1735 dtype: str
1736
1737 When `pat` is a string and `regex` is False, every `pat` is replaced with
1738 `repl` as with :meth:`str.replace`:
1739
1740 >>> pd.Series(["f.o", "fuz", np.nan]).str.replace("f.", "ba", regex=False)
1741 0 bao
1742 1 fuz
1743 2 NaN
1744 dtype: str
1745
1746 When `repl` is a callable, it is called on every `pat` using
1747 :func:`re.sub`. The callable should expect one positional argument
1748 (a regex object) and return a string.
1749
1750 To get the idea:
1751
1752 >>> pd.Series(["foo", "fuz", np.nan]).str.replace("f", repr, regex=True)
1753 0 <re.Match object; span=(0, 1), match='f'>oo
1754 1 <re.Match object; span=(0, 1), match='f'>uz
1755 2 NaN
1756 dtype: str
1757
1758 Reverse every lowercase alphabetic word:
1759
1760 >>> repl = lambda m: m.group(0)[::-1]
1761 >>> ser = pd.Series(["foo 123", "bar baz", np.nan])
1762 >>> ser.str.replace(r"[a-z]+", repl, regex=True)
1763 0 oof 123
1764 1 rab zab
1765 2 NaN
1766 dtype: str
1767
1768 Using regex groups (extract second group and swap case):
1769
1770 >>> pat = r"(?P<one>\w+) (?P<two>\w+) (?P<three>\w+)"
1771 >>> repl = lambda m: m.group("two").swapcase()
1772 >>> ser = pd.Series(["One Two Three", "Foo Bar Baz"])
1773 >>> ser.str.replace(pat, repl, regex=True)
1774 0 tWO
1775 1 bAR
1776 dtype: str
1777
1778 Using a compiled regex with flags
1779
1780 >>> import re
1781 >>> regex_pat = re.compile(r"FUZ", flags=re.IGNORECASE)
1782 >>> pd.Series(["foo", "fuz", np.nan]).str.replace(regex_pat, "bar", regex=True)
1783 0 foo
1784 1 bar
1785 2 NaN
1786 dtype: str
1787 """
1788 if isinstance(pat, dict) and repl is not None:
1789 raise ValueError("repl cannot be used when pat is a dictionary")
1790
1791 # Check whether repl is valid (GH 13438, GH 15055)
1792 if not isinstance(pat, dict) and not (isinstance(repl, str) or callable(repl)):
1793 raise TypeError("repl must be a string or callable")
1794
1795 is_compiled_re = is_re(pat)
1796 if regex or regex is None:
1797 if is_compiled_re and (case is not None or flags != 0):
1798 raise ValueError(
1799 "case and flags cannot be set when pat is a compiled regex"
1800 )
1801
1802 elif is_compiled_re:
1803 raise ValueError(
1804 "Cannot use a compiled regex as replacement pattern with regex=False"
1805 )
1806 elif callable(repl):
1807 raise ValueError("Cannot use a callable replacement when regex=False")
1808
1809 if case is None:
1810 case = True
1811
1812 res_output = self._data
1813 if not isinstance(pat, dict):
1814 pat = {pat: repl}
1815
1816 for key, value in pat.items():
1817 result = res_output.array._str_replace(
1818 key, value, n=n, case=case, flags=flags, regex=regex
1819 )
1820 res_output = self._wrap_result(result)
1821
1822 return res_output
1823
1824 @forbid_nonstring_types(["bytes"])
1825 def repeat(self, repeats):
1826 """
1827 Duplicate each string in the Series or Index.
1828
1829 Duplicates each string in the Series or Index, either by applying the
1830 same repeat count to all elements or by using different repeat values
1831 for each element.
1832
1833 Parameters
1834 ----------
1835 repeats : int or sequence of int
1836 Same value for all (int) or different value per (sequence).
1837
1838 Returns
1839 -------
1840 Series or pandas.Index
1841 Series or Index of repeated string objects specified by
1842 input parameter repeats.
1843
1844 See Also
1845 --------
1846 Series.str.lower : Convert all characters in each string to lowercase.
1847 Series.str.upper : Convert all characters in each string to uppercase.
1848 Series.str.title : Convert each string to title case (capitalizing the first
1849 letter of each word).
1850 Series.str.strip : Remove leading and trailing whitespace from each string.
1851 Series.str.replace : Replace occurrences of a substring with another substring
1852 in each string.
1853 Series.str.ljust : Left-justify each string in the Series/Index by padding with
1854 a specified character.
1855 Series.str.rjust : Right-justify each string in the Series/Index by padding with
1856 a specified character.
1857
1858 Examples
1859 --------
1860 >>> s = pd.Series(["a", "b", "c"])
1861 >>> s
1862 0 a
1863 1 b
1864 2 c
1865 dtype: str
1866
1867 Single int repeats string in Series
1868
1869 >>> s.str.repeat(repeats=2)
1870 0 aa
1871 1 bb
1872 2 cc
1873 dtype: str
1874
1875 Sequence of int repeats corresponding string in Series
1876
1877 >>> s.str.repeat(repeats=[1, 2, 3])
1878 0 a
1879 1 bb
1880 2 ccc
1881 dtype: str
1882 """
1883 result = self._data.array._str_repeat(repeats)
1884 return self._wrap_result(result)
1885
1886 @forbid_nonstring_types(["bytes"])
1887 def pad(
1888 self,
1889 width: int,
1890 side: Literal["left", "right", "both"] = "left",
1891 fillchar: str = " ",
1892 ):
1893 """
1894 Pad strings in the Series/Index up to width.
1895
1896 This function pads strings in a Series or Index to a specified width,
1897 filling the extra space with a character of your choice. It provides
1898 flexibility in positioning the padding, allowing it to be added to the
1899 left, right, or both sides. This is useful for formatting strings to
1900 align text or ensure consistent string lengths in data processing.
1901
1902 Parameters
1903 ----------
1904 width : int
1905 Minimum width of resulting string; additional characters will be filled
1906 with character defined in `fillchar`.
1907 side : {'left', 'right', 'both'}, default 'left'
1908 Side from which to fill resulting string.
1909 fillchar : str, default ' '
1910 Additional character for filling, default is whitespace.
1911
1912 Returns
1913 -------
1914 Series or Index of object
1915 Returns Series or Index with minimum number of char in object.
1916
1917 See Also
1918 --------
1919 Series.str.rjust : Fills the left side of strings with an arbitrary
1920 character. Equivalent to ``Series.str.pad(side='left')``.
1921 Series.str.ljust : Fills the right side of strings with an arbitrary
1922 character. Equivalent to ``Series.str.pad(side='right')``.
1923 Series.str.center : Fills both sides of strings with an arbitrary
1924 character. Equivalent to ``Series.str.pad(side='both')``.
1925 Series.str.zfill : Pad strings in the Series/Index by prepending '0'
1926 character. Equivalent to ``Series.str.pad(side='left', fillchar='0')``.
1927
1928 Examples
1929 --------
1930 >>> s = pd.Series(["caribou", "tiger"])
1931 >>> s
1932 0 caribou
1933 1 tiger
1934 dtype: str
1935
1936 >>> s.str.pad(width=10)
1937 0 caribou
1938 1 tiger
1939 dtype: str
1940
1941 >>> s.str.pad(width=10, side="right", fillchar="-")
1942 0 caribou---
1943 1 tiger-----
1944 dtype: str
1945
1946 >>> s.str.pad(width=10, side="both", fillchar="-")
1947 0 -caribou--
1948 1 --tiger---
1949 dtype: str
1950 """
1951 if not isinstance(fillchar, str):
1952 msg = f"fillchar must be a character, not {type(fillchar).__name__}"
1953 raise TypeError(msg)
1954
1955 if len(fillchar) != 1:
1956 raise TypeError("fillchar must be a character, not str")
1957
1958 if not is_integer(width):
1959 msg = f"width must be of integer type, not {type(width).__name__}"
1960 raise TypeError(msg)
1961
1962 result = self._data.array._str_pad(width, side=side, fillchar=fillchar)
1963 return self._wrap_result(result)
1964
1965 @forbid_nonstring_types(["bytes"])
1966 def center(self, width: int, fillchar: str = " "):
1967 """
1968 Pad left and right side of strings in the Series/Index.
1969
1970 Equivalent to :meth:`str.center`.
1971
1972 Parameters
1973 ----------
1974 width : int
1975 Minimum width of resulting string; additional characters will be filled
1976 with ``fillchar``.
1977 fillchar : str
1978 Additional character for filling, default is whitespace.
1979
1980 Returns
1981 -------
1982 Series/Index of objects.
1983 A Series or Index where the strings are modified by :meth:`str.center`.
1984
1985 See Also
1986 --------
1987 Series.str.rjust : Fills the left side of strings with an arbitrary
1988 character.
1989 Series.str.ljust : Fills the right side of strings with an arbitrary
1990 character.
1991 Series.str.center : Fills both sides of strings with an arbitrary
1992 character.
1993 Series.str.zfill : Pad strings in the Series/Index by prepending '0'
1994 character.
1995
1996 Examples
1997 --------
1998 For Series.str.center:
1999
2000 >>> ser = pd.Series(["dog", "bird", "mouse"])
2001 >>> ser.str.center(8, fillchar=".")
2002 0 ..dog...
2003 1 ..bird..
2004 2 .mouse..
2005 dtype: str
2006
2007 For Series.str.ljust:
2008
2009 >>> ser = pd.Series(["dog", "bird", "mouse"])
2010 >>> ser.str.ljust(8, fillchar=".")
2011 0 dog.....
2012 1 bird....
2013 2 mouse...
2014 dtype: str
2015
2016 For Series.str.rjust:
2017
2018 >>> ser = pd.Series(["dog", "bird", "mouse"])
2019 >>> ser.str.rjust(8, fillchar=".")
2020 0 .....dog
2021 1 ....bird
2022 2 ...mouse
2023 dtype: str
2024 """
2025 return self.pad(width, side="both", fillchar=fillchar)
2026
2027 @forbid_nonstring_types(["bytes"])
2028 def ljust(self, width: int, fillchar: str = " "):
2029 """
2030 Pad right side of strings in the Series/Index.
2031
2032 Equivalent to :meth:`str.ljust`.
2033
2034 Parameters
2035 ----------
2036 width : int
2037 Minimum width of resulting string; additional characters will be filled
2038 with ``fillchar``.
2039 fillchar : str
2040 Additional character for filling, default is whitespace.
2041
2042 Returns
2043 -------
2044 Series/Index of objects.
2045 A Series or Index where the strings are modified by :meth:`str.ljust`.
2046
2047 See Also
2048 --------
2049 Series.str.rjust : Fills the left side of strings with an arbitrary
2050 character.
2051 Series.str.ljust : Fills the right side of strings with an arbitrary
2052 character.
2053 Series.str.center : Fills both sides of strings with an arbitrary
2054 character.
2055 Series.str.zfill : Pad strings in the Series/Index by prepending '0'
2056 character.
2057
2058 Examples
2059 --------
2060 For Series.str.center:
2061
2062 >>> ser = pd.Series(["dog", "bird", "mouse"])
2063 >>> ser.str.center(8, fillchar=".")
2064 0 ..dog...
2065 1 ..bird..
2066 2 .mouse..
2067 dtype: str
2068
2069 For Series.str.ljust:
2070
2071 >>> ser = pd.Series(["dog", "bird", "mouse"])
2072 >>> ser.str.ljust(8, fillchar=".")
2073 0 dog.....
2074 1 bird....
2075 2 mouse...
2076 dtype: str
2077
2078 For Series.str.rjust:
2079
2080 >>> ser = pd.Series(["dog", "bird", "mouse"])
2081 >>> ser.str.rjust(8, fillchar=".")
2082 0 .....dog
2083 1 ....bird
2084 2 ...mouse
2085 dtype: str
2086 """
2087 return self.pad(width, side="right", fillchar=fillchar)
2088
2089 @forbid_nonstring_types(["bytes"])
2090 def rjust(self, width: int, fillchar: str = " "):
2091 """
2092 Pad left side of strings in the Series/Index.
2093
2094 Equivalent to :meth:`str.rjust`.
2095
2096 Parameters
2097 ----------
2098 width : int
2099 Minimum width of resulting string; additional characters will be filled
2100 with ``fillchar``.
2101 fillchar : str
2102 Additional character for filling, default is whitespace.
2103
2104 Returns
2105 -------
2106 Series/Index of objects.
2107 A Series or Index where the strings are modified by :meth:`str.rjust`.
2108
2109 See Also
2110 --------
2111 Series.str.rjust : Fills the left side of strings with an arbitrary
2112 character.
2113 Series.str.ljust : Fills the right side of strings with an arbitrary
2114 character.
2115 Series.str.center : Fills both sides of strings with an arbitrary
2116 character.
2117 Series.str.zfill : Pad strings in the Series/Index by prepending '0'
2118 character.
2119
2120 Examples
2121 --------
2122 For Series.str.center:
2123
2124 >>> ser = pd.Series(["dog", "bird", "mouse"])
2125 >>> ser.str.center(8, fillchar=".")
2126 0 ..dog...
2127 1 ..bird..
2128 2 .mouse..
2129 dtype: str
2130
2131 For Series.str.ljust:
2132
2133 >>> ser = pd.Series(["dog", "bird", "mouse"])
2134 >>> ser.str.ljust(8, fillchar=".")
2135 0 dog.....
2136 1 bird....
2137 2 mouse...
2138 dtype: str
2139
2140 For Series.str.rjust:
2141
2142 >>> ser = pd.Series(["dog", "bird", "mouse"])
2143 >>> ser.str.rjust(8, fillchar=".")
2144 0 .....dog
2145 1 ....bird
2146 2 ...mouse
2147 dtype: str
2148 """
2149 return self.pad(width, side="left", fillchar=fillchar)
2150
2151 @forbid_nonstring_types(["bytes"])
2152 def zfill(self, width: int):
2153 """
2154 Pad strings in the Series/Index by prepending '0' characters.
2155
2156 Strings in the Series/Index are padded with '0' characters on the
2157 left of the string to reach a total string length `width`. Strings
2158 in the Series/Index with length greater or equal to `width` are
2159 unchanged.
2160
2161 Parameters
2162 ----------
2163 width : int
2164 Minimum length of resulting string; strings with length less
2165 than `width` be prepended with '0' characters.
2166
2167 Returns
2168 -------
2169 Series/Index of objects.
2170 A Series or Index where the strings are prepended with '0' characters.
2171
2172 See Also
2173 --------
2174 Series.str.rjust : Fills the left side of strings with an arbitrary
2175 character.
2176 Series.str.ljust : Fills the right side of strings with an arbitrary
2177 character.
2178 Series.str.pad : Fills the specified sides of strings with an arbitrary
2179 character.
2180 Series.str.center : Fills both sides of strings with an arbitrary
2181 character.
2182
2183 Notes
2184 -----
2185 Differs from :meth:`str.zfill` which has special handling
2186 for '+'/'-' in the string.
2187
2188 Examples
2189 --------
2190 >>> s = pd.Series(["-1", "1", "1000", 10, np.nan])
2191 >>> s
2192 0 -1
2193 1 1
2194 2 1000
2195 3 10
2196 4 NaN
2197 dtype: object
2198
2199 Note that ``10`` and ``NaN`` are not strings, therefore they are
2200 converted to ``NaN``. The minus sign in ``'-1'`` is treated as a
2201 special character and the zero is added to the right of it
2202 (:meth:`str.zfill` would have moved it to the left). ``1000``
2203 remains unchanged as it is longer than `width`.
2204
2205 >>> s.str.zfill(3)
2206 0 -01
2207 1 001
2208 2 1000
2209 3 NaN
2210 4 NaN
2211 dtype: object
2212 """
2213 if not is_integer(width):
2214 msg = f"width must be of integer type, not {type(width).__name__}"
2215 raise TypeError(msg)
2216
2217 result = self._data.array._str_zfill(width)
2218 return self._wrap_result(result)
2219
2220 def slice(self, start=None, stop=None, step=None):
2221 """
2222 Slice substrings from each element in the Series or Index.
2223
2224 Slicing substrings from strings in a Series or Index helps extract
2225 specific portions of data, making it easier to analyze or manipulate
2226 text. This is useful for tasks like parsing structured text fields or
2227 isolating parts of strings with a consistent format.
2228
2229 Parameters
2230 ----------
2231 start : int, optional
2232 Start position for slice operation.
2233 stop : int, optional
2234 Stop position for slice operation.
2235 step : int, optional
2236 Step size for slice operation.
2237
2238 Returns
2239 -------
2240 Series or Index of object
2241 Series or Index from sliced substring from original string object.
2242
2243 See Also
2244 --------
2245 Series.str.slice_replace : Replace a slice with a string.
2246 Series.str.get : Return element at position.
2247 Equivalent to `Series.str.slice(start=i, stop=i+1)` with `i`
2248 being the position.
2249
2250 Examples
2251 --------
2252 >>> s = pd.Series(["koala", "dog", "chameleon"])
2253 >>> s
2254 0 koala
2255 1 dog
2256 2 chameleon
2257 dtype: str
2258
2259 >>> s.str.slice(start=1)
2260 0 oala
2261 1 og
2262 2 hameleon
2263 dtype: str
2264
2265 >>> s.str.slice(start=-1)
2266 0 a
2267 1 g
2268 2 n
2269 dtype: str
2270
2271 >>> s.str.slice(stop=2)
2272 0 ko
2273 1 do
2274 2 ch
2275 dtype: str
2276
2277 >>> s.str.slice(step=2)
2278 0 kaa
2279 1 dg
2280 2 caeen
2281 dtype: str
2282
2283 >>> s.str.slice(start=0, stop=5, step=3)
2284 0 kl
2285 1 d
2286 2 cm
2287 dtype: str
2288
2289 Equivalent behaviour to:
2290
2291 >>> s.str[0:5:3]
2292 0 kl
2293 1 d
2294 2 cm
2295 dtype: str
2296 """
2297 result = self._data.array._str_slice(start, stop, step)
2298 return self._wrap_result(result)
2299
2300 @forbid_nonstring_types(["bytes"])
2301 def slice_replace(self, start=None, stop=None, repl=None):
2302 """
2303 Replace a positional slice of a string with another value.
2304
2305 This function allows replacing specific parts of a string in a Series
2306 or Index by specifying start and stop positions. It is useful for
2307 modifying substrings in a controlled way, such as updating sections of
2308 text based on their positions or patterns.
2309
2310 Parameters
2311 ----------
2312 start : int, optional
2313 Left index position to use for the slice. If not specified (None),
2314 the slice is unbounded on the left, i.e. slice from the start
2315 of the string.
2316 stop : int, optional
2317 Right index position to use for the slice. If not specified (None),
2318 the slice is unbounded on the right, i.e. slice until the
2319 end of the string.
2320 repl : str, optional
2321 String for replacement. If not specified (None), the sliced region
2322 is replaced with an empty string.
2323
2324 Returns
2325 -------
2326 Series or Index
2327 Same type as the original object.
2328
2329 See Also
2330 --------
2331 Series.str.slice : Just slicing without replacement.
2332
2333 Examples
2334 --------
2335 >>> s = pd.Series(["a", "ab", "abc", "abdc", "abcde"])
2336 >>> s
2337 0 a
2338 1 ab
2339 2 abc
2340 3 abdc
2341 4 abcde
2342 dtype: str
2343
2344 Specify just `start`, meaning replace `start` until the end of the
2345 string with `repl`.
2346
2347 >>> s.str.slice_replace(1, repl="X")
2348 0 aX
2349 1 aX
2350 2 aX
2351 3 aX
2352 4 aX
2353 dtype: str
2354
2355 Specify just `stop`, meaning the start of the string to `stop` is replaced
2356 with `repl`, and the rest of the string is included.
2357
2358 >>> s.str.slice_replace(stop=2, repl="X")
2359 0 X
2360 1 X
2361 2 Xc
2362 3 Xdc
2363 4 Xcde
2364 dtype: str
2365
2366 Specify `start` and `stop`, meaning the slice from `start` to `stop` is
2367 replaced with `repl`. Everything before or after `start` and `stop` is
2368 included as is.
2369
2370 >>> s.str.slice_replace(start=1, stop=3, repl="X")
2371 0 aX
2372 1 aX
2373 2 aX
2374 3 aXc
2375 4 aXde
2376 dtype: str
2377 """
2378 result = self._data.array._str_slice_replace(start, stop, repl)
2379 return self._wrap_result(result)
2380
2381 def decode(
2382 self, encoding, errors: str = "strict", dtype: str | DtypeObj | None = None
2383 ):
2384 """
2385 Decode character string in the Series/Index using indicated encoding.
2386
2387 Equivalent to :meth:`str.decode` in python2 and :meth:`bytes.decode` in
2388 python3.
2389
2390 Parameters
2391 ----------
2392 encoding : str
2393 Specifies the encoding to be used.
2394 errors : str, optional
2395 Specifies the error handling scheme.
2396 Possible values are those supported by :meth:`bytes.decode`.
2397 dtype : str or dtype, optional
2398 The dtype of the result. When not ``None``, must be either a string or
2399 object dtype. When ``None``, the dtype of the result is determined by
2400 ``pd.options.future.infer_string``.
2401
2402 .. versionadded:: 2.3.0
2403
2404 Returns
2405 -------
2406 Series or Index
2407 A Series or Index with decoded strings.
2408
2409 See Also
2410 --------
2411 Series.str.encode : Encodes strings into bytes in a Series/Index.
2412
2413 Examples
2414 --------
2415 For Series:
2416
2417 >>> ser = pd.Series([b"cow", b"123", b"()"])
2418 >>> ser.str.decode("ascii")
2419 0 cow
2420 1 123
2421 2 ()
2422 dtype: str
2423 """
2424 if dtype is not None and not is_string_dtype(dtype):
2425 raise ValueError(f"dtype must be string or object, got {dtype=}")
2426 if dtype is None and using_string_dtype():
2427 dtype = "str"
2428 # TODO: Add a similar _bytes interface.
2429 if encoding in _cpython_optimized_decoders:
2430 # CPython optimized implementation
2431 f = lambda x: x.decode(encoding, errors)
2432 else:
2433 decoder = codecs.getdecoder(encoding)
2434 f = lambda x: decoder(x, errors)[0]
2435 arr = self._data.array
2436 result = arr._str_map(f)
2437 return self._wrap_result(result, dtype=dtype)
2438
2439 @forbid_nonstring_types(["bytes"])
2440 def encode(self, encoding, errors: str = "strict"):
2441 """
2442 Encode character string in the Series/Index using indicated encoding.
2443
2444 Equivalent to :meth:`str.encode`.
2445
2446 Parameters
2447 ----------
2448 encoding : str
2449 Specifies the encoding to be used.
2450 errors : str, optional
2451 Specifies the error handling scheme.
2452 Possible values are those supported by :meth:`str.encode`.
2453
2454 Returns
2455 -------
2456 Series/Index of objects
2457 A Series or Index with strings encoded into bytes.
2458
2459 See Also
2460 --------
2461 Series.str.decode : Decodes bytes into strings in a Series/Index.
2462
2463 Examples
2464 --------
2465 >>> ser = pd.Series(["cow", "123", "()"])
2466 >>> ser.str.encode(encoding="ascii")
2467 0 b'cow'
2468 1 b'123'
2469 2 b'()'
2470 dtype: object
2471 """
2472 result = self._data.array._str_encode(encoding, errors)
2473 return self._wrap_result(result, returns_string=False)
2474
2475 @forbid_nonstring_types(["bytes"])
2476 def strip(self, to_strip=None):
2477 """
2478 Remove leading and trailing characters.
2479
2480 Strip whitespaces (including newlines) or a set of specified characters
2481 from each string in the Series/Index from left and right sides.
2482 Replaces any non-strings in Series with NaNs.
2483 Equivalent to :meth:`str.strip`.
2484
2485 Parameters
2486 ----------
2487 to_strip : str or None, default None
2488 Specifying the set of characters to be removed.
2489 All combinations of this set of characters will be stripped.
2490 If None then whitespaces are removed.
2491
2492 Returns
2493 -------
2494 Series or Index of object
2495 Series or Index with the strings being stripped from the left and
2496 right sides.
2497
2498 See Also
2499 --------
2500 Series.str.strip : Remove leading and trailing characters in Series/Index.
2501 Series.str.lstrip : Remove leading characters in Series/Index.
2502 Series.str.rstrip : Remove trailing characters in Series/Index.
2503
2504 Examples
2505 --------
2506 >>> s = pd.Series(["1. Ant. ", "2. Bee!\\n", "3. Cat?\\t", np.nan, 10, True])
2507 >>> s
2508 0 1. Ant.
2509 1 2. Bee!\\n
2510 2 3. Cat?\\t
2511 3 NaN
2512 4 10
2513 5 True
2514 dtype: object
2515
2516 >>> s.str.strip()
2517 0 1. Ant.
2518 1 2. Bee!
2519 2 3. Cat?
2520 3 NaN
2521 4 NaN
2522 5 NaN
2523 dtype: object
2524
2525 >>> s.str.lstrip("123.")
2526 0 Ant.
2527 1 Bee!\\n
2528 2 Cat?\\t
2529 3 NaN
2530 4 NaN
2531 5 NaN
2532 dtype: object
2533
2534 >>> s.str.rstrip(".!? \\n\\t")
2535 0 1. Ant
2536 1 2. Bee
2537 2 3. Cat
2538 3 NaN
2539 4 NaN
2540 5 NaN
2541 dtype: object
2542
2543 >>> s.str.strip("123.!? \\n\\t")
2544 0 Ant
2545 1 Bee
2546 2 Cat
2547 3 NaN
2548 4 NaN
2549 5 NaN
2550 dtype: object
2551 """
2552 result = self._data.array._str_strip(to_strip)
2553 return self._wrap_result(result)
2554
2555 @forbid_nonstring_types(["bytes"])
2556 def lstrip(self, to_strip=None):
2557 """
2558 Remove leading characters.
2559
2560 Strip whitespaces (including newlines) or a set of specified characters
2561 from each string in the Series/Index from left side.
2562 Replaces any non-strings in Series with NaNs.
2563 Equivalent to :meth:`str.lstrip`.
2564
2565 Parameters
2566 ----------
2567 to_strip : str or None, default None
2568 Specifying the set of characters to be removed.
2569 All combinations of this set of characters will be stripped.
2570 If None then whitespaces are removed.
2571
2572 Returns
2573 -------
2574 Series or Index of object
2575 Series or Index with the strings being stripped from the left side.
2576
2577 See Also
2578 --------
2579 Series.str.strip : Remove leading and trailing characters in Series/Index.
2580 Series.str.lstrip : Remove leading characters in Series/Index.
2581 Series.str.rstrip : Remove trailing characters in Series/Index.
2582
2583 Examples
2584 --------
2585 >>> s = pd.Series(["1. Ant. ", "2. Bee!\\n", "3. Cat?\\t", np.nan, 10, True])
2586 >>> s
2587 0 1. Ant.
2588 1 2. Bee!\\n
2589 2 3. Cat?\\t
2590 3 NaN
2591 4 10
2592 5 True
2593 dtype: object
2594
2595 >>> s.str.strip()
2596 0 1. Ant.
2597 1 2. Bee!
2598 2 3. Cat?
2599 3 NaN
2600 4 NaN
2601 5 NaN
2602 dtype: object
2603
2604 >>> s.str.lstrip("123.")
2605 0 Ant.
2606 1 Bee!\\n
2607 2 Cat?\\t
2608 3 NaN
2609 4 NaN
2610 5 NaN
2611 dtype: object
2612
2613 >>> s.str.rstrip(".!? \\n\\t")
2614 0 1. Ant
2615 1 2. Bee
2616 2 3. Cat
2617 3 NaN
2618 4 NaN
2619 5 NaN
2620 dtype: object
2621
2622 >>> s.str.strip("123.!? \\n\\t")
2623 0 Ant
2624 1 Bee
2625 2 Cat
2626 3 NaN
2627 4 NaN
2628 5 NaN
2629 dtype: object
2630 """
2631 result = self._data.array._str_lstrip(to_strip)
2632 return self._wrap_result(result)
2633
2634 @forbid_nonstring_types(["bytes"])
2635 def rstrip(self, to_strip=None):
2636 """
2637 Remove trailing characters.
2638
2639 Strip whitespaces (including newlines) or a set of specified characters
2640 from each string in the Series/Index from right side.
2641 Replaces any non-strings in Series with NaNs.
2642 Equivalent to :meth:`str.rstrip`.
2643
2644 Parameters
2645 ----------
2646 to_strip : str or None, default None
2647 Specifying the set of characters to be removed.
2648 All combinations of this set of characters will be stripped.
2649 If None then whitespaces are removed.
2650
2651 Returns
2652 -------
2653 Series or Index of object
2654 Series or Index with the strings being stripped from the right side.
2655
2656 See Also
2657 --------
2658 Series.str.strip : Remove leading and trailing characters in Series/Index.
2659 Series.str.lstrip : Remove leading characters in Series/Index.
2660 Series.str.rstrip : Remove trailing characters in Series/Index.
2661
2662 Examples
2663 --------
2664 >>> s = pd.Series(["1. Ant. ", "2. Bee!\\n", "3. Cat?\\t", np.nan, 10, True])
2665 >>> s
2666 0 1. Ant.
2667 1 2. Bee!\\n
2668 2 3. Cat?\\t
2669 3 NaN
2670 4 10
2671 5 True
2672 dtype: object
2673
2674 >>> s.str.strip()
2675 0 1. Ant.
2676 1 2. Bee!
2677 2 3. Cat?
2678 3 NaN
2679 4 NaN
2680 5 NaN
2681 dtype: object
2682
2683 >>> s.str.lstrip("123.")
2684 0 Ant.
2685 1 Bee!\\n
2686 2 Cat?\\t
2687 3 NaN
2688 4 NaN
2689 5 NaN
2690 dtype: object
2691
2692 >>> s.str.rstrip(".!? \\n\\t")
2693 0 1. Ant
2694 1 2. Bee
2695 2 3. Cat
2696 3 NaN
2697 4 NaN
2698 5 NaN
2699 dtype: object
2700
2701 >>> s.str.strip("123.!? \\n\\t")
2702 0 Ant
2703 1 Bee
2704 2 Cat
2705 3 NaN
2706 4 NaN
2707 5 NaN
2708 dtype: object
2709 """
2710 result = self._data.array._str_rstrip(to_strip)
2711 return self._wrap_result(result)
2712
2713 @forbid_nonstring_types(["bytes"])
2714 def removeprefix(self, prefix: str):
2715 """
2716 Remove a prefix from an object series.
2717
2718 If the prefix is not present, the original string will be returned.
2719
2720 Parameters
2721 ----------
2722 prefix : str
2723 Remove the prefix of the string.
2724
2725 Returns
2726 -------
2727 Series/Index: object
2728 The Series or Index with given prefix removed.
2729
2730 See Also
2731 --------
2732 Series.str.removesuffix : Remove a suffix from an object series.
2733
2734 Examples
2735 --------
2736 >>> s = pd.Series(["str_foo", "str_bar", "no_prefix"])
2737 >>> s
2738 0 str_foo
2739 1 str_bar
2740 2 no_prefix
2741 dtype: str
2742 >>> s.str.removeprefix("str_")
2743 0 foo
2744 1 bar
2745 2 no_prefix
2746 dtype: str
2747
2748 >>> s = pd.Series(["foo_str", "bar_str", "no_suffix"])
2749 >>> s
2750 0 foo_str
2751 1 bar_str
2752 2 no_suffix
2753 dtype: str
2754 >>> s.str.removesuffix("_str")
2755 0 foo
2756 1 bar
2757 2 no_suffix
2758 dtype: str
2759 """
2760 result = self._data.array._str_removeprefix(prefix)
2761 return self._wrap_result(result)
2762
2763 @forbid_nonstring_types(["bytes"])
2764 def removesuffix(self, suffix: str):
2765 """
2766 Remove a suffix from an object series.
2767
2768 If the suffix is not present, the original string will be returned.
2769
2770 Parameters
2771 ----------
2772 suffix : str
2773 Remove the suffix of the string.
2774
2775 Returns
2776 -------
2777 Series/Index: object
2778 The Series or Index with given suffix removed.
2779
2780 See Also
2781 --------
2782 Series.str.removeprefix : Remove a prefix from an object series.
2783
2784 Examples
2785 --------
2786 >>> s = pd.Series(["str_foo", "str_bar", "no_prefix"])
2787 >>> s
2788 0 str_foo
2789 1 str_bar
2790 2 no_prefix
2791 dtype: str
2792 >>> s.str.removeprefix("str_")
2793 0 foo
2794 1 bar
2795 2 no_prefix
2796 dtype: str
2797
2798 >>> s = pd.Series(["foo_str", "bar_str", "no_suffix"])
2799 >>> s
2800 0 foo_str
2801 1 bar_str
2802 2 no_suffix
2803 dtype: str
2804 >>> s.str.removesuffix("_str")
2805 0 foo
2806 1 bar
2807 2 no_suffix
2808 dtype: str
2809 """
2810 result = self._data.array._str_removesuffix(suffix)
2811 return self._wrap_result(result)
2812
2813 @forbid_nonstring_types(["bytes"])
2814 def wrap(
2815 self,
2816 width: int,
2817 expand_tabs: bool = True,
2818 tabsize: int = 8,
2819 replace_whitespace: bool = True,
2820 drop_whitespace: bool = True,
2821 initial_indent: str = "",
2822 subsequent_indent: str = "",
2823 fix_sentence_endings: bool = False,
2824 break_long_words: bool = True,
2825 break_on_hyphens: bool = True,
2826 max_lines: int | None = None,
2827 placeholder: str = " [...]",
2828 ):
2829 r"""
2830 Wrap strings in Series/Index at specified line width.
2831
2832 This method has the same keyword parameters and defaults as
2833 :class:`textwrap.TextWrapper`.
2834
2835 Parameters
2836 ----------
2837 width : int, optional
2838 Maximum line width.
2839 expand_tabs : bool, optional
2840 If True, tab characters will be expanded to spaces (default: True).
2841 tabsize : int, optional
2842 If expand_tabs is true, then all tab characters in text will be
2843 expanded to zero or more spaces, depending on the current column
2844 and the given tab size (default: 8).
2845 replace_whitespace : bool, optional
2846 If True, each whitespace character (as defined by string.whitespace)
2847 remaining after tab expansion will be replaced by a single space
2848 (default: True).
2849 drop_whitespace : bool, optional
2850 If True, whitespace that, after wrapping, happens to end up at the
2851 beginning or end of a line is dropped (default: True).
2852 initial_indent : str, optional
2853 String that will be prepended to the first line of wrapped output.
2854 Counts towards the length of the first line. The empty string is
2855 not indented (default: '').
2856 subsequent_indent : str, optional
2857 String that will be prepended to all lines of wrapped output except
2858 the first. Counts towards the length of each line except the first
2859 (default: '').
2860 fix_sentence_endings : bool, optional
2861 If true, TextWrapper attempts to detect sentence endings and ensure
2862 that sentences are always separated by exactly two spaces. This is
2863 generally desired for text in a monospaced font. However, the sentence
2864 detection algorithm is imperfect: it assumes that a sentence ending
2865 consists of a lowercase letter followed by one of '.', '!', or '?',
2866 possibly followed by one of '"' or "'", followed by a space. One
2867 problem with this algorithm is that it is unable to detect the
2868 difference between “Dr.” in `[...] Dr. Frankenstein's monster [...]`
2869 and “Spot.” in `[...] See Spot. See Spot run [...]`
2870 Since the sentence detection algorithm relies on string.lowercase
2871 for the definition of “lowercase letter”, and a convention of using
2872 two spaces after a period to separate sentences on the same line,
2873 it is specific to English-language texts (default: False).
2874 break_long_words : bool, optional
2875 If True, then words longer than width will be broken in order to ensure
2876 that no lines are longer than width. If it is false, long words will
2877 not be broken, and some lines may be longer than width (default: True).
2878 break_on_hyphens : bool, optional
2879 If True, wrapping will occur preferably on whitespace and right after
2880 hyphens in compound words, as it is customary in English. If false,
2881 only whitespaces will be considered as potentially good places for line
2882 breaks, but you need to set break_long_words to false if you want truly
2883 insecable words (default: True).
2884 max_lines : int, optional
2885 If not None, then the output will contain at most max_lines lines, with
2886 placeholder appearing at the end of the output (default: None).
2887 placeholder : str, optional
2888 String that will appear at the end of the output text if it has been
2889 truncated (default: ' [...]').
2890
2891 Returns
2892 -------
2893 Series or Index
2894 A Series or Index where the strings are wrapped at the specified line width.
2895
2896 See Also
2897 --------
2898 Series.str.strip : Remove leading and trailing characters in Series/Index.
2899 Series.str.lstrip : Remove leading characters in Series/Index.
2900 Series.str.rstrip : Remove trailing characters in Series/Index.
2901
2902 Notes
2903 -----
2904 Internally, this method uses a :class:`textwrap.TextWrapper` instance with
2905 default settings. To achieve behavior matching R's stringr library str_wrap
2906 function, use the arguments:
2907
2908 - expand_tabs = False
2909 - replace_whitespace = True
2910 - drop_whitespace = True
2911 - break_long_words = False
2912 - break_on_hyphens = False
2913
2914 Examples
2915 --------
2916 >>> s = pd.Series(["line to be wrapped", "another line to be wrapped"])
2917 >>> s.str.wrap(12)
2918 0 line to be\nwrapped
2919 1 another line\nto be\nwrapped
2920 dtype: str
2921 """
2922 result = self._data.array._str_wrap(
2923 width=width,
2924 expand_tabs=expand_tabs,
2925 tabsize=tabsize,
2926 replace_whitespace=replace_whitespace,
2927 drop_whitespace=drop_whitespace,
2928 initial_indent=initial_indent,
2929 subsequent_indent=subsequent_indent,
2930 fix_sentence_endings=fix_sentence_endings,
2931 break_long_words=break_long_words,
2932 break_on_hyphens=break_on_hyphens,
2933 max_lines=max_lines,
2934 placeholder=placeholder,
2935 )
2936 return self._wrap_result(result)
2937
2938 @forbid_nonstring_types(["bytes"])
2939 def get_dummies(
2940 self,
2941 sep: str = "|",
2942 dtype: NpDtype | None = None,
2943 ):
2944 """
2945 Return DataFrame of dummy/indicator variables for Series.
2946
2947 Each string in Series is split by sep and returned as a DataFrame
2948 of dummy/indicator variables.
2949
2950 Parameters
2951 ----------
2952 sep : str, default "|"
2953 String to split on.
2954 dtype : dtype, default np.int64
2955 Data type for new columns. Only a single dtype is allowed.
2956
2957 Returns
2958 -------
2959 DataFrame
2960 Dummy variables corresponding to values of the Series.
2961
2962 See Also
2963 --------
2964 get_dummies : Convert categorical variable into dummy/indicator
2965 variables.
2966
2967 Examples
2968 --------
2969 >>> pd.Series(["a|b", "a", "a|c"]).str.get_dummies()
2970 a b c
2971 0 1 1 0
2972 1 1 0 0
2973 2 1 0 1
2974
2975 >>> pd.Series(["a|b", np.nan, "a|c"]).str.get_dummies()
2976 a b c
2977 0 1 1 0
2978 1 0 0 0
2979 2 1 0 1
2980
2981 >>> pd.Series(["a|b", np.nan, "a|c"]).str.get_dummies(dtype=bool)
2982 a b c
2983 0 True True False
2984 1 False False False
2985 2 True False True
2986 """
2987 from pandas.core.frame import DataFrame
2988
2989 if dtype is not None and not (is_numeric_dtype(dtype) or is_bool_dtype(dtype)):
2990 raise ValueError("Only numeric or boolean dtypes are supported for 'dtype'")
2991 # we need to cast to Series of strings as only that has all
2992 # methods available for making the dummies...
2993 result, name = self._data.array._str_get_dummies(sep, dtype)
2994 if is_extension_array_dtype(dtype):
2995 return self._wrap_result(
2996 DataFrame(result, columns=name, dtype=dtype),
2997 name=name,
2998 returns_string=False,
2999 )
3000 return self._wrap_result(
3001 result,
3002 name=name,
3003 expand=True,
3004 returns_string=False,
3005 )
3006
3007 @forbid_nonstring_types(["bytes"])
3008 def translate(self, table):
3009 """
3010 Map all characters in the string through the given mapping table.
3011
3012 This method is equivalent to the standard :meth:`str.translate`
3013 method for strings. It maps each character in the string to a new
3014 character according to the translation table provided. Unmapped
3015 characters are left unchanged, while characters mapped to None
3016 are removed.
3017
3018 Parameters
3019 ----------
3020 table : dict
3021 Table is a mapping of Unicode ordinals to Unicode ordinals, strings, or
3022 None. Unmapped characters are left untouched.
3023 Characters mapped to None are deleted. :meth:`str.maketrans` is a
3024 helper function for making translation tables.
3025
3026 Returns
3027 -------
3028 Series or Index
3029 A new Series or Index with translated strings.
3030
3031 See Also
3032 --------
3033 Series.str.replace : Replace occurrences of pattern/regex in the
3034 Series with some other string.
3035 Index.str.replace : Replace occurrences of pattern/regex in the
3036 Index with some other string.
3037
3038 Examples
3039 --------
3040 >>> ser = pd.Series(["El niño", "Françoise"])
3041 >>> mytable = str.maketrans({"ñ": "n", "ç": "c"})
3042 >>> ser.str.translate(mytable)
3043 0 El nino
3044 1 Francoise
3045 dtype: str
3046 """
3047 result = self._data.array._str_translate(table)
3048 dtype = object if self._data.dtype == "object" else None
3049 return self._wrap_result(result, dtype=dtype)
3050
3051 @forbid_nonstring_types(["bytes"])
3052 def count(self, pat, flags: int = 0):
3053 r"""
3054 Count occurrences of pattern in each string of the Series/Index.
3055
3056 This function is used to count the number of times a particular regex
3057 pattern is repeated in each of the string elements of the
3058 :class:`~pandas.Series`.
3059
3060 Parameters
3061 ----------
3062 pat : str
3063 Valid regular expression.
3064 flags : int, default 0, meaning no flags
3065 Flags for the `re` module. For a complete list, `see here
3066 <https://docs.python.org/3/howto/regex.html#compilation-flags>`_.
3067
3068 Returns
3069 -------
3070 Series or Index
3071 Same type as the calling object containing the integer counts.
3072
3073 See Also
3074 --------
3075 re : Standard library module for regular expressions.
3076 str.count : Standard library version, without regular expression support.
3077
3078 Notes
3079 -----
3080 Some characters need to be escaped when passing in `pat`.
3081 eg. ``'$'`` has a special meaning in regex and must be escaped when
3082 finding this literal character.
3083
3084 Examples
3085 --------
3086 >>> s = pd.Series(["A", "B", "Aaba", "Baca", np.nan, "CABA", "cat"])
3087 >>> s.str.count("a")
3088 0 0.0
3089 1 0.0
3090 2 2.0
3091 3 2.0
3092 4 NaN
3093 5 0.0
3094 6 1.0
3095 dtype: float64
3096
3097 Escape ``'$'`` to find the literal dollar sign.
3098
3099 >>> s = pd.Series(["$", "B", "Aab$", "$$ca", "C$B$", "cat"])
3100 >>> s.str.count("\\$")
3101 0 1
3102 1 0
3103 2 1
3104 3 2
3105 4 2
3106 5 0
3107 dtype: int64
3108
3109 This is also available on Index
3110
3111 >>> pd.Index(["A", "A", "Aaba", "cat"]).str.count("a")
3112 Index([0, 0, 2, 1], dtype='int64')
3113 """
3114 result = self._data.array._str_count(pat, flags)
3115 return self._wrap_result(result, returns_string=False)
3116
3117 @forbid_nonstring_types(["bytes"])
3118 def startswith(
3119 self, pat: str | tuple[str, ...], na: Scalar | lib.NoDefault = lib.no_default
3120 ) -> Series | Index:
3121 """
3122 Test if the start of each string element matches a pattern.
3123
3124 Equivalent to :meth:`str.startswith`.
3125
3126 Parameters
3127 ----------
3128 pat : str or tuple[str, ...]
3129 Character sequence or tuple of strings. Regular expressions are not
3130 accepted.
3131 na : scalar, optional
3132 Object shown if element tested is not a string. The default depends
3133 on dtype of the array. For the ``"str"`` dtype, ``False`` is used.
3134 For object dtype, ``numpy.nan`` is used. For the nullable
3135 ``StringDtype``, ``pandas.NA`` is used.
3136
3137 Returns
3138 -------
3139 Series or Index of bool
3140 A Series of booleans indicating whether the given pattern matches
3141 the start of each string element.
3142
3143 See Also
3144 --------
3145 str.startswith : Python standard library string method.
3146 Series.str.endswith : Same as startswith, but tests the end of string.
3147 Series.str.contains : Tests if string element contains a pattern.
3148
3149 Examples
3150 --------
3151 >>> s = pd.Series(["bat", "Bear", "cat", np.nan])
3152 >>> s
3153 0 bat
3154 1 Bear
3155 2 cat
3156 3 NaN
3157 dtype: str
3158
3159 >>> s.str.startswith("b")
3160 0 True
3161 1 False
3162 2 False
3163 3 False
3164 dtype: bool
3165
3166 >>> s.str.startswith(("b", "B"))
3167 0 True
3168 1 True
3169 2 False
3170 3 False
3171 dtype: bool
3172 """
3173 if not isinstance(pat, (str, tuple)):
3174 msg = f"expected a string or tuple, not {type(pat).__name__}"
3175 raise TypeError(msg)
3176 result = self._data.array._str_startswith(pat, na=na)
3177 return self._wrap_result(result, returns_string=False)
3178
3179 @forbid_nonstring_types(["bytes"])
3180 def endswith(
3181 self, pat: str | tuple[str, ...], na: Scalar | lib.NoDefault = lib.no_default
3182 ) -> Series | Index:
3183 """
3184 Test if the end of each string element matches a pattern.
3185
3186 Equivalent to :meth:`str.endswith`.
3187
3188 Parameters
3189 ----------
3190 pat : str or tuple[str, ...]
3191 Character sequence or tuple of strings. Regular expressions are not
3192 accepted.
3193 na : scalar, optional
3194 Object shown if element tested is not a string. The default depends
3195 on dtype of the array. For the ``"str"`` dtype, ``False`` is used.
3196 For object dtype, ``numpy.nan`` is used. For the nullable
3197 ``StringDtype``, ``pandas.NA`` is used.
3198
3199 Returns
3200 -------
3201 Series or Index of bool
3202 A Series of booleans indicating whether the given pattern matches
3203 the end of each string element.
3204
3205 See Also
3206 --------
3207 str.endswith : Python standard library string method.
3208 Series.str.startswith : Same as endswith, but tests the start of string.
3209 Series.str.contains : Tests if string element contains a pattern.
3210
3211 Examples
3212 --------
3213 >>> s = pd.Series(["bat", "bear", "caT", np.nan])
3214 >>> s
3215 0 bat
3216 1 bear
3217 2 caT
3218 3 NaN
3219 dtype: str
3220
3221 >>> s.str.endswith("t")
3222 0 True
3223 1 False
3224 2 False
3225 3 False
3226 dtype: bool
3227
3228 >>> s.str.endswith(("t", "T"))
3229 0 True
3230 1 False
3231 2 True
3232 3 False
3233 dtype: bool
3234 """
3235 if not isinstance(pat, (str, tuple)):
3236 msg = f"expected a string or tuple, not {type(pat).__name__}"
3237 raise TypeError(msg)
3238 result = self._data.array._str_endswith(pat, na=na)
3239 return self._wrap_result(result, returns_string=False)
3240
3241 @forbid_nonstring_types(["bytes"])
3242 def findall(self, pat, flags: int = 0):
3243 """
3244 Find all occurrences of pattern or regular expression in the Series/Index.
3245
3246 Equivalent to applying :func:`re.findall` to all the elements in the
3247 Series/Index.
3248
3249 Parameters
3250 ----------
3251 pat : str
3252 Pattern or regular expression.
3253 flags : int, default 0
3254 Flags from ``re`` module, e.g. `re.IGNORECASE` (default is 0, which
3255 means no flags).
3256
3257 Returns
3258 -------
3259 Series/Index of lists of strings
3260 All non-overlapping matches of pattern or regular expression in each
3261 string of this Series/Index.
3262
3263 See Also
3264 --------
3265 count : Count occurrences of pattern or regular expression in each string
3266 of the Series/Index.
3267 extractall : For each string in the Series, extract groups from all matches
3268 of regular expression and return a DataFrame with one row for each
3269 match and one column for each group.
3270 re.findall : The equivalent ``re`` function to all non-overlapping matches
3271 of pattern or regular expression in string, as a list of strings.
3272
3273 Examples
3274 --------
3275 >>> s = pd.Series(["Lion", "Monkey", "Rabbit"])
3276
3277 The search for the pattern 'Monkey' returns one match:
3278
3279 >>> s.str.findall("Monkey")
3280 0 []
3281 1 [Monkey]
3282 2 []
3283 dtype: object
3284
3285 On the other hand, the search for the pattern 'MONKEY' doesn't return any
3286 match:
3287
3288 >>> s.str.findall("MONKEY")
3289 0 []
3290 1 []
3291 2 []
3292 dtype: object
3293
3294 Flags can be added to the pattern or regular expression. For instance,
3295 to find the pattern 'MONKEY' ignoring the case:
3296
3297 >>> import re
3298 >>> s.str.findall("MONKEY", flags=re.IGNORECASE)
3299 0 []
3300 1 [Monkey]
3301 2 []
3302 dtype: object
3303
3304 When the pattern matches more than one string in the Series, all matches
3305 are returned:
3306
3307 >>> s.str.findall("on")
3308 0 [on]
3309 1 [on]
3310 2 []
3311 dtype: object
3312
3313 Regular expressions are supported too. For instance, the search for all the
3314 strings ending with the word 'on' is shown next:
3315
3316 >>> s.str.findall("on$")
3317 0 [on]
3318 1 []
3319 2 []
3320 dtype: object
3321
3322 If the pattern is found more than once in the same string, then a list of
3323 multiple strings is returned:
3324
3325 >>> s.str.findall("b")
3326 0 []
3327 1 []
3328 2 [b, b]
3329 dtype: object
3330 """
3331 result = self._data.array._str_findall(pat, flags)
3332 return self._wrap_result(result, returns_string=False)
3333
3334 @forbid_nonstring_types(["bytes"])
3335 def extract(
3336 self, pat: str, flags: int = 0, expand: bool = True
3337 ) -> DataFrame | Series | Index:
3338 r"""
3339 Extract capture groups in the regex `pat` as columns in a DataFrame.
3340
3341 For each subject string in the Series, extract groups from the
3342 first match of regular expression `pat`.
3343
3344 Parameters
3345 ----------
3346 pat : str
3347 Regular expression pattern with capturing groups.
3348 flags : int, default 0 (no flags)
3349 Flags from the ``re`` module, e.g. ``re.IGNORECASE``, that
3350 modify regular expression matching for things like case,
3351 spaces, etc. For more details, see :mod:`re`.
3352 expand : bool, default True
3353 If True, return DataFrame with one column per capture group.
3354 If False, return a Series/Index if there is one capture group
3355 or DataFrame if there are multiple capture groups.
3356
3357 Returns
3358 -------
3359 DataFrame or Series or Index
3360 A DataFrame with one row for each subject string, and one
3361 column for each group. Any capture group names in regular
3362 expression pat will be used for column names; otherwise
3363 capture group numbers will be used. The dtype of each result
3364 column is always object, even when no match is found. If
3365 ``expand=False`` and pat has only one capture group, then
3366 return a Series (if subject is a Series) or Index (if subject
3367 is an Index).
3368
3369 See Also
3370 --------
3371 extractall : Returns all matches (not just the first match).
3372
3373 Examples
3374 --------
3375 A pattern with two groups will return a DataFrame with two columns.
3376 Non-matches will be NaN.
3377
3378 >>> s = pd.Series(["a1", "b2", "c3"])
3379 >>> s.str.extract(r"([ab])(\d)")
3380 0 1
3381 0 a 1
3382 1 b 2
3383 2 NaN NaN
3384
3385 A pattern may contain optional groups.
3386
3387 >>> s.str.extract(r"([ab])?(\d)")
3388 0 1
3389 0 a 1
3390 1 b 2
3391 2 NaN 3
3392
3393 Named groups will become column names in the result.
3394
3395 >>> s.str.extract(r"(?P<letter>[ab])(?P<digit>\d)")
3396 letter digit
3397 0 a 1
3398 1 b 2
3399 2 NaN NaN
3400
3401 A pattern with one group will return a DataFrame with one column
3402 if expand=True.
3403
3404 >>> s.str.extract(r"[ab](\d)", expand=True)
3405 0
3406 0 1
3407 1 2
3408 2 NaN
3409
3410 A pattern with one group will return a Series if expand=False.
3411
3412 >>> s.str.extract(r"[ab](\d)", expand=False)
3413 0 1
3414 1 2
3415 2 NaN
3416 dtype: str
3417 """
3418 from pandas import DataFrame
3419
3420 if not isinstance(expand, bool):
3421 raise ValueError("expand must be True or False")
3422
3423 regex = re.compile(pat, flags=flags)
3424 if regex.groups == 0:
3425 raise ValueError("pattern contains no capture groups")
3426
3427 if not expand and regex.groups > 1 and isinstance(self._data, ABCIndex):
3428 raise ValueError("only one regex group is supported with Index")
3429
3430 obj = self._data
3431 result_dtype = _result_dtype(obj)
3432
3433 returns_df = regex.groups > 1 or expand
3434
3435 if returns_df:
3436 name = None
3437 columns = _get_group_names(regex)
3438
3439 if obj.array.size == 0:
3440 result = DataFrame(columns=columns, dtype=result_dtype)
3441
3442 else:
3443 result_list = self._data.array._str_extract(
3444 pat, flags=flags, expand=returns_df
3445 )
3446
3447 result_index: Index | None
3448 if isinstance(obj, ABCSeries):
3449 result_index = obj.index
3450 else:
3451 result_index = None
3452
3453 result = DataFrame(
3454 result_list, columns=columns, index=result_index, dtype=result_dtype
3455 )
3456
3457 else:
3458 name = _get_single_group_name(regex)
3459 result = self._data.array._str_extract(pat, flags=flags, expand=returns_df)
3460 return self._wrap_result(result, name=name, dtype=result_dtype)
3461
3462 @forbid_nonstring_types(["bytes"])
3463 def extractall(self, pat, flags: int = 0) -> DataFrame:
3464 r"""
3465 Extract capture groups in the regex `pat` as columns in DataFrame.
3466
3467 For each subject string in the Series, extract groups from all
3468 matches of regular expression pat. When each subject string in the
3469 Series has exactly one match, extractall(pat).xs(0, level='match')
3470 is the same as extract(pat).
3471
3472 Parameters
3473 ----------
3474 pat : str
3475 Regular expression pattern with capturing groups.
3476 flags : int, default 0 (no flags)
3477 A ``re`` module flag, for example ``re.IGNORECASE``. These allow
3478 to modify regular expression matching for things like case, spaces,
3479 etc. Multiple flags can be combined with the bitwise OR operator,
3480 for example ``re.IGNORECASE | re.MULTILINE``.
3481
3482 Returns
3483 -------
3484 DataFrame
3485 A ``DataFrame`` with one row for each match, and one column for each
3486 group. Its rows have a ``MultiIndex`` with first levels that come from
3487 the subject ``Series``. The last level is named 'match' and indexes the
3488 matches in each item of the ``Series``. Any capture group names in
3489 regular expression pat will be used for column names; otherwise capture
3490 group numbers will be used.
3491
3492 See Also
3493 --------
3494 extract : Returns first match only (not all matches).
3495
3496 Examples
3497 --------
3498 A pattern with one group will return a DataFrame with one column.
3499 Indices with no matches will not appear in the result.
3500
3501 >>> s = pd.Series(["a1a2", "b1", "c1"], index=["A", "B", "C"])
3502 >>> s.str.extractall(r"[ab](\d)")
3503 0
3504 match
3505 A 0 1
3506 1 2
3507 B 0 1
3508
3509 Capture group names are used for column names of the result.
3510
3511 >>> s.str.extractall(r"[ab](?P<digit>\d)")
3512 digit
3513 match
3514 A 0 1
3515 1 2
3516 B 0 1
3517
3518 A pattern with two groups will return a DataFrame with two columns.
3519
3520 >>> s.str.extractall(r"(?P<letter>[ab])(?P<digit>\d)")
3521 letter digit
3522 match
3523 A 0 a 1
3524 1 a 2
3525 B 0 b 1
3526
3527 Optional groups that do not match are NaN in the result.
3528
3529 >>> s.str.extractall(r"(?P<letter>[ab])?(?P<digit>\d)")
3530 letter digit
3531 match
3532 A 0 a 1
3533 1 a 2
3534 B 0 b 1
3535 C 0 NaN 1
3536 """
3537 # TODO: dispatch
3538 return str_extractall(self._orig, pat, flags)
3539
3540 @forbid_nonstring_types(["bytes"])
3541 def find(self, sub, start: int = 0, end=None):
3542 """
3543 Return lowest indexes in each strings in the Series/Index.
3544
3545 Each of returned indexes corresponds to the position where the
3546 substring is fully contained between [start:end]. Return -1 on
3547 failure. Equivalent to standard :meth:`str.find`.
3548
3549 Parameters
3550 ----------
3551 sub : str
3552 Substring being searched.
3553 start : int
3554 Left edge index.
3555 end : int
3556 Right edge index.
3557
3558 Returns
3559 -------
3560 Series or Index of int.
3561 A Series (if the input is a Series) or an Index (if the input is an
3562 Index) of the lowest indexes corresponding to the positions where the
3563 substring is found in each string of the input.
3564
3565 See Also
3566 --------
3567 rfind : Return highest indexes in each strings.
3568
3569 Examples
3570 --------
3571 For Series.str.find:
3572
3573 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"])
3574 >>> ser.str.find("_")
3575 0 0
3576 1 4
3577 2 2
3578 dtype: int64
3579
3580 For Series.str.rfind:
3581
3582 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"])
3583 >>> ser.str.rfind("_")
3584 0 4
3585 1 4
3586 2 4
3587 dtype: int64
3588 """
3589 if not isinstance(sub, str):
3590 msg = f"expected a string object, not {type(sub).__name__}"
3591 raise TypeError(msg)
3592
3593 result = self._data.array._str_find(sub, start, end)
3594 return self._wrap_result(result, returns_string=False)
3595
3596 @forbid_nonstring_types(["bytes"])
3597 def rfind(self, sub, start: int = 0, end=None):
3598 """
3599 Return highest indexes in each strings in the Series/Index.
3600
3601 Each of returned indexes corresponds to the position where the
3602 substring is fully contained between [start:end]. Return -1 on
3603 failure. Equivalent to standard :meth:`str.rfind`.
3604
3605 Parameters
3606 ----------
3607 sub : str
3608 Substring being searched.
3609 start : int
3610 Left edge index.
3611 end : int
3612 Right edge index.
3613
3614 Returns
3615 -------
3616 Series or Index of int.
3617 A Series (if the input is a Series) or an Index (if the input is an
3618 Index) of the highest indexes corresponding to the positions where the
3619 substring is found in each string of the input.
3620
3621 See Also
3622 --------
3623 find : Return lowest indexes in each strings.
3624
3625 Examples
3626 --------
3627 For Series.str.find:
3628
3629 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"])
3630 >>> ser.str.find("_")
3631 0 0
3632 1 4
3633 2 2
3634 dtype: int64
3635
3636 For Series.str.rfind:
3637
3638 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"])
3639 >>> ser.str.rfind("_")
3640 0 4
3641 1 4
3642 2 4
3643 dtype: int64
3644 """
3645 if not isinstance(sub, str):
3646 msg = f"expected a string object, not {type(sub).__name__}"
3647 raise TypeError(msg)
3648
3649 result = self._data.array._str_rfind(sub, start=start, end=end)
3650 return self._wrap_result(result, returns_string=False)
3651
3652 @forbid_nonstring_types(["bytes"])
3653 def normalize(self, form):
3654 """
3655 Return the Unicode normal form for the strings in the Series/Index.
3656
3657 For more information on the forms, see the
3658 :func:`unicodedata.normalize`.
3659
3660 Parameters
3661 ----------
3662 form : {'NFC', 'NFKC', 'NFD', 'NFKD'}
3663 Unicode form.
3664
3665 Returns
3666 -------
3667 Series/Index of objects
3668 A Series or Index of strings in the same Unicode form specified by `form`.
3669 The returned object retains the same type as the input (Series or Index),
3670 and contains the normalized strings.
3671
3672 See Also
3673 --------
3674 Series.str.upper : Convert all characters in each string to uppercase.
3675 Series.str.lower : Convert all characters in each string to lowercase.
3676 Series.str.title : Convert each string to title case (capitalizing the
3677 first letter of each word).
3678 Series.str.strip : Remove leading and trailing whitespace from each string.
3679 Series.str.replace : Replace occurrences of a substring with another substring
3680 in each string.
3681
3682 Examples
3683 --------
3684 >>> ser = pd.Series(["ñ"])
3685 >>> ser.str.normalize("NFC") == ser.str.normalize("NFD")
3686 0 False
3687 dtype: bool
3688 """
3689 result = self._data.array._str_normalize(form)
3690 return self._wrap_result(result)
3691
3692 @forbid_nonstring_types(["bytes"])
3693 def index(self, sub, start: int = 0, end=None):
3694 """
3695 Return lowest indexes in each string in Series/Index.
3696
3697 Each of the returned indexes corresponds to the position where the
3698 substring is fully contained between [start:end]. This is the same
3699 as ``str.find`` except instead of returning -1, it raises a
3700 ValueError when the substring is not found. Equivalent to standard
3701 ``str.index``.
3702
3703 Parameters
3704 ----------
3705 sub : str
3706 Substring being searched.
3707 start : int
3708 Left edge index.
3709 end : int
3710 Right edge index.
3711
3712 Returns
3713 -------
3714 Series or Index of object
3715 Returns a Series or an Index of the lowest indexes
3716 in each string of the input.
3717
3718 See Also
3719 --------
3720 rindex : Return highest indexes in each strings.
3721
3722 Examples
3723 --------
3724 For Series.str.index:
3725
3726 >>> ser = pd.Series(["horse", "eagle", "donkey"])
3727 >>> ser.str.index("e")
3728 0 4
3729 1 0
3730 2 4
3731 dtype: int64
3732
3733 For Series.str.rindex:
3734
3735 >>> ser = pd.Series(["Deer", "eagle", "Sheep"])
3736 >>> ser.str.rindex("e")
3737 0 2
3738 1 4
3739 2 3
3740 dtype: int64
3741 """
3742 if not isinstance(sub, str):
3743 msg = f"expected a string object, not {type(sub).__name__}"
3744 raise TypeError(msg)
3745
3746 result = self._data.array._str_index(sub, start=start, end=end)
3747 return self._wrap_result(result, returns_string=False)
3748
3749 @forbid_nonstring_types(["bytes"])
3750 def rindex(self, sub, start: int = 0, end=None):
3751 """
3752 Return highest indexes in each string in Series/Index.
3753
3754 Each of the returned indexes corresponds to the position where the
3755 substring is fully contained between [start:end]. This is the same
3756 as ``str.rfind`` except instead of returning -1, it raises a
3757 ValueError when the substring is not found. Equivalent to standard
3758 ``str.rindex``.
3759
3760 Parameters
3761 ----------
3762 sub : str
3763 Substring being searched.
3764 start : int
3765 Left edge index.
3766 end : int
3767 Right edge index.
3768
3769 Returns
3770 -------
3771 Series or Index of object
3772 Returns a Series or an Index of the highest indexes
3773 in each string of the input.
3774
3775 See Also
3776 --------
3777 index : Return lowest indexes in each strings.
3778
3779 Examples
3780 --------
3781 For Series.str.index:
3782
3783 >>> ser = pd.Series(["horse", "eagle", "donkey"])
3784 >>> ser.str.index("e")
3785 0 4
3786 1 0
3787 2 4
3788 dtype: int64
3789
3790 For Series.str.rindex:
3791
3792 >>> ser = pd.Series(["Deer", "eagle", "Sheep"])
3793 >>> ser.str.rindex("e")
3794 0 2
3795 1 4
3796 2 3
3797 dtype: int64
3798 """
3799 if not isinstance(sub, str):
3800 msg = f"expected a string object, not {type(sub).__name__}"
3801 raise TypeError(msg)
3802
3803 result = self._data.array._str_rindex(sub, start=start, end=end)
3804 return self._wrap_result(result, returns_string=False)
3805
3806 def len(self):
3807 """
3808 Compute the length of each element in the Series/Index.
3809
3810 The element may be a sequence (such as a string, tuple or list) or a collection
3811 (such as a dictionary).
3812
3813 Returns
3814 -------
3815 Series or Index of int
3816 A Series or Index of integer values indicating the length of each
3817 element in the Series or Index.
3818
3819 See Also
3820 --------
3821 str.len : Python built-in function returning the length of an object.
3822 Series.size : Returns the length of the Series.
3823
3824 Examples
3825 --------
3826 Returns the length (number of characters) in a string. Returns the
3827 number of entries for dictionaries, lists or tuples.
3828
3829 >>> s = pd.Series(
3830 ... ["dog", "", 5, {"foo": "bar"}, [2, 3, 5, 7], ("one", "two", "three")]
3831 ... )
3832 >>> s
3833 0 dog
3834 1
3835 2 5
3836 3 {'foo': 'bar'}
3837 4 [2, 3, 5, 7]
3838 5 (one, two, three)
3839 dtype: object
3840 >>> s.str.len()
3841 0 3.0
3842 1 0.0
3843 2 NaN
3844 3 1.0
3845 4 4.0
3846 5 3.0
3847 dtype: float64
3848 """
3849 result = self._data.array._str_len()
3850 return self._wrap_result(result, returns_string=False)
3851
3852 @forbid_nonstring_types(["bytes"])
3853 def lower(self):
3854 """
3855 Convert strings in the Series/Index to lowercase.
3856
3857 Equivalent to :meth:`str.lower`.
3858
3859 Returns
3860 -------
3861 Series or Index of objects
3862 A Series or Index where the strings are modified by :meth:`str.lower`.
3863
3864 See Also
3865 --------
3866 Series.str.lower : Converts all characters to lowercase.
3867 Series.str.upper : Converts all characters to uppercase.
3868 Series.str.title : Converts first character of each word to uppercase and
3869 remaining to lowercase.
3870 Series.str.capitalize : Converts first character to uppercase and
3871 remaining to lowercase.
3872 Series.str.swapcase : Converts uppercase to lowercase and lowercase to
3873 uppercase.
3874 Series.str.casefold: Removes all case distinctions in the string.
3875
3876 Examples
3877 --------
3878 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"])
3879 >>> s
3880 0 lower
3881 1 CAPITALS
3882 2 this is a sentence
3883 3 SwApCaSe
3884 dtype: str
3885
3886 >>> s.str.lower()
3887 0 lower
3888 1 capitals
3889 2 this is a sentence
3890 3 swapcase
3891 dtype: str
3892
3893 >>> s.str.upper()
3894 0 LOWER
3895 1 CAPITALS
3896 2 THIS IS A SENTENCE
3897 3 SWAPCASE
3898 dtype: str
3899
3900 >>> s.str.title()
3901 0 Lower
3902 1 Capitals
3903 2 This Is A Sentence
3904 3 Swapcase
3905 dtype: str
3906
3907 >>> s.str.capitalize()
3908 0 Lower
3909 1 Capitals
3910 2 This is a sentence
3911 3 Swapcase
3912 dtype: str
3913
3914 >>> s.str.swapcase()
3915 0 LOWER
3916 1 capitals
3917 2 THIS IS A SENTENCE
3918 3 sWaPcAsE
3919 dtype: str
3920 """
3921 result = self._data.array._str_lower()
3922 return self._wrap_result(result)
3923
3924 @forbid_nonstring_types(["bytes"])
3925 def upper(self):
3926 """
3927 Convert strings in the Series/Index to uppercase.
3928
3929 Equivalent to :meth:`str.upper`.
3930
3931 Returns
3932 -------
3933 Series or Index of objects
3934 A Series or Index where the strings are modified by :meth:`str.upper`.
3935
3936 See Also
3937 --------
3938 Series.str.lower : Converts all characters to lowercase.
3939 Series.str.upper : Converts all characters to uppercase.
3940 Series.str.title : Converts first character of each word to uppercase and
3941 remaining to lowercase.
3942 Series.str.capitalize : Converts first character to uppercase and
3943 remaining to lowercase.
3944 Series.str.swapcase : Converts uppercase to lowercase and lowercase to
3945 uppercase.
3946 Series.str.casefold: Removes all case distinctions in the string.
3947
3948 Examples
3949 --------
3950 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"])
3951 >>> s
3952 0 lower
3953 1 CAPITALS
3954 2 this is a sentence
3955 3 SwApCaSe
3956 dtype: str
3957
3958 >>> s.str.lower()
3959 0 lower
3960 1 capitals
3961 2 this is a sentence
3962 3 swapcase
3963 dtype: str
3964
3965 >>> s.str.upper()
3966 0 LOWER
3967 1 CAPITALS
3968 2 THIS IS A SENTENCE
3969 3 SWAPCASE
3970 dtype: str
3971
3972 >>> s.str.title()
3973 0 Lower
3974 1 Capitals
3975 2 This Is A Sentence
3976 3 Swapcase
3977 dtype: str
3978
3979 >>> s.str.capitalize()
3980 0 Lower
3981 1 Capitals
3982 2 This is a sentence
3983 3 Swapcase
3984 dtype: str
3985
3986 >>> s.str.swapcase()
3987 0 LOWER
3988 1 capitals
3989 2 THIS IS A SENTENCE
3990 3 sWaPcAsE
3991 dtype: str
3992 """
3993 result = self._data.array._str_upper()
3994 return self._wrap_result(result)
3995
3996 @forbid_nonstring_types(["bytes"])
3997 def title(self):
3998 """
3999 Convert strings in the Series/Index to titlecase.
4000
4001 Equivalent to :meth:`str.title`.
4002
4003 Returns
4004 -------
4005 Series or Index of objects
4006 A Series or Index where the strings are modified by :meth:`str.title`.
4007
4008 See Also
4009 --------
4010 Series.str.lower : Converts all characters to lowercase.
4011 Series.str.upper : Converts all characters to uppercase.
4012 Series.str.title : Converts first character of each word to uppercase and
4013 remaining to lowercase.
4014 Series.str.capitalize : Converts first character to uppercase and
4015 remaining to lowercase.
4016 Series.str.swapcase : Converts uppercase to lowercase and lowercase to
4017 uppercase.
4018 Series.str.casefold: Removes all case distinctions in the string.
4019
4020 Examples
4021 --------
4022 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"])
4023 >>> s
4024 0 lower
4025 1 CAPITALS
4026 2 this is a sentence
4027 3 SwApCaSe
4028 dtype: str
4029
4030 >>> s.str.lower()
4031 0 lower
4032 1 capitals
4033 2 this is a sentence
4034 3 swapcase
4035 dtype: str
4036
4037 >>> s.str.upper()
4038 0 LOWER
4039 1 CAPITALS
4040 2 THIS IS A SENTENCE
4041 3 SWAPCASE
4042 dtype: str
4043
4044 >>> s.str.title()
4045 0 Lower
4046 1 Capitals
4047 2 This Is A Sentence
4048 3 Swapcase
4049 dtype: str
4050
4051 >>> s.str.capitalize()
4052 0 Lower
4053 1 Capitals
4054 2 This is a sentence
4055 3 Swapcase
4056 dtype: str
4057
4058 >>> s.str.swapcase()
4059 0 LOWER
4060 1 capitals
4061 2 THIS IS A SENTENCE
4062 3 sWaPcAsE
4063 dtype: str
4064 """
4065 result = self._data.array._str_title()
4066 return self._wrap_result(result)
4067
4068 @forbid_nonstring_types(["bytes"])
4069 def capitalize(self):
4070 """
4071 Convert strings in the Series/Index to be capitalized.
4072
4073 Equivalent to :meth:`str.capitalize`.
4074
4075 Returns
4076 -------
4077 Series or Index of objects
4078 A Series or Index where the strings are modified by :meth:`str.capitalize`.
4079
4080 See Also
4081 --------
4082 Series.str.lower : Converts all characters to lowercase.
4083 Series.str.upper : Converts all characters to uppercase.
4084 Series.str.title : Converts first character of each word to uppercase and
4085 remaining to lowercase.
4086 Series.str.capitalize : Converts first character to uppercase and
4087 remaining to lowercase.
4088 Series.str.swapcase : Converts uppercase to lowercase and lowercase to
4089 uppercase.
4090 Series.str.casefold: Removes all case distinctions in the string.
4091
4092 Examples
4093 --------
4094 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"])
4095 >>> s
4096 0 lower
4097 1 CAPITALS
4098 2 this is a sentence
4099 3 SwApCaSe
4100 dtype: str
4101
4102 >>> s.str.lower()
4103 0 lower
4104 1 capitals
4105 2 this is a sentence
4106 3 swapcase
4107 dtype: str
4108
4109 >>> s.str.upper()
4110 0 LOWER
4111 1 CAPITALS
4112 2 THIS IS A SENTENCE
4113 3 SWAPCASE
4114 dtype: str
4115
4116 >>> s.str.title()
4117 0 Lower
4118 1 Capitals
4119 2 This Is A Sentence
4120 3 Swapcase
4121 dtype: str
4122
4123 >>> s.str.capitalize()
4124 0 Lower
4125 1 Capitals
4126 2 This is a sentence
4127 3 Swapcase
4128 dtype: str
4129
4130 >>> s.str.swapcase()
4131 0 LOWER
4132 1 capitals
4133 2 THIS IS A SENTENCE
4134 3 sWaPcAsE
4135 dtype: str
4136 """
4137 result = self._data.array._str_capitalize()
4138 return self._wrap_result(result)
4139
4140 @forbid_nonstring_types(["bytes"])
4141 def swapcase(self):
4142 """
4143 Convert strings in the Series/Index to be swapcased.
4144
4145 Equivalent to :meth:`str.swapcase`.
4146
4147 Returns
4148 -------
4149 Series or Index of objects
4150 A Series or Index where the strings are modified by :meth:`str.swapcase`.
4151
4152 See Also
4153 --------
4154 Series.str.lower : Converts all characters to lowercase.
4155 Series.str.upper : Converts all characters to uppercase.
4156 Series.str.title : Converts first character of each word to uppercase and
4157 remaining to lowercase.
4158 Series.str.capitalize : Converts first character to uppercase and
4159 remaining to lowercase.
4160 Series.str.swapcase : Converts uppercase to lowercase and lowercase to
4161 uppercase.
4162 Series.str.casefold: Removes all case distinctions in the string.
4163
4164 Examples
4165 --------
4166 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"])
4167 >>> s
4168 0 lower
4169 1 CAPITALS
4170 2 this is a sentence
4171 3 SwApCaSe
4172 dtype: str
4173
4174 >>> s.str.lower()
4175 0 lower
4176 1 capitals
4177 2 this is a sentence
4178 3 swapcase
4179 dtype: str
4180
4181 >>> s.str.upper()
4182 0 LOWER
4183 1 CAPITALS
4184 2 THIS IS A SENTENCE
4185 3 SWAPCASE
4186 dtype: str
4187
4188 >>> s.str.title()
4189 0 Lower
4190 1 Capitals
4191 2 This Is A Sentence
4192 3 Swapcase
4193 dtype: str
4194
4195 >>> s.str.capitalize()
4196 0 Lower
4197 1 Capitals
4198 2 This is a sentence
4199 3 Swapcase
4200 dtype: str
4201
4202 >>> s.str.swapcase()
4203 0 LOWER
4204 1 capitals
4205 2 THIS IS A SENTENCE
4206 3 sWaPcAsE
4207 dtype: str
4208 """
4209 result = self._data.array._str_swapcase()
4210 return self._wrap_result(result)
4211
4212 @forbid_nonstring_types(["bytes"])
4213 def casefold(self):
4214 """
4215 Convert strings in the Series/Index to be casefolded.
4216
4217 Equivalent to :meth:`str.casefold`.
4218
4219 Returns
4220 -------
4221 Series or Index of objects
4222 A Series or Index where the strings are modified by :meth:`str.casefold`.
4223
4224 See Also
4225 --------
4226 Series.str.lower : Converts all characters to lowercase.
4227 Series.str.upper : Converts all characters to uppercase.
4228 Series.str.title : Converts first character of each word to uppercase and
4229 remaining to lowercase.
4230 Series.str.capitalize : Converts first character to uppercase and
4231 remaining to lowercase.
4232 Series.str.swapcase : Converts uppercase to lowercase and lowercase to
4233 uppercase.
4234 Series.str.casefold: Removes all case distinctions in the string.
4235
4236 Examples
4237 --------
4238 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"])
4239 >>> s
4240 0 lower
4241 1 CAPITALS
4242 2 this is a sentence
4243 3 SwApCaSe
4244 dtype: str
4245
4246 >>> s.str.lower()
4247 0 lower
4248 1 capitals
4249 2 this is a sentence
4250 3 swapcase
4251 dtype: str
4252
4253 >>> s.str.upper()
4254 0 LOWER
4255 1 CAPITALS
4256 2 THIS IS A SENTENCE
4257 3 SWAPCASE
4258 dtype: str
4259
4260 >>> s.str.title()
4261 0 Lower
4262 1 Capitals
4263 2 This Is A Sentence
4264 3 Swapcase
4265 dtype: str
4266
4267 >>> s.str.capitalize()
4268 0 Lower
4269 1 Capitals
4270 2 This is a sentence
4271 3 Swapcase
4272 dtype: str
4273
4274 >>> s.str.swapcase()
4275 0 LOWER
4276 1 capitals
4277 2 THIS IS A SENTENCE
4278 3 sWaPcAsE
4279 dtype: str
4280 """
4281 result = self._data.array._str_casefold()
4282 return self._wrap_result(result)
4283
4284 @forbid_nonstring_types(["bytes"])
4285 def isalnum(self):
4286 """
4287 Check whether all characters in each string are alphanumeric.
4288
4289 This is equivalent to running the Python string method
4290 :meth:`str.isalnum` for each element of the Series/Index. If a string
4291 has zero characters, ``False`` is returned for that check.
4292
4293 Returns
4294 -------
4295 Series or Index of bool
4296 Series or Index of boolean values with the same length as the original
4297 Series/Index.
4298
4299 See Also
4300 --------
4301 Series.str.isalpha : Check whether all characters are alphabetic.
4302 Series.str.isnumeric : Check whether all characters are numeric.
4303 Series.str.isdigit : Check whether all characters are digits.
4304 Series.str.isdecimal : Check whether all characters are decimal.
4305 Series.str.isspace : Check whether all characters are whitespace.
4306 Series.str.islower : Check whether all characters are lowercase.
4307 Series.str.isascii : Check whether all characters are ascii.
4308 Series.str.isupper : Check whether all characters are uppercase.
4309 Series.str.istitle : Check whether all characters are titlecase.
4310
4311 Examples
4312 --------
4313 >>> s1 = pd.Series(["one", "one1", "1", ""])
4314 >>> s1.str.isalnum()
4315 0 True
4316 1 True
4317 2 True
4318 3 False
4319 dtype: bool
4320
4321 Note that checks against characters mixed with any additional punctuation
4322 or whitespace will evaluate to false for an alphanumeric check.
4323
4324 >>> s2 = pd.Series(["A B", "1.5", "3,000"])
4325 >>> s2.str.isalnum()
4326 0 False
4327 1 False
4328 2 False
4329 dtype: bool
4330 """
4331 result = self._data.array._str_isalnum()
4332 return self._wrap_result(result, returns_string=False)
4333
4334 @forbid_nonstring_types(["bytes"])
4335 def isalpha(self):
4336 """
4337 Check whether all characters in each string are alphabetic.
4338
4339 This is equivalent to running the Python string method
4340 :meth:`str.isalpha` for each element of the Series/Index. If a string
4341 has zero characters, ``False`` is returned for that check.
4342
4343 Returns
4344 -------
4345 Series or Index of bool
4346 Series or Index of boolean values with the same length as the original
4347 Series/Index.
4348
4349 See Also
4350 --------
4351 Series.str.isnumeric : Check whether all characters are numeric.
4352 Series.str.isalnum : Check whether all characters are alphanumeric.
4353 Series.str.isdigit : Check whether all characters are digits.
4354 Series.str.isdecimal : Check whether all characters are decimal.
4355 Series.str.isspace : Check whether all characters are whitespace.
4356 Series.str.islower : Check whether all characters are lowercase.
4357 Series.str.isascii : Check whether all characters are ascii.
4358 Series.str.isupper : Check whether all characters are uppercase.
4359 Series.str.istitle : Check whether all characters are titlecase.
4360
4361 Examples
4362 --------
4363
4364 >>> s1 = pd.Series(["one", "one1", "1", ""])
4365 >>> s1.str.isalpha()
4366 0 True
4367 1 False
4368 2 False
4369 3 False
4370 dtype: bool
4371 """
4372 result = self._data.array._str_isalpha()
4373 return self._wrap_result(result, returns_string=False)
4374
4375 @forbid_nonstring_types(["bytes"])
4376 def isdigit(self):
4377 """
4378 Check whether all characters in each string are digits.
4379
4380 This is equivalent to running the Python string method
4381 :meth:`str.isdigit` for each element of the Series/Index. If a string
4382 has zero characters, ``False`` is returned for that check.
4383
4384 Returns
4385 -------
4386 Series or Index of bool
4387 Series or Index of boolean values with the same length as the original
4388 Series/Index.
4389
4390 See Also
4391 --------
4392 Series.str.isalpha : Check whether all characters are alphabetic.
4393 Series.str.isnumeric : Check whether all characters are numeric.
4394 Series.str.isalnum : Check whether all characters are alphanumeric.
4395 Series.str.isdecimal : Check whether all characters are decimal.
4396 Series.str.isspace : Check whether all characters are whitespace.
4397 Series.str.islower : Check whether all characters are lowercase.
4398 Series.str.isascii : Check whether all characters are ascii.
4399 Series.str.isupper : Check whether all characters are uppercase.
4400 Series.str.istitle : Check whether all characters are titlecase.
4401
4402 Notes
4403 -----
4404 Similar to ``str.isdecimal`` but also includes special digits, like
4405 superscripted and subscripted digits in unicode.
4406
4407 The exact behavior of this method, i.e. which unicode characters are
4408 considered as digits, depends on the backend used for string operations,
4409 and there can be small differences.
4410 For example, Python considers the ³ superscript character as a digit, but
4411 not the ⅕ fraction character, while PyArrow considers both as digits. For
4412 simple (ascii) decimal numbers, the behaviour is consistent.
4413
4414 Examples
4415 --------
4416
4417 >>> s3 = pd.Series(["23", "³", "⅕", ""])
4418 >>> s3.str.isdigit()
4419 0 True
4420 1 True
4421 2 True
4422 3 False
4423 dtype: bool
4424 """
4425 result = self._data.array._str_isdigit()
4426 return self._wrap_result(result, returns_string=False)
4427
4428 @forbid_nonstring_types(["bytes"])
4429 def isspace(self):
4430 """
4431 Check whether all characters in each string are whitespace.
4432
4433 This is equivalent to running the Python string method
4434 :meth:`str.isspace` for each element of the Series/Index. If a string
4435 has zero characters, ``False`` is returned for that check.
4436
4437 Returns
4438 -------
4439 Series or Index of bool
4440 Series or Index of boolean values with the same length as the original
4441 Series/Index.
4442
4443 See Also
4444 --------
4445 Series.str.isalpha : Check whether all characters are alphabetic.
4446 Series.str.isnumeric : Check whether all characters are numeric.
4447 Series.str.isalnum : Check whether all characters are alphanumeric.
4448 Series.str.isdigit : Check whether all characters are digits.
4449 Series.str.isdecimal : Check whether all characters are decimal.
4450 Series.str.islower : Check whether all characters are lowercase.
4451 Series.str.isascii : Check whether all characters are ascii.
4452 Series.str.isupper : Check whether all characters are uppercase.
4453 Series.str.istitle : Check whether all characters are titlecase.
4454
4455 Examples
4456 --------
4457
4458 >>> s4 = pd.Series([" ", "\\t\\r\\n ", ""])
4459 >>> s4.str.isspace()
4460 0 True
4461 1 True
4462 2 False
4463 dtype: bool
4464 """
4465 result = self._data.array._str_isspace()
4466 return self._wrap_result(result, returns_string=False)
4467
4468 @forbid_nonstring_types(["bytes"])
4469 def islower(self):
4470 """
4471 Check whether all characters in each string are lowercase.
4472
4473 This is equivalent to running the Python string method
4474 :meth:`str.islower` for each element of the Series/Index. If a string
4475 has zero characters, ``False`` is returned for that check.
4476
4477 Returns
4478 -------
4479 Series or Index of bool
4480 Series or Index of boolean values with the same length as the original
4481 Series/Index.
4482
4483 See Also
4484 --------
4485 Series.str.isalpha : Check whether all characters are alphabetic.
4486 Series.str.isnumeric : Check whether all characters are numeric.
4487 Series.str.isalnum : Check whether all characters are alphanumeric.
4488 Series.str.isdigit : Check whether all characters are digits.
4489 Series.str.isdecimal : Check whether all characters are decimal.
4490 Series.str.isspace : Check whether all characters are whitespace.
4491 Series.str.isascii : Check whether all characters are ascii.
4492 Series.str.isupper : Check whether all characters are uppercase.
4493 Series.str.istitle : Check whether all characters are titlecase.
4494
4495 Examples
4496 --------
4497
4498 >>> s5 = pd.Series(["leopard", "Golden Eagle", "SNAKE", ""])
4499 >>> s5.str.islower()
4500 0 True
4501 1 False
4502 2 False
4503 3 False
4504 dtype: bool
4505 """
4506 result = self._data.array._str_islower()
4507 return self._wrap_result(result, returns_string=False)
4508
4509 @forbid_nonstring_types(["bytes"])
4510 def isascii(self):
4511 """
4512 Check whether all characters in each string are ascii.
4513
4514 This is equivalent to running the Python string method
4515 :meth:`str.isascii` for each element of the Series/Index. If a string
4516 has zero characters, ``False`` is returned for that check.
4517
4518 Returns
4519 -------
4520 Series or Index of bool
4521 Series or Index of boolean values with the same length as the original
4522 Series/Index.
4523
4524 See Also
4525 --------
4526 Series.str.isalpha : Check whether all characters are alphabetic.
4527 Series.str.isnumeric : Check whether all characters are numeric.
4528 Series.str.isalnum : Check whether all characters are alphanumeric.
4529 Series.str.isdigit : Check whether all characters are digits.
4530 Series.str.isdecimal : Check whether all characters are decimal.
4531 Series.str.isspace : Check whether all characters are whitespace.
4532 Series.str.islower : Check whether all characters are lowercase.
4533 Series.str.isupper : Check whether all characters are uppercase.
4534 Series.str.istitle : Check whether all characters are titlecase.
4535
4536 Examples
4537 --------
4538 The ``s5.str.isascii`` method checks for whether all characters are ascii
4539 characters, which includes digits 0-9, capital and lowercase letters A-Z,
4540 and some other special characters.
4541
4542 >>> s5 = pd.Series(["ö", "see123", "hello world", ""])
4543 >>> s5.str.isascii()
4544 0 False
4545 1 True
4546 2 True
4547 3 True
4548 dtype: bool
4549 """
4550 result = self._data.array._str_isascii()
4551 return self._wrap_result(result, returns_string=False)
4552
4553 @forbid_nonstring_types(["bytes"])
4554 def isupper(self):
4555 """
4556 Check whether all characters in each string are uppercase.
4557
4558 This is equivalent to running the Python string method
4559 :meth:`str.isupper` for each element of the Series/Index. If a string
4560 has zero characters, ``False`` is returned for that check.
4561
4562 Returns
4563 -------
4564 Series or Index of bool
4565 Series or Index of boolean values with the same length as the original
4566 Series/Index.
4567
4568 See Also
4569 --------
4570 Series.str.isalpha : Check whether all characters are alphabetic.
4571 Series.str.isnumeric : Check whether all characters are numeric.
4572 Series.str.isalnum : Check whether all characters are alphanumeric.
4573 Series.str.isdigit : Check whether all characters are digits.
4574 Series.str.isdecimal : Check whether all characters are decimal.
4575 Series.str.isspace : Check whether all characters are whitespace.
4576 Series.str.islower : Check whether all characters are lowercase.
4577 Series.str.isascii : Check whether all characters are ascii.
4578 Series.str.istitle : Check whether all characters are titlecase.
4579
4580 Examples
4581 --------
4582
4583 >>> s5 = pd.Series(["leopard", "Golden Eagle", "SNAKE", ""])
4584 >>> s5.str.isupper()
4585 0 False
4586 1 False
4587 2 True
4588 3 False
4589 dtype: bool
4590 """
4591 result = self._data.array._str_isupper()
4592 return self._wrap_result(result, returns_string=False)
4593
4594 @forbid_nonstring_types(["bytes"])
4595 def istitle(self):
4596 """
4597 Check whether all characters in each string are titlecase.
4598
4599 This is equivalent to running the Python string method
4600 :meth:`str.istitle` for each element of the Series/Index. If a string
4601 has zero characters, ``False`` is returned for that check.
4602
4603 Returns
4604 -------
4605 Series or Index of bool
4606 Series or Index of boolean values with the same length as the original
4607 Series/Index.
4608
4609 See Also
4610 --------
4611 Series.str.isalpha : Check whether all characters are alphabetic.
4612 Series.str.isnumeric : Check whether all characters are numeric.
4613 Series.str.isalnum : Check whether all characters are alphanumeric.
4614 Series.str.isdigit : Check whether all characters are digits.
4615 Series.str.isdecimal : Check whether all characters are decimal.
4616 Series.str.isspace : Check whether all characters are whitespace.
4617 Series.str.islower : Check whether all characters are lowercase.
4618 Series.str.isascii : Check whether all characters are ascii.
4619 Series.str.isupper : Check whether all characters are uppercase.
4620
4621 Examples
4622 --------
4623 The ``s5.str.istitle`` method checks for whether all words are in title
4624 case (whether only the first letter of each word is capitalized). Words are
4625 assumed to be as any sequence of non-numeric characters separated by
4626 whitespace characters.
4627
4628 >>> s5 = pd.Series(["leopard", "Golden Eagle", "SNAKE", ""])
4629 >>> s5.str.istitle()
4630 0 False
4631 1 True
4632 2 False
4633 3 False
4634 dtype: bool
4635 """
4636 result = self._data.array._str_istitle()
4637 return self._wrap_result(result, returns_string=False)
4638
4639 @forbid_nonstring_types(["bytes"])
4640 def isnumeric(self):
4641 """
4642 Check whether all characters in each string are numeric.
4643
4644 This is equivalent to running the Python string method
4645 :meth:`str.isnumeric` for each element of the Series/Index. If a string
4646 has zero characters, ``False`` is returned for that check.
4647
4648 Returns
4649 -------
4650 Series or Index of bool
4651 Series or Index of boolean values with the same length as the original
4652 Series/Index.
4653
4654 See Also
4655 --------
4656 Series.str.isalpha : Check whether all characters are alphabetic.
4657 Series.str.isalnum : Check whether all characters are alphanumeric.
4658 Series.str.isdigit : Check whether all characters are digits.
4659 Series.str.isdecimal : Check whether all characters are decimal.
4660 Series.str.isspace : Check whether all characters are whitespace.
4661 Series.str.islower : Check whether all characters are lowercase.
4662 Series.str.isascii : Check whether all characters are ascii.
4663 Series.str.isupper : Check whether all characters are uppercase.
4664 Series.str.istitle : Check whether all characters are titlecase.
4665
4666 Examples
4667 --------
4668 The ``s.str.isnumeric`` method is the same as ``s3.str.isdigit`` but
4669 also includes other characters that can represent quantities such as
4670 unicode fractions.
4671
4672 >>> s1 = pd.Series(["one", "one1", "1", "", "³", "⅕"])
4673 >>> s1.str.isnumeric()
4674 0 False
4675 1 False
4676 2 True
4677 3 False
4678 4 True
4679 5 True
4680 dtype: bool
4681
4682 For a string to be considered numeric, all its characters must have a Unicode
4683 numeric property matching :py:meth:`str.is_numeric`. As a consequence,
4684 the following cases are **not** recognized as numeric:
4685
4686 - **Decimal numbers** (e.g., "1.1"): due to period ``"."``
4687 - **Negative numbers** (e.g., "-5"): due to minus sign ``"-"``
4688 - **Scientific notation** (e.g., "1e3"): due to characters like ``"e"``
4689
4690 >>> s2 = pd.Series(["1.1", "-5", "1e3"])
4691 >>> s2.str.isnumeric()
4692 0 False
4693 1 False
4694 2 False
4695 dtype: bool
4696 """
4697 result = self._data.array._str_isnumeric()
4698 return self._wrap_result(result, returns_string=False)
4699
4700 @forbid_nonstring_types(["bytes"])
4701 def isdecimal(self):
4702 """
4703 Check whether all characters in each string are decimal.
4704
4705 This is equivalent to running the Python string method
4706 :meth:`str.isdecimal` for each element of the Series/Index. If a string
4707 has zero characters, ``False`` is returned for that check.
4708
4709 Returns
4710 -------
4711 Series or Index of bool
4712 Series or Index of boolean values with the same length as the original
4713 Series/Index.
4714
4715 See Also
4716 --------
4717 Series.str.isalpha : Check whether all characters are alphabetic.
4718 Series.str.isnumeric : Check whether all characters are numeric.
4719 Series.str.isalnum : Check whether all characters are alphanumeric.
4720 Series.str.isdigit : Check whether all characters are digits.
4721 Series.str.isspace : Check whether all characters are whitespace.
4722 Series.str.islower : Check whether all characters are lowercase.
4723 Series.str.isascii : Check whether all characters are ascii.
4724 Series.str.isupper : Check whether all characters are uppercase.
4725 Series.str.istitle : Check whether all characters are titlecase.
4726
4727 Examples
4728 --------
4729 The ``s3.str.isdecimal`` method checks for characters used to form
4730 numbers in base 10.
4731
4732 >>> s3 = pd.Series(["23", "³", "⅕", ""])
4733 >>> s3.str.isdecimal()
4734 0 True
4735 1 False
4736 2 False
4737 3 False
4738 dtype: bool
4739 """
4740 result = self._data.array._str_isdecimal()
4741 return self._wrap_result(result, returns_string=False)
4742
4743
4744def cat_safe(list_of_columns: list[npt.NDArray[np.object_]], sep: str):
4745 """
4746 Auxiliary function for :meth:`str.cat`.
4747
4748 Same signature as cat_core, but handles TypeErrors in concatenation, which
4749 happen if the arrays in list_of columns have the wrong dtypes or content.
4750
4751 Parameters
4752 ----------
4753 list_of_columns : list of numpy arrays
4754 List of arrays to be concatenated with sep;
4755 these arrays may not contain NaNs!
4756 sep : string
4757 The separator string for concatenating the columns.
4758
4759 Returns
4760 -------
4761 nd.array
4762 The concatenation of list_of_columns with sep.
4763 """
4764 try:
4765 result = cat_core(list_of_columns, sep)
4766 except TypeError:
4767 # if there are any non-string values (wrong dtype or hidden behind
4768 # object dtype), np.sum will fail; catch and return with better message
4769 for column in list_of_columns:
4770 dtype = lib.infer_dtype(column, skipna=True)
4771 if dtype not in ["string", "empty"]:
4772 raise TypeError(
4773 "Concatenation requires list-likes containing only "
4774 "strings (or missing values). Offending values found in "
4775 f"column {dtype}"
4776 ) from None
4777 return result
4778
4779
4780def cat_core(list_of_columns: list, sep: str):
4781 """
4782 Auxiliary function for :meth:`str.cat`
4783
4784 Parameters
4785 ----------
4786 list_of_columns : list of numpy arrays
4787 List of arrays to be concatenated with sep;
4788 these arrays may not contain NaNs!
4789 sep : string
4790 The separator string for concatenating the columns.
4791
4792 Returns
4793 -------
4794 nd.array
4795 The concatenation of list_of_columns with sep.
4796 """
4797 if sep == "":
4798 # no need to interleave sep if it is empty
4799 arr_of_cols = np.asarray(list_of_columns, dtype=object)
4800 return np.sum(arr_of_cols, axis=0)
4801 list_with_sep = [sep] * (2 * len(list_of_columns) - 1)
4802 list_with_sep[::2] = list_of_columns
4803 arr_with_sep = np.asarray(list_with_sep, dtype=object)
4804 return np.sum(arr_with_sep, axis=0)
4805
4806
4807def _result_dtype(arr):
4808 # workaround #27953
4809 # ideally we just pass `dtype=arr.dtype` unconditionally, but this fails
4810 # when the list of values is empty.
4811 from pandas.core.arrays.string_ import StringDtype
4812
4813 if isinstance(arr.dtype, (ArrowDtype, StringDtype)):
4814 return arr.dtype
4815 return object
4816
4817
4818def _get_single_group_name(regex: re.Pattern) -> Hashable:
4819 if regex.groupindex:
4820 return next(iter(regex.groupindex))
4821 else:
4822 return None
4823
4824
4825def _get_group_names(regex: re.Pattern) -> list[Hashable] | range:
4826 """
4827 Get named groups from compiled regex.
4828
4829 Unnamed groups are numbered.
4830
4831 Parameters
4832 ----------
4833 regex : compiled regex
4834
4835 Returns
4836 -------
4837 list of column labels
4838 """
4839 rng = range(regex.groups)
4840 names = {v: k for k, v in regex.groupindex.items()}
4841 if not names:
4842 return rng
4843 result: list[Hashable] = [names.get(1 + i, i) for i in rng]
4844 arr = np.array(result)
4845 if arr.dtype.kind == "i" and lib.is_range_indexer(arr, len(arr)):
4846 return rng
4847 return result
4848
4849
4850def str_extractall(arr, pat, flags: int = 0) -> DataFrame:
4851 regex = re.compile(pat, flags=flags)
4852 # the regex must contain capture groups.
4853 if regex.groups == 0:
4854 raise ValueError("pattern contains no capture groups")
4855
4856 if isinstance(arr, ABCIndex):
4857 arr = arr.to_series().reset_index(drop=True).astype(arr.dtype)
4858
4859 columns = _get_group_names(regex)
4860 match_list = []
4861 index_list = []
4862 is_mi = arr.index.nlevels > 1
4863
4864 for subject_key, subject in arr.items():
4865 if isinstance(subject, str):
4866 if not is_mi:
4867 subject_key = (subject_key,)
4868
4869 for match_i, match_tuple in enumerate(regex.findall(subject)):
4870 if isinstance(match_tuple, str):
4871 match_tuple = (match_tuple,)
4872 na_tuple = [np.nan if group == "" else group for group in match_tuple]
4873 match_list.append(na_tuple)
4874 result_key = (*subject_key, match_i)
4875 index_list.append(result_key)
4876
4877 from pandas import MultiIndex
4878
4879 index = MultiIndex.from_tuples(index_list, names=[*arr.index.names, "match"])
4880 dtype = _result_dtype(arr)
4881
4882 result = arr._constructor_expanddim(
4883 match_list, index=index, columns=columns, dtype=dtype
4884 )
4885 return result