Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/pandas/core/strings/accessor.py: 25%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

615 statements  

1from __future__ import annotations 

2 

3import codecs 

4from functools import wraps 

5import re 

6from typing import ( 

7 TYPE_CHECKING, 

8 Literal, 

9 cast, 

10) 

11import warnings 

12 

13import numpy as np 

14 

15from pandas._config import using_string_dtype 

16 

17from pandas._libs import lib 

18from pandas._typing import ( 

19 AlignJoin, 

20 DtypeObj, 

21 F, 

22 Scalar, 

23 npt, 

24) 

25from pandas.util._exceptions import find_stack_level 

26 

27from pandas.core.dtypes.common import ( 

28 ensure_object, 

29 is_bool_dtype, 

30 is_extension_array_dtype, 

31 is_integer, 

32 is_list_like, 

33 is_numeric_dtype, 

34 is_object_dtype, 

35 is_re, 

36 is_string_dtype, 

37) 

38from pandas.core.dtypes.dtypes import ( 

39 ArrowDtype, 

40 CategoricalDtype, 

41) 

42from pandas.core.dtypes.generic import ( 

43 ABCDataFrame, 

44 ABCIndex, 

45 ABCMultiIndex, 

46 ABCSeries, 

47) 

48from pandas.core.dtypes.missing import isna 

49 

50from pandas.core.arrays import ExtensionArray 

51from pandas.core.base import NoNewAttributesMixin 

52from pandas.core.construction import extract_array 

53 

54if TYPE_CHECKING: 

55 from collections.abc import ( 

56 Callable, 

57 Hashable, 

58 Iterator, 

59 ) 

60 

61 from pandas._typing import NpDtype 

62 

63 from pandas import ( 

64 DataFrame, 

65 Index, 

66 Series, 

67 ) 

68 

69_cpython_optimized_encoders = ( 

70 "utf-8", 

71 "utf8", 

72 "latin-1", 

73 "latin1", 

74 "iso-8859-1", 

75 "mbcs", 

76 "ascii", 

77) 

78_cpython_optimized_decoders = (*_cpython_optimized_encoders, "utf-16", "utf-32") 

79 

80 

81def forbid_nonstring_types( 

82 forbidden: list[str] | None, name: str | None = None 

83) -> Callable[[F], F]: 

84 """ 

85 Decorator to forbid specific types for a method of StringMethods. 

86 

87 For calling `.str.{method}` on a Series or Index, it is necessary to first 

88 initialize the :class:`StringMethods` object, and then call the method. 

89 However, different methods allow different input types, and so this can not 

90 be checked during :meth:`StringMethods.__init__`, but must be done on a 

91 per-method basis. This decorator exists to facilitate this process, and 

92 make it explicit which (inferred) types are disallowed by the method. 

93 

94 :meth:`StringMethods.__init__` allows the *union* of types its different 

95 methods allow (after skipping NaNs; see :meth:`StringMethods._validate`), 

96 namely: ['string', 'empty', 'bytes', 'mixed', 'mixed-integer']. 

97 

98 The default string types ['string', 'empty'] are allowed for all methods. 

99 For the additional types ['bytes', 'mixed', 'mixed-integer'], each method 

100 then needs to forbid the types it is not intended for. 

101 

102 Parameters 

103 ---------- 

104 forbidden : list-of-str or None 

105 List of forbidden non-string types, may be one or more of 

106 `['bytes', 'mixed', 'mixed-integer']`. 

107 name : str, default None 

108 Name of the method to use in the error message. By default, this is 

109 None, in which case the name from the method being wrapped will be 

110 copied. However, for working with further wrappers (like _pat_wrapper 

111 and _noarg_wrapper), it is necessary to specify the name. 

112 

113 Returns 

114 ------- 

115 func : wrapper 

116 The method to which the decorator is applied, with an added check that 

117 enforces the inferred type to not be in the list of forbidden types. 

118 

119 Raises 

120 ------ 

121 TypeError 

122 If the inferred type of the underlying data is in `forbidden`. 

123 """ 

124 # deal with None 

125 forbidden = [] if forbidden is None else forbidden 

126 

127 allowed_types = {"string", "empty", "bytes", "mixed", "mixed-integer"} - set( 

128 forbidden 

129 ) 

130 

131 def _forbid_nonstring_types(func: F) -> F: 

132 func_name = func.__name__ if name is None else name 

133 

134 @wraps(func) 

135 def wrapper(self, *args, **kwargs): 

136 if self._inferred_dtype not in allowed_types: 

137 msg = ( 

138 f"Cannot use .str.{func_name} with values of " 

139 f"inferred dtype '{self._inferred_dtype}'." 

140 ) 

141 raise TypeError(msg) 

142 return func(self, *args, **kwargs) 

143 

144 wrapper.__name__ = func_name 

145 return cast(F, wrapper) 

146 

147 return _forbid_nonstring_types 

148 

149 

150class StringMethods(NoNewAttributesMixin): 

151 """ 

152 Vectorized string functions for Series and Index. 

153 

154 NAs stay NA unless handled otherwise by a particular method. 

155 Patterned after Python's string methods, with some inspiration from 

156 R's stringr package. 

157 

158 Parameters 

159 ---------- 

160 data : Series or Index 

161 The content of the Series or Index. 

162 

163 See Also 

164 -------- 

165 Series.str : Vectorized string functions for Series. 

166 Index.str : Vectorized string functions for Index. 

167 

168 Examples 

169 -------- 

170 >>> s = pd.Series(["A_Str_Series"]) 

171 >>> s 

172 0 A_Str_Series 

173 dtype: str 

174 

175 >>> s.str.split("_") 

176 0 [A, Str, Series] 

177 dtype: object 

178 

179 >>> s.str.replace("_", "") 

180 0 AStrSeries 

181 dtype: str 

182 """ 

183 

184 # Note: see the docstring in pandas.core.strings.__init__ 

185 # for an explanation of the implementation. 

186 # TODO: Dispatch all the methods 

187 # Currently the following are not dispatched to the array 

188 # * cat 

189 # * extractall 

190 

191 def __init__(self, data) -> None: 

192 from pandas.core.arrays.string_ import StringDtype 

193 

194 self._inferred_dtype = self._validate(data) 

195 self._is_categorical = isinstance(data.dtype, CategoricalDtype) 

196 self._is_string = isinstance(data.dtype, StringDtype) 

197 self._data = data 

198 

199 self._index = self._name = None 

200 if isinstance(data, ABCSeries): 

201 self._index = data.index 

202 self._name = data.name 

203 

204 # ._values.categories works for both Series/Index 

205 self._parent = data._values.categories if self._is_categorical else data 

206 # save orig to blow up categoricals to the right type 

207 self._orig = data 

208 self._freeze() 

209 

210 @staticmethod 

211 def _validate(data): 

212 """ 

213 Auxiliary function for StringMethods, infers and checks dtype of data. 

214 

215 This is a "first line of defence" at the creation of the StringMethods- 

216 object, and just checks that the dtype is in the 

217 *union* of the allowed types over all string methods below; this 

218 restriction is then refined on a per-method basis using the decorator 

219 @forbid_nonstring_types (more info in the corresponding docstring). 

220 

221 This really should exclude all series/index with any non-string values, 

222 but that isn't practical for performance reasons until we have a str 

223 dtype (GH 9343 / 13877) 

224 

225 Parameters 

226 ---------- 

227 data : The content of the Series 

228 

229 Returns 

230 ------- 

231 dtype : inferred dtype of data 

232 """ 

233 if isinstance(data, ABCMultiIndex): 

234 raise AttributeError( 

235 "Can only use .str accessor with Index, not MultiIndex" 

236 ) 

237 

238 # see _libs/lib.pyx for list of inferred types 

239 allowed_types = ["string", "empty", "bytes", "mixed", "mixed-integer"] 

240 

241 data = extract_array(data) 

242 

243 values = getattr(data, "categories", data) # categorical / normal 

244 

245 inferred_dtype = lib.infer_dtype(values, skipna=True) 

246 

247 if inferred_dtype not in allowed_types: 

248 raise AttributeError( 

249 f"Can only use .str accessor with string values, not {inferred_dtype}" 

250 ) 

251 return inferred_dtype 

252 

253 def __getitem__(self, key): 

254 result = self._data.array._str_getitem(key) 

255 return self._wrap_result(result) 

256 

257 def __iter__(self) -> Iterator: 

258 raise TypeError(f"'{type(self).__name__}' object is not iterable") 

259 

260 def _wrap_result( 

261 self, 

262 result, 

263 name=None, 

264 expand: bool | None = None, 

265 fill_value=np.nan, 

266 returns_string: bool = True, 

267 dtype=None, 

268 ): 

269 from pandas import ( 

270 Index, 

271 MultiIndex, 

272 ) 

273 

274 if not hasattr(result, "ndim") or not hasattr(result, "dtype"): 

275 if isinstance(result, ABCDataFrame): 

276 result = result.__finalize__(self._orig, name="str") 

277 return result 

278 assert result.ndim < 3 

279 

280 # We can be wrapping a string / object / categorical result, in which 

281 # case we'll want to return the same dtype as the input. 

282 # Or we can be wrapping a numeric output, in which case we don't want 

283 # to return a StringArray. 

284 # Ideally the array method returns the right array type. 

285 if expand is None: 

286 # infer from ndim if expand is not specified 

287 expand = result.ndim != 1 

288 elif expand is True and not isinstance(self._orig, ABCIndex): 

289 # required when expand=True is explicitly specified 

290 # not needed when inferred 

291 if isinstance(result.dtype, ArrowDtype): 

292 import pyarrow as pa 

293 

294 from pandas.core.arrays.arrow.array import ArrowExtensionArray 

295 

296 value_lengths = pa.compute.list_value_length(result._pa_array) 

297 max_len = pa.compute.max(value_lengths).as_py() 

298 min_len = pa.compute.min(value_lengths).as_py() 

299 if result._hasna: 

300 # ArrowExtensionArray.fillna doesn't work for list scalars 

301 result = ArrowExtensionArray( 

302 result._pa_array.fill_null([None] * max_len) 

303 ) 

304 if min_len < max_len: 

305 # append nulls to each scalar list element up to max_len 

306 result = ArrowExtensionArray( 

307 pa.compute.list_slice( 

308 result._pa_array, 

309 start=0, 

310 stop=max_len, 

311 return_fixed_size_list=True, 

312 ) 

313 ) 

314 if name is None: 

315 name = range(max_len) 

316 result = ( 

317 pa.compute.list_flatten(result._pa_array) 

318 .to_numpy() 

319 .reshape(len(result), max_len) 

320 ) 

321 result = { 

322 label: ArrowExtensionArray(pa.array(res)) 

323 for label, res in zip(name, result.T, strict=True) 

324 } 

325 elif is_object_dtype(result): 

326 

327 def cons_row(x): 

328 if is_list_like(x): 

329 return x 

330 else: 

331 return [x] 

332 

333 result = [cons_row(x) for x in result] 

334 if result and not self._is_string: 

335 # propagate nan values to match longest sequence (GH 18450) 

336 max_len = max(len(x) for x in result) 

337 result = [ 

338 x * max_len if len(x) == 0 or x[0] is np.nan else x 

339 for x in result 

340 ] 

341 

342 if not isinstance(expand, bool): 

343 raise ValueError("expand must be True or False") 

344 

345 if expand is False: 

346 # if expand is False, result should have the same name 

347 # as the original otherwise specified 

348 if name is None: 

349 name = getattr(result, "name", None) 

350 if name is None: 

351 # do not use logical or, _orig may be a DataFrame 

352 # which has "name" column 

353 name = self._orig.name 

354 

355 # Wait until we are sure result is a Series or Index before 

356 # checking attributes (GH 12180) 

357 if isinstance(self._orig, ABCIndex): 

358 # if result is a boolean np.array, return the np.array 

359 # instead of wrapping it into a boolean Index (GH 8875) 

360 if is_bool_dtype(result): 

361 return result 

362 

363 if expand: 

364 result = list(result) 

365 out: Index = MultiIndex.from_tuples(result, names=name) 

366 if out.nlevels == 1: 

367 # We had all tuples of length-one, which are 

368 # better represented as a regular Index. 

369 out = out.get_level_values(0) 

370 return out 

371 else: 

372 return Index(result, name=name, dtype=dtype, copy=False) 

373 else: 

374 index = self._orig.index 

375 # This is a mess. 

376 _dtype: DtypeObj | str | None = dtype 

377 vdtype = getattr(result, "dtype", None) 

378 if _dtype is not None: 

379 pass 

380 elif self._is_string: 

381 if is_bool_dtype(vdtype): 

382 _dtype = result.dtype 

383 elif returns_string: 

384 _dtype = self._orig.dtype 

385 else: 

386 _dtype = vdtype 

387 elif vdtype is not None: 

388 _dtype = vdtype 

389 

390 if expand: 

391 cons = self._orig._constructor_expanddim 

392 result = cons(result, columns=name, index=index, dtype=_dtype) 

393 else: 

394 # Must be a Series 

395 cons = self._orig._constructor 

396 result = cons(result, name=name, index=index, dtype=_dtype) 

397 result = result.__finalize__(self._orig, method="str") 

398 if name is not None and result.ndim == 1: 

399 # __finalize__ might copy over the original name, but we may 

400 # want the new name (e.g. str.extract). 

401 result.name = name 

402 return result 

403 

404 def _get_series_list(self, others): 

405 """ 

406 Auxiliary function for :meth:`str.cat`. Turn potentially mixed input 

407 into a list of Series (elements without an index must match the length 

408 of the calling Series/Index). 

409 

410 Parameters 

411 ---------- 

412 others : Series, DataFrame, np.ndarray, list-like or list-like of 

413 Objects that are either Series, Index or np.ndarray (1-dim). 

414 

415 Returns 

416 ------- 

417 list of Series 

418 Others transformed into list of Series. 

419 """ 

420 from pandas import ( 

421 DataFrame, 

422 Series, 

423 ) 

424 

425 # self._orig is either Series or Index 

426 idx = self._orig if isinstance(self._orig, ABCIndex) else self._orig.index 

427 

428 # Generally speaking, all objects without an index inherit the index 

429 # `idx` of the calling Series/Index - i.e. must have matching length. 

430 # Objects with an index (i.e. Series/Index/DataFrame) keep their own. 

431 if isinstance(others, ABCSeries): 

432 return [others] 

433 elif isinstance(others, ABCIndex): 

434 return [Series(others, index=idx, dtype=others.dtype)] 

435 elif isinstance(others, ABCDataFrame): 

436 return [others[x] for x in others] 

437 elif isinstance(others, np.ndarray) and others.ndim == 2: 

438 others = DataFrame(others, index=idx) 

439 return [others[x] for x in others] 

440 elif is_list_like(others, allow_sets=False): 

441 try: 

442 others = list(others) # ensure iterators do not get read twice etc 

443 except TypeError: 

444 # e.g. ser.str, raise below 

445 pass 

446 else: 

447 # in case of list-like `others`, all elements must be 

448 # either Series/Index/np.ndarray (1-dim)... 

449 if all( 

450 isinstance(x, (ABCSeries, ABCIndex, ExtensionArray)) 

451 or (isinstance(x, np.ndarray) and x.ndim == 1) 

452 for x in others 

453 ): 

454 los: list[Series] = [] 

455 while others: # iterate through list and append each element 

456 los = los + self._get_series_list(others.pop(0)) 

457 return los 

458 # ... or just strings 

459 elif all(not is_list_like(x) for x in others): 

460 return [Series(others, index=idx)] 

461 raise TypeError( 

462 "others must be Series, Index, DataFrame, np.ndarray " 

463 "or list-like (either containing only strings or " 

464 "containing only objects of type Series/Index/" 

465 "np.ndarray[1-dim])" 

466 ) 

467 

468 @forbid_nonstring_types(["bytes", "mixed", "mixed-integer"]) 

469 def cat( 

470 self, 

471 others=None, 

472 sep: str | None = None, 

473 na_rep=None, 

474 join: AlignJoin = "left", 

475 ) -> str | Series | Index: 

476 """ 

477 Concatenate strings in the Series/Index with given separator. 

478 

479 If `others` is specified, this function concatenates the Series/Index 

480 and elements of `others` element-wise. 

481 If `others` is not passed, then all values in the Series/Index are 

482 concatenated into a single string with a given `sep`. 

483 

484 Parameters 

485 ---------- 

486 others : Series, Index, DataFrame, np.ndarray or list-like 

487 Series, Index, DataFrame, np.ndarray (one- or two-dimensional) and 

488 other list-likes of strings must have the same length as the 

489 calling Series/Index, with the exception of indexed objects (i.e. 

490 Series/Index/DataFrame) if `join` is not None. 

491 

492 If others is a list-like that contains a combination of Series, 

493 Index or np.ndarray (1-dim), then all elements will be unpacked and 

494 must satisfy the above criteria individually. 

495 

496 If others is None, the method returns the concatenation of all 

497 strings in the calling Series/Index. 

498 sep : str, default '' 

499 The separator between the different elements/columns. By default 

500 the empty string `''` is used. 

501 na_rep : str or None, default None 

502 Representation that is inserted for all missing values: 

503 

504 - If `na_rep` is None, and `others` is None, missing values in the 

505 Series/Index are omitted from the result. 

506 - If `na_rep` is None, and `others` is not None, a row containing a 

507 missing value in any of the columns (before concatenation) will 

508 have a missing value in the result. 

509 join : {'left', 'right', 'outer', 'inner'}, default 'left' 

510 Determines the join-style between the calling Series/Index and any 

511 Series/Index/DataFrame in `others` (objects without an index need 

512 to match the length of the calling Series/Index). To disable 

513 alignment, use `.values` on any Series/Index/DataFrame in `others`. 

514 

515 Returns 

516 ------- 

517 str, Series or Index 

518 If `others` is None, `str` is returned, otherwise a `Series/Index` 

519 (same type as caller) of objects is returned. 

520 

521 See Also 

522 -------- 

523 split : Split each string in the Series/Index. 

524 join : Join lists contained as elements in the Series/Index. 

525 

526 Examples 

527 -------- 

528 When not passing `others`, all values are concatenated into a single 

529 string: 

530 

531 >>> s = pd.Series(["a", "b", np.nan, "d"]) 

532 >>> s.str.cat(sep=" ") 

533 'a b d' 

534 

535 By default, NA values in the Series are ignored. Using `na_rep`, they 

536 can be given a representation: 

537 

538 >>> s.str.cat(sep=" ", na_rep="?") 

539 'a b ? d' 

540 

541 If `others` is specified, corresponding values are concatenated with 

542 the separator. Result will be a Series of strings. 

543 

544 >>> s.str.cat(["A", "B", "C", "D"], sep=",") 

545 0 a,A 

546 1 b,B 

547 2 NaN 

548 3 d,D 

549 dtype: str 

550 

551 Missing values will remain missing in the result, but can again be 

552 represented using `na_rep` 

553 

554 >>> s.str.cat(["A", "B", "C", "D"], sep=",", na_rep="-") 

555 0 a,A 

556 1 b,B 

557 2 -,C 

558 3 d,D 

559 dtype: str 

560 

561 If `sep` is not specified, the values are concatenated without 

562 separation. 

563 

564 >>> s.str.cat(["A", "B", "C", "D"], na_rep="-") 

565 0 aA 

566 1 bB 

567 2 -C 

568 3 dD 

569 dtype: str 

570 

571 Series with different indexes can be aligned before concatenation. The 

572 `join`-keyword works as in other methods. 

573 

574 >>> t = pd.Series(["d", "a", "e", "c"], index=[3, 0, 4, 2]) 

575 >>> s.str.cat(t, join="left", na_rep="-") 

576 0 aa 

577 1 b- 

578 2 -c 

579 3 dd 

580 dtype: str 

581 >>> 

582 >>> s.str.cat(t, join="outer", na_rep="-") 

583 0 aa 

584 1 b- 

585 2 -c 

586 3 dd 

587 4 -e 

588 dtype: str 

589 >>> 

590 >>> s.str.cat(t, join="inner", na_rep="-") 

591 0 aa 

592 2 -c 

593 3 dd 

594 dtype: str 

595 >>> 

596 >>> s.str.cat(t, join="right", na_rep="-") 

597 3 dd 

598 0 aa 

599 4 -e 

600 2 -c 

601 dtype: str 

602 

603 For more examples, see :ref:`here <text.concatenate>`. 

604 """ 

605 # TODO: dispatch 

606 from pandas import ( 

607 Index, 

608 Series, 

609 concat, 

610 ) 

611 

612 if isinstance(others, str): 

613 raise ValueError("Did you mean to supply a `sep` keyword?") 

614 if sep is None: 

615 sep = "" 

616 

617 if isinstance(self._orig, ABCIndex): 

618 data = Series(self._orig, index=self._orig, dtype=self._orig.dtype) 

619 else: # Series 

620 data = self._orig 

621 

622 # concatenate Series/Index with itself if no "others" 

623 if others is None: 

624 # error: Incompatible types in assignment (expression has type 

625 # "ndarray", variable has type "Series") 

626 data = ensure_object(data) # type: ignore[assignment] 

627 na_mask = isna(data) 

628 if na_rep is None and na_mask.any(): 

629 return sep.join(data[~na_mask]) 

630 elif na_rep is not None and na_mask.any(): 

631 return sep.join(np.where(na_mask, na_rep, data)) 

632 else: 

633 return sep.join(data) 

634 

635 try: 

636 # turn anything in "others" into lists of Series 

637 others = self._get_series_list(others) 

638 except ValueError as err: # do not catch TypeError raised by _get_series_list 

639 raise ValueError( 

640 "If `others` contains arrays or lists (or other " 

641 "list-likes without an index), these must all be " 

642 "of the same length as the calling Series/Index." 

643 ) from err 

644 

645 # align if required 

646 if any(not data.index.equals(x.index) for x in others): 

647 # Need to add keys for uniqueness in case of duplicate columns 

648 others = concat( 

649 others, 

650 axis=1, 

651 join=(join if join == "inner" else "outer"), 

652 keys=range(len(others)), 

653 sort=False, 

654 ) 

655 data, others = data.align(others, join=join) 

656 others = [others[x] for x in others] # again list of Series 

657 

658 all_cols = [ensure_object(x) for x in [data, *others]] 

659 na_masks = np.array([isna(x) for x in all_cols]) 

660 union_mask = np.logical_or.reduce(na_masks, axis=0) 

661 

662 if na_rep is None and union_mask.any(): 

663 # no na_rep means NaNs for all rows where any column has a NaN 

664 # only necessary if there are actually any NaNs 

665 result = np.empty(len(data), dtype=object) 

666 np.putmask(result, union_mask, np.nan) 

667 

668 not_masked = ~union_mask 

669 result[not_masked] = cat_safe([x[not_masked] for x in all_cols], sep) 

670 elif na_rep is not None and union_mask.any(): 

671 # fill NaNs with na_rep in case there are actually any NaNs 

672 all_cols = [ 

673 np.where(nm, na_rep, col) 

674 for nm, col in zip(na_masks, all_cols, strict=True) 

675 ] 

676 result = cat_safe(all_cols, sep) 

677 else: 

678 # no NaNs - can just concatenate 

679 result = cat_safe(all_cols, sep) 

680 

681 out: Index | Series 

682 if isinstance(self._orig.dtype, CategoricalDtype): 

683 # We need to infer the new categories. 

684 dtype = self._orig.dtype.categories.dtype 

685 else: 

686 dtype = self._orig.dtype 

687 if isinstance(self._orig, ABCIndex): 

688 # add dtype for case that result is all-NA 

689 if isna(result).all(): 

690 dtype = object # type: ignore[assignment] 

691 

692 out = Index(result, dtype=dtype, name=self._orig.name, copy=False) 

693 else: # Series 

694 res_ser = Series( 

695 result, dtype=dtype, index=data.index, name=self._orig.name, copy=False 

696 ) 

697 out = res_ser.__finalize__(self._orig, method="str_cat") 

698 return out 

699 

700 @forbid_nonstring_types(["bytes"]) 

701 def split( 

702 self, 

703 pat: str | re.Pattern | None = None, 

704 *, 

705 n=-1, 

706 expand: bool = False, 

707 regex: bool | None = None, 

708 ): 

709 r""" 

710 Split strings around given separator/delimiter. 

711 

712 Splits the string in the Series/Index from the beginning, 

713 at the specified delimiter string. 

714 

715 Parameters 

716 ---------- 

717 pat : str or compiled regex, optional 

718 String or regular expression to split on. 

719 If not specified, split on whitespace. 

720 n : int, default -1 (all) 

721 Limit number of splits in output. 

722 ``None``, 0 and -1 will be interpreted as return all splits. 

723 expand : bool, default False 

724 Expand the split strings into separate columns. 

725 

726 - If ``True``, return DataFrame/MultiIndex expanding dimensionality. 

727 - If ``False``, return Series/Index, containing lists of strings. 

728 

729 regex : bool, default None 

730 Determines if the passed-in pattern is a regular expression: 

731 

732 - If ``True``, assumes the passed-in pattern is a regular expression 

733 - If ``False``, treats the pattern as a literal string. 

734 - If ``None`` and `pat` length is 1, treats `pat` as a literal string. 

735 - If ``None`` and `pat` length is not 1, treats `pat` as a regular 

736 expression. 

737 - Cannot be set to False if `pat` is a compiled regex 

738 

739 Returns 

740 ------- 

741 Series, Index, DataFrame or MultiIndex 

742 Type matches caller unless ``expand=True`` (see Notes). 

743 

744 Raises 

745 ------ 

746 ValueError 

747 * if `regex` is False and `pat` is a compiled regex 

748 

749 See Also 

750 -------- 

751 Series.str.split : Split strings around given separator/delimiter. 

752 Series.str.rsplit : Splits string around given separator/delimiter, 

753 starting from the right. 

754 Series.str.join : Join lists contained as elements in the Series/Index 

755 with passed delimiter. 

756 str.split : Standard library version for split. 

757 str.rsplit : Standard library version for rsplit. 

758 

759 Notes 

760 ----- 

761 The handling of the `n` keyword depends on the number of found splits: 

762 

763 - If found splits > `n`, make first `n` splits only 

764 - If found splits <= `n`, make all splits 

765 - If for a certain row the number of found splits < `n`, 

766 append `None` for padding up to `n` if ``expand=True`` 

767 

768 If using ``expand=True``, Series and Index callers return DataFrame and 

769 MultiIndex objects, respectively. 

770 

771 Use of `regex =False` with a `pat` as a compiled regex will raise an error. 

772 

773 Examples 

774 -------- 

775 >>> s = pd.Series( 

776 ... [ 

777 ... "this is a regular sentence", 

778 ... "https://docs.python.org/3/tutorial/index.html", 

779 ... np.nan, 

780 ... ] 

781 ... ) 

782 >>> s 

783 0 this is a regular sentence 

784 1 https://docs.python.org/3/tutorial/index.html 

785 2 NaN 

786 dtype: str 

787 

788 In the default setting, the string is split by whitespace. 

789 

790 >>> s.str.split() 

791 0 [this, is, a, regular, sentence] 

792 1 [https://docs.python.org/3/tutorial/index.html] 

793 2 NaN 

794 dtype: object 

795 

796 Without the `n` parameter, the outputs of `rsplit` and `split` 

797 are identical. 

798 

799 >>> s.str.rsplit() 

800 0 [this, is, a, regular, sentence] 

801 1 [https://docs.python.org/3/tutorial/index.html] 

802 2 NaN 

803 dtype: object 

804 

805 The `n` parameter can be used to limit the number of splits on the 

806 delimiter. The outputs of `split` and `rsplit` are different. 

807 

808 >>> s.str.split(n=2) 

809 0 [this, is, a regular sentence] 

810 1 [https://docs.python.org/3/tutorial/index.html] 

811 2 NaN 

812 dtype: object 

813 

814 >>> s.str.rsplit(n=2) 

815 0 [this is a, regular, sentence] 

816 1 [https://docs.python.org/3/tutorial/index.html] 

817 2 NaN 

818 dtype: object 

819 

820 The `pat` parameter can be used to split by other characters. 

821 

822 >>> s.str.split(pat="/") 

823 0 [this is a regular sentence] 

824 1 [https:, , docs.python.org, 3, tutorial, index... 

825 2 NaN 

826 dtype: object 

827 

828 When using ``expand=True``, the split elements will expand out into 

829 separate columns. If NaN is present, it is propagated throughout 

830 the columns during the split. 

831 

832 >>> s.str.split(expand=True) 

833 0 1 2 3 4 

834 0 this is a regular sentence 

835 1 https://docs.python.org/3/tutorial/index.html NaN NaN NaN NaN 

836 2 NaN NaN NaN NaN NaN 

837 

838 For slightly more complex use cases like splitting the html document name 

839 from a url, a combination of parameter settings can be used. 

840 

841 >>> s.str.rsplit("/", n=1, expand=True) 

842 0 1 

843 0 this is a regular sentence NaN 

844 1 https://docs.python.org/3/tutorial index.html 

845 2 NaN NaN 

846 

847 Remember to escape special characters when explicitly using regular expressions. 

848 

849 >>> s = pd.Series(["foo and bar plus baz"]) 

850 >>> s.str.split(r"and|plus", expand=True) 

851 0 1 2 

852 0 foo bar baz 

853 

854 Regular expressions can be used to handle urls or file names. 

855 When `pat` is a string and ``regex=None`` (the default), the given `pat` is 

856 compiled as a regex only if ``len(pat) != 1``. 

857 

858 >>> s = pd.Series(["foojpgbar.jpg"]) 

859 >>> s.str.split(r".", expand=True) 

860 0 1 

861 0 foojpgbar jpg 

862 

863 >>> s.str.split(r"\.jpg", expand=True) 

864 0 1 

865 0 foojpgbar 

866 

867 When ``regex=True``, `pat` is interpreted as a regex 

868 

869 >>> s.str.split(r"\.jpg", regex=True, expand=True) 

870 0 1 

871 0 foojpgbar 

872 

873 A compiled regex can be passed as `pat` 

874 

875 >>> import re 

876 >>> s.str.split(re.compile(r"\.jpg"), expand=True) 

877 0 1 

878 0 foojpgbar 

879 

880 When ``regex=False``, `pat` is interpreted as the string itself 

881 

882 >>> s.str.split(r"\.jpg", regex=False, expand=True) 

883 0 

884 0 foojpgbar.jpg 

885 """ 

886 if regex is False and is_re(pat): 

887 raise ValueError( 

888 "Cannot use a compiled regex as replacement pattern with regex=False" 

889 ) 

890 if is_re(pat): 

891 regex = True 

892 result = self._data.array._str_split(pat, n, expand, regex) 

893 if self._data.dtype == "category": 

894 dtype = self._data.dtype.categories.dtype if expand else object 

895 else: 

896 dtype = object if self._data.dtype == object else None 

897 return self._wrap_result( 

898 result, expand=expand, returns_string=expand, dtype=dtype 

899 ) 

900 

901 @forbid_nonstring_types(["bytes"]) 

902 def rsplit(self, pat=None, *, n=-1, expand: bool = False): 

903 """ 

904 Split strings around given separator/delimiter. 

905 

906 Splits the string in the Series/Index from the end, 

907 at the specified delimiter string. 

908 

909 Parameters 

910 ---------- 

911 pat : str, optional 

912 String to split on. 

913 If not specified, split on whitespace. 

914 n : int, default -1 (all) 

915 Limit number of splits in output. 

916 ``None``, 0 and -1 will be interpreted as return all splits. 

917 expand : bool, default False 

918 Expand the split strings into separate columns. 

919 

920 - If ``True``, return DataFrame/MultiIndex expanding dimensionality. 

921 - If ``False``, return Series/Index, containing lists of strings. 

922 

923 Returns 

924 ------- 

925 Series, Index, DataFrame or MultiIndex 

926 Type matches caller unless ``expand=True`` (see Notes). 

927 

928 See Also 

929 -------- 

930 Series.str.split : Split strings around given separator/delimiter. 

931 Series.str.rsplit : Splits string around given separator/delimiter, 

932 starting from the right. 

933 Series.str.join : Join lists contained as elements in the Series/Index 

934 with passed delimiter. 

935 str.split : Standard library version for split. 

936 str.rsplit : Standard library version for rsplit. 

937 

938 Notes 

939 ----- 

940 The handling of the `n` keyword depends on the number of found splits: 

941 

942 - If found splits > `n`, make first `n` splits only 

943 - If found splits <= `n`, make all splits 

944 - If for a certain row the number of found splits < `n`, 

945 append `None` for padding up to `n` if ``expand=True`` 

946 

947 If using ``expand=True``, Series and Index callers return DataFrame and 

948 MultiIndex objects, respectively. 

949 

950 Examples 

951 -------- 

952 >>> s = pd.Series( 

953 ... [ 

954 ... "this is a regular sentence", 

955 ... "https://docs.python.org/3/tutorial/index.html", 

956 ... np.nan, 

957 ... ] 

958 ... ) 

959 >>> s 

960 0 this is a regular sentence 

961 1 https://docs.python.org/3/tutorial/index.html 

962 2 NaN 

963 dtype: str 

964 

965 In the default setting, the string is split by whitespace. 

966 

967 >>> s.str.split() 

968 0 [this, is, a, regular, sentence] 

969 1 [https://docs.python.org/3/tutorial/index.html] 

970 2 NaN 

971 dtype: object 

972 

973 Without the `n` parameter, the outputs of `rsplit` and `split` 

974 are identical. 

975 

976 >>> s.str.rsplit() 

977 0 [this, is, a, regular, sentence] 

978 1 [https://docs.python.org/3/tutorial/index.html] 

979 2 NaN 

980 dtype: object 

981 

982 The `n` parameter can be used to limit the number of splits on the 

983 delimiter. The outputs of `split` and `rsplit` are different. 

984 

985 >>> s.str.split(n=2) 

986 0 [this, is, a regular sentence] 

987 1 [https://docs.python.org/3/tutorial/index.html] 

988 2 NaN 

989 dtype: object 

990 

991 >>> s.str.rsplit(n=2) 

992 0 [this is a, regular, sentence] 

993 1 [https://docs.python.org/3/tutorial/index.html] 

994 2 NaN 

995 dtype: object 

996 

997 The `pat` parameter can be used to split by other characters. 

998 

999 >>> s.str.split(pat="/") 

1000 0 [this is a regular sentence] 

1001 1 [https:, , docs.python.org, 3, tutorial, index... 

1002 2 NaN 

1003 dtype: object 

1004 

1005 When using ``expand=True``, the split elements will expand out into 

1006 separate columns. If NaN is present, it is propagated throughout 

1007 the columns during the split. 

1008 

1009 >>> s.str.split(expand=True) 

1010 0 1 2 3 4 

1011 0 this is a regular sentence 

1012 1 https://docs.python.org/3/tutorial/index.html NaN NaN NaN NaN 

1013 2 NaN NaN NaN NaN NaN 

1014 

1015 For slightly more complex use cases like splitting the html document name 

1016 from a url, a combination of parameter settings can be used. 

1017 

1018 >>> s.str.rsplit("/", n=1, expand=True) 

1019 0 1 

1020 0 this is a regular sentence NaN 

1021 1 https://docs.python.org/3/tutorial index.html 

1022 2 NaN NaN 

1023 """ 

1024 result = self._data.array._str_rsplit(pat, n=n) 

1025 if self._data.dtype == "category": 

1026 dtype = self._data.dtype.categories.dtype if expand else object 

1027 else: 

1028 dtype = object if self._data.dtype == object else None 

1029 return self._wrap_result( 

1030 result, expand=expand, returns_string=expand, dtype=dtype 

1031 ) 

1032 

1033 @forbid_nonstring_types(["bytes"]) 

1034 def partition(self, sep: str = " ", expand: bool = True): 

1035 """ 

1036 Split the string at the first occurrence of `sep`. 

1037 

1038 This method splits the string at the first occurrence of `sep`, 

1039 and returns 3 elements containing the part before the separator, 

1040 the separator itself, and the part after the separator. 

1041 If the separator is not found, return 3 elements containing the string itself, 

1042 followed by two empty strings. 

1043 

1044 Parameters 

1045 ---------- 

1046 sep : str, default whitespace 

1047 String to split on. 

1048 expand : bool, default True 

1049 If True, return DataFrame/MultiIndex expanding dimensionality. 

1050 If False, return Series/Index. 

1051 

1052 Returns 

1053 ------- 

1054 DataFrame/MultiIndex or Series/Index of objects 

1055 Returns appropriate type based on `expand` parameter with strings 

1056 split based on the `sep` parameter. 

1057 

1058 See Also 

1059 -------- 

1060 rpartition : Split the string at the last occurrence of `sep`. 

1061 Series.str.split : Split strings around given separators. 

1062 str.partition : Standard library version. 

1063 

1064 Examples 

1065 -------- 

1066 >>> s = pd.Series(["Linda van der Berg", "George Pitt-Rivers"]) 

1067 >>> s 

1068 0 Linda van der Berg 

1069 1 George Pitt-Rivers 

1070 dtype: str 

1071 

1072 >>> s.str.partition() 

1073 0 1 2 

1074 0 Linda van der Berg 

1075 1 George Pitt-Rivers 

1076 

1077 To partition by the last space instead of the first one: 

1078 

1079 >>> s.str.rpartition() 

1080 0 1 2 

1081 0 Linda van der Berg 

1082 1 George Pitt-Rivers 

1083 

1084 To partition by something different than a space: 

1085 

1086 >>> s.str.partition("-") 

1087 0 1 2 

1088 0 Linda van der Berg 

1089 1 George Pitt - Rivers 

1090 

1091 To return a Series containing tuples instead of a DataFrame: 

1092 

1093 >>> s.str.partition("-", expand=False) 

1094 0 (Linda van der Berg, , ) 

1095 1 (George Pitt, -, Rivers) 

1096 dtype: object 

1097 

1098 Also available on indices: 

1099 

1100 >>> idx = pd.Index(["X 123", "Y 999"]) 

1101 >>> idx 

1102 Index(['X 123', 'Y 999'], dtype='str') 

1103 

1104 Which will create a MultiIndex: 

1105 

1106 >>> idx.str.partition() 

1107 MultiIndex([('X', ' ', '123'), 

1108 ('Y', ' ', '999')], 

1109 ) 

1110 

1111 Or an index with tuples with ``expand=False``: 

1112 

1113 >>> idx.str.partition(expand=False) 

1114 Index([('X', ' ', '123'), ('Y', ' ', '999')], dtype='object') 

1115 """ 

1116 result = self._data.array._str_partition(sep, expand) 

1117 if self._data.dtype == "category": 

1118 dtype = self._data.dtype.categories.dtype if expand else object 

1119 else: 

1120 dtype = object if self._data.dtype == object else None 

1121 return self._wrap_result( 

1122 result, expand=expand, returns_string=expand, dtype=dtype 

1123 ) 

1124 

1125 @forbid_nonstring_types(["bytes"]) 

1126 def rpartition(self, sep: str = " ", expand: bool = True): 

1127 """ 

1128 Split the string at the last occurrence of `sep`. 

1129 

1130 This method splits the string at the last occurrence of `sep`, 

1131 and returns 3 elements containing the part before the separator, 

1132 the separator itself, and the part after the separator. 

1133 If the separator is not found, return 3 elements containing two empty strings, 

1134 followed by the string itself. 

1135 

1136 Parameters 

1137 ---------- 

1138 sep : str, default " " 

1139 String to split on. 

1140 expand : bool, default True 

1141 If True, return DataFrame/MultiIndex expanding dimensionality. 

1142 If False, return Series/Index. 

1143 

1144 Returns 

1145 ------- 

1146 DataFrame/MultiIndex or Series/Index of objects 

1147 Returns appropriate type based on `expand` parameter with strings 

1148 split based on the `sep` parameter. 

1149 

1150 See Also 

1151 -------- 

1152 partition : Split the string at the first occurrence of `sep`. 

1153 Series.str.split : Split strings around given separators. 

1154 str.partition : Standard library version. 

1155 

1156 Examples 

1157 -------- 

1158 >>> s = pd.Series(["Linda van der Berg", "George Pitt-Rivers"]) 

1159 >>> s 

1160 0 Linda van der Berg 

1161 1 George Pitt-Rivers 

1162 dtype: str 

1163 

1164 >>> s.str.partition() 

1165 0 1 2 

1166 0 Linda van der Berg 

1167 1 George Pitt-Rivers 

1168 

1169 To partition by the last space instead of the first one: 

1170 

1171 >>> s.str.rpartition() 

1172 0 1 2 

1173 0 Linda van der Berg 

1174 1 George Pitt-Rivers 

1175 

1176 To partition by something different than a space: 

1177 

1178 >>> s.str.partition("-") 

1179 0 1 2 

1180 0 Linda van der Berg 

1181 1 George Pitt - Rivers 

1182 

1183 To return a Series containing tuples instead of a DataFrame: 

1184 

1185 >>> s.str.partition("-", expand=False) 

1186 0 (Linda van der Berg, , ) 

1187 1 (George Pitt, -, Rivers) 

1188 dtype: object 

1189 

1190 Also available on indices: 

1191 

1192 >>> idx = pd.Index(["X 123", "Y 999"]) 

1193 >>> idx 

1194 Index(['X 123', 'Y 999'], dtype='str') 

1195 

1196 Which will create a MultiIndex: 

1197 

1198 >>> idx.str.partition() 

1199 MultiIndex([('X', ' ', '123'), 

1200 ('Y', ' ', '999')], 

1201 ) 

1202 

1203 Or an index with tuples with ``expand=False``: 

1204 

1205 >>> idx.str.partition(expand=False) 

1206 Index([('X', ' ', '123'), ('Y', ' ', '999')], dtype='object') 

1207 """ 

1208 result = self._data.array._str_rpartition(sep, expand) 

1209 if self._data.dtype == "category": 

1210 dtype = self._data.dtype.categories.dtype if expand else object 

1211 else: 

1212 dtype = object if self._data.dtype == object else None 

1213 return self._wrap_result( 

1214 result, expand=expand, returns_string=expand, dtype=dtype 

1215 ) 

1216 

1217 def get(self, i): 

1218 """ 

1219 Extract element from each component at specified position or with specified key. 

1220 

1221 Extract element from lists, tuples, dict, or strings in each element in the 

1222 Series/Index. 

1223 

1224 Parameters 

1225 ---------- 

1226 i : int or hashable dict label 

1227 Position or key of element to extract. 

1228 

1229 Returns 

1230 ------- 

1231 Series or Index 

1232 Series or Index where each value is the extracted element from 

1233 the corresponding input component. 

1234 

1235 See Also 

1236 -------- 

1237 Series.str.extract : Extract capture groups in the regex as columns 

1238 in a DataFrame. 

1239 

1240 Examples 

1241 -------- 

1242 >>> s = pd.Series( 

1243 ... [ 

1244 ... "String", 

1245 ... (1, 2, 3), 

1246 ... ["a", "b", "c"], 

1247 ... 123, 

1248 ... -456, 

1249 ... {1: "Hello", "2": "World"}, 

1250 ... ] 

1251 ... ) 

1252 >>> s 

1253 0 String 

1254 1 (1, 2, 3) 

1255 2 [a, b, c] 

1256 3 123 

1257 4 -456 

1258 5 {1: 'Hello', '2': 'World'} 

1259 dtype: object 

1260 

1261 >>> s.str.get(1) 

1262 0 t 

1263 1 2 

1264 2 b 

1265 3 NaN 

1266 4 NaN 

1267 5 Hello 

1268 dtype: object 

1269 

1270 >>> s.str.get(-1) 

1271 0 g 

1272 1 3 

1273 2 c 

1274 3 NaN 

1275 4 NaN 

1276 5 None 

1277 dtype: object 

1278 

1279 Return element with given key 

1280 

1281 >>> s = pd.Series( 

1282 ... [ 

1283 ... {"name": "Hello", "value": "World"}, 

1284 ... {"name": "Goodbye", "value": "Planet"}, 

1285 ... ] 

1286 ... ) 

1287 >>> s.str.get("name") 

1288 0 Hello 

1289 1 Goodbye 

1290 dtype: object 

1291 """ 

1292 result = self._data.array._str_get(i) 

1293 return self._wrap_result(result) 

1294 

1295 @forbid_nonstring_types(["bytes"]) 

1296 def join(self, sep: str): 

1297 """ 

1298 Join lists contained as elements in the Series/Index with passed delimiter. 

1299 

1300 If the elements of a Series are lists themselves, join the content of these 

1301 lists using the delimiter passed to the function. 

1302 This function is an equivalent to :meth:`str.join`. 

1303 

1304 Parameters 

1305 ---------- 

1306 sep : str 

1307 Delimiter to use between list entries. 

1308 

1309 Returns 

1310 ------- 

1311 Series/Index: object 

1312 The list entries concatenated by intervening occurrences of the 

1313 delimiter. 

1314 

1315 Raises 

1316 ------ 

1317 AttributeError 

1318 If the supplied Series contains neither strings nor lists. 

1319 

1320 See Also 

1321 -------- 

1322 str.join : Standard library version of this method. 

1323 Series.str.split : Split strings around given separator/delimiter. 

1324 

1325 Notes 

1326 ----- 

1327 If any of the list items is not a string object, the result of the join 

1328 will be `NaN`. 

1329 

1330 Examples 

1331 -------- 

1332 Example with a list that contains non-string elements. 

1333 

1334 >>> s = pd.Series( 

1335 ... [ 

1336 ... ["lion", "elephant", "zebra"], 

1337 ... [1.1, 2.2, 3.3], 

1338 ... ["cat", np.nan, "dog"], 

1339 ... ["cow", 4.5, "goat"], 

1340 ... ["duck", ["swan", "fish"], "guppy"], 

1341 ... ] 

1342 ... ) 

1343 >>> s 

1344 0 [lion, elephant, zebra] 

1345 1 [1.1, 2.2, 3.3] 

1346 2 [cat, nan, dog] 

1347 3 [cow, 4.5, goat] 

1348 4 [duck, [swan, fish], guppy] 

1349 dtype: object 

1350 

1351 Join all lists using a '-'. The lists containing object(s) of types other 

1352 than str will produce a NaN. 

1353 

1354 >>> s.str.join("-") 

1355 0 lion-elephant-zebra 

1356 1 NaN 

1357 2 NaN 

1358 3 NaN 

1359 4 NaN 

1360 dtype: object 

1361 """ 

1362 result = self._data.array._str_join(sep) 

1363 return self._wrap_result(result) 

1364 

1365 @forbid_nonstring_types(["bytes"]) 

1366 def contains( 

1367 self, 

1368 pat, 

1369 case: bool = True, 

1370 flags: int = 0, 

1371 na=lib.no_default, 

1372 regex: bool = True, 

1373 ): 

1374 r""" 

1375 Test if pattern or regex is contained within a string of a Series or Index. 

1376 

1377 Return boolean Series or Index based on whether a given pattern or regex is 

1378 contained within a string of a Series or Index. 

1379 

1380 Parameters 

1381 ---------- 

1382 pat : str 

1383 Character sequence or regular expression. 

1384 case : bool, default True 

1385 If True, case sensitive. 

1386 flags : int, default 0 (no flags) 

1387 Flags to pass through to the re module, e.g. re.IGNORECASE. 

1388 na : scalar, optional 

1389 Fill value for missing values. The default depends on dtype of the 

1390 array. For the ``"str"`` dtype, ``False`` is used. For object 

1391 dtype, ``numpy.nan`` is used. For the nullable ``StringDtype``, 

1392 ``pandas.NA`` is used. 

1393 regex : bool, default True 

1394 If True, assumes the pat is a regular expression. 

1395 

1396 If False, treats the pat as a literal string. 

1397 

1398 Returns 

1399 ------- 

1400 Series or Index of boolean values 

1401 A Series or Index of boolean values indicating whether the 

1402 given pattern is contained within the string of each element 

1403 of the Series or Index. 

1404 

1405 See Also 

1406 -------- 

1407 match : Analogous, but stricter, relying on re.match instead of re.search. 

1408 Series.str.startswith : Test if the start of each string element matches a 

1409 pattern. 

1410 Series.str.endswith : Same as startswith, but tests the end of string. 

1411 

1412 Examples 

1413 -------- 

1414 Returning a Series of booleans using only a literal pattern. 

1415 

1416 >>> s1 = pd.Series(["Mouse", "dog", "house and parrot", "23", np.nan]) 

1417 >>> s1.str.contains("og", regex=False) 

1418 0 False 

1419 1 True 

1420 2 False 

1421 3 False 

1422 4 False 

1423 dtype: bool 

1424 

1425 Returning an Index of booleans using only a literal pattern. 

1426 

1427 >>> ind = pd.Index(["Mouse", "dog", "house and parrot", "23.0", np.nan]) 

1428 >>> ind.str.contains("23", regex=False) 

1429 array([False, False, False, True, False]) 

1430 

1431 Specifying case sensitivity using `case`. 

1432 

1433 >>> s1.str.contains("oG", case=True, regex=True) 

1434 0 False 

1435 1 False 

1436 2 False 

1437 3 False 

1438 4 False 

1439 dtype: bool 

1440 

1441 Returning 'house' or 'dog' when either expression occurs in a string. 

1442 

1443 >>> s1.str.contains("house|dog", regex=True) 

1444 0 False 

1445 1 True 

1446 2 True 

1447 3 False 

1448 4 False 

1449 dtype: bool 

1450 

1451 Ignoring case sensitivity using `flags` with regex. 

1452 

1453 >>> import re 

1454 >>> s1.str.contains("PARROT", flags=re.IGNORECASE, regex=True) 

1455 0 False 

1456 1 False 

1457 2 True 

1458 3 False 

1459 4 False 

1460 dtype: bool 

1461 

1462 Returning any digit using regular expression. 

1463 

1464 >>> s1.str.contains("\\d", regex=True) 

1465 0 False 

1466 1 False 

1467 2 False 

1468 3 True 

1469 4 False 

1470 dtype: bool 

1471 

1472 Ensure `pat` is a not a literal pattern when `regex` is set to True. 

1473 Note in the following example one might expect only `s2[1]` and `s2[3]` to 

1474 return `True`. However, '.0' as a regex matches any character 

1475 followed by a 0. 

1476 

1477 >>> s2 = pd.Series(["40", "40.0", "41", "41.0", "35"]) 

1478 >>> s2.str.contains(".0", regex=True) 

1479 0 True 

1480 1 True 

1481 2 False 

1482 3 True 

1483 4 False 

1484 dtype: bool 

1485 """ 

1486 if regex: 

1487 try: 

1488 has_groups = re.compile(pat).groups 

1489 except re.error: 

1490 has_groups = False 

1491 if has_groups: 

1492 warnings.warn( 

1493 "This pattern is interpreted as a regular expression, and has " 

1494 "match groups. To actually get the groups, use str.extract.", 

1495 UserWarning, 

1496 stacklevel=find_stack_level(), 

1497 ) 

1498 

1499 result = self._data.array._str_contains(pat, case, flags, na, regex) 

1500 return self._wrap_result(result, fill_value=na, returns_string=False) 

1501 

1502 @forbid_nonstring_types(["bytes"]) 

1503 def match( 

1504 self, 

1505 pat: str | re.Pattern, 

1506 case: bool | lib.NoDefault = lib.no_default, 

1507 flags: int | lib.NoDefault = lib.no_default, 

1508 na=lib.no_default, 

1509 ): 

1510 """ 

1511 Determine if each string starts with a match of a regular expression. 

1512 

1513 Determines whether each string in the Series or Index starts with a 

1514 match to a specified regular expression. This function is especially 

1515 useful for validating prefixes, such as ensuring that codes, tags, or 

1516 identifiers begin with a specific pattern. 

1517 

1518 Parameters 

1519 ---------- 

1520 pat : str or compiled regex 

1521 Character sequence or regular expression. 

1522 case : bool, default True 

1523 If True, case sensitive. 

1524 flags : int, default 0 (no flags) 

1525 Regex module flags, e.g. re.IGNORECASE. 

1526 na : scalar, optional 

1527 Fill value for missing values. The default depends on dtype of the 

1528 array. For the ``"str"`` dtype, ``False`` is used. For object 

1529 dtype, ``numpy.nan`` is used. For the nullable ``StringDtype``, 

1530 ``pandas.NA`` is used. 

1531 

1532 Returns 

1533 ------- 

1534 Series/Index/array of boolean values 

1535 A Series, Index, or array of boolean values indicating whether the start 

1536 of each string matches the pattern. The result will be of the same type 

1537 as the input. 

1538 

1539 See Also 

1540 -------- 

1541 fullmatch : Stricter matching that requires the entire string to match. 

1542 contains : Analogous, but less strict, relying on re.search instead of 

1543 re.match. 

1544 extract : Extract matched groups. 

1545 

1546 Examples 

1547 -------- 

1548 >>> ser = pd.Series(["horse", "eagle", "donkey"]) 

1549 >>> ser.str.match("e") 

1550 0 False 

1551 1 True 

1552 2 False 

1553 dtype: bool 

1554 """ 

1555 if flags is not lib.no_default: 

1556 # pat.flags will have re.U regardless, so we need to add it here 

1557 # before checking for a match 

1558 flags = flags | re.U 

1559 if is_re(pat): 

1560 if pat.flags != flags: 

1561 raise ValueError( 

1562 "Cannot both specify 'flags' and pass a compiled regexp " 

1563 "object with conflicting flags" 

1564 ) 

1565 else: 

1566 pat = re.compile(pat, flags=flags) 

1567 # set flags=0 to ensure that when we call 

1568 # re.compile(pat, flags=flags) the constructor does not raise. 

1569 flags = 0 

1570 else: 

1571 flags = 0 

1572 

1573 if case is lib.no_default: 

1574 if is_re(pat): 

1575 case = not bool(pat.flags & re.IGNORECASE) 

1576 else: 

1577 # Case-sensitive default 

1578 case = True 

1579 elif is_re(pat): 

1580 implicit_case = not bool(pat.flags & re.IGNORECASE) 

1581 if implicit_case != case: 

1582 # GH#62240 

1583 raise ValueError( 

1584 "Cannot both specify 'case' and pass a compiled regexp " 

1585 "object with conflicting case-sensitivity" 

1586 ) 

1587 

1588 result = self._data.array._str_match(pat, case=case, flags=flags, na=na) 

1589 return self._wrap_result(result, fill_value=na, returns_string=False) 

1590 

1591 @forbid_nonstring_types(["bytes"]) 

1592 def fullmatch(self, pat, case: bool = True, flags: int = 0, na=lib.no_default): 

1593 """ 

1594 Determine if each string entirely matches a regular expression. 

1595 

1596 Checks if each string in the Series or Index fully matches the 

1597 specified regular expression pattern. This function is useful when the 

1598 requirement is for an entire string to conform to a pattern, such as 

1599 validating formats like phone numbers or email addresses. 

1600 

1601 Parameters 

1602 ---------- 

1603 pat : str 

1604 Character sequence or regular expression. 

1605 case : bool, default True 

1606 If True, case sensitive. 

1607 flags : int, default 0 (no flags) 

1608 Regex module flags, e.g. re.IGNORECASE. 

1609 na : scalar, optional 

1610 Fill value for missing values. The default depends on dtype of the 

1611 array. For the ``"str"`` dtype, ``False`` is used. For object 

1612 dtype, ``numpy.nan`` is used. For the nullable ``StringDtype``, 

1613 ``pandas.NA`` is used. 

1614 

1615 Returns 

1616 ------- 

1617 Series/Index/array of boolean values 

1618 The function returns a Series, Index, or array of boolean values, 

1619 where True indicates that the entire string matches the regular 

1620 expression pattern and False indicates that it does not. 

1621 

1622 See Also 

1623 -------- 

1624 match : Similar, but also returns `True` when only a *prefix* of the string 

1625 matches the regular expression. 

1626 extract : Extract matched groups. 

1627 

1628 Examples 

1629 -------- 

1630 >>> ser = pd.Series(["cat", "duck", "dove"]) 

1631 >>> ser.str.fullmatch(r"d.+") 

1632 0 False 

1633 1 True 

1634 2 True 

1635 dtype: bool 

1636 """ 

1637 result = self._data.array._str_fullmatch(pat, case=case, flags=flags, na=na) 

1638 return self._wrap_result(result, fill_value=na, returns_string=False) 

1639 

1640 @forbid_nonstring_types(["bytes"]) 

1641 def replace( 

1642 self, 

1643 pat: str | re.Pattern | dict, 

1644 repl: str | Callable | None = None, 

1645 n: int = -1, 

1646 case: bool | None = None, 

1647 flags: int = 0, 

1648 regex: bool = False, 

1649 ): 

1650 r""" 

1651 Replace each occurrence of pattern/regex in the Series/Index. 

1652 

1653 Equivalent to :meth:`str.replace` or :func:`re.sub`, depending on 

1654 the regex value. 

1655 

1656 Parameters 

1657 ---------- 

1658 pat : str, compiled regex, or a dict 

1659 String can be a character sequence or regular expression. 

1660 Dictionary contains <key : value> pairs of strings to be replaced 

1661 along with the updated value. 

1662 repl : str or callable 

1663 Replacement string or a callable. The callable is passed the regex 

1664 match object and must return a replacement string to be used. 

1665 Must have a value of None if `pat` is a dict 

1666 See :func:`re.sub`. 

1667 n : int, default -1 (all) 

1668 Number of replacements to make from start. 

1669 case : bool, default None 

1670 Determines if replace is case sensitive: 

1671 

1672 - If True, case sensitive (the default if `pat` is a string) 

1673 - Set to False for case insensitive 

1674 - Cannot be set if `pat` is a compiled regex. 

1675 

1676 flags : int, default 0 (no flags) 

1677 Regex module flags, e.g. re.IGNORECASE. Cannot be set if `pat` is a compiled 

1678 regex. 

1679 regex : bool, default False 

1680 Determines if the passed-in pattern is a regular expression: 

1681 

1682 - If True, assumes the passed-in pattern is a regular expression. 

1683 - If False, treats the pattern as a literal string 

1684 - Cannot be set to False if `pat` is a compiled regex or `repl` is 

1685 a callable. 

1686 

1687 Returns 

1688 ------- 

1689 Series or Index of object 

1690 A copy of the object with all matching occurrences of `pat` replaced by 

1691 `repl`. 

1692 

1693 Raises 

1694 ------ 

1695 ValueError 

1696 * if `regex` is False and `repl` is a callable or `pat` is a compiled 

1697 regex 

1698 * if `pat` is a compiled regex and `case` or `flags` is set 

1699 * if `pat` is a dictionary and `repl` is not None. 

1700 

1701 See Also 

1702 -------- 

1703 Series.str.replace : Method to replace occurrences of a substring with another 

1704 substring. 

1705 Series.str.extract : Extract substrings using a regular expression. 

1706 Series.str.findall : Find all occurrences of a pattern or regex in each string. 

1707 Series.str.split : Split each string by a specified delimiter or pattern. 

1708 

1709 Notes 

1710 ----- 

1711 When `pat` is a compiled regex, all flags should be included in the 

1712 compiled regex. Use of `case`, `flags`, or `regex=False` with a compiled 

1713 regex will raise an error. 

1714 

1715 Examples 

1716 -------- 

1717 When `pat` is a dictionary, every key in `pat` is replaced 

1718 with its corresponding value: 

1719 

1720 >>> pd.Series(["A", "B", np.nan]).str.replace(pat={"A": "a", "B": "b"}) 

1721 0 a 

1722 1 b 

1723 2 NaN 

1724 dtype: str 

1725 

1726 When `pat` is a string and `regex` is True, the given `pat` 

1727 is compiled as a regex. When `repl` is a string, it replaces matching 

1728 regex patterns as with :meth:`re.sub`. NaN value(s) in the Series are 

1729 left as is: 

1730 

1731 >>> pd.Series(["foo", "fuz", np.nan]).str.replace("f.", "ba", regex=True) 

1732 0 bao 

1733 1 baz 

1734 2 NaN 

1735 dtype: str 

1736 

1737 When `pat` is a string and `regex` is False, every `pat` is replaced with 

1738 `repl` as with :meth:`str.replace`: 

1739 

1740 >>> pd.Series(["f.o", "fuz", np.nan]).str.replace("f.", "ba", regex=False) 

1741 0 bao 

1742 1 fuz 

1743 2 NaN 

1744 dtype: str 

1745 

1746 When `repl` is a callable, it is called on every `pat` using 

1747 :func:`re.sub`. The callable should expect one positional argument 

1748 (a regex object) and return a string. 

1749 

1750 To get the idea: 

1751 

1752 >>> pd.Series(["foo", "fuz", np.nan]).str.replace("f", repr, regex=True) 

1753 0 <re.Match object; span=(0, 1), match='f'>oo 

1754 1 <re.Match object; span=(0, 1), match='f'>uz 

1755 2 NaN 

1756 dtype: str 

1757 

1758 Reverse every lowercase alphabetic word: 

1759 

1760 >>> repl = lambda m: m.group(0)[::-1] 

1761 >>> ser = pd.Series(["foo 123", "bar baz", np.nan]) 

1762 >>> ser.str.replace(r"[a-z]+", repl, regex=True) 

1763 0 oof 123 

1764 1 rab zab 

1765 2 NaN 

1766 dtype: str 

1767 

1768 Using regex groups (extract second group and swap case): 

1769 

1770 >>> pat = r"(?P<one>\w+) (?P<two>\w+) (?P<three>\w+)" 

1771 >>> repl = lambda m: m.group("two").swapcase() 

1772 >>> ser = pd.Series(["One Two Three", "Foo Bar Baz"]) 

1773 >>> ser.str.replace(pat, repl, regex=True) 

1774 0 tWO 

1775 1 bAR 

1776 dtype: str 

1777 

1778 Using a compiled regex with flags 

1779 

1780 >>> import re 

1781 >>> regex_pat = re.compile(r"FUZ", flags=re.IGNORECASE) 

1782 >>> pd.Series(["foo", "fuz", np.nan]).str.replace(regex_pat, "bar", regex=True) 

1783 0 foo 

1784 1 bar 

1785 2 NaN 

1786 dtype: str 

1787 """ 

1788 if isinstance(pat, dict) and repl is not None: 

1789 raise ValueError("repl cannot be used when pat is a dictionary") 

1790 

1791 # Check whether repl is valid (GH 13438, GH 15055) 

1792 if not isinstance(pat, dict) and not (isinstance(repl, str) or callable(repl)): 

1793 raise TypeError("repl must be a string or callable") 

1794 

1795 is_compiled_re = is_re(pat) 

1796 if regex or regex is None: 

1797 if is_compiled_re and (case is not None or flags != 0): 

1798 raise ValueError( 

1799 "case and flags cannot be set when pat is a compiled regex" 

1800 ) 

1801 

1802 elif is_compiled_re: 

1803 raise ValueError( 

1804 "Cannot use a compiled regex as replacement pattern with regex=False" 

1805 ) 

1806 elif callable(repl): 

1807 raise ValueError("Cannot use a callable replacement when regex=False") 

1808 

1809 if case is None: 

1810 case = True 

1811 

1812 res_output = self._data 

1813 if not isinstance(pat, dict): 

1814 pat = {pat: repl} 

1815 

1816 for key, value in pat.items(): 

1817 result = res_output.array._str_replace( 

1818 key, value, n=n, case=case, flags=flags, regex=regex 

1819 ) 

1820 res_output = self._wrap_result(result) 

1821 

1822 return res_output 

1823 

1824 @forbid_nonstring_types(["bytes"]) 

1825 def repeat(self, repeats): 

1826 """ 

1827 Duplicate each string in the Series or Index. 

1828 

1829 Duplicates each string in the Series or Index, either by applying the 

1830 same repeat count to all elements or by using different repeat values 

1831 for each element. 

1832 

1833 Parameters 

1834 ---------- 

1835 repeats : int or sequence of int 

1836 Same value for all (int) or different value per (sequence). 

1837 

1838 Returns 

1839 ------- 

1840 Series or pandas.Index 

1841 Series or Index of repeated string objects specified by 

1842 input parameter repeats. 

1843 

1844 See Also 

1845 -------- 

1846 Series.str.lower : Convert all characters in each string to lowercase. 

1847 Series.str.upper : Convert all characters in each string to uppercase. 

1848 Series.str.title : Convert each string to title case (capitalizing the first 

1849 letter of each word). 

1850 Series.str.strip : Remove leading and trailing whitespace from each string. 

1851 Series.str.replace : Replace occurrences of a substring with another substring 

1852 in each string. 

1853 Series.str.ljust : Left-justify each string in the Series/Index by padding with 

1854 a specified character. 

1855 Series.str.rjust : Right-justify each string in the Series/Index by padding with 

1856 a specified character. 

1857 

1858 Examples 

1859 -------- 

1860 >>> s = pd.Series(["a", "b", "c"]) 

1861 >>> s 

1862 0 a 

1863 1 b 

1864 2 c 

1865 dtype: str 

1866 

1867 Single int repeats string in Series 

1868 

1869 >>> s.str.repeat(repeats=2) 

1870 0 aa 

1871 1 bb 

1872 2 cc 

1873 dtype: str 

1874 

1875 Sequence of int repeats corresponding string in Series 

1876 

1877 >>> s.str.repeat(repeats=[1, 2, 3]) 

1878 0 a 

1879 1 bb 

1880 2 ccc 

1881 dtype: str 

1882 """ 

1883 result = self._data.array._str_repeat(repeats) 

1884 return self._wrap_result(result) 

1885 

1886 @forbid_nonstring_types(["bytes"]) 

1887 def pad( 

1888 self, 

1889 width: int, 

1890 side: Literal["left", "right", "both"] = "left", 

1891 fillchar: str = " ", 

1892 ): 

1893 """ 

1894 Pad strings in the Series/Index up to width. 

1895 

1896 This function pads strings in a Series or Index to a specified width, 

1897 filling the extra space with a character of your choice. It provides 

1898 flexibility in positioning the padding, allowing it to be added to the 

1899 left, right, or both sides. This is useful for formatting strings to 

1900 align text or ensure consistent string lengths in data processing. 

1901 

1902 Parameters 

1903 ---------- 

1904 width : int 

1905 Minimum width of resulting string; additional characters will be filled 

1906 with character defined in `fillchar`. 

1907 side : {'left', 'right', 'both'}, default 'left' 

1908 Side from which to fill resulting string. 

1909 fillchar : str, default ' ' 

1910 Additional character for filling, default is whitespace. 

1911 

1912 Returns 

1913 ------- 

1914 Series or Index of object 

1915 Returns Series or Index with minimum number of char in object. 

1916 

1917 See Also 

1918 -------- 

1919 Series.str.rjust : Fills the left side of strings with an arbitrary 

1920 character. Equivalent to ``Series.str.pad(side='left')``. 

1921 Series.str.ljust : Fills the right side of strings with an arbitrary 

1922 character. Equivalent to ``Series.str.pad(side='right')``. 

1923 Series.str.center : Fills both sides of strings with an arbitrary 

1924 character. Equivalent to ``Series.str.pad(side='both')``. 

1925 Series.str.zfill : Pad strings in the Series/Index by prepending '0' 

1926 character. Equivalent to ``Series.str.pad(side='left', fillchar='0')``. 

1927 

1928 Examples 

1929 -------- 

1930 >>> s = pd.Series(["caribou", "tiger"]) 

1931 >>> s 

1932 0 caribou 

1933 1 tiger 

1934 dtype: str 

1935 

1936 >>> s.str.pad(width=10) 

1937 0 caribou 

1938 1 tiger 

1939 dtype: str 

1940 

1941 >>> s.str.pad(width=10, side="right", fillchar="-") 

1942 0 caribou--- 

1943 1 tiger----- 

1944 dtype: str 

1945 

1946 >>> s.str.pad(width=10, side="both", fillchar="-") 

1947 0 -caribou-- 

1948 1 --tiger--- 

1949 dtype: str 

1950 """ 

1951 if not isinstance(fillchar, str): 

1952 msg = f"fillchar must be a character, not {type(fillchar).__name__}" 

1953 raise TypeError(msg) 

1954 

1955 if len(fillchar) != 1: 

1956 raise TypeError("fillchar must be a character, not str") 

1957 

1958 if not is_integer(width): 

1959 msg = f"width must be of integer type, not {type(width).__name__}" 

1960 raise TypeError(msg) 

1961 

1962 result = self._data.array._str_pad(width, side=side, fillchar=fillchar) 

1963 return self._wrap_result(result) 

1964 

1965 @forbid_nonstring_types(["bytes"]) 

1966 def center(self, width: int, fillchar: str = " "): 

1967 """ 

1968 Pad left and right side of strings in the Series/Index. 

1969 

1970 Equivalent to :meth:`str.center`. 

1971 

1972 Parameters 

1973 ---------- 

1974 width : int 

1975 Minimum width of resulting string; additional characters will be filled 

1976 with ``fillchar``. 

1977 fillchar : str 

1978 Additional character for filling, default is whitespace. 

1979 

1980 Returns 

1981 ------- 

1982 Series/Index of objects. 

1983 A Series or Index where the strings are modified by :meth:`str.center`. 

1984 

1985 See Also 

1986 -------- 

1987 Series.str.rjust : Fills the left side of strings with an arbitrary 

1988 character. 

1989 Series.str.ljust : Fills the right side of strings with an arbitrary 

1990 character. 

1991 Series.str.center : Fills both sides of strings with an arbitrary 

1992 character. 

1993 Series.str.zfill : Pad strings in the Series/Index by prepending '0' 

1994 character. 

1995 

1996 Examples 

1997 -------- 

1998 For Series.str.center: 

1999 

2000 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2001 >>> ser.str.center(8, fillchar=".") 

2002 0 ..dog... 

2003 1 ..bird.. 

2004 2 .mouse.. 

2005 dtype: str 

2006 

2007 For Series.str.ljust: 

2008 

2009 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2010 >>> ser.str.ljust(8, fillchar=".") 

2011 0 dog..... 

2012 1 bird.... 

2013 2 mouse... 

2014 dtype: str 

2015 

2016 For Series.str.rjust: 

2017 

2018 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2019 >>> ser.str.rjust(8, fillchar=".") 

2020 0 .....dog 

2021 1 ....bird 

2022 2 ...mouse 

2023 dtype: str 

2024 """ 

2025 return self.pad(width, side="both", fillchar=fillchar) 

2026 

2027 @forbid_nonstring_types(["bytes"]) 

2028 def ljust(self, width: int, fillchar: str = " "): 

2029 """ 

2030 Pad right side of strings in the Series/Index. 

2031 

2032 Equivalent to :meth:`str.ljust`. 

2033 

2034 Parameters 

2035 ---------- 

2036 width : int 

2037 Minimum width of resulting string; additional characters will be filled 

2038 with ``fillchar``. 

2039 fillchar : str 

2040 Additional character for filling, default is whitespace. 

2041 

2042 Returns 

2043 ------- 

2044 Series/Index of objects. 

2045 A Series or Index where the strings are modified by :meth:`str.ljust`. 

2046 

2047 See Also 

2048 -------- 

2049 Series.str.rjust : Fills the left side of strings with an arbitrary 

2050 character. 

2051 Series.str.ljust : Fills the right side of strings with an arbitrary 

2052 character. 

2053 Series.str.center : Fills both sides of strings with an arbitrary 

2054 character. 

2055 Series.str.zfill : Pad strings in the Series/Index by prepending '0' 

2056 character. 

2057 

2058 Examples 

2059 -------- 

2060 For Series.str.center: 

2061 

2062 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2063 >>> ser.str.center(8, fillchar=".") 

2064 0 ..dog... 

2065 1 ..bird.. 

2066 2 .mouse.. 

2067 dtype: str 

2068 

2069 For Series.str.ljust: 

2070 

2071 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2072 >>> ser.str.ljust(8, fillchar=".") 

2073 0 dog..... 

2074 1 bird.... 

2075 2 mouse... 

2076 dtype: str 

2077 

2078 For Series.str.rjust: 

2079 

2080 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2081 >>> ser.str.rjust(8, fillchar=".") 

2082 0 .....dog 

2083 1 ....bird 

2084 2 ...mouse 

2085 dtype: str 

2086 """ 

2087 return self.pad(width, side="right", fillchar=fillchar) 

2088 

2089 @forbid_nonstring_types(["bytes"]) 

2090 def rjust(self, width: int, fillchar: str = " "): 

2091 """ 

2092 Pad left side of strings in the Series/Index. 

2093 

2094 Equivalent to :meth:`str.rjust`. 

2095 

2096 Parameters 

2097 ---------- 

2098 width : int 

2099 Minimum width of resulting string; additional characters will be filled 

2100 with ``fillchar``. 

2101 fillchar : str 

2102 Additional character for filling, default is whitespace. 

2103 

2104 Returns 

2105 ------- 

2106 Series/Index of objects. 

2107 A Series or Index where the strings are modified by :meth:`str.rjust`. 

2108 

2109 See Also 

2110 -------- 

2111 Series.str.rjust : Fills the left side of strings with an arbitrary 

2112 character. 

2113 Series.str.ljust : Fills the right side of strings with an arbitrary 

2114 character. 

2115 Series.str.center : Fills both sides of strings with an arbitrary 

2116 character. 

2117 Series.str.zfill : Pad strings in the Series/Index by prepending '0' 

2118 character. 

2119 

2120 Examples 

2121 -------- 

2122 For Series.str.center: 

2123 

2124 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2125 >>> ser.str.center(8, fillchar=".") 

2126 0 ..dog... 

2127 1 ..bird.. 

2128 2 .mouse.. 

2129 dtype: str 

2130 

2131 For Series.str.ljust: 

2132 

2133 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2134 >>> ser.str.ljust(8, fillchar=".") 

2135 0 dog..... 

2136 1 bird.... 

2137 2 mouse... 

2138 dtype: str 

2139 

2140 For Series.str.rjust: 

2141 

2142 >>> ser = pd.Series(["dog", "bird", "mouse"]) 

2143 >>> ser.str.rjust(8, fillchar=".") 

2144 0 .....dog 

2145 1 ....bird 

2146 2 ...mouse 

2147 dtype: str 

2148 """ 

2149 return self.pad(width, side="left", fillchar=fillchar) 

2150 

2151 @forbid_nonstring_types(["bytes"]) 

2152 def zfill(self, width: int): 

2153 """ 

2154 Pad strings in the Series/Index by prepending '0' characters. 

2155 

2156 Strings in the Series/Index are padded with '0' characters on the 

2157 left of the string to reach a total string length `width`. Strings 

2158 in the Series/Index with length greater or equal to `width` are 

2159 unchanged. 

2160 

2161 Parameters 

2162 ---------- 

2163 width : int 

2164 Minimum length of resulting string; strings with length less 

2165 than `width` be prepended with '0' characters. 

2166 

2167 Returns 

2168 ------- 

2169 Series/Index of objects. 

2170 A Series or Index where the strings are prepended with '0' characters. 

2171 

2172 See Also 

2173 -------- 

2174 Series.str.rjust : Fills the left side of strings with an arbitrary 

2175 character. 

2176 Series.str.ljust : Fills the right side of strings with an arbitrary 

2177 character. 

2178 Series.str.pad : Fills the specified sides of strings with an arbitrary 

2179 character. 

2180 Series.str.center : Fills both sides of strings with an arbitrary 

2181 character. 

2182 

2183 Notes 

2184 ----- 

2185 Differs from :meth:`str.zfill` which has special handling 

2186 for '+'/'-' in the string. 

2187 

2188 Examples 

2189 -------- 

2190 >>> s = pd.Series(["-1", "1", "1000", 10, np.nan]) 

2191 >>> s 

2192 0 -1 

2193 1 1 

2194 2 1000 

2195 3 10 

2196 4 NaN 

2197 dtype: object 

2198 

2199 Note that ``10`` and ``NaN`` are not strings, therefore they are 

2200 converted to ``NaN``. The minus sign in ``'-1'`` is treated as a 

2201 special character and the zero is added to the right of it 

2202 (:meth:`str.zfill` would have moved it to the left). ``1000`` 

2203 remains unchanged as it is longer than `width`. 

2204 

2205 >>> s.str.zfill(3) 

2206 0 -01 

2207 1 001 

2208 2 1000 

2209 3 NaN 

2210 4 NaN 

2211 dtype: object 

2212 """ 

2213 if not is_integer(width): 

2214 msg = f"width must be of integer type, not {type(width).__name__}" 

2215 raise TypeError(msg) 

2216 

2217 result = self._data.array._str_zfill(width) 

2218 return self._wrap_result(result) 

2219 

2220 def slice(self, start=None, stop=None, step=None): 

2221 """ 

2222 Slice substrings from each element in the Series or Index. 

2223 

2224 Slicing substrings from strings in a Series or Index helps extract 

2225 specific portions of data, making it easier to analyze or manipulate 

2226 text. This is useful for tasks like parsing structured text fields or 

2227 isolating parts of strings with a consistent format. 

2228 

2229 Parameters 

2230 ---------- 

2231 start : int, optional 

2232 Start position for slice operation. 

2233 stop : int, optional 

2234 Stop position for slice operation. 

2235 step : int, optional 

2236 Step size for slice operation. 

2237 

2238 Returns 

2239 ------- 

2240 Series or Index of object 

2241 Series or Index from sliced substring from original string object. 

2242 

2243 See Also 

2244 -------- 

2245 Series.str.slice_replace : Replace a slice with a string. 

2246 Series.str.get : Return element at position. 

2247 Equivalent to `Series.str.slice(start=i, stop=i+1)` with `i` 

2248 being the position. 

2249 

2250 Examples 

2251 -------- 

2252 >>> s = pd.Series(["koala", "dog", "chameleon"]) 

2253 >>> s 

2254 0 koala 

2255 1 dog 

2256 2 chameleon 

2257 dtype: str 

2258 

2259 >>> s.str.slice(start=1) 

2260 0 oala 

2261 1 og 

2262 2 hameleon 

2263 dtype: str 

2264 

2265 >>> s.str.slice(start=-1) 

2266 0 a 

2267 1 g 

2268 2 n 

2269 dtype: str 

2270 

2271 >>> s.str.slice(stop=2) 

2272 0 ko 

2273 1 do 

2274 2 ch 

2275 dtype: str 

2276 

2277 >>> s.str.slice(step=2) 

2278 0 kaa 

2279 1 dg 

2280 2 caeen 

2281 dtype: str 

2282 

2283 >>> s.str.slice(start=0, stop=5, step=3) 

2284 0 kl 

2285 1 d 

2286 2 cm 

2287 dtype: str 

2288 

2289 Equivalent behaviour to: 

2290 

2291 >>> s.str[0:5:3] 

2292 0 kl 

2293 1 d 

2294 2 cm 

2295 dtype: str 

2296 """ 

2297 result = self._data.array._str_slice(start, stop, step) 

2298 return self._wrap_result(result) 

2299 

2300 @forbid_nonstring_types(["bytes"]) 

2301 def slice_replace(self, start=None, stop=None, repl=None): 

2302 """ 

2303 Replace a positional slice of a string with another value. 

2304 

2305 This function allows replacing specific parts of a string in a Series 

2306 or Index by specifying start and stop positions. It is useful for 

2307 modifying substrings in a controlled way, such as updating sections of 

2308 text based on their positions or patterns. 

2309 

2310 Parameters 

2311 ---------- 

2312 start : int, optional 

2313 Left index position to use for the slice. If not specified (None), 

2314 the slice is unbounded on the left, i.e. slice from the start 

2315 of the string. 

2316 stop : int, optional 

2317 Right index position to use for the slice. If not specified (None), 

2318 the slice is unbounded on the right, i.e. slice until the 

2319 end of the string. 

2320 repl : str, optional 

2321 String for replacement. If not specified (None), the sliced region 

2322 is replaced with an empty string. 

2323 

2324 Returns 

2325 ------- 

2326 Series or Index 

2327 Same type as the original object. 

2328 

2329 See Also 

2330 -------- 

2331 Series.str.slice : Just slicing without replacement. 

2332 

2333 Examples 

2334 -------- 

2335 >>> s = pd.Series(["a", "ab", "abc", "abdc", "abcde"]) 

2336 >>> s 

2337 0 a 

2338 1 ab 

2339 2 abc 

2340 3 abdc 

2341 4 abcde 

2342 dtype: str 

2343 

2344 Specify just `start`, meaning replace `start` until the end of the 

2345 string with `repl`. 

2346 

2347 >>> s.str.slice_replace(1, repl="X") 

2348 0 aX 

2349 1 aX 

2350 2 aX 

2351 3 aX 

2352 4 aX 

2353 dtype: str 

2354 

2355 Specify just `stop`, meaning the start of the string to `stop` is replaced 

2356 with `repl`, and the rest of the string is included. 

2357 

2358 >>> s.str.slice_replace(stop=2, repl="X") 

2359 0 X 

2360 1 X 

2361 2 Xc 

2362 3 Xdc 

2363 4 Xcde 

2364 dtype: str 

2365 

2366 Specify `start` and `stop`, meaning the slice from `start` to `stop` is 

2367 replaced with `repl`. Everything before or after `start` and `stop` is 

2368 included as is. 

2369 

2370 >>> s.str.slice_replace(start=1, stop=3, repl="X") 

2371 0 aX 

2372 1 aX 

2373 2 aX 

2374 3 aXc 

2375 4 aXde 

2376 dtype: str 

2377 """ 

2378 result = self._data.array._str_slice_replace(start, stop, repl) 

2379 return self._wrap_result(result) 

2380 

2381 def decode( 

2382 self, encoding, errors: str = "strict", dtype: str | DtypeObj | None = None 

2383 ): 

2384 """ 

2385 Decode character string in the Series/Index using indicated encoding. 

2386 

2387 Equivalent to :meth:`str.decode` in python2 and :meth:`bytes.decode` in 

2388 python3. 

2389 

2390 Parameters 

2391 ---------- 

2392 encoding : str 

2393 Specifies the encoding to be used. 

2394 errors : str, optional 

2395 Specifies the error handling scheme. 

2396 Possible values are those supported by :meth:`bytes.decode`. 

2397 dtype : str or dtype, optional 

2398 The dtype of the result. When not ``None``, must be either a string or 

2399 object dtype. When ``None``, the dtype of the result is determined by 

2400 ``pd.options.future.infer_string``. 

2401 

2402 .. versionadded:: 2.3.0 

2403 

2404 Returns 

2405 ------- 

2406 Series or Index 

2407 A Series or Index with decoded strings. 

2408 

2409 See Also 

2410 -------- 

2411 Series.str.encode : Encodes strings into bytes in a Series/Index. 

2412 

2413 Examples 

2414 -------- 

2415 For Series: 

2416 

2417 >>> ser = pd.Series([b"cow", b"123", b"()"]) 

2418 >>> ser.str.decode("ascii") 

2419 0 cow 

2420 1 123 

2421 2 () 

2422 dtype: str 

2423 """ 

2424 if dtype is not None and not is_string_dtype(dtype): 

2425 raise ValueError(f"dtype must be string or object, got {dtype=}") 

2426 if dtype is None and using_string_dtype(): 

2427 dtype = "str" 

2428 # TODO: Add a similar _bytes interface. 

2429 if encoding in _cpython_optimized_decoders: 

2430 # CPython optimized implementation 

2431 f = lambda x: x.decode(encoding, errors) 

2432 else: 

2433 decoder = codecs.getdecoder(encoding) 

2434 f = lambda x: decoder(x, errors)[0] 

2435 arr = self._data.array 

2436 result = arr._str_map(f) 

2437 return self._wrap_result(result, dtype=dtype) 

2438 

2439 @forbid_nonstring_types(["bytes"]) 

2440 def encode(self, encoding, errors: str = "strict"): 

2441 """ 

2442 Encode character string in the Series/Index using indicated encoding. 

2443 

2444 Equivalent to :meth:`str.encode`. 

2445 

2446 Parameters 

2447 ---------- 

2448 encoding : str 

2449 Specifies the encoding to be used. 

2450 errors : str, optional 

2451 Specifies the error handling scheme. 

2452 Possible values are those supported by :meth:`str.encode`. 

2453 

2454 Returns 

2455 ------- 

2456 Series/Index of objects 

2457 A Series or Index with strings encoded into bytes. 

2458 

2459 See Also 

2460 -------- 

2461 Series.str.decode : Decodes bytes into strings in a Series/Index. 

2462 

2463 Examples 

2464 -------- 

2465 >>> ser = pd.Series(["cow", "123", "()"]) 

2466 >>> ser.str.encode(encoding="ascii") 

2467 0 b'cow' 

2468 1 b'123' 

2469 2 b'()' 

2470 dtype: object 

2471 """ 

2472 result = self._data.array._str_encode(encoding, errors) 

2473 return self._wrap_result(result, returns_string=False) 

2474 

2475 @forbid_nonstring_types(["bytes"]) 

2476 def strip(self, to_strip=None): 

2477 """ 

2478 Remove leading and trailing characters. 

2479 

2480 Strip whitespaces (including newlines) or a set of specified characters 

2481 from each string in the Series/Index from left and right sides. 

2482 Replaces any non-strings in Series with NaNs. 

2483 Equivalent to :meth:`str.strip`. 

2484 

2485 Parameters 

2486 ---------- 

2487 to_strip : str or None, default None 

2488 Specifying the set of characters to be removed. 

2489 All combinations of this set of characters will be stripped. 

2490 If None then whitespaces are removed. 

2491 

2492 Returns 

2493 ------- 

2494 Series or Index of object 

2495 Series or Index with the strings being stripped from the left and 

2496 right sides. 

2497 

2498 See Also 

2499 -------- 

2500 Series.str.strip : Remove leading and trailing characters in Series/Index. 

2501 Series.str.lstrip : Remove leading characters in Series/Index. 

2502 Series.str.rstrip : Remove trailing characters in Series/Index. 

2503 

2504 Examples 

2505 -------- 

2506 >>> s = pd.Series(["1. Ant. ", "2. Bee!\\n", "3. Cat?\\t", np.nan, 10, True]) 

2507 >>> s 

2508 0 1. Ant. 

2509 1 2. Bee!\\n 

2510 2 3. Cat?\\t 

2511 3 NaN 

2512 4 10 

2513 5 True 

2514 dtype: object 

2515 

2516 >>> s.str.strip() 

2517 0 1. Ant. 

2518 1 2. Bee! 

2519 2 3. Cat? 

2520 3 NaN 

2521 4 NaN 

2522 5 NaN 

2523 dtype: object 

2524 

2525 >>> s.str.lstrip("123.") 

2526 0 Ant. 

2527 1 Bee!\\n 

2528 2 Cat?\\t 

2529 3 NaN 

2530 4 NaN 

2531 5 NaN 

2532 dtype: object 

2533 

2534 >>> s.str.rstrip(".!? \\n\\t") 

2535 0 1. Ant 

2536 1 2. Bee 

2537 2 3. Cat 

2538 3 NaN 

2539 4 NaN 

2540 5 NaN 

2541 dtype: object 

2542 

2543 >>> s.str.strip("123.!? \\n\\t") 

2544 0 Ant 

2545 1 Bee 

2546 2 Cat 

2547 3 NaN 

2548 4 NaN 

2549 5 NaN 

2550 dtype: object 

2551 """ 

2552 result = self._data.array._str_strip(to_strip) 

2553 return self._wrap_result(result) 

2554 

2555 @forbid_nonstring_types(["bytes"]) 

2556 def lstrip(self, to_strip=None): 

2557 """ 

2558 Remove leading characters. 

2559 

2560 Strip whitespaces (including newlines) or a set of specified characters 

2561 from each string in the Series/Index from left side. 

2562 Replaces any non-strings in Series with NaNs. 

2563 Equivalent to :meth:`str.lstrip`. 

2564 

2565 Parameters 

2566 ---------- 

2567 to_strip : str or None, default None 

2568 Specifying the set of characters to be removed. 

2569 All combinations of this set of characters will be stripped. 

2570 If None then whitespaces are removed. 

2571 

2572 Returns 

2573 ------- 

2574 Series or Index of object 

2575 Series or Index with the strings being stripped from the left side. 

2576 

2577 See Also 

2578 -------- 

2579 Series.str.strip : Remove leading and trailing characters in Series/Index. 

2580 Series.str.lstrip : Remove leading characters in Series/Index. 

2581 Series.str.rstrip : Remove trailing characters in Series/Index. 

2582 

2583 Examples 

2584 -------- 

2585 >>> s = pd.Series(["1. Ant. ", "2. Bee!\\n", "3. Cat?\\t", np.nan, 10, True]) 

2586 >>> s 

2587 0 1. Ant. 

2588 1 2. Bee!\\n 

2589 2 3. Cat?\\t 

2590 3 NaN 

2591 4 10 

2592 5 True 

2593 dtype: object 

2594 

2595 >>> s.str.strip() 

2596 0 1. Ant. 

2597 1 2. Bee! 

2598 2 3. Cat? 

2599 3 NaN 

2600 4 NaN 

2601 5 NaN 

2602 dtype: object 

2603 

2604 >>> s.str.lstrip("123.") 

2605 0 Ant. 

2606 1 Bee!\\n 

2607 2 Cat?\\t 

2608 3 NaN 

2609 4 NaN 

2610 5 NaN 

2611 dtype: object 

2612 

2613 >>> s.str.rstrip(".!? \\n\\t") 

2614 0 1. Ant 

2615 1 2. Bee 

2616 2 3. Cat 

2617 3 NaN 

2618 4 NaN 

2619 5 NaN 

2620 dtype: object 

2621 

2622 >>> s.str.strip("123.!? \\n\\t") 

2623 0 Ant 

2624 1 Bee 

2625 2 Cat 

2626 3 NaN 

2627 4 NaN 

2628 5 NaN 

2629 dtype: object 

2630 """ 

2631 result = self._data.array._str_lstrip(to_strip) 

2632 return self._wrap_result(result) 

2633 

2634 @forbid_nonstring_types(["bytes"]) 

2635 def rstrip(self, to_strip=None): 

2636 """ 

2637 Remove trailing characters. 

2638 

2639 Strip whitespaces (including newlines) or a set of specified characters 

2640 from each string in the Series/Index from right side. 

2641 Replaces any non-strings in Series with NaNs. 

2642 Equivalent to :meth:`str.rstrip`. 

2643 

2644 Parameters 

2645 ---------- 

2646 to_strip : str or None, default None 

2647 Specifying the set of characters to be removed. 

2648 All combinations of this set of characters will be stripped. 

2649 If None then whitespaces are removed. 

2650 

2651 Returns 

2652 ------- 

2653 Series or Index of object 

2654 Series or Index with the strings being stripped from the right side. 

2655 

2656 See Also 

2657 -------- 

2658 Series.str.strip : Remove leading and trailing characters in Series/Index. 

2659 Series.str.lstrip : Remove leading characters in Series/Index. 

2660 Series.str.rstrip : Remove trailing characters in Series/Index. 

2661 

2662 Examples 

2663 -------- 

2664 >>> s = pd.Series(["1. Ant. ", "2. Bee!\\n", "3. Cat?\\t", np.nan, 10, True]) 

2665 >>> s 

2666 0 1. Ant. 

2667 1 2. Bee!\\n 

2668 2 3. Cat?\\t 

2669 3 NaN 

2670 4 10 

2671 5 True 

2672 dtype: object 

2673 

2674 >>> s.str.strip() 

2675 0 1. Ant. 

2676 1 2. Bee! 

2677 2 3. Cat? 

2678 3 NaN 

2679 4 NaN 

2680 5 NaN 

2681 dtype: object 

2682 

2683 >>> s.str.lstrip("123.") 

2684 0 Ant. 

2685 1 Bee!\\n 

2686 2 Cat?\\t 

2687 3 NaN 

2688 4 NaN 

2689 5 NaN 

2690 dtype: object 

2691 

2692 >>> s.str.rstrip(".!? \\n\\t") 

2693 0 1. Ant 

2694 1 2. Bee 

2695 2 3. Cat 

2696 3 NaN 

2697 4 NaN 

2698 5 NaN 

2699 dtype: object 

2700 

2701 >>> s.str.strip("123.!? \\n\\t") 

2702 0 Ant 

2703 1 Bee 

2704 2 Cat 

2705 3 NaN 

2706 4 NaN 

2707 5 NaN 

2708 dtype: object 

2709 """ 

2710 result = self._data.array._str_rstrip(to_strip) 

2711 return self._wrap_result(result) 

2712 

2713 @forbid_nonstring_types(["bytes"]) 

2714 def removeprefix(self, prefix: str): 

2715 """ 

2716 Remove a prefix from an object series. 

2717 

2718 If the prefix is not present, the original string will be returned. 

2719 

2720 Parameters 

2721 ---------- 

2722 prefix : str 

2723 Remove the prefix of the string. 

2724 

2725 Returns 

2726 ------- 

2727 Series/Index: object 

2728 The Series or Index with given prefix removed. 

2729 

2730 See Also 

2731 -------- 

2732 Series.str.removesuffix : Remove a suffix from an object series. 

2733 

2734 Examples 

2735 -------- 

2736 >>> s = pd.Series(["str_foo", "str_bar", "no_prefix"]) 

2737 >>> s 

2738 0 str_foo 

2739 1 str_bar 

2740 2 no_prefix 

2741 dtype: str 

2742 >>> s.str.removeprefix("str_") 

2743 0 foo 

2744 1 bar 

2745 2 no_prefix 

2746 dtype: str 

2747 

2748 >>> s = pd.Series(["foo_str", "bar_str", "no_suffix"]) 

2749 >>> s 

2750 0 foo_str 

2751 1 bar_str 

2752 2 no_suffix 

2753 dtype: str 

2754 >>> s.str.removesuffix("_str") 

2755 0 foo 

2756 1 bar 

2757 2 no_suffix 

2758 dtype: str 

2759 """ 

2760 result = self._data.array._str_removeprefix(prefix) 

2761 return self._wrap_result(result) 

2762 

2763 @forbid_nonstring_types(["bytes"]) 

2764 def removesuffix(self, suffix: str): 

2765 """ 

2766 Remove a suffix from an object series. 

2767 

2768 If the suffix is not present, the original string will be returned. 

2769 

2770 Parameters 

2771 ---------- 

2772 suffix : str 

2773 Remove the suffix of the string. 

2774 

2775 Returns 

2776 ------- 

2777 Series/Index: object 

2778 The Series or Index with given suffix removed. 

2779 

2780 See Also 

2781 -------- 

2782 Series.str.removeprefix : Remove a prefix from an object series. 

2783 

2784 Examples 

2785 -------- 

2786 >>> s = pd.Series(["str_foo", "str_bar", "no_prefix"]) 

2787 >>> s 

2788 0 str_foo 

2789 1 str_bar 

2790 2 no_prefix 

2791 dtype: str 

2792 >>> s.str.removeprefix("str_") 

2793 0 foo 

2794 1 bar 

2795 2 no_prefix 

2796 dtype: str 

2797 

2798 >>> s = pd.Series(["foo_str", "bar_str", "no_suffix"]) 

2799 >>> s 

2800 0 foo_str 

2801 1 bar_str 

2802 2 no_suffix 

2803 dtype: str 

2804 >>> s.str.removesuffix("_str") 

2805 0 foo 

2806 1 bar 

2807 2 no_suffix 

2808 dtype: str 

2809 """ 

2810 result = self._data.array._str_removesuffix(suffix) 

2811 return self._wrap_result(result) 

2812 

2813 @forbid_nonstring_types(["bytes"]) 

2814 def wrap( 

2815 self, 

2816 width: int, 

2817 expand_tabs: bool = True, 

2818 tabsize: int = 8, 

2819 replace_whitespace: bool = True, 

2820 drop_whitespace: bool = True, 

2821 initial_indent: str = "", 

2822 subsequent_indent: str = "", 

2823 fix_sentence_endings: bool = False, 

2824 break_long_words: bool = True, 

2825 break_on_hyphens: bool = True, 

2826 max_lines: int | None = None, 

2827 placeholder: str = " [...]", 

2828 ): 

2829 r""" 

2830 Wrap strings in Series/Index at specified line width. 

2831 

2832 This method has the same keyword parameters and defaults as 

2833 :class:`textwrap.TextWrapper`. 

2834 

2835 Parameters 

2836 ---------- 

2837 width : int, optional 

2838 Maximum line width. 

2839 expand_tabs : bool, optional 

2840 If True, tab characters will be expanded to spaces (default: True). 

2841 tabsize : int, optional 

2842 If expand_tabs is true, then all tab characters in text will be 

2843 expanded to zero or more spaces, depending on the current column 

2844 and the given tab size (default: 8). 

2845 replace_whitespace : bool, optional 

2846 If True, each whitespace character (as defined by string.whitespace) 

2847 remaining after tab expansion will be replaced by a single space 

2848 (default: True). 

2849 drop_whitespace : bool, optional 

2850 If True, whitespace that, after wrapping, happens to end up at the 

2851 beginning or end of a line is dropped (default: True). 

2852 initial_indent : str, optional 

2853 String that will be prepended to the first line of wrapped output. 

2854 Counts towards the length of the first line. The empty string is 

2855 not indented (default: ''). 

2856 subsequent_indent : str, optional 

2857 String that will be prepended to all lines of wrapped output except 

2858 the first. Counts towards the length of each line except the first 

2859 (default: ''). 

2860 fix_sentence_endings : bool, optional 

2861 If true, TextWrapper attempts to detect sentence endings and ensure 

2862 that sentences are always separated by exactly two spaces. This is 

2863 generally desired for text in a monospaced font. However, the sentence 

2864 detection algorithm is imperfect: it assumes that a sentence ending 

2865 consists of a lowercase letter followed by one of '.', '!', or '?', 

2866 possibly followed by one of '"' or "'", followed by a space. One 

2867 problem with this algorithm is that it is unable to detect the 

2868 difference between “Dr.” in `[...] Dr. Frankenstein's monster [...]` 

2869 and “Spot.” in `[...] See Spot. See Spot run [...]` 

2870 Since the sentence detection algorithm relies on string.lowercase 

2871 for the definition of “lowercase letter”, and a convention of using 

2872 two spaces after a period to separate sentences on the same line, 

2873 it is specific to English-language texts (default: False). 

2874 break_long_words : bool, optional 

2875 If True, then words longer than width will be broken in order to ensure 

2876 that no lines are longer than width. If it is false, long words will 

2877 not be broken, and some lines may be longer than width (default: True). 

2878 break_on_hyphens : bool, optional 

2879 If True, wrapping will occur preferably on whitespace and right after 

2880 hyphens in compound words, as it is customary in English. If false, 

2881 only whitespaces will be considered as potentially good places for line 

2882 breaks, but you need to set break_long_words to false if you want truly 

2883 insecable words (default: True). 

2884 max_lines : int, optional 

2885 If not None, then the output will contain at most max_lines lines, with 

2886 placeholder appearing at the end of the output (default: None). 

2887 placeholder : str, optional 

2888 String that will appear at the end of the output text if it has been 

2889 truncated (default: ' [...]'). 

2890 

2891 Returns 

2892 ------- 

2893 Series or Index 

2894 A Series or Index where the strings are wrapped at the specified line width. 

2895 

2896 See Also 

2897 -------- 

2898 Series.str.strip : Remove leading and trailing characters in Series/Index. 

2899 Series.str.lstrip : Remove leading characters in Series/Index. 

2900 Series.str.rstrip : Remove trailing characters in Series/Index. 

2901 

2902 Notes 

2903 ----- 

2904 Internally, this method uses a :class:`textwrap.TextWrapper` instance with 

2905 default settings. To achieve behavior matching R's stringr library str_wrap 

2906 function, use the arguments: 

2907 

2908 - expand_tabs = False 

2909 - replace_whitespace = True 

2910 - drop_whitespace = True 

2911 - break_long_words = False 

2912 - break_on_hyphens = False 

2913 

2914 Examples 

2915 -------- 

2916 >>> s = pd.Series(["line to be wrapped", "another line to be wrapped"]) 

2917 >>> s.str.wrap(12) 

2918 0 line to be\nwrapped 

2919 1 another line\nto be\nwrapped 

2920 dtype: str 

2921 """ 

2922 result = self._data.array._str_wrap( 

2923 width=width, 

2924 expand_tabs=expand_tabs, 

2925 tabsize=tabsize, 

2926 replace_whitespace=replace_whitespace, 

2927 drop_whitespace=drop_whitespace, 

2928 initial_indent=initial_indent, 

2929 subsequent_indent=subsequent_indent, 

2930 fix_sentence_endings=fix_sentence_endings, 

2931 break_long_words=break_long_words, 

2932 break_on_hyphens=break_on_hyphens, 

2933 max_lines=max_lines, 

2934 placeholder=placeholder, 

2935 ) 

2936 return self._wrap_result(result) 

2937 

2938 @forbid_nonstring_types(["bytes"]) 

2939 def get_dummies( 

2940 self, 

2941 sep: str = "|", 

2942 dtype: NpDtype | None = None, 

2943 ): 

2944 """ 

2945 Return DataFrame of dummy/indicator variables for Series. 

2946 

2947 Each string in Series is split by sep and returned as a DataFrame 

2948 of dummy/indicator variables. 

2949 

2950 Parameters 

2951 ---------- 

2952 sep : str, default "|" 

2953 String to split on. 

2954 dtype : dtype, default np.int64 

2955 Data type for new columns. Only a single dtype is allowed. 

2956 

2957 Returns 

2958 ------- 

2959 DataFrame 

2960 Dummy variables corresponding to values of the Series. 

2961 

2962 See Also 

2963 -------- 

2964 get_dummies : Convert categorical variable into dummy/indicator 

2965 variables. 

2966 

2967 Examples 

2968 -------- 

2969 >>> pd.Series(["a|b", "a", "a|c"]).str.get_dummies() 

2970 a b c 

2971 0 1 1 0 

2972 1 1 0 0 

2973 2 1 0 1 

2974 

2975 >>> pd.Series(["a|b", np.nan, "a|c"]).str.get_dummies() 

2976 a b c 

2977 0 1 1 0 

2978 1 0 0 0 

2979 2 1 0 1 

2980 

2981 >>> pd.Series(["a|b", np.nan, "a|c"]).str.get_dummies(dtype=bool) 

2982 a b c 

2983 0 True True False 

2984 1 False False False 

2985 2 True False True 

2986 """ 

2987 from pandas.core.frame import DataFrame 

2988 

2989 if dtype is not None and not (is_numeric_dtype(dtype) or is_bool_dtype(dtype)): 

2990 raise ValueError("Only numeric or boolean dtypes are supported for 'dtype'") 

2991 # we need to cast to Series of strings as only that has all 

2992 # methods available for making the dummies... 

2993 result, name = self._data.array._str_get_dummies(sep, dtype) 

2994 if is_extension_array_dtype(dtype): 

2995 return self._wrap_result( 

2996 DataFrame(result, columns=name, dtype=dtype), 

2997 name=name, 

2998 returns_string=False, 

2999 ) 

3000 return self._wrap_result( 

3001 result, 

3002 name=name, 

3003 expand=True, 

3004 returns_string=False, 

3005 ) 

3006 

3007 @forbid_nonstring_types(["bytes"]) 

3008 def translate(self, table): 

3009 """ 

3010 Map all characters in the string through the given mapping table. 

3011 

3012 This method is equivalent to the standard :meth:`str.translate` 

3013 method for strings. It maps each character in the string to a new 

3014 character according to the translation table provided. Unmapped 

3015 characters are left unchanged, while characters mapped to None 

3016 are removed. 

3017 

3018 Parameters 

3019 ---------- 

3020 table : dict 

3021 Table is a mapping of Unicode ordinals to Unicode ordinals, strings, or 

3022 None. Unmapped characters are left untouched. 

3023 Characters mapped to None are deleted. :meth:`str.maketrans` is a 

3024 helper function for making translation tables. 

3025 

3026 Returns 

3027 ------- 

3028 Series or Index 

3029 A new Series or Index with translated strings. 

3030 

3031 See Also 

3032 -------- 

3033 Series.str.replace : Replace occurrences of pattern/regex in the 

3034 Series with some other string. 

3035 Index.str.replace : Replace occurrences of pattern/regex in the 

3036 Index with some other string. 

3037 

3038 Examples 

3039 -------- 

3040 >>> ser = pd.Series(["El niño", "Françoise"]) 

3041 >>> mytable = str.maketrans({"ñ": "n", "ç": "c"}) 

3042 >>> ser.str.translate(mytable) 

3043 0 El nino 

3044 1 Francoise 

3045 dtype: str 

3046 """ 

3047 result = self._data.array._str_translate(table) 

3048 dtype = object if self._data.dtype == "object" else None 

3049 return self._wrap_result(result, dtype=dtype) 

3050 

3051 @forbid_nonstring_types(["bytes"]) 

3052 def count(self, pat, flags: int = 0): 

3053 r""" 

3054 Count occurrences of pattern in each string of the Series/Index. 

3055 

3056 This function is used to count the number of times a particular regex 

3057 pattern is repeated in each of the string elements of the 

3058 :class:`~pandas.Series`. 

3059 

3060 Parameters 

3061 ---------- 

3062 pat : str 

3063 Valid regular expression. 

3064 flags : int, default 0, meaning no flags 

3065 Flags for the `re` module. For a complete list, `see here 

3066 <https://docs.python.org/3/howto/regex.html#compilation-flags>`_. 

3067 

3068 Returns 

3069 ------- 

3070 Series or Index 

3071 Same type as the calling object containing the integer counts. 

3072 

3073 See Also 

3074 -------- 

3075 re : Standard library module for regular expressions. 

3076 str.count : Standard library version, without regular expression support. 

3077 

3078 Notes 

3079 ----- 

3080 Some characters need to be escaped when passing in `pat`. 

3081 eg. ``'$'`` has a special meaning in regex and must be escaped when 

3082 finding this literal character. 

3083 

3084 Examples 

3085 -------- 

3086 >>> s = pd.Series(["A", "B", "Aaba", "Baca", np.nan, "CABA", "cat"]) 

3087 >>> s.str.count("a") 

3088 0 0.0 

3089 1 0.0 

3090 2 2.0 

3091 3 2.0 

3092 4 NaN 

3093 5 0.0 

3094 6 1.0 

3095 dtype: float64 

3096 

3097 Escape ``'$'`` to find the literal dollar sign. 

3098 

3099 >>> s = pd.Series(["$", "B", "Aab$", "$$ca", "C$B$", "cat"]) 

3100 >>> s.str.count("\\$") 

3101 0 1 

3102 1 0 

3103 2 1 

3104 3 2 

3105 4 2 

3106 5 0 

3107 dtype: int64 

3108 

3109 This is also available on Index 

3110 

3111 >>> pd.Index(["A", "A", "Aaba", "cat"]).str.count("a") 

3112 Index([0, 0, 2, 1], dtype='int64') 

3113 """ 

3114 result = self._data.array._str_count(pat, flags) 

3115 return self._wrap_result(result, returns_string=False) 

3116 

3117 @forbid_nonstring_types(["bytes"]) 

3118 def startswith( 

3119 self, pat: str | tuple[str, ...], na: Scalar | lib.NoDefault = lib.no_default 

3120 ) -> Series | Index: 

3121 """ 

3122 Test if the start of each string element matches a pattern. 

3123 

3124 Equivalent to :meth:`str.startswith`. 

3125 

3126 Parameters 

3127 ---------- 

3128 pat : str or tuple[str, ...] 

3129 Character sequence or tuple of strings. Regular expressions are not 

3130 accepted. 

3131 na : scalar, optional 

3132 Object shown if element tested is not a string. The default depends 

3133 on dtype of the array. For the ``"str"`` dtype, ``False`` is used. 

3134 For object dtype, ``numpy.nan`` is used. For the nullable 

3135 ``StringDtype``, ``pandas.NA`` is used. 

3136 

3137 Returns 

3138 ------- 

3139 Series or Index of bool 

3140 A Series of booleans indicating whether the given pattern matches 

3141 the start of each string element. 

3142 

3143 See Also 

3144 -------- 

3145 str.startswith : Python standard library string method. 

3146 Series.str.endswith : Same as startswith, but tests the end of string. 

3147 Series.str.contains : Tests if string element contains a pattern. 

3148 

3149 Examples 

3150 -------- 

3151 >>> s = pd.Series(["bat", "Bear", "cat", np.nan]) 

3152 >>> s 

3153 0 bat 

3154 1 Bear 

3155 2 cat 

3156 3 NaN 

3157 dtype: str 

3158 

3159 >>> s.str.startswith("b") 

3160 0 True 

3161 1 False 

3162 2 False 

3163 3 False 

3164 dtype: bool 

3165 

3166 >>> s.str.startswith(("b", "B")) 

3167 0 True 

3168 1 True 

3169 2 False 

3170 3 False 

3171 dtype: bool 

3172 """ 

3173 if not isinstance(pat, (str, tuple)): 

3174 msg = f"expected a string or tuple, not {type(pat).__name__}" 

3175 raise TypeError(msg) 

3176 result = self._data.array._str_startswith(pat, na=na) 

3177 return self._wrap_result(result, returns_string=False) 

3178 

3179 @forbid_nonstring_types(["bytes"]) 

3180 def endswith( 

3181 self, pat: str | tuple[str, ...], na: Scalar | lib.NoDefault = lib.no_default 

3182 ) -> Series | Index: 

3183 """ 

3184 Test if the end of each string element matches a pattern. 

3185 

3186 Equivalent to :meth:`str.endswith`. 

3187 

3188 Parameters 

3189 ---------- 

3190 pat : str or tuple[str, ...] 

3191 Character sequence or tuple of strings. Regular expressions are not 

3192 accepted. 

3193 na : scalar, optional 

3194 Object shown if element tested is not a string. The default depends 

3195 on dtype of the array. For the ``"str"`` dtype, ``False`` is used. 

3196 For object dtype, ``numpy.nan`` is used. For the nullable 

3197 ``StringDtype``, ``pandas.NA`` is used. 

3198 

3199 Returns 

3200 ------- 

3201 Series or Index of bool 

3202 A Series of booleans indicating whether the given pattern matches 

3203 the end of each string element. 

3204 

3205 See Also 

3206 -------- 

3207 str.endswith : Python standard library string method. 

3208 Series.str.startswith : Same as endswith, but tests the start of string. 

3209 Series.str.contains : Tests if string element contains a pattern. 

3210 

3211 Examples 

3212 -------- 

3213 >>> s = pd.Series(["bat", "bear", "caT", np.nan]) 

3214 >>> s 

3215 0 bat 

3216 1 bear 

3217 2 caT 

3218 3 NaN 

3219 dtype: str 

3220 

3221 >>> s.str.endswith("t") 

3222 0 True 

3223 1 False 

3224 2 False 

3225 3 False 

3226 dtype: bool 

3227 

3228 >>> s.str.endswith(("t", "T")) 

3229 0 True 

3230 1 False 

3231 2 True 

3232 3 False 

3233 dtype: bool 

3234 """ 

3235 if not isinstance(pat, (str, tuple)): 

3236 msg = f"expected a string or tuple, not {type(pat).__name__}" 

3237 raise TypeError(msg) 

3238 result = self._data.array._str_endswith(pat, na=na) 

3239 return self._wrap_result(result, returns_string=False) 

3240 

3241 @forbid_nonstring_types(["bytes"]) 

3242 def findall(self, pat, flags: int = 0): 

3243 """ 

3244 Find all occurrences of pattern or regular expression in the Series/Index. 

3245 

3246 Equivalent to applying :func:`re.findall` to all the elements in the 

3247 Series/Index. 

3248 

3249 Parameters 

3250 ---------- 

3251 pat : str 

3252 Pattern or regular expression. 

3253 flags : int, default 0 

3254 Flags from ``re`` module, e.g. `re.IGNORECASE` (default is 0, which 

3255 means no flags). 

3256 

3257 Returns 

3258 ------- 

3259 Series/Index of lists of strings 

3260 All non-overlapping matches of pattern or regular expression in each 

3261 string of this Series/Index. 

3262 

3263 See Also 

3264 -------- 

3265 count : Count occurrences of pattern or regular expression in each string 

3266 of the Series/Index. 

3267 extractall : For each string in the Series, extract groups from all matches 

3268 of regular expression and return a DataFrame with one row for each 

3269 match and one column for each group. 

3270 re.findall : The equivalent ``re`` function to all non-overlapping matches 

3271 of pattern or regular expression in string, as a list of strings. 

3272 

3273 Examples 

3274 -------- 

3275 >>> s = pd.Series(["Lion", "Monkey", "Rabbit"]) 

3276 

3277 The search for the pattern 'Monkey' returns one match: 

3278 

3279 >>> s.str.findall("Monkey") 

3280 0 [] 

3281 1 [Monkey] 

3282 2 [] 

3283 dtype: object 

3284 

3285 On the other hand, the search for the pattern 'MONKEY' doesn't return any 

3286 match: 

3287 

3288 >>> s.str.findall("MONKEY") 

3289 0 [] 

3290 1 [] 

3291 2 [] 

3292 dtype: object 

3293 

3294 Flags can be added to the pattern or regular expression. For instance, 

3295 to find the pattern 'MONKEY' ignoring the case: 

3296 

3297 >>> import re 

3298 >>> s.str.findall("MONKEY", flags=re.IGNORECASE) 

3299 0 [] 

3300 1 [Monkey] 

3301 2 [] 

3302 dtype: object 

3303 

3304 When the pattern matches more than one string in the Series, all matches 

3305 are returned: 

3306 

3307 >>> s.str.findall("on") 

3308 0 [on] 

3309 1 [on] 

3310 2 [] 

3311 dtype: object 

3312 

3313 Regular expressions are supported too. For instance, the search for all the 

3314 strings ending with the word 'on' is shown next: 

3315 

3316 >>> s.str.findall("on$") 

3317 0 [on] 

3318 1 [] 

3319 2 [] 

3320 dtype: object 

3321 

3322 If the pattern is found more than once in the same string, then a list of 

3323 multiple strings is returned: 

3324 

3325 >>> s.str.findall("b") 

3326 0 [] 

3327 1 [] 

3328 2 [b, b] 

3329 dtype: object 

3330 """ 

3331 result = self._data.array._str_findall(pat, flags) 

3332 return self._wrap_result(result, returns_string=False) 

3333 

3334 @forbid_nonstring_types(["bytes"]) 

3335 def extract( 

3336 self, pat: str, flags: int = 0, expand: bool = True 

3337 ) -> DataFrame | Series | Index: 

3338 r""" 

3339 Extract capture groups in the regex `pat` as columns in a DataFrame. 

3340 

3341 For each subject string in the Series, extract groups from the 

3342 first match of regular expression `pat`. 

3343 

3344 Parameters 

3345 ---------- 

3346 pat : str 

3347 Regular expression pattern with capturing groups. 

3348 flags : int, default 0 (no flags) 

3349 Flags from the ``re`` module, e.g. ``re.IGNORECASE``, that 

3350 modify regular expression matching for things like case, 

3351 spaces, etc. For more details, see :mod:`re`. 

3352 expand : bool, default True 

3353 If True, return DataFrame with one column per capture group. 

3354 If False, return a Series/Index if there is one capture group 

3355 or DataFrame if there are multiple capture groups. 

3356 

3357 Returns 

3358 ------- 

3359 DataFrame or Series or Index 

3360 A DataFrame with one row for each subject string, and one 

3361 column for each group. Any capture group names in regular 

3362 expression pat will be used for column names; otherwise 

3363 capture group numbers will be used. The dtype of each result 

3364 column is always object, even when no match is found. If 

3365 ``expand=False`` and pat has only one capture group, then 

3366 return a Series (if subject is a Series) or Index (if subject 

3367 is an Index). 

3368 

3369 See Also 

3370 -------- 

3371 extractall : Returns all matches (not just the first match). 

3372 

3373 Examples 

3374 -------- 

3375 A pattern with two groups will return a DataFrame with two columns. 

3376 Non-matches will be NaN. 

3377 

3378 >>> s = pd.Series(["a1", "b2", "c3"]) 

3379 >>> s.str.extract(r"([ab])(\d)") 

3380 0 1 

3381 0 a 1 

3382 1 b 2 

3383 2 NaN NaN 

3384 

3385 A pattern may contain optional groups. 

3386 

3387 >>> s.str.extract(r"([ab])?(\d)") 

3388 0 1 

3389 0 a 1 

3390 1 b 2 

3391 2 NaN 3 

3392 

3393 Named groups will become column names in the result. 

3394 

3395 >>> s.str.extract(r"(?P<letter>[ab])(?P<digit>\d)") 

3396 letter digit 

3397 0 a 1 

3398 1 b 2 

3399 2 NaN NaN 

3400 

3401 A pattern with one group will return a DataFrame with one column 

3402 if expand=True. 

3403 

3404 >>> s.str.extract(r"[ab](\d)", expand=True) 

3405 0 

3406 0 1 

3407 1 2 

3408 2 NaN 

3409 

3410 A pattern with one group will return a Series if expand=False. 

3411 

3412 >>> s.str.extract(r"[ab](\d)", expand=False) 

3413 0 1 

3414 1 2 

3415 2 NaN 

3416 dtype: str 

3417 """ 

3418 from pandas import DataFrame 

3419 

3420 if not isinstance(expand, bool): 

3421 raise ValueError("expand must be True or False") 

3422 

3423 regex = re.compile(pat, flags=flags) 

3424 if regex.groups == 0: 

3425 raise ValueError("pattern contains no capture groups") 

3426 

3427 if not expand and regex.groups > 1 and isinstance(self._data, ABCIndex): 

3428 raise ValueError("only one regex group is supported with Index") 

3429 

3430 obj = self._data 

3431 result_dtype = _result_dtype(obj) 

3432 

3433 returns_df = regex.groups > 1 or expand 

3434 

3435 if returns_df: 

3436 name = None 

3437 columns = _get_group_names(regex) 

3438 

3439 if obj.array.size == 0: 

3440 result = DataFrame(columns=columns, dtype=result_dtype) 

3441 

3442 else: 

3443 result_list = self._data.array._str_extract( 

3444 pat, flags=flags, expand=returns_df 

3445 ) 

3446 

3447 result_index: Index | None 

3448 if isinstance(obj, ABCSeries): 

3449 result_index = obj.index 

3450 else: 

3451 result_index = None 

3452 

3453 result = DataFrame( 

3454 result_list, columns=columns, index=result_index, dtype=result_dtype 

3455 ) 

3456 

3457 else: 

3458 name = _get_single_group_name(regex) 

3459 result = self._data.array._str_extract(pat, flags=flags, expand=returns_df) 

3460 return self._wrap_result(result, name=name, dtype=result_dtype) 

3461 

3462 @forbid_nonstring_types(["bytes"]) 

3463 def extractall(self, pat, flags: int = 0) -> DataFrame: 

3464 r""" 

3465 Extract capture groups in the regex `pat` as columns in DataFrame. 

3466 

3467 For each subject string in the Series, extract groups from all 

3468 matches of regular expression pat. When each subject string in the 

3469 Series has exactly one match, extractall(pat).xs(0, level='match') 

3470 is the same as extract(pat). 

3471 

3472 Parameters 

3473 ---------- 

3474 pat : str 

3475 Regular expression pattern with capturing groups. 

3476 flags : int, default 0 (no flags) 

3477 A ``re`` module flag, for example ``re.IGNORECASE``. These allow 

3478 to modify regular expression matching for things like case, spaces, 

3479 etc. Multiple flags can be combined with the bitwise OR operator, 

3480 for example ``re.IGNORECASE | re.MULTILINE``. 

3481 

3482 Returns 

3483 ------- 

3484 DataFrame 

3485 A ``DataFrame`` with one row for each match, and one column for each 

3486 group. Its rows have a ``MultiIndex`` with first levels that come from 

3487 the subject ``Series``. The last level is named 'match' and indexes the 

3488 matches in each item of the ``Series``. Any capture group names in 

3489 regular expression pat will be used for column names; otherwise capture 

3490 group numbers will be used. 

3491 

3492 See Also 

3493 -------- 

3494 extract : Returns first match only (not all matches). 

3495 

3496 Examples 

3497 -------- 

3498 A pattern with one group will return a DataFrame with one column. 

3499 Indices with no matches will not appear in the result. 

3500 

3501 >>> s = pd.Series(["a1a2", "b1", "c1"], index=["A", "B", "C"]) 

3502 >>> s.str.extractall(r"[ab](\d)") 

3503 0 

3504 match 

3505 A 0 1 

3506 1 2 

3507 B 0 1 

3508 

3509 Capture group names are used for column names of the result. 

3510 

3511 >>> s.str.extractall(r"[ab](?P<digit>\d)") 

3512 digit 

3513 match 

3514 A 0 1 

3515 1 2 

3516 B 0 1 

3517 

3518 A pattern with two groups will return a DataFrame with two columns. 

3519 

3520 >>> s.str.extractall(r"(?P<letter>[ab])(?P<digit>\d)") 

3521 letter digit 

3522 match 

3523 A 0 a 1 

3524 1 a 2 

3525 B 0 b 1 

3526 

3527 Optional groups that do not match are NaN in the result. 

3528 

3529 >>> s.str.extractall(r"(?P<letter>[ab])?(?P<digit>\d)") 

3530 letter digit 

3531 match 

3532 A 0 a 1 

3533 1 a 2 

3534 B 0 b 1 

3535 C 0 NaN 1 

3536 """ 

3537 # TODO: dispatch 

3538 return str_extractall(self._orig, pat, flags) 

3539 

3540 @forbid_nonstring_types(["bytes"]) 

3541 def find(self, sub, start: int = 0, end=None): 

3542 """ 

3543 Return lowest indexes in each strings in the Series/Index. 

3544 

3545 Each of returned indexes corresponds to the position where the 

3546 substring is fully contained between [start:end]. Return -1 on 

3547 failure. Equivalent to standard :meth:`str.find`. 

3548 

3549 Parameters 

3550 ---------- 

3551 sub : str 

3552 Substring being searched. 

3553 start : int 

3554 Left edge index. 

3555 end : int 

3556 Right edge index. 

3557 

3558 Returns 

3559 ------- 

3560 Series or Index of int. 

3561 A Series (if the input is a Series) or an Index (if the input is an 

3562 Index) of the lowest indexes corresponding to the positions where the 

3563 substring is found in each string of the input. 

3564 

3565 See Also 

3566 -------- 

3567 rfind : Return highest indexes in each strings. 

3568 

3569 Examples 

3570 -------- 

3571 For Series.str.find: 

3572 

3573 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"]) 

3574 >>> ser.str.find("_") 

3575 0 0 

3576 1 4 

3577 2 2 

3578 dtype: int64 

3579 

3580 For Series.str.rfind: 

3581 

3582 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"]) 

3583 >>> ser.str.rfind("_") 

3584 0 4 

3585 1 4 

3586 2 4 

3587 dtype: int64 

3588 """ 

3589 if not isinstance(sub, str): 

3590 msg = f"expected a string object, not {type(sub).__name__}" 

3591 raise TypeError(msg) 

3592 

3593 result = self._data.array._str_find(sub, start, end) 

3594 return self._wrap_result(result, returns_string=False) 

3595 

3596 @forbid_nonstring_types(["bytes"]) 

3597 def rfind(self, sub, start: int = 0, end=None): 

3598 """ 

3599 Return highest indexes in each strings in the Series/Index. 

3600 

3601 Each of returned indexes corresponds to the position where the 

3602 substring is fully contained between [start:end]. Return -1 on 

3603 failure. Equivalent to standard :meth:`str.rfind`. 

3604 

3605 Parameters 

3606 ---------- 

3607 sub : str 

3608 Substring being searched. 

3609 start : int 

3610 Left edge index. 

3611 end : int 

3612 Right edge index. 

3613 

3614 Returns 

3615 ------- 

3616 Series or Index of int. 

3617 A Series (if the input is a Series) or an Index (if the input is an 

3618 Index) of the highest indexes corresponding to the positions where the 

3619 substring is found in each string of the input. 

3620 

3621 See Also 

3622 -------- 

3623 find : Return lowest indexes in each strings. 

3624 

3625 Examples 

3626 -------- 

3627 For Series.str.find: 

3628 

3629 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"]) 

3630 >>> ser.str.find("_") 

3631 0 0 

3632 1 4 

3633 2 2 

3634 dtype: int64 

3635 

3636 For Series.str.rfind: 

3637 

3638 >>> ser = pd.Series(["_cow_", "duck_", "do_v_e"]) 

3639 >>> ser.str.rfind("_") 

3640 0 4 

3641 1 4 

3642 2 4 

3643 dtype: int64 

3644 """ 

3645 if not isinstance(sub, str): 

3646 msg = f"expected a string object, not {type(sub).__name__}" 

3647 raise TypeError(msg) 

3648 

3649 result = self._data.array._str_rfind(sub, start=start, end=end) 

3650 return self._wrap_result(result, returns_string=False) 

3651 

3652 @forbid_nonstring_types(["bytes"]) 

3653 def normalize(self, form): 

3654 """ 

3655 Return the Unicode normal form for the strings in the Series/Index. 

3656 

3657 For more information on the forms, see the 

3658 :func:`unicodedata.normalize`. 

3659 

3660 Parameters 

3661 ---------- 

3662 form : {'NFC', 'NFKC', 'NFD', 'NFKD'} 

3663 Unicode form. 

3664 

3665 Returns 

3666 ------- 

3667 Series/Index of objects 

3668 A Series or Index of strings in the same Unicode form specified by `form`. 

3669 The returned object retains the same type as the input (Series or Index), 

3670 and contains the normalized strings. 

3671 

3672 See Also 

3673 -------- 

3674 Series.str.upper : Convert all characters in each string to uppercase. 

3675 Series.str.lower : Convert all characters in each string to lowercase. 

3676 Series.str.title : Convert each string to title case (capitalizing the 

3677 first letter of each word). 

3678 Series.str.strip : Remove leading and trailing whitespace from each string. 

3679 Series.str.replace : Replace occurrences of a substring with another substring 

3680 in each string. 

3681 

3682 Examples 

3683 -------- 

3684 >>> ser = pd.Series(["ñ"]) 

3685 >>> ser.str.normalize("NFC") == ser.str.normalize("NFD") 

3686 0 False 

3687 dtype: bool 

3688 """ 

3689 result = self._data.array._str_normalize(form) 

3690 return self._wrap_result(result) 

3691 

3692 @forbid_nonstring_types(["bytes"]) 

3693 def index(self, sub, start: int = 0, end=None): 

3694 """ 

3695 Return lowest indexes in each string in Series/Index. 

3696 

3697 Each of the returned indexes corresponds to the position where the 

3698 substring is fully contained between [start:end]. This is the same 

3699 as ``str.find`` except instead of returning -1, it raises a 

3700 ValueError when the substring is not found. Equivalent to standard 

3701 ``str.index``. 

3702 

3703 Parameters 

3704 ---------- 

3705 sub : str 

3706 Substring being searched. 

3707 start : int 

3708 Left edge index. 

3709 end : int 

3710 Right edge index. 

3711 

3712 Returns 

3713 ------- 

3714 Series or Index of object 

3715 Returns a Series or an Index of the lowest indexes 

3716 in each string of the input. 

3717 

3718 See Also 

3719 -------- 

3720 rindex : Return highest indexes in each strings. 

3721 

3722 Examples 

3723 -------- 

3724 For Series.str.index: 

3725 

3726 >>> ser = pd.Series(["horse", "eagle", "donkey"]) 

3727 >>> ser.str.index("e") 

3728 0 4 

3729 1 0 

3730 2 4 

3731 dtype: int64 

3732 

3733 For Series.str.rindex: 

3734 

3735 >>> ser = pd.Series(["Deer", "eagle", "Sheep"]) 

3736 >>> ser.str.rindex("e") 

3737 0 2 

3738 1 4 

3739 2 3 

3740 dtype: int64 

3741 """ 

3742 if not isinstance(sub, str): 

3743 msg = f"expected a string object, not {type(sub).__name__}" 

3744 raise TypeError(msg) 

3745 

3746 result = self._data.array._str_index(sub, start=start, end=end) 

3747 return self._wrap_result(result, returns_string=False) 

3748 

3749 @forbid_nonstring_types(["bytes"]) 

3750 def rindex(self, sub, start: int = 0, end=None): 

3751 """ 

3752 Return highest indexes in each string in Series/Index. 

3753 

3754 Each of the returned indexes corresponds to the position where the 

3755 substring is fully contained between [start:end]. This is the same 

3756 as ``str.rfind`` except instead of returning -1, it raises a 

3757 ValueError when the substring is not found. Equivalent to standard 

3758 ``str.rindex``. 

3759 

3760 Parameters 

3761 ---------- 

3762 sub : str 

3763 Substring being searched. 

3764 start : int 

3765 Left edge index. 

3766 end : int 

3767 Right edge index. 

3768 

3769 Returns 

3770 ------- 

3771 Series or Index of object 

3772 Returns a Series or an Index of the highest indexes 

3773 in each string of the input. 

3774 

3775 See Also 

3776 -------- 

3777 index : Return lowest indexes in each strings. 

3778 

3779 Examples 

3780 -------- 

3781 For Series.str.index: 

3782 

3783 >>> ser = pd.Series(["horse", "eagle", "donkey"]) 

3784 >>> ser.str.index("e") 

3785 0 4 

3786 1 0 

3787 2 4 

3788 dtype: int64 

3789 

3790 For Series.str.rindex: 

3791 

3792 >>> ser = pd.Series(["Deer", "eagle", "Sheep"]) 

3793 >>> ser.str.rindex("e") 

3794 0 2 

3795 1 4 

3796 2 3 

3797 dtype: int64 

3798 """ 

3799 if not isinstance(sub, str): 

3800 msg = f"expected a string object, not {type(sub).__name__}" 

3801 raise TypeError(msg) 

3802 

3803 result = self._data.array._str_rindex(sub, start=start, end=end) 

3804 return self._wrap_result(result, returns_string=False) 

3805 

3806 def len(self): 

3807 """ 

3808 Compute the length of each element in the Series/Index. 

3809 

3810 The element may be a sequence (such as a string, tuple or list) or a collection 

3811 (such as a dictionary). 

3812 

3813 Returns 

3814 ------- 

3815 Series or Index of int 

3816 A Series or Index of integer values indicating the length of each 

3817 element in the Series or Index. 

3818 

3819 See Also 

3820 -------- 

3821 str.len : Python built-in function returning the length of an object. 

3822 Series.size : Returns the length of the Series. 

3823 

3824 Examples 

3825 -------- 

3826 Returns the length (number of characters) in a string. Returns the 

3827 number of entries for dictionaries, lists or tuples. 

3828 

3829 >>> s = pd.Series( 

3830 ... ["dog", "", 5, {"foo": "bar"}, [2, 3, 5, 7], ("one", "two", "three")] 

3831 ... ) 

3832 >>> s 

3833 0 dog 

3834 1 

3835 2 5 

3836 3 {'foo': 'bar'} 

3837 4 [2, 3, 5, 7] 

3838 5 (one, two, three) 

3839 dtype: object 

3840 >>> s.str.len() 

3841 0 3.0 

3842 1 0.0 

3843 2 NaN 

3844 3 1.0 

3845 4 4.0 

3846 5 3.0 

3847 dtype: float64 

3848 """ 

3849 result = self._data.array._str_len() 

3850 return self._wrap_result(result, returns_string=False) 

3851 

3852 @forbid_nonstring_types(["bytes"]) 

3853 def lower(self): 

3854 """ 

3855 Convert strings in the Series/Index to lowercase. 

3856 

3857 Equivalent to :meth:`str.lower`. 

3858 

3859 Returns 

3860 ------- 

3861 Series or Index of objects 

3862 A Series or Index where the strings are modified by :meth:`str.lower`. 

3863 

3864 See Also 

3865 -------- 

3866 Series.str.lower : Converts all characters to lowercase. 

3867 Series.str.upper : Converts all characters to uppercase. 

3868 Series.str.title : Converts first character of each word to uppercase and 

3869 remaining to lowercase. 

3870 Series.str.capitalize : Converts first character to uppercase and 

3871 remaining to lowercase. 

3872 Series.str.swapcase : Converts uppercase to lowercase and lowercase to 

3873 uppercase. 

3874 Series.str.casefold: Removes all case distinctions in the string. 

3875 

3876 Examples 

3877 -------- 

3878 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"]) 

3879 >>> s 

3880 0 lower 

3881 1 CAPITALS 

3882 2 this is a sentence 

3883 3 SwApCaSe 

3884 dtype: str 

3885 

3886 >>> s.str.lower() 

3887 0 lower 

3888 1 capitals 

3889 2 this is a sentence 

3890 3 swapcase 

3891 dtype: str 

3892 

3893 >>> s.str.upper() 

3894 0 LOWER 

3895 1 CAPITALS 

3896 2 THIS IS A SENTENCE 

3897 3 SWAPCASE 

3898 dtype: str 

3899 

3900 >>> s.str.title() 

3901 0 Lower 

3902 1 Capitals 

3903 2 This Is A Sentence 

3904 3 Swapcase 

3905 dtype: str 

3906 

3907 >>> s.str.capitalize() 

3908 0 Lower 

3909 1 Capitals 

3910 2 This is a sentence 

3911 3 Swapcase 

3912 dtype: str 

3913 

3914 >>> s.str.swapcase() 

3915 0 LOWER 

3916 1 capitals 

3917 2 THIS IS A SENTENCE 

3918 3 sWaPcAsE 

3919 dtype: str 

3920 """ 

3921 result = self._data.array._str_lower() 

3922 return self._wrap_result(result) 

3923 

3924 @forbid_nonstring_types(["bytes"]) 

3925 def upper(self): 

3926 """ 

3927 Convert strings in the Series/Index to uppercase. 

3928 

3929 Equivalent to :meth:`str.upper`. 

3930 

3931 Returns 

3932 ------- 

3933 Series or Index of objects 

3934 A Series or Index where the strings are modified by :meth:`str.upper`. 

3935 

3936 See Also 

3937 -------- 

3938 Series.str.lower : Converts all characters to lowercase. 

3939 Series.str.upper : Converts all characters to uppercase. 

3940 Series.str.title : Converts first character of each word to uppercase and 

3941 remaining to lowercase. 

3942 Series.str.capitalize : Converts first character to uppercase and 

3943 remaining to lowercase. 

3944 Series.str.swapcase : Converts uppercase to lowercase and lowercase to 

3945 uppercase. 

3946 Series.str.casefold: Removes all case distinctions in the string. 

3947 

3948 Examples 

3949 -------- 

3950 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"]) 

3951 >>> s 

3952 0 lower 

3953 1 CAPITALS 

3954 2 this is a sentence 

3955 3 SwApCaSe 

3956 dtype: str 

3957 

3958 >>> s.str.lower() 

3959 0 lower 

3960 1 capitals 

3961 2 this is a sentence 

3962 3 swapcase 

3963 dtype: str 

3964 

3965 >>> s.str.upper() 

3966 0 LOWER 

3967 1 CAPITALS 

3968 2 THIS IS A SENTENCE 

3969 3 SWAPCASE 

3970 dtype: str 

3971 

3972 >>> s.str.title() 

3973 0 Lower 

3974 1 Capitals 

3975 2 This Is A Sentence 

3976 3 Swapcase 

3977 dtype: str 

3978 

3979 >>> s.str.capitalize() 

3980 0 Lower 

3981 1 Capitals 

3982 2 This is a sentence 

3983 3 Swapcase 

3984 dtype: str 

3985 

3986 >>> s.str.swapcase() 

3987 0 LOWER 

3988 1 capitals 

3989 2 THIS IS A SENTENCE 

3990 3 sWaPcAsE 

3991 dtype: str 

3992 """ 

3993 result = self._data.array._str_upper() 

3994 return self._wrap_result(result) 

3995 

3996 @forbid_nonstring_types(["bytes"]) 

3997 def title(self): 

3998 """ 

3999 Convert strings in the Series/Index to titlecase. 

4000 

4001 Equivalent to :meth:`str.title`. 

4002 

4003 Returns 

4004 ------- 

4005 Series or Index of objects 

4006 A Series or Index where the strings are modified by :meth:`str.title`. 

4007 

4008 See Also 

4009 -------- 

4010 Series.str.lower : Converts all characters to lowercase. 

4011 Series.str.upper : Converts all characters to uppercase. 

4012 Series.str.title : Converts first character of each word to uppercase and 

4013 remaining to lowercase. 

4014 Series.str.capitalize : Converts first character to uppercase and 

4015 remaining to lowercase. 

4016 Series.str.swapcase : Converts uppercase to lowercase and lowercase to 

4017 uppercase. 

4018 Series.str.casefold: Removes all case distinctions in the string. 

4019 

4020 Examples 

4021 -------- 

4022 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"]) 

4023 >>> s 

4024 0 lower 

4025 1 CAPITALS 

4026 2 this is a sentence 

4027 3 SwApCaSe 

4028 dtype: str 

4029 

4030 >>> s.str.lower() 

4031 0 lower 

4032 1 capitals 

4033 2 this is a sentence 

4034 3 swapcase 

4035 dtype: str 

4036 

4037 >>> s.str.upper() 

4038 0 LOWER 

4039 1 CAPITALS 

4040 2 THIS IS A SENTENCE 

4041 3 SWAPCASE 

4042 dtype: str 

4043 

4044 >>> s.str.title() 

4045 0 Lower 

4046 1 Capitals 

4047 2 This Is A Sentence 

4048 3 Swapcase 

4049 dtype: str 

4050 

4051 >>> s.str.capitalize() 

4052 0 Lower 

4053 1 Capitals 

4054 2 This is a sentence 

4055 3 Swapcase 

4056 dtype: str 

4057 

4058 >>> s.str.swapcase() 

4059 0 LOWER 

4060 1 capitals 

4061 2 THIS IS A SENTENCE 

4062 3 sWaPcAsE 

4063 dtype: str 

4064 """ 

4065 result = self._data.array._str_title() 

4066 return self._wrap_result(result) 

4067 

4068 @forbid_nonstring_types(["bytes"]) 

4069 def capitalize(self): 

4070 """ 

4071 Convert strings in the Series/Index to be capitalized. 

4072 

4073 Equivalent to :meth:`str.capitalize`. 

4074 

4075 Returns 

4076 ------- 

4077 Series or Index of objects 

4078 A Series or Index where the strings are modified by :meth:`str.capitalize`. 

4079 

4080 See Also 

4081 -------- 

4082 Series.str.lower : Converts all characters to lowercase. 

4083 Series.str.upper : Converts all characters to uppercase. 

4084 Series.str.title : Converts first character of each word to uppercase and 

4085 remaining to lowercase. 

4086 Series.str.capitalize : Converts first character to uppercase and 

4087 remaining to lowercase. 

4088 Series.str.swapcase : Converts uppercase to lowercase and lowercase to 

4089 uppercase. 

4090 Series.str.casefold: Removes all case distinctions in the string. 

4091 

4092 Examples 

4093 -------- 

4094 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"]) 

4095 >>> s 

4096 0 lower 

4097 1 CAPITALS 

4098 2 this is a sentence 

4099 3 SwApCaSe 

4100 dtype: str 

4101 

4102 >>> s.str.lower() 

4103 0 lower 

4104 1 capitals 

4105 2 this is a sentence 

4106 3 swapcase 

4107 dtype: str 

4108 

4109 >>> s.str.upper() 

4110 0 LOWER 

4111 1 CAPITALS 

4112 2 THIS IS A SENTENCE 

4113 3 SWAPCASE 

4114 dtype: str 

4115 

4116 >>> s.str.title() 

4117 0 Lower 

4118 1 Capitals 

4119 2 This Is A Sentence 

4120 3 Swapcase 

4121 dtype: str 

4122 

4123 >>> s.str.capitalize() 

4124 0 Lower 

4125 1 Capitals 

4126 2 This is a sentence 

4127 3 Swapcase 

4128 dtype: str 

4129 

4130 >>> s.str.swapcase() 

4131 0 LOWER 

4132 1 capitals 

4133 2 THIS IS A SENTENCE 

4134 3 sWaPcAsE 

4135 dtype: str 

4136 """ 

4137 result = self._data.array._str_capitalize() 

4138 return self._wrap_result(result) 

4139 

4140 @forbid_nonstring_types(["bytes"]) 

4141 def swapcase(self): 

4142 """ 

4143 Convert strings in the Series/Index to be swapcased. 

4144 

4145 Equivalent to :meth:`str.swapcase`. 

4146 

4147 Returns 

4148 ------- 

4149 Series or Index of objects 

4150 A Series or Index where the strings are modified by :meth:`str.swapcase`. 

4151 

4152 See Also 

4153 -------- 

4154 Series.str.lower : Converts all characters to lowercase. 

4155 Series.str.upper : Converts all characters to uppercase. 

4156 Series.str.title : Converts first character of each word to uppercase and 

4157 remaining to lowercase. 

4158 Series.str.capitalize : Converts first character to uppercase and 

4159 remaining to lowercase. 

4160 Series.str.swapcase : Converts uppercase to lowercase and lowercase to 

4161 uppercase. 

4162 Series.str.casefold: Removes all case distinctions in the string. 

4163 

4164 Examples 

4165 -------- 

4166 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"]) 

4167 >>> s 

4168 0 lower 

4169 1 CAPITALS 

4170 2 this is a sentence 

4171 3 SwApCaSe 

4172 dtype: str 

4173 

4174 >>> s.str.lower() 

4175 0 lower 

4176 1 capitals 

4177 2 this is a sentence 

4178 3 swapcase 

4179 dtype: str 

4180 

4181 >>> s.str.upper() 

4182 0 LOWER 

4183 1 CAPITALS 

4184 2 THIS IS A SENTENCE 

4185 3 SWAPCASE 

4186 dtype: str 

4187 

4188 >>> s.str.title() 

4189 0 Lower 

4190 1 Capitals 

4191 2 This Is A Sentence 

4192 3 Swapcase 

4193 dtype: str 

4194 

4195 >>> s.str.capitalize() 

4196 0 Lower 

4197 1 Capitals 

4198 2 This is a sentence 

4199 3 Swapcase 

4200 dtype: str 

4201 

4202 >>> s.str.swapcase() 

4203 0 LOWER 

4204 1 capitals 

4205 2 THIS IS A SENTENCE 

4206 3 sWaPcAsE 

4207 dtype: str 

4208 """ 

4209 result = self._data.array._str_swapcase() 

4210 return self._wrap_result(result) 

4211 

4212 @forbid_nonstring_types(["bytes"]) 

4213 def casefold(self): 

4214 """ 

4215 Convert strings in the Series/Index to be casefolded. 

4216 

4217 Equivalent to :meth:`str.casefold`. 

4218 

4219 Returns 

4220 ------- 

4221 Series or Index of objects 

4222 A Series or Index where the strings are modified by :meth:`str.casefold`. 

4223 

4224 See Also 

4225 -------- 

4226 Series.str.lower : Converts all characters to lowercase. 

4227 Series.str.upper : Converts all characters to uppercase. 

4228 Series.str.title : Converts first character of each word to uppercase and 

4229 remaining to lowercase. 

4230 Series.str.capitalize : Converts first character to uppercase and 

4231 remaining to lowercase. 

4232 Series.str.swapcase : Converts uppercase to lowercase and lowercase to 

4233 uppercase. 

4234 Series.str.casefold: Removes all case distinctions in the string. 

4235 

4236 Examples 

4237 -------- 

4238 >>> s = pd.Series(["lower", "CAPITALS", "this is a sentence", "SwApCaSe"]) 

4239 >>> s 

4240 0 lower 

4241 1 CAPITALS 

4242 2 this is a sentence 

4243 3 SwApCaSe 

4244 dtype: str 

4245 

4246 >>> s.str.lower() 

4247 0 lower 

4248 1 capitals 

4249 2 this is a sentence 

4250 3 swapcase 

4251 dtype: str 

4252 

4253 >>> s.str.upper() 

4254 0 LOWER 

4255 1 CAPITALS 

4256 2 THIS IS A SENTENCE 

4257 3 SWAPCASE 

4258 dtype: str 

4259 

4260 >>> s.str.title() 

4261 0 Lower 

4262 1 Capitals 

4263 2 This Is A Sentence 

4264 3 Swapcase 

4265 dtype: str 

4266 

4267 >>> s.str.capitalize() 

4268 0 Lower 

4269 1 Capitals 

4270 2 This is a sentence 

4271 3 Swapcase 

4272 dtype: str 

4273 

4274 >>> s.str.swapcase() 

4275 0 LOWER 

4276 1 capitals 

4277 2 THIS IS A SENTENCE 

4278 3 sWaPcAsE 

4279 dtype: str 

4280 """ 

4281 result = self._data.array._str_casefold() 

4282 return self._wrap_result(result) 

4283 

4284 @forbid_nonstring_types(["bytes"]) 

4285 def isalnum(self): 

4286 """ 

4287 Check whether all characters in each string are alphanumeric. 

4288 

4289 This is equivalent to running the Python string method 

4290 :meth:`str.isalnum` for each element of the Series/Index. If a string 

4291 has zero characters, ``False`` is returned for that check. 

4292 

4293 Returns 

4294 ------- 

4295 Series or Index of bool 

4296 Series or Index of boolean values with the same length as the original 

4297 Series/Index. 

4298 

4299 See Also 

4300 -------- 

4301 Series.str.isalpha : Check whether all characters are alphabetic. 

4302 Series.str.isnumeric : Check whether all characters are numeric. 

4303 Series.str.isdigit : Check whether all characters are digits. 

4304 Series.str.isdecimal : Check whether all characters are decimal. 

4305 Series.str.isspace : Check whether all characters are whitespace. 

4306 Series.str.islower : Check whether all characters are lowercase. 

4307 Series.str.isascii : Check whether all characters are ascii. 

4308 Series.str.isupper : Check whether all characters are uppercase. 

4309 Series.str.istitle : Check whether all characters are titlecase. 

4310 

4311 Examples 

4312 -------- 

4313 >>> s1 = pd.Series(["one", "one1", "1", ""]) 

4314 >>> s1.str.isalnum() 

4315 0 True 

4316 1 True 

4317 2 True 

4318 3 False 

4319 dtype: bool 

4320 

4321 Note that checks against characters mixed with any additional punctuation 

4322 or whitespace will evaluate to false for an alphanumeric check. 

4323 

4324 >>> s2 = pd.Series(["A B", "1.5", "3,000"]) 

4325 >>> s2.str.isalnum() 

4326 0 False 

4327 1 False 

4328 2 False 

4329 dtype: bool 

4330 """ 

4331 result = self._data.array._str_isalnum() 

4332 return self._wrap_result(result, returns_string=False) 

4333 

4334 @forbid_nonstring_types(["bytes"]) 

4335 def isalpha(self): 

4336 """ 

4337 Check whether all characters in each string are alphabetic. 

4338 

4339 This is equivalent to running the Python string method 

4340 :meth:`str.isalpha` for each element of the Series/Index. If a string 

4341 has zero characters, ``False`` is returned for that check. 

4342 

4343 Returns 

4344 ------- 

4345 Series or Index of bool 

4346 Series or Index of boolean values with the same length as the original 

4347 Series/Index. 

4348 

4349 See Also 

4350 -------- 

4351 Series.str.isnumeric : Check whether all characters are numeric. 

4352 Series.str.isalnum : Check whether all characters are alphanumeric. 

4353 Series.str.isdigit : Check whether all characters are digits. 

4354 Series.str.isdecimal : Check whether all characters are decimal. 

4355 Series.str.isspace : Check whether all characters are whitespace. 

4356 Series.str.islower : Check whether all characters are lowercase. 

4357 Series.str.isascii : Check whether all characters are ascii. 

4358 Series.str.isupper : Check whether all characters are uppercase. 

4359 Series.str.istitle : Check whether all characters are titlecase. 

4360 

4361 Examples 

4362 -------- 

4363 

4364 >>> s1 = pd.Series(["one", "one1", "1", ""]) 

4365 >>> s1.str.isalpha() 

4366 0 True 

4367 1 False 

4368 2 False 

4369 3 False 

4370 dtype: bool 

4371 """ 

4372 result = self._data.array._str_isalpha() 

4373 return self._wrap_result(result, returns_string=False) 

4374 

4375 @forbid_nonstring_types(["bytes"]) 

4376 def isdigit(self): 

4377 """ 

4378 Check whether all characters in each string are digits. 

4379 

4380 This is equivalent to running the Python string method 

4381 :meth:`str.isdigit` for each element of the Series/Index. If a string 

4382 has zero characters, ``False`` is returned for that check. 

4383 

4384 Returns 

4385 ------- 

4386 Series or Index of bool 

4387 Series or Index of boolean values with the same length as the original 

4388 Series/Index. 

4389 

4390 See Also 

4391 -------- 

4392 Series.str.isalpha : Check whether all characters are alphabetic. 

4393 Series.str.isnumeric : Check whether all characters are numeric. 

4394 Series.str.isalnum : Check whether all characters are alphanumeric. 

4395 Series.str.isdecimal : Check whether all characters are decimal. 

4396 Series.str.isspace : Check whether all characters are whitespace. 

4397 Series.str.islower : Check whether all characters are lowercase. 

4398 Series.str.isascii : Check whether all characters are ascii. 

4399 Series.str.isupper : Check whether all characters are uppercase. 

4400 Series.str.istitle : Check whether all characters are titlecase. 

4401 

4402 Notes 

4403 ----- 

4404 Similar to ``str.isdecimal`` but also includes special digits, like 

4405 superscripted and subscripted digits in unicode. 

4406 

4407 The exact behavior of this method, i.e. which unicode characters are 

4408 considered as digits, depends on the backend used for string operations, 

4409 and there can be small differences. 

4410 For example, Python considers the ³ superscript character as a digit, but 

4411 not the ⅕ fraction character, while PyArrow considers both as digits. For 

4412 simple (ascii) decimal numbers, the behaviour is consistent. 

4413 

4414 Examples 

4415 -------- 

4416 

4417 >>> s3 = pd.Series(["23", "³", "⅕", ""]) 

4418 >>> s3.str.isdigit() 

4419 0 True 

4420 1 True 

4421 2 True 

4422 3 False 

4423 dtype: bool 

4424 """ 

4425 result = self._data.array._str_isdigit() 

4426 return self._wrap_result(result, returns_string=False) 

4427 

4428 @forbid_nonstring_types(["bytes"]) 

4429 def isspace(self): 

4430 """ 

4431 Check whether all characters in each string are whitespace. 

4432 

4433 This is equivalent to running the Python string method 

4434 :meth:`str.isspace` for each element of the Series/Index. If a string 

4435 has zero characters, ``False`` is returned for that check. 

4436 

4437 Returns 

4438 ------- 

4439 Series or Index of bool 

4440 Series or Index of boolean values with the same length as the original 

4441 Series/Index. 

4442 

4443 See Also 

4444 -------- 

4445 Series.str.isalpha : Check whether all characters are alphabetic. 

4446 Series.str.isnumeric : Check whether all characters are numeric. 

4447 Series.str.isalnum : Check whether all characters are alphanumeric. 

4448 Series.str.isdigit : Check whether all characters are digits. 

4449 Series.str.isdecimal : Check whether all characters are decimal. 

4450 Series.str.islower : Check whether all characters are lowercase. 

4451 Series.str.isascii : Check whether all characters are ascii. 

4452 Series.str.isupper : Check whether all characters are uppercase. 

4453 Series.str.istitle : Check whether all characters are titlecase. 

4454 

4455 Examples 

4456 -------- 

4457 

4458 >>> s4 = pd.Series([" ", "\\t\\r\\n ", ""]) 

4459 >>> s4.str.isspace() 

4460 0 True 

4461 1 True 

4462 2 False 

4463 dtype: bool 

4464 """ 

4465 result = self._data.array._str_isspace() 

4466 return self._wrap_result(result, returns_string=False) 

4467 

4468 @forbid_nonstring_types(["bytes"]) 

4469 def islower(self): 

4470 """ 

4471 Check whether all characters in each string are lowercase. 

4472 

4473 This is equivalent to running the Python string method 

4474 :meth:`str.islower` for each element of the Series/Index. If a string 

4475 has zero characters, ``False`` is returned for that check. 

4476 

4477 Returns 

4478 ------- 

4479 Series or Index of bool 

4480 Series or Index of boolean values with the same length as the original 

4481 Series/Index. 

4482 

4483 See Also 

4484 -------- 

4485 Series.str.isalpha : Check whether all characters are alphabetic. 

4486 Series.str.isnumeric : Check whether all characters are numeric. 

4487 Series.str.isalnum : Check whether all characters are alphanumeric. 

4488 Series.str.isdigit : Check whether all characters are digits. 

4489 Series.str.isdecimal : Check whether all characters are decimal. 

4490 Series.str.isspace : Check whether all characters are whitespace. 

4491 Series.str.isascii : Check whether all characters are ascii. 

4492 Series.str.isupper : Check whether all characters are uppercase. 

4493 Series.str.istitle : Check whether all characters are titlecase. 

4494 

4495 Examples 

4496 -------- 

4497 

4498 >>> s5 = pd.Series(["leopard", "Golden Eagle", "SNAKE", ""]) 

4499 >>> s5.str.islower() 

4500 0 True 

4501 1 False 

4502 2 False 

4503 3 False 

4504 dtype: bool 

4505 """ 

4506 result = self._data.array._str_islower() 

4507 return self._wrap_result(result, returns_string=False) 

4508 

4509 @forbid_nonstring_types(["bytes"]) 

4510 def isascii(self): 

4511 """ 

4512 Check whether all characters in each string are ascii. 

4513 

4514 This is equivalent to running the Python string method 

4515 :meth:`str.isascii` for each element of the Series/Index. If a string 

4516 has zero characters, ``False`` is returned for that check. 

4517 

4518 Returns 

4519 ------- 

4520 Series or Index of bool 

4521 Series or Index of boolean values with the same length as the original 

4522 Series/Index. 

4523 

4524 See Also 

4525 -------- 

4526 Series.str.isalpha : Check whether all characters are alphabetic. 

4527 Series.str.isnumeric : Check whether all characters are numeric. 

4528 Series.str.isalnum : Check whether all characters are alphanumeric. 

4529 Series.str.isdigit : Check whether all characters are digits. 

4530 Series.str.isdecimal : Check whether all characters are decimal. 

4531 Series.str.isspace : Check whether all characters are whitespace. 

4532 Series.str.islower : Check whether all characters are lowercase. 

4533 Series.str.isupper : Check whether all characters are uppercase. 

4534 Series.str.istitle : Check whether all characters are titlecase. 

4535 

4536 Examples 

4537 -------- 

4538 The ``s5.str.isascii`` method checks for whether all characters are ascii 

4539 characters, which includes digits 0-9, capital and lowercase letters A-Z, 

4540 and some other special characters. 

4541 

4542 >>> s5 = pd.Series(["ö", "see123", "hello world", ""]) 

4543 >>> s5.str.isascii() 

4544 0 False 

4545 1 True 

4546 2 True 

4547 3 True 

4548 dtype: bool 

4549 """ 

4550 result = self._data.array._str_isascii() 

4551 return self._wrap_result(result, returns_string=False) 

4552 

4553 @forbid_nonstring_types(["bytes"]) 

4554 def isupper(self): 

4555 """ 

4556 Check whether all characters in each string are uppercase. 

4557 

4558 This is equivalent to running the Python string method 

4559 :meth:`str.isupper` for each element of the Series/Index. If a string 

4560 has zero characters, ``False`` is returned for that check. 

4561 

4562 Returns 

4563 ------- 

4564 Series or Index of bool 

4565 Series or Index of boolean values with the same length as the original 

4566 Series/Index. 

4567 

4568 See Also 

4569 -------- 

4570 Series.str.isalpha : Check whether all characters are alphabetic. 

4571 Series.str.isnumeric : Check whether all characters are numeric. 

4572 Series.str.isalnum : Check whether all characters are alphanumeric. 

4573 Series.str.isdigit : Check whether all characters are digits. 

4574 Series.str.isdecimal : Check whether all characters are decimal. 

4575 Series.str.isspace : Check whether all characters are whitespace. 

4576 Series.str.islower : Check whether all characters are lowercase. 

4577 Series.str.isascii : Check whether all characters are ascii. 

4578 Series.str.istitle : Check whether all characters are titlecase. 

4579 

4580 Examples 

4581 -------- 

4582 

4583 >>> s5 = pd.Series(["leopard", "Golden Eagle", "SNAKE", ""]) 

4584 >>> s5.str.isupper() 

4585 0 False 

4586 1 False 

4587 2 True 

4588 3 False 

4589 dtype: bool 

4590 """ 

4591 result = self._data.array._str_isupper() 

4592 return self._wrap_result(result, returns_string=False) 

4593 

4594 @forbid_nonstring_types(["bytes"]) 

4595 def istitle(self): 

4596 """ 

4597 Check whether all characters in each string are titlecase. 

4598 

4599 This is equivalent to running the Python string method 

4600 :meth:`str.istitle` for each element of the Series/Index. If a string 

4601 has zero characters, ``False`` is returned for that check. 

4602 

4603 Returns 

4604 ------- 

4605 Series or Index of bool 

4606 Series or Index of boolean values with the same length as the original 

4607 Series/Index. 

4608 

4609 See Also 

4610 -------- 

4611 Series.str.isalpha : Check whether all characters are alphabetic. 

4612 Series.str.isnumeric : Check whether all characters are numeric. 

4613 Series.str.isalnum : Check whether all characters are alphanumeric. 

4614 Series.str.isdigit : Check whether all characters are digits. 

4615 Series.str.isdecimal : Check whether all characters are decimal. 

4616 Series.str.isspace : Check whether all characters are whitespace. 

4617 Series.str.islower : Check whether all characters are lowercase. 

4618 Series.str.isascii : Check whether all characters are ascii. 

4619 Series.str.isupper : Check whether all characters are uppercase. 

4620 

4621 Examples 

4622 -------- 

4623 The ``s5.str.istitle`` method checks for whether all words are in title 

4624 case (whether only the first letter of each word is capitalized). Words are 

4625 assumed to be as any sequence of non-numeric characters separated by 

4626 whitespace characters. 

4627 

4628 >>> s5 = pd.Series(["leopard", "Golden Eagle", "SNAKE", ""]) 

4629 >>> s5.str.istitle() 

4630 0 False 

4631 1 True 

4632 2 False 

4633 3 False 

4634 dtype: bool 

4635 """ 

4636 result = self._data.array._str_istitle() 

4637 return self._wrap_result(result, returns_string=False) 

4638 

4639 @forbid_nonstring_types(["bytes"]) 

4640 def isnumeric(self): 

4641 """ 

4642 Check whether all characters in each string are numeric. 

4643 

4644 This is equivalent to running the Python string method 

4645 :meth:`str.isnumeric` for each element of the Series/Index. If a string 

4646 has zero characters, ``False`` is returned for that check. 

4647 

4648 Returns 

4649 ------- 

4650 Series or Index of bool 

4651 Series or Index of boolean values with the same length as the original 

4652 Series/Index. 

4653 

4654 See Also 

4655 -------- 

4656 Series.str.isalpha : Check whether all characters are alphabetic. 

4657 Series.str.isalnum : Check whether all characters are alphanumeric. 

4658 Series.str.isdigit : Check whether all characters are digits. 

4659 Series.str.isdecimal : Check whether all characters are decimal. 

4660 Series.str.isspace : Check whether all characters are whitespace. 

4661 Series.str.islower : Check whether all characters are lowercase. 

4662 Series.str.isascii : Check whether all characters are ascii. 

4663 Series.str.isupper : Check whether all characters are uppercase. 

4664 Series.str.istitle : Check whether all characters are titlecase. 

4665 

4666 Examples 

4667 -------- 

4668 The ``s.str.isnumeric`` method is the same as ``s3.str.isdigit`` but 

4669 also includes other characters that can represent quantities such as 

4670 unicode fractions. 

4671 

4672 >>> s1 = pd.Series(["one", "one1", "1", "", "³", "⅕"]) 

4673 >>> s1.str.isnumeric() 

4674 0 False 

4675 1 False 

4676 2 True 

4677 3 False 

4678 4 True 

4679 5 True 

4680 dtype: bool 

4681 

4682 For a string to be considered numeric, all its characters must have a Unicode 

4683 numeric property matching :py:meth:`str.is_numeric`. As a consequence, 

4684 the following cases are **not** recognized as numeric: 

4685 

4686 - **Decimal numbers** (e.g., "1.1"): due to period ``"."`` 

4687 - **Negative numbers** (e.g., "-5"): due to minus sign ``"-"`` 

4688 - **Scientific notation** (e.g., "1e3"): due to characters like ``"e"`` 

4689 

4690 >>> s2 = pd.Series(["1.1", "-5", "1e3"]) 

4691 >>> s2.str.isnumeric() 

4692 0 False 

4693 1 False 

4694 2 False 

4695 dtype: bool 

4696 """ 

4697 result = self._data.array._str_isnumeric() 

4698 return self._wrap_result(result, returns_string=False) 

4699 

4700 @forbid_nonstring_types(["bytes"]) 

4701 def isdecimal(self): 

4702 """ 

4703 Check whether all characters in each string are decimal. 

4704 

4705 This is equivalent to running the Python string method 

4706 :meth:`str.isdecimal` for each element of the Series/Index. If a string 

4707 has zero characters, ``False`` is returned for that check. 

4708 

4709 Returns 

4710 ------- 

4711 Series or Index of bool 

4712 Series or Index of boolean values with the same length as the original 

4713 Series/Index. 

4714 

4715 See Also 

4716 -------- 

4717 Series.str.isalpha : Check whether all characters are alphabetic. 

4718 Series.str.isnumeric : Check whether all characters are numeric. 

4719 Series.str.isalnum : Check whether all characters are alphanumeric. 

4720 Series.str.isdigit : Check whether all characters are digits. 

4721 Series.str.isspace : Check whether all characters are whitespace. 

4722 Series.str.islower : Check whether all characters are lowercase. 

4723 Series.str.isascii : Check whether all characters are ascii. 

4724 Series.str.isupper : Check whether all characters are uppercase. 

4725 Series.str.istitle : Check whether all characters are titlecase. 

4726 

4727 Examples 

4728 -------- 

4729 The ``s3.str.isdecimal`` method checks for characters used to form 

4730 numbers in base 10. 

4731 

4732 >>> s3 = pd.Series(["23", "³", "⅕", ""]) 

4733 >>> s3.str.isdecimal() 

4734 0 True 

4735 1 False 

4736 2 False 

4737 3 False 

4738 dtype: bool 

4739 """ 

4740 result = self._data.array._str_isdecimal() 

4741 return self._wrap_result(result, returns_string=False) 

4742 

4743 

4744def cat_safe(list_of_columns: list[npt.NDArray[np.object_]], sep: str): 

4745 """ 

4746 Auxiliary function for :meth:`str.cat`. 

4747 

4748 Same signature as cat_core, but handles TypeErrors in concatenation, which 

4749 happen if the arrays in list_of columns have the wrong dtypes or content. 

4750 

4751 Parameters 

4752 ---------- 

4753 list_of_columns : list of numpy arrays 

4754 List of arrays to be concatenated with sep; 

4755 these arrays may not contain NaNs! 

4756 sep : string 

4757 The separator string for concatenating the columns. 

4758 

4759 Returns 

4760 ------- 

4761 nd.array 

4762 The concatenation of list_of_columns with sep. 

4763 """ 

4764 try: 

4765 result = cat_core(list_of_columns, sep) 

4766 except TypeError: 

4767 # if there are any non-string values (wrong dtype or hidden behind 

4768 # object dtype), np.sum will fail; catch and return with better message 

4769 for column in list_of_columns: 

4770 dtype = lib.infer_dtype(column, skipna=True) 

4771 if dtype not in ["string", "empty"]: 

4772 raise TypeError( 

4773 "Concatenation requires list-likes containing only " 

4774 "strings (or missing values). Offending values found in " 

4775 f"column {dtype}" 

4776 ) from None 

4777 return result 

4778 

4779 

4780def cat_core(list_of_columns: list, sep: str): 

4781 """ 

4782 Auxiliary function for :meth:`str.cat` 

4783 

4784 Parameters 

4785 ---------- 

4786 list_of_columns : list of numpy arrays 

4787 List of arrays to be concatenated with sep; 

4788 these arrays may not contain NaNs! 

4789 sep : string 

4790 The separator string for concatenating the columns. 

4791 

4792 Returns 

4793 ------- 

4794 nd.array 

4795 The concatenation of list_of_columns with sep. 

4796 """ 

4797 if sep == "": 

4798 # no need to interleave sep if it is empty 

4799 arr_of_cols = np.asarray(list_of_columns, dtype=object) 

4800 return np.sum(arr_of_cols, axis=0) 

4801 list_with_sep = [sep] * (2 * len(list_of_columns) - 1) 

4802 list_with_sep[::2] = list_of_columns 

4803 arr_with_sep = np.asarray(list_with_sep, dtype=object) 

4804 return np.sum(arr_with_sep, axis=0) 

4805 

4806 

4807def _result_dtype(arr): 

4808 # workaround #27953 

4809 # ideally we just pass `dtype=arr.dtype` unconditionally, but this fails 

4810 # when the list of values is empty. 

4811 from pandas.core.arrays.string_ import StringDtype 

4812 

4813 if isinstance(arr.dtype, (ArrowDtype, StringDtype)): 

4814 return arr.dtype 

4815 return object 

4816 

4817 

4818def _get_single_group_name(regex: re.Pattern) -> Hashable: 

4819 if regex.groupindex: 

4820 return next(iter(regex.groupindex)) 

4821 else: 

4822 return None 

4823 

4824 

4825def _get_group_names(regex: re.Pattern) -> list[Hashable] | range: 

4826 """ 

4827 Get named groups from compiled regex. 

4828 

4829 Unnamed groups are numbered. 

4830 

4831 Parameters 

4832 ---------- 

4833 regex : compiled regex 

4834 

4835 Returns 

4836 ------- 

4837 list of column labels 

4838 """ 

4839 rng = range(regex.groups) 

4840 names = {v: k for k, v in regex.groupindex.items()} 

4841 if not names: 

4842 return rng 

4843 result: list[Hashable] = [names.get(1 + i, i) for i in rng] 

4844 arr = np.array(result) 

4845 if arr.dtype.kind == "i" and lib.is_range_indexer(arr, len(arr)): 

4846 return rng 

4847 return result 

4848 

4849 

4850def str_extractall(arr, pat, flags: int = 0) -> DataFrame: 

4851 regex = re.compile(pat, flags=flags) 

4852 # the regex must contain capture groups. 

4853 if regex.groups == 0: 

4854 raise ValueError("pattern contains no capture groups") 

4855 

4856 if isinstance(arr, ABCIndex): 

4857 arr = arr.to_series().reset_index(drop=True).astype(arr.dtype) 

4858 

4859 columns = _get_group_names(regex) 

4860 match_list = [] 

4861 index_list = [] 

4862 is_mi = arr.index.nlevels > 1 

4863 

4864 for subject_key, subject in arr.items(): 

4865 if isinstance(subject, str): 

4866 if not is_mi: 

4867 subject_key = (subject_key,) 

4868 

4869 for match_i, match_tuple in enumerate(regex.findall(subject)): 

4870 if isinstance(match_tuple, str): 

4871 match_tuple = (match_tuple,) 

4872 na_tuple = [np.nan if group == "" else group for group in match_tuple] 

4873 match_list.append(na_tuple) 

4874 result_key = (*subject_key, match_i) 

4875 index_list.append(result_key) 

4876 

4877 from pandas import MultiIndex 

4878 

4879 index = MultiIndex.from_tuples(index_list, names=[*arr.index.names, "match"]) 

4880 dtype = _result_dtype(arr) 

4881 

4882 result = arr._constructor_expanddim( 

4883 match_list, index=index, columns=columns, dtype=dtype 

4884 ) 

4885 return result