Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/fsspec/utils.py: 15%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

409 statements  

1from __future__ import annotations 

2 

3import contextlib 

4import ctypes 

5import logging 

6import math 

7import os 

8import re 

9import sys 

10import tempfile 

11from collections.abc import Callable, Iterable, Iterator, Sequence 

12from functools import partial 

13from hashlib import md5 

14from importlib.metadata import version 

15from typing import IO, TYPE_CHECKING, Any, TypeVar 

16from urllib.parse import urlsplit 

17 

18if TYPE_CHECKING: 

19 import pathlib 

20 from typing import TypeGuard 

21 

22 from fsspec.spec import AbstractFileSystem 

23 

24 

25DEFAULT_BLOCK_SIZE = 5 * 2**20 

26 

27T = TypeVar("T") 

28 

29 

30def infer_storage_options( 

31 urlpath: str, inherit_storage_options: dict[str, Any] | None = None 

32) -> dict[str, Any]: 

33 """Infer storage options from URL path and merge it with existing storage 

34 options. 

35 

36 Parameters 

37 ---------- 

38 urlpath: str or unicode 

39 Either local absolute file path or URL (hdfs://namenode:8020/file.csv) 

40 inherit_storage_options: dict (optional) 

41 Its contents will get merged with the inferred information from the 

42 given path 

43 

44 Returns 

45 ------- 

46 Storage options dict. 

47 

48 Examples 

49 -------- 

50 >>> infer_storage_options('/mnt/datasets/test.csv') # doctest: +SKIP 

51 {"protocol": "file", "path", "/mnt/datasets/test.csv"} 

52 >>> infer_storage_options( 

53 ... 'hdfs://username:pwd@node:123/mnt/datasets/test.csv?q=1', 

54 ... inherit_storage_options={'extra': 'value'}, 

55 ... ) # doctest: +SKIP 

56 {"protocol": "hdfs", "username": "username", "password": "pwd", 

57 "host": "node", "port": 123, "path": "/mnt/datasets/test.csv", 

58 "url_query": "q=1", "extra": "value"} 

59 """ 

60 

61 # Discover Windows paths including disk name in this special case. 

62 is_filesystem = re.match(r"^[a-zA-Z]:[\\/]", urlpath) 

63 

64 # Discover URI according to RFC 3986: Scheme names consist of a 

65 # sequence of characters beginning with a letter and followed by 

66 # any combination of letters, digits, plus ("+"), period ("."), 

67 # or hyphen ("-"). 

68 # https://datatracker.ietf.org/doc/html/rfc3986#section-3.1 

69 is_uri = re.match(r"^[a-zA-Z0-9+.-]+://", urlpath) 

70 

71 if is_filesystem or is_uri is None: 

72 return {"protocol": "file", "path": urlpath} 

73 

74 parsed_path = urlsplit(urlpath) 

75 protocol = parsed_path.scheme or "file" 

76 if parsed_path.fragment: 

77 path = "#".join([parsed_path.path, parsed_path.fragment]) 

78 else: 

79 path = parsed_path.path 

80 if protocol == "file": 

81 # Special case parsing file protocol URL on Windows according to: 

82 # https://msdn.microsoft.com/en-us/library/jj710207.aspx 

83 windows_path = re.match(r"^/([a-zA-Z])[:|]([\\/].*)$", path) 

84 if windows_path: 

85 drive, path = windows_path.groups() 

86 path = f"{drive}:{path}" 

87 

88 if protocol in ["http", "https"]: 

89 # for HTTP, we don't want to parse, as requests will anyway 

90 return {"protocol": protocol, "path": urlpath} 

91 

92 options: dict[str, Any] = {"protocol": protocol, "path": path} 

93 

94 if parsed_path.netloc: 

95 # Parse `hostname` from netloc manually because `parsed_path.hostname` 

96 # lowercases the hostname which is not always desirable (e.g. in S3): 

97 # https://github.com/dask/dask/issues/1417 

98 host = parsed_path.netloc.rsplit("@", 1)[-1] 

99 if host.startswith("[") and "]" in host: 

100 # An IPv6 literal carries colons of its own, so only a colon after 

101 # the closing bracket separates the port. 

102 options["host"] = host[: host.index("]") + 1] 

103 else: 

104 options["host"] = host.rsplit(":", 1)[0] 

105 

106 if protocol in ("s3", "s3a", "gcs", "gs"): 

107 options["path"] = options["host"] + options["path"] 

108 else: 

109 options["host"] = options["host"] 

110 if parsed_path.port: 

111 options["port"] = parsed_path.port 

112 if parsed_path.username: 

113 options["username"] = parsed_path.username 

114 if parsed_path.password: 

115 options["password"] = parsed_path.password 

116 

117 if parsed_path.query: 

118 options["url_query"] = parsed_path.query 

119 if parsed_path.fragment: 

120 options["url_fragment"] = parsed_path.fragment 

121 

122 if inherit_storage_options: 

123 update_storage_options(options, inherit_storage_options) 

124 

125 return options 

126 

127 

128def update_storage_options( 

129 options: dict[str, Any], inherited: dict[str, Any] | None = None 

130) -> None: 

131 if not inherited: 

132 inherited = {} 

133 collisions = set(options) & set(inherited) 

134 if collisions: 

135 for collision in collisions: 

136 if options.get(collision) != inherited.get(collision): 

137 raise KeyError( 

138 f"Collision between inferred and specified storage " 

139 f"option:\n{collision}" 

140 ) 

141 options.update(inherited) 

142 

143 

144# Compression extensions registered via fsspec.compression.register_compression 

145compressions: dict[str, str] = {} 

146 

147 

148def infer_compression(filename: str) -> str | None: 

149 """Infer compression, if available, from filename. 

150 

151 Infer a named compression type, if registered and available, from filename 

152 extension. This includes builtin (gz, bz2, zip) compressions, as well as 

153 optional compressions. See fsspec.compression.register_compression. 

154 """ 

155 extension = os.path.splitext(filename)[-1].strip(".").lower() 

156 if extension in compressions: 

157 return compressions[extension] 

158 return None 

159 

160 

161def build_name_function(max_int: float) -> Callable[[int], str]: 

162 """Returns a function that receives a single integer 

163 and returns it as a string padded by enough zero characters 

164 to align with maximum possible integer 

165 

166 >>> name_f = build_name_function(57) 

167 

168 >>> name_f(7) 

169 '07' 

170 >>> name_f(31) 

171 '31' 

172 >>> build_name_function(1000)(42) 

173 '0042' 

174 >>> build_name_function(999)(42) 

175 '042' 

176 >>> build_name_function(0)(0) 

177 '0' 

178 """ 

179 # handle corner cases max_int is 0 or exact power of 10 

180 max_int += 1e-8 

181 

182 pad_length = int(math.ceil(math.log10(max_int))) 

183 

184 def name_function(i: int) -> str: 

185 return str(i).zfill(pad_length) 

186 

187 return name_function 

188 

189 

190def seek_delimiter(file: IO[bytes], delimiter: bytes, blocksize: int) -> bool: 

191 r"""Seek current file to file start, file end, or byte after delimiter seq. 

192 

193 Seeks file to next chunk delimiter, where chunks are defined on file start, 

194 a delimiting sequence, and file end. Use file.tell() to see location afterwards. 

195 Note that file start is a valid split, so must be at offset > 0 to seek for 

196 delimiter. 

197 

198 Parameters 

199 ---------- 

200 file: a file 

201 delimiter: bytes 

202 a delimiter like ``b'\n'`` or message sentinel, matching file .read() type 

203 blocksize: int 

204 Number of bytes to read from the file at once. 

205 

206 

207 Returns 

208 ------- 

209 Returns True if a delimiter was found, False if at file start or end. 

210 

211 """ 

212 

213 if file.tell() == 0: 

214 # beginning-of-file, return without seek 

215 return False 

216 

217 # Interface is for binary IO, with delimiter as bytes, but initialize last 

218 # with result of file.read to preserve compatibility with text IO. 

219 last: bytes | None = None 

220 while True: 

221 current = file.read(blocksize) 

222 if not current: 

223 # end-of-file without delimiter 

224 return False 

225 full = last + current if last else current 

226 try: 

227 if delimiter in full: 

228 i = full.index(delimiter) 

229 file.seek(file.tell() - (len(full) - i) + len(delimiter)) 

230 return True 

231 elif len(current) < blocksize: 

232 # end-of-file without delimiter 

233 return False 

234 except (OSError, ValueError): 

235 pass 

236 last = full[-len(delimiter) :] 

237 

238 

239def read_block( 

240 f: IO[bytes], 

241 offset: int, 

242 length: int | None, 

243 delimiter: bytes | None = None, 

244 split_before: bool = False, 

245) -> bytes: 

246 """Read a block of bytes from a file 

247 

248 Parameters 

249 ---------- 

250 f: File 

251 Open file 

252 offset: int 

253 Byte offset to start read 

254 length: int 

255 Number of bytes to read, read through end of file if None 

256 delimiter: bytes (optional) 

257 Ensure reading starts and stops at delimiter bytestring 

258 split_before: bool (optional) 

259 Start/stop read *before* delimiter bytestring. 

260 

261 

262 If using the ``delimiter=`` keyword argument we ensure that the read 

263 starts and stops at delimiter boundaries that follow the locations 

264 ``offset`` and ``offset + length``. If ``offset`` is zero then we 

265 start at zero, regardless of delimiter. The bytestring returned WILL 

266 include the terminating delimiter string. 

267 

268 Examples 

269 -------- 

270 

271 >>> from io import BytesIO # doctest: +SKIP 

272 >>> f = BytesIO(b'Alice, 100\\nBob, 200\\nCharlie, 300') # doctest: +SKIP 

273 >>> read_block(f, 0, 13) # doctest: +SKIP 

274 b'Alice, 100\\nBo' 

275 

276 >>> read_block(f, 0, 13, delimiter=b'\\n') # doctest: +SKIP 

277 b'Alice, 100\\nBob, 200\\n' 

278 

279 >>> read_block(f, 10, 10, delimiter=b'\\n') # doctest: +SKIP 

280 b'Bob, 200\\nCharlie, 300' 

281 """ 

282 if delimiter: 

283 f.seek(offset) 

284 found_start_delim = seek_delimiter(f, delimiter, 2**16) 

285 if length is None: 

286 return f.read() 

287 start = f.tell() 

288 length -= start - offset 

289 

290 f.seek(start + length) 

291 found_end_delim = seek_delimiter(f, delimiter, 2**16) 

292 end = f.tell() 

293 

294 # Adjust split location to before delimiter if seek found the 

295 # delimiter sequence, not start or end of file. 

296 if found_start_delim and split_before: 

297 start -= len(delimiter) 

298 

299 if found_end_delim and split_before: 

300 end -= len(delimiter) 

301 

302 offset = start 

303 length = end - start 

304 

305 f.seek(offset) 

306 

307 # TODO: allow length to be None and read to the end of the file? 

308 assert length is not None 

309 b = f.read(length) 

310 return b 

311 

312 

313def tokenize(*args: Any, **kwargs: Any) -> str: 

314 """Deterministic token 

315 

316 (modified from dask.base) 

317 

318 >>> tokenize([1, 2, '3']) 

319 '9d71491b50023b06fc76928e6eddb952' 

320 

321 >>> tokenize('Hello') == tokenize('Hello') 

322 True 

323 """ 

324 if kwargs: 

325 args += (kwargs,) 

326 try: 

327 h = md5(str(args).encode()) 

328 except ValueError: 

329 # FIPS systems: https://github.com/fsspec/filesystem_spec/issues/380 

330 h = md5(str(args).encode(), usedforsecurity=False) 

331 return h.hexdigest() 

332 

333 

334def stringify_path(filepath: str | os.PathLike[str] | pathlib.Path) -> str: 

335 """Attempt to convert a path-like object to a string. 

336 

337 Parameters 

338 ---------- 

339 filepath: object to be converted 

340 

341 Returns 

342 ------- 

343 filepath_str: maybe a string version of the object 

344 

345 Notes 

346 ----- 

347 Objects supporting the fspath protocol are coerced according to its 

348 __fspath__ method. 

349 

350 For backwards compatibility with older Python version, pathlib.Path 

351 objects are specially coerced. 

352 

353 Any other object is passed through unchanged, which includes bytes, 

354 strings, buffers, or anything else that's not even path-like. 

355 """ 

356 if isinstance(filepath, str): 

357 return filepath 

358 elif hasattr(filepath, "__fspath__"): 

359 return filepath.__fspath__() 

360 elif hasattr(filepath, "path"): 

361 return filepath.path 

362 else: 

363 return filepath # type: ignore[return-value] 

364 

365 

366def make_instance( 

367 cls: Callable[..., T], args: Sequence[Any], kwargs: dict[str, Any] 

368) -> T: 

369 inst = cls(*args, **kwargs) 

370 inst._determine_worker() # type: ignore[attr-defined] 

371 return inst 

372 

373 

374def common_prefix(paths: Iterable[str]) -> str: 

375 """For a list of paths, find the shortest prefix common to all""" 

376 parts = [p.split("/") for p in paths] 

377 lmax = min(len(p) for p in parts) 

378 end = 0 

379 for i in range(lmax): 

380 end = all(p[i] == parts[0][i] for p in parts) 

381 if not end: 

382 break 

383 i += end 

384 return "/".join(parts[0][:i]) 

385 

386 

387def other_paths( 

388 paths: list[str], 

389 path2: str | list[str], 

390 exists: bool = False, 

391 flatten: bool = False, 

392) -> list[str]: 

393 """In bulk file operations, construct a new file tree from a list of files 

394 

395 Parameters 

396 ---------- 

397 paths: list of str 

398 The input file tree 

399 path2: str or list of str 

400 Root to construct the new list in. If this is already a list of str, we just 

401 assert it has the right number of elements. 

402 exists: bool (optional) 

403 For a str destination, it is already exists (and is a dir), files should 

404 end up inside. 

405 flatten: bool (optional) 

406 Whether to flatten the input directory tree structure so that the output files 

407 are in the same directory. 

408 

409 Returns 

410 ------- 

411 list of str 

412 """ 

413 

414 if isinstance(path2, str): 

415 path2 = path2.rstrip("/") 

416 

417 if flatten: 

418 path2 = ["/".join((path2, p.split("/")[-1])) for p in paths] 

419 else: 

420 cp = common_prefix(paths) 

421 if exists: 

422 cp = cp.rsplit("/", 1)[0] 

423 if not cp and all(not s.startswith("/") for s in paths): 

424 path2 = ["/".join([path2, p]) for p in paths] 

425 else: 

426 path2 = [p.replace(cp, path2, 1) for p in paths] 

427 else: 

428 assert len(paths) == len(path2) 

429 return path2 

430 

431 

432def check_contained(root: str, paths: list[str]) -> None: 

433 """Raise if any of ``paths`` lies outside the destination ``root``. 

434 

435 Bulk copies build their destination names by joining source names onto a 

436 destination root. Those names come from the source listing, so a name 

437 holding ".." segments resolves above the root and writes outside the 

438 destination the caller asked for. 

439 

440 Parameters 

441 ---------- 

442 root: str 

443 The destination the caller passed. 

444 paths: list of str 

445 The destination names built for that root. 

446 """ 

447 root_abs = os.path.abspath(root) 

448 # normcase so that a case-insensitive platform does not report a false 

449 # escape, while the message keeps the paths as the caller would see them. 

450 root_key = os.path.normcase(root_abs) 

451 prefix = root_key.rstrip(os.sep) + os.sep 

452 for path in paths: 

453 path_abs = os.path.abspath(path) 

454 path_key = os.path.normcase(path_abs) 

455 if path_key != root_key and not path_key.startswith(prefix): 

456 raise ValueError( 

457 f"path {path!r} would be copied to {path_abs!r}, which is " 

458 f"outside the destination {root!r}" 

459 ) 

460 

461 

462def is_exception(obj: Any) -> bool: 

463 return isinstance(obj, BaseException) 

464 

465 

466def isfilelike(f: Any) -> TypeGuard[IO[bytes]]: 

467 return all(hasattr(f, attr) for attr in ["read", "close", "tell"]) 

468 

469 

470def get_protocol(url: str) -> str: 

471 url = stringify_path(url) 

472 parts = re.split(r"(\:\:|\://)", url, maxsplit=1) 

473 if len(parts) > 1: 

474 return parts[0] 

475 return "file" 

476 

477 

478def get_file_extension(url: str) -> str: 

479 url = stringify_path(url) 

480 # Only consider the final path component: a "." in a parent directory name 

481 # (e.g. "/path/to.dir/file") is not the file's extension. 

482 name = url.rsplit("/", 1)[-1] 

483 # A leading dot marks a hidden file rather than an extension, so ".bashrc" 

484 # has none, while ".hidden.txt" still has "txt". 

485 stem, dot, extension = name.lstrip(".").rpartition(".") 

486 if stem and dot: 

487 return extension 

488 return "" 

489 

490 

491def can_be_local(path: str) -> bool: 

492 """Can the given URL be used with open_local?""" 

493 from fsspec import get_filesystem_class 

494 

495 try: 

496 return getattr(get_filesystem_class(get_protocol(path)), "local_file", False) 

497 except (ValueError, ImportError): 

498 # not in registry or import failed 

499 return False 

500 

501 

502def get_package_version_without_import(name: str) -> str | None: 

503 """For given package name, try to find the version without importing it 

504 

505 Import and package.__version__ is still the backup here, so an import 

506 *might* happen. 

507 

508 Returns either the version string, or None if the package 

509 or the version was not readily found. 

510 """ 

511 if name in sys.modules: 

512 mod = sys.modules[name] 

513 if hasattr(mod, "__version__"): 

514 return mod.__version__ 

515 try: 

516 return version(name) 

517 except: # noqa: E722 

518 pass 

519 try: 

520 import importlib 

521 

522 mod = importlib.import_module(name) 

523 return mod.__version__ 

524 except (ImportError, AttributeError): 

525 return None 

526 

527 

528def setup_logging( 

529 logger: logging.Logger | None = None, 

530 logger_name: str | None = None, 

531 level: str = "DEBUG", 

532 clear: bool = True, 

533) -> logging.Logger: 

534 if logger is None and logger_name is None: 

535 raise ValueError("Provide either logger object or logger name") 

536 logger = logger or logging.getLogger(logger_name) 

537 handle = logging.StreamHandler() 

538 formatter = logging.Formatter( 

539 "%(asctime)s - %(name)s - %(levelname)s - %(funcName)s -- %(message)s" 

540 ) 

541 handle.setFormatter(formatter) 

542 if clear: 

543 logger.handlers.clear() 

544 logger.addHandler(handle) 

545 logger.setLevel(level) 

546 return logger 

547 

548 

549def _unstrip_protocol(name: str, fs: AbstractFileSystem) -> str: 

550 return fs.unstrip_protocol(name) 

551 

552 

553def mirror_from( 

554 origin_name: str, methods: Iterable[str] 

555) -> Callable[[type[T]], type[T]]: 

556 """Mirror attributes and methods from the given 

557 origin_name attribute of the instance to the 

558 decorated class""" 

559 

560 def origin_getter(method: str, self: Any) -> Any: 

561 origin = getattr(self, origin_name) 

562 return getattr(origin, method) 

563 

564 def wrapper(cls: type[T]) -> type[T]: 

565 for method in methods: 

566 wrapped_method = partial(origin_getter, method) 

567 setattr(cls, method, property(wrapped_method)) 

568 return cls 

569 

570 return wrapper 

571 

572 

573@contextlib.contextmanager 

574def nullcontext(obj: T) -> Iterator[T]: 

575 yield obj 

576 

577 

578def merge_offset_ranges( 

579 paths: list[str], 

580 starts: list[int | None] | int | None, 

581 ends: list[int | None] | int | None, 

582 max_gap: int = 0, 

583 max_block: int | None = None, 

584 sort: bool = True, 

585) -> tuple[list[str], list[int], list[int | None]]: 

586 """Merge adjacent byte-offset ranges when the inter-range 

587 gap is <= `max_gap`, and when the merged byte range does not 

588 exceed `max_block` (if specified). Every input range is covered by 

589 at least one returned range. Overlapping input ranges are merged 

590 where `max_block` allows it, so returned ranges may overlap once a 

591 chain of overlapping inputs reaches `max_block`; a single input 

592 range larger than `max_block` is still returned whole. 

593 

594 An `end` of `None` means to the end of the file. By default, this 

595 function will re-order the input paths and byte ranges to ensure 

596 sorted order. 

597 

598 Passing `sort=False` skips the re-ordering, which is only worthwhile 

599 when the inputs are already grouped by path and ascending by start 

600 within each path. Ranges that break that order still appear in the 

601 output, as their own range rather than merged, so coverage holds for 

602 any input order. 

603 """ 

604 # Check input 

605 if not isinstance(paths, list): 

606 raise TypeError 

607 if not isinstance(starts, list): 

608 starts = [starts] * len(paths) 

609 if not isinstance(ends, list): 

610 ends = [ends] * len(paths) 

611 if len(starts) != len(paths) or len(ends) != len(paths): 

612 raise ValueError 

613 

614 starts_i: list[int] = [s or 0 for s in starts] 

615 ends_i: list[int | None] = ends 

616 

617 # Early Return 

618 if len(starts_i) <= 1: 

619 return paths, starts_i, ends_i 

620 

621 # Sort by paths and then ranges if `sort=True` 

622 if sort: 

623 ranges = sorted( 

624 zip(paths, starts_i, ends_i), 

625 # None end sorts last (covers furthest into the file) 

626 key=lambda pse: (pse[0], pse[1], math.inf if pse[2] is None else pse[2]), 

627 ) 

628 paths = [r[0] for r in ranges] 

629 starts_i = [r[1] for r in ranges] 

630 ends_i = [r[2] for r in ranges] 

631 

632 # Loop through the coupled `paths`, `starts`, and 

633 # `ends`, and merge adjacent blocks when appropriate 

634 new_paths = paths[:1] 

635 new_starts = starts_i[:1] 

636 new_ends = ends_i[:1] 

637 for path, start, end in zip(paths[1:], starts_i[1:], ends_i[1:]): 

638 prev_end = new_ends[-1] 

639 if path != new_paths[-1]: 

640 # Cannot merge with previous block 

641 new_paths.append(path) 

642 new_starts.append(start) 

643 new_ends.append(end) 

644 elif start < new_starts[-1]: 

645 # Out of order (only possible when `sort=False`). Walking the 

646 # current block start backwards would uncover bytes already 

647 # attributed to it, so give this range its own block 

648 new_paths.append(path) 

649 new_starts.append(start) 

650 new_ends.append(end) 

651 elif prev_end is None: 

652 # Previous block already covers the rest of the file 

653 continue 

654 elif start < prev_end: 

655 # Overlap / nested 

656 if end is not None and end <= prev_end: 

657 # Already covered by the current block 

658 continue 

659 elif ( 

660 end is not None 

661 and max_block is not None 

662 and (end - new_starts[-1]) > max_block 

663 ): 

664 # Extending would exceed `max_block`. Start a new block, 

665 # which overlaps the previous one, rather than letting a 

666 # chain of overlaps grow the block without bound 

667 new_paths.append(path) 

668 new_starts.append(start) 

669 new_ends.append(end) 

670 else: 

671 # An `end` of None extends the block to EOF; a separate 

672 # block would subsume the current one anyway 

673 new_ends[-1] = end 

674 elif (start - prev_end) > max_gap or ( 

675 max_block is not None 

676 and (end is None or (end - new_starts[-1]) > max_block) 

677 ): 

678 # Gap too large, or merging would exceed `max_block` 

679 new_paths.append(path) 

680 new_starts.append(start) 

681 new_ends.append(end) 

682 else: 

683 # Merge with the previous block 

684 new_ends[-1] = end 

685 

686 return new_paths, new_starts, new_ends 

687 

688 

689def file_size(filelike: IO[bytes]) -> int: 

690 """Find length of any open read-mode file-like""" 

691 pos = filelike.tell() 

692 try: 

693 return filelike.seek(0, 2) 

694 finally: 

695 filelike.seek(pos) 

696 

697 

698@contextlib.contextmanager 

699def atomic_write(path: str, mode: str = "wb"): 

700 """ 

701 A context manager that opens a temporary file next to `path` and, on exit, 

702 replaces `path` with the temporary file, thereby updating `path` 

703 atomically. 

704 """ 

705 fd, fn = tempfile.mkstemp( 

706 dir=os.path.dirname(path), prefix=os.path.basename(path) + "-" 

707 ) 

708 try: 

709 with open(fd, mode) as fp: 

710 yield fp 

711 except BaseException: 

712 with contextlib.suppress(FileNotFoundError): 

713 os.unlink(fn) 

714 raise 

715 else: 

716 os.replace(fn, path) 

717 

718 

719def _translate(pat, STAR, QUESTION_MARK): 

720 # Copied from: https://github.com/python/cpython/pull/106703. 

721 res: list[str] = [] 

722 add = res.append 

723 i, n = 0, len(pat) 

724 while i < n: 

725 c = pat[i] 

726 i = i + 1 

727 if c == "*": 

728 # compress consecutive `*` into one 

729 if (not res) or res[-1] is not STAR: 

730 add(STAR) 

731 elif c == "?": 

732 add(QUESTION_MARK) 

733 elif c == "[": 

734 j = i 

735 if j < n and pat[j] == "!": 

736 j = j + 1 

737 if j < n and pat[j] == "]": 

738 j = j + 1 

739 while j < n and pat[j] != "]": 

740 j = j + 1 

741 if j >= n: 

742 add("\\[") 

743 else: 

744 stuff = pat[i:j] 

745 if "-" not in stuff: 

746 stuff = stuff.replace("\\", r"\\") 

747 else: 

748 chunks = [] 

749 k = i + 2 if pat[i] == "!" else i + 1 

750 while True: 

751 k = pat.find("-", k, j) 

752 if k < 0: 

753 break 

754 chunks.append(pat[i:k]) 

755 i = k + 1 

756 k = k + 3 

757 chunk = pat[i:j] 

758 if chunk: 

759 chunks.append(chunk) 

760 else: 

761 chunks[-1] += "-" 

762 # Remove empty ranges -- invalid in RE. 

763 for k in range(len(chunks) - 1, 0, -1): 

764 if chunks[k - 1][-1] > chunks[k][0]: 

765 chunks[k - 1] = chunks[k - 1][:-1] + chunks[k][1:] 

766 del chunks[k] 

767 # Escape backslashes and hyphens for set difference (--). 

768 # Hyphens that create ranges shouldn't be escaped. 

769 stuff = "-".join( 

770 s.replace("\\", r"\\").replace("-", r"\-") for s in chunks 

771 ) 

772 # Escape set operations (&&, ~~ and ||). 

773 stuff = re.sub(r"([&~|])", r"\\\1", stuff) 

774 i = j + 1 

775 if not stuff: 

776 # Empty range: never match. 

777 add("(?!)") 

778 elif stuff == "!": 

779 # Negated empty range: match any character. 

780 add(".") 

781 else: 

782 if stuff[0] == "!": 

783 stuff = "^" + stuff[1:] 

784 elif stuff[0] in ("^", "["): 

785 stuff = "\\" + stuff 

786 add(f"[{stuff}]") 

787 else: 

788 add(re.escape(c)) 

789 assert i == n 

790 return res 

791 

792 

793def glob_translate(pat): 

794 # Copied from: https://github.com/python/cpython/pull/106703. 

795 # The keyword parameters' values are fixed to: 

796 # recursive=True, include_hidden=True, seps="/" 

797 """Translate a pathname with shell wildcards to a regular expression.""" 

798 # fsspec paths always use "/" as their separator (AbstractFileSystem.sep), on every 

799 # platform, and glob() has already put the pattern through _strip_protocol by the 

800 # time it reaches here. Taking the separators from os.path instead would make a 

801 # backslash a separator on Windows only, so a key containing one, which is an 

802 # ordinary character on an object store, would stop matching there. 

803 seps = "/" 

804 escaped_seps = "".join(map(re.escape, seps)) 

805 any_sep = f"[{escaped_seps}]" if len(seps) > 1 else escaped_seps 

806 not_sep = f"[^{escaped_seps}]" 

807 one_last_segment = f"{not_sep}+" 

808 one_segment = f"{one_last_segment}{any_sep}" 

809 any_segments = f"(?:.+{any_sep})?" 

810 any_last_segments = ".*" 

811 results = [] 

812 parts = re.split(any_sep, pat) 

813 last_part_idx = len(parts) - 1 

814 for idx, part in enumerate(parts): 

815 if part == "*": 

816 results.append(one_segment if idx < last_part_idx else one_last_segment) 

817 continue 

818 if part == "**": 

819 results.append(any_segments if idx < last_part_idx else any_last_segments) 

820 continue 

821 elif "**" in part: 

822 raise ValueError( 

823 "Invalid pattern: '**' can only be an entire path component" 

824 ) 

825 if part: 

826 results.extend(_translate(part, f"{not_sep}*", not_sep)) 

827 if idx < last_part_idx: 

828 results.append(any_sep) 

829 res = "".join(results) 

830 return rf"(?s:{res})\Z" 

831 

832 

833try: 

834 PyBytes_FromStringAndSize = ctypes.pythonapi.PyBytes_FromStringAndSize 

835 PyBytes_FromStringAndSize.argtypes = (ctypes.c_void_p, ctypes.c_ssize_t) 

836 PyBytes_FromStringAndSize.restype = ctypes.py_object 

837 

838 PyBytes_AsString = ctypes.pythonapi.PyBytes_AsString 

839 PyBytes_AsString.argtypes = (ctypes.py_object,) 

840 PyBytes_AsString.restype = ctypes.c_void_p 

841 HAS_CPYTHON_API = True 

842except Exception: 

843 PyBytes_FromStringAndSize = None 

844 PyBytes_AsString = None 

845 HAS_CPYTHON_API = False 

846 

847 

848# Please refer to following discussion to understand why this is required at this point 

849# Discussion = https://github.com/fsspec/gcsfs/pull/795#discussion_r3032749881 

850def _fast_slice(src_bytes: bytes, offset: int, read_size: int) -> bytes: 

851 if read_size == 0: 

852 return b"" 

853 if offset < 0 or offset + read_size > len(src_bytes): 

854 raise ValueError("Slice indices out of bounds") 

855 

856 if HAS_CPYTHON_API: 

857 dest_bytes = PyBytes_FromStringAndSize(None, read_size) 

858 src_ptr = PyBytes_AsString(src_bytes) 

859 dest_ptr = PyBytes_AsString(dest_bytes) 

860 # Releases the GIL 

861 ctypes.memmove(dest_ptr, src_ptr + offset, read_size) 

862 return dest_bytes 

863 else: 

864 # Standard fallback for PyPy/non-CPython 

865 return src_bytes[offset : offset + read_size]