Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/fsspec/utils.py: 15%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1from __future__ import annotations
3import contextlib
4import ctypes
5import logging
6import math
7import os
8import re
9import sys
10import tempfile
11from collections.abc import Callable, Iterable, Iterator, Sequence
12from functools import partial
13from hashlib import md5
14from importlib.metadata import version
15from typing import IO, TYPE_CHECKING, Any, TypeVar
16from urllib.parse import urlsplit
18if TYPE_CHECKING:
19 import pathlib
20 from typing import TypeGuard
22 from fsspec.spec import AbstractFileSystem
25DEFAULT_BLOCK_SIZE = 5 * 2**20
27T = TypeVar("T")
30def infer_storage_options(
31 urlpath: str, inherit_storage_options: dict[str, Any] | None = None
32) -> dict[str, Any]:
33 """Infer storage options from URL path and merge it with existing storage
34 options.
36 Parameters
37 ----------
38 urlpath: str or unicode
39 Either local absolute file path or URL (hdfs://namenode:8020/file.csv)
40 inherit_storage_options: dict (optional)
41 Its contents will get merged with the inferred information from the
42 given path
44 Returns
45 -------
46 Storage options dict.
48 Examples
49 --------
50 >>> infer_storage_options('/mnt/datasets/test.csv') # doctest: +SKIP
51 {"protocol": "file", "path", "/mnt/datasets/test.csv"}
52 >>> infer_storage_options(
53 ... 'hdfs://username:pwd@node:123/mnt/datasets/test.csv?q=1',
54 ... inherit_storage_options={'extra': 'value'},
55 ... ) # doctest: +SKIP
56 {"protocol": "hdfs", "username": "username", "password": "pwd",
57 "host": "node", "port": 123, "path": "/mnt/datasets/test.csv",
58 "url_query": "q=1", "extra": "value"}
59 """
61 # Discover Windows paths including disk name in this special case.
62 is_filesystem = re.match(r"^[a-zA-Z]:[\\/]", urlpath)
64 # Discover URI according to RFC 3986: Scheme names consist of a
65 # sequence of characters beginning with a letter and followed by
66 # any combination of letters, digits, plus ("+"), period ("."),
67 # or hyphen ("-").
68 # https://datatracker.ietf.org/doc/html/rfc3986#section-3.1
69 is_uri = re.match(r"^[a-zA-Z0-9+.-]+://", urlpath)
71 if is_filesystem or is_uri is None:
72 return {"protocol": "file", "path": urlpath}
74 parsed_path = urlsplit(urlpath)
75 protocol = parsed_path.scheme or "file"
76 if parsed_path.fragment:
77 path = "#".join([parsed_path.path, parsed_path.fragment])
78 else:
79 path = parsed_path.path
80 if protocol == "file":
81 # Special case parsing file protocol URL on Windows according to:
82 # https://msdn.microsoft.com/en-us/library/jj710207.aspx
83 windows_path = re.match(r"^/([a-zA-Z])[:|]([\\/].*)$", path)
84 if windows_path:
85 drive, path = windows_path.groups()
86 path = f"{drive}:{path}"
88 if protocol in ["http", "https"]:
89 # for HTTP, we don't want to parse, as requests will anyway
90 return {"protocol": protocol, "path": urlpath}
92 options: dict[str, Any] = {"protocol": protocol, "path": path}
94 if parsed_path.netloc:
95 # Parse `hostname` from netloc manually because `parsed_path.hostname`
96 # lowercases the hostname which is not always desirable (e.g. in S3):
97 # https://github.com/dask/dask/issues/1417
98 host = parsed_path.netloc.rsplit("@", 1)[-1]
99 if host.startswith("[") and "]" in host:
100 # An IPv6 literal carries colons of its own, so only a colon after
101 # the closing bracket separates the port.
102 options["host"] = host[: host.index("]") + 1]
103 else:
104 options["host"] = host.rsplit(":", 1)[0]
106 if protocol in ("s3", "s3a", "gcs", "gs"):
107 options["path"] = options["host"] + options["path"]
108 else:
109 options["host"] = options["host"]
110 if parsed_path.port:
111 options["port"] = parsed_path.port
112 if parsed_path.username:
113 options["username"] = parsed_path.username
114 if parsed_path.password:
115 options["password"] = parsed_path.password
117 if parsed_path.query:
118 options["url_query"] = parsed_path.query
119 if parsed_path.fragment:
120 options["url_fragment"] = parsed_path.fragment
122 if inherit_storage_options:
123 update_storage_options(options, inherit_storage_options)
125 return options
128def update_storage_options(
129 options: dict[str, Any], inherited: dict[str, Any] | None = None
130) -> None:
131 if not inherited:
132 inherited = {}
133 collisions = set(options) & set(inherited)
134 if collisions:
135 for collision in collisions:
136 if options.get(collision) != inherited.get(collision):
137 raise KeyError(
138 f"Collision between inferred and specified storage "
139 f"option:\n{collision}"
140 )
141 options.update(inherited)
144# Compression extensions registered via fsspec.compression.register_compression
145compressions: dict[str, str] = {}
148def infer_compression(filename: str) -> str | None:
149 """Infer compression, if available, from filename.
151 Infer a named compression type, if registered and available, from filename
152 extension. This includes builtin (gz, bz2, zip) compressions, as well as
153 optional compressions. See fsspec.compression.register_compression.
154 """
155 extension = os.path.splitext(filename)[-1].strip(".").lower()
156 if extension in compressions:
157 return compressions[extension]
158 return None
161def build_name_function(max_int: float) -> Callable[[int], str]:
162 """Returns a function that receives a single integer
163 and returns it as a string padded by enough zero characters
164 to align with maximum possible integer
166 >>> name_f = build_name_function(57)
168 >>> name_f(7)
169 '07'
170 >>> name_f(31)
171 '31'
172 >>> build_name_function(1000)(42)
173 '0042'
174 >>> build_name_function(999)(42)
175 '042'
176 >>> build_name_function(0)(0)
177 '0'
178 """
179 # handle corner cases max_int is 0 or exact power of 10
180 max_int += 1e-8
182 pad_length = int(math.ceil(math.log10(max_int)))
184 def name_function(i: int) -> str:
185 return str(i).zfill(pad_length)
187 return name_function
190def seek_delimiter(file: IO[bytes], delimiter: bytes, blocksize: int) -> bool:
191 r"""Seek current file to file start, file end, or byte after delimiter seq.
193 Seeks file to next chunk delimiter, where chunks are defined on file start,
194 a delimiting sequence, and file end. Use file.tell() to see location afterwards.
195 Note that file start is a valid split, so must be at offset > 0 to seek for
196 delimiter.
198 Parameters
199 ----------
200 file: a file
201 delimiter: bytes
202 a delimiter like ``b'\n'`` or message sentinel, matching file .read() type
203 blocksize: int
204 Number of bytes to read from the file at once.
207 Returns
208 -------
209 Returns True if a delimiter was found, False if at file start or end.
211 """
213 if file.tell() == 0:
214 # beginning-of-file, return without seek
215 return False
217 # Interface is for binary IO, with delimiter as bytes, but initialize last
218 # with result of file.read to preserve compatibility with text IO.
219 last: bytes | None = None
220 while True:
221 current = file.read(blocksize)
222 if not current:
223 # end-of-file without delimiter
224 return False
225 full = last + current if last else current
226 try:
227 if delimiter in full:
228 i = full.index(delimiter)
229 file.seek(file.tell() - (len(full) - i) + len(delimiter))
230 return True
231 elif len(current) < blocksize:
232 # end-of-file without delimiter
233 return False
234 except (OSError, ValueError):
235 pass
236 last = full[-len(delimiter) :]
239def read_block(
240 f: IO[bytes],
241 offset: int,
242 length: int | None,
243 delimiter: bytes | None = None,
244 split_before: bool = False,
245) -> bytes:
246 """Read a block of bytes from a file
248 Parameters
249 ----------
250 f: File
251 Open file
252 offset: int
253 Byte offset to start read
254 length: int
255 Number of bytes to read, read through end of file if None
256 delimiter: bytes (optional)
257 Ensure reading starts and stops at delimiter bytestring
258 split_before: bool (optional)
259 Start/stop read *before* delimiter bytestring.
262 If using the ``delimiter=`` keyword argument we ensure that the read
263 starts and stops at delimiter boundaries that follow the locations
264 ``offset`` and ``offset + length``. If ``offset`` is zero then we
265 start at zero, regardless of delimiter. The bytestring returned WILL
266 include the terminating delimiter string.
268 Examples
269 --------
271 >>> from io import BytesIO # doctest: +SKIP
272 >>> f = BytesIO(b'Alice, 100\\nBob, 200\\nCharlie, 300') # doctest: +SKIP
273 >>> read_block(f, 0, 13) # doctest: +SKIP
274 b'Alice, 100\\nBo'
276 >>> read_block(f, 0, 13, delimiter=b'\\n') # doctest: +SKIP
277 b'Alice, 100\\nBob, 200\\n'
279 >>> read_block(f, 10, 10, delimiter=b'\\n') # doctest: +SKIP
280 b'Bob, 200\\nCharlie, 300'
281 """
282 if delimiter:
283 f.seek(offset)
284 found_start_delim = seek_delimiter(f, delimiter, 2**16)
285 if length is None:
286 return f.read()
287 start = f.tell()
288 length -= start - offset
290 f.seek(start + length)
291 found_end_delim = seek_delimiter(f, delimiter, 2**16)
292 end = f.tell()
294 # Adjust split location to before delimiter if seek found the
295 # delimiter sequence, not start or end of file.
296 if found_start_delim and split_before:
297 start -= len(delimiter)
299 if found_end_delim and split_before:
300 end -= len(delimiter)
302 offset = start
303 length = end - start
305 f.seek(offset)
307 # TODO: allow length to be None and read to the end of the file?
308 assert length is not None
309 b = f.read(length)
310 return b
313def tokenize(*args: Any, **kwargs: Any) -> str:
314 """Deterministic token
316 (modified from dask.base)
318 >>> tokenize([1, 2, '3'])
319 '9d71491b50023b06fc76928e6eddb952'
321 >>> tokenize('Hello') == tokenize('Hello')
322 True
323 """
324 if kwargs:
325 args += (kwargs,)
326 try:
327 h = md5(str(args).encode())
328 except ValueError:
329 # FIPS systems: https://github.com/fsspec/filesystem_spec/issues/380
330 h = md5(str(args).encode(), usedforsecurity=False)
331 return h.hexdigest()
334def stringify_path(filepath: str | os.PathLike[str] | pathlib.Path) -> str:
335 """Attempt to convert a path-like object to a string.
337 Parameters
338 ----------
339 filepath: object to be converted
341 Returns
342 -------
343 filepath_str: maybe a string version of the object
345 Notes
346 -----
347 Objects supporting the fspath protocol are coerced according to its
348 __fspath__ method.
350 For backwards compatibility with older Python version, pathlib.Path
351 objects are specially coerced.
353 Any other object is passed through unchanged, which includes bytes,
354 strings, buffers, or anything else that's not even path-like.
355 """
356 if isinstance(filepath, str):
357 return filepath
358 elif hasattr(filepath, "__fspath__"):
359 return filepath.__fspath__()
360 elif hasattr(filepath, "path"):
361 return filepath.path
362 else:
363 return filepath # type: ignore[return-value]
366def make_instance(
367 cls: Callable[..., T], args: Sequence[Any], kwargs: dict[str, Any]
368) -> T:
369 inst = cls(*args, **kwargs)
370 inst._determine_worker() # type: ignore[attr-defined]
371 return inst
374def common_prefix(paths: Iterable[str]) -> str:
375 """For a list of paths, find the shortest prefix common to all"""
376 parts = [p.split("/") for p in paths]
377 lmax = min(len(p) for p in parts)
378 end = 0
379 for i in range(lmax):
380 end = all(p[i] == parts[0][i] for p in parts)
381 if not end:
382 break
383 i += end
384 return "/".join(parts[0][:i])
387def other_paths(
388 paths: list[str],
389 path2: str | list[str],
390 exists: bool = False,
391 flatten: bool = False,
392) -> list[str]:
393 """In bulk file operations, construct a new file tree from a list of files
395 Parameters
396 ----------
397 paths: list of str
398 The input file tree
399 path2: str or list of str
400 Root to construct the new list in. If this is already a list of str, we just
401 assert it has the right number of elements.
402 exists: bool (optional)
403 For a str destination, it is already exists (and is a dir), files should
404 end up inside.
405 flatten: bool (optional)
406 Whether to flatten the input directory tree structure so that the output files
407 are in the same directory.
409 Returns
410 -------
411 list of str
412 """
414 if isinstance(path2, str):
415 path2 = path2.rstrip("/")
417 if flatten:
418 path2 = ["/".join((path2, p.split("/")[-1])) for p in paths]
419 else:
420 cp = common_prefix(paths)
421 if exists:
422 cp = cp.rsplit("/", 1)[0]
423 if not cp and all(not s.startswith("/") for s in paths):
424 path2 = ["/".join([path2, p]) for p in paths]
425 else:
426 path2 = [p.replace(cp, path2, 1) for p in paths]
427 else:
428 assert len(paths) == len(path2)
429 return path2
432def check_contained(root: str, paths: list[str]) -> None:
433 """Raise if any of ``paths`` lies outside the destination ``root``.
435 Bulk copies build their destination names by joining source names onto a
436 destination root. Those names come from the source listing, so a name
437 holding ".." segments resolves above the root and writes outside the
438 destination the caller asked for.
440 Parameters
441 ----------
442 root: str
443 The destination the caller passed.
444 paths: list of str
445 The destination names built for that root.
446 """
447 root_abs = os.path.abspath(root)
448 # normcase so that a case-insensitive platform does not report a false
449 # escape, while the message keeps the paths as the caller would see them.
450 root_key = os.path.normcase(root_abs)
451 prefix = root_key.rstrip(os.sep) + os.sep
452 for path in paths:
453 path_abs = os.path.abspath(path)
454 path_key = os.path.normcase(path_abs)
455 if path_key != root_key and not path_key.startswith(prefix):
456 raise ValueError(
457 f"path {path!r} would be copied to {path_abs!r}, which is "
458 f"outside the destination {root!r}"
459 )
462def is_exception(obj: Any) -> bool:
463 return isinstance(obj, BaseException)
466def isfilelike(f: Any) -> TypeGuard[IO[bytes]]:
467 return all(hasattr(f, attr) for attr in ["read", "close", "tell"])
470def get_protocol(url: str) -> str:
471 url = stringify_path(url)
472 parts = re.split(r"(\:\:|\://)", url, maxsplit=1)
473 if len(parts) > 1:
474 return parts[0]
475 return "file"
478def get_file_extension(url: str) -> str:
479 url = stringify_path(url)
480 # Only consider the final path component: a "." in a parent directory name
481 # (e.g. "/path/to.dir/file") is not the file's extension.
482 name = url.rsplit("/", 1)[-1]
483 # A leading dot marks a hidden file rather than an extension, so ".bashrc"
484 # has none, while ".hidden.txt" still has "txt".
485 stem, dot, extension = name.lstrip(".").rpartition(".")
486 if stem and dot:
487 return extension
488 return ""
491def can_be_local(path: str) -> bool:
492 """Can the given URL be used with open_local?"""
493 from fsspec import get_filesystem_class
495 try:
496 return getattr(get_filesystem_class(get_protocol(path)), "local_file", False)
497 except (ValueError, ImportError):
498 # not in registry or import failed
499 return False
502def get_package_version_without_import(name: str) -> str | None:
503 """For given package name, try to find the version without importing it
505 Import and package.__version__ is still the backup here, so an import
506 *might* happen.
508 Returns either the version string, or None if the package
509 or the version was not readily found.
510 """
511 if name in sys.modules:
512 mod = sys.modules[name]
513 if hasattr(mod, "__version__"):
514 return mod.__version__
515 try:
516 return version(name)
517 except: # noqa: E722
518 pass
519 try:
520 import importlib
522 mod = importlib.import_module(name)
523 return mod.__version__
524 except (ImportError, AttributeError):
525 return None
528def setup_logging(
529 logger: logging.Logger | None = None,
530 logger_name: str | None = None,
531 level: str = "DEBUG",
532 clear: bool = True,
533) -> logging.Logger:
534 if logger is None and logger_name is None:
535 raise ValueError("Provide either logger object or logger name")
536 logger = logger or logging.getLogger(logger_name)
537 handle = logging.StreamHandler()
538 formatter = logging.Formatter(
539 "%(asctime)s - %(name)s - %(levelname)s - %(funcName)s -- %(message)s"
540 )
541 handle.setFormatter(formatter)
542 if clear:
543 logger.handlers.clear()
544 logger.addHandler(handle)
545 logger.setLevel(level)
546 return logger
549def _unstrip_protocol(name: str, fs: AbstractFileSystem) -> str:
550 return fs.unstrip_protocol(name)
553def mirror_from(
554 origin_name: str, methods: Iterable[str]
555) -> Callable[[type[T]], type[T]]:
556 """Mirror attributes and methods from the given
557 origin_name attribute of the instance to the
558 decorated class"""
560 def origin_getter(method: str, self: Any) -> Any:
561 origin = getattr(self, origin_name)
562 return getattr(origin, method)
564 def wrapper(cls: type[T]) -> type[T]:
565 for method in methods:
566 wrapped_method = partial(origin_getter, method)
567 setattr(cls, method, property(wrapped_method))
568 return cls
570 return wrapper
573@contextlib.contextmanager
574def nullcontext(obj: T) -> Iterator[T]:
575 yield obj
578def merge_offset_ranges(
579 paths: list[str],
580 starts: list[int | None] | int | None,
581 ends: list[int | None] | int | None,
582 max_gap: int = 0,
583 max_block: int | None = None,
584 sort: bool = True,
585) -> tuple[list[str], list[int], list[int | None]]:
586 """Merge adjacent byte-offset ranges when the inter-range
587 gap is <= `max_gap`, and when the merged byte range does not
588 exceed `max_block` (if specified). Every input range is covered by
589 at least one returned range. Overlapping input ranges are merged
590 where `max_block` allows it, so returned ranges may overlap once a
591 chain of overlapping inputs reaches `max_block`; a single input
592 range larger than `max_block` is still returned whole.
594 An `end` of `None` means to the end of the file. By default, this
595 function will re-order the input paths and byte ranges to ensure
596 sorted order.
598 Passing `sort=False` skips the re-ordering, which is only worthwhile
599 when the inputs are already grouped by path and ascending by start
600 within each path. Ranges that break that order still appear in the
601 output, as their own range rather than merged, so coverage holds for
602 any input order.
603 """
604 # Check input
605 if not isinstance(paths, list):
606 raise TypeError
607 if not isinstance(starts, list):
608 starts = [starts] * len(paths)
609 if not isinstance(ends, list):
610 ends = [ends] * len(paths)
611 if len(starts) != len(paths) or len(ends) != len(paths):
612 raise ValueError
614 starts_i: list[int] = [s or 0 for s in starts]
615 ends_i: list[int | None] = ends
617 # Early Return
618 if len(starts_i) <= 1:
619 return paths, starts_i, ends_i
621 # Sort by paths and then ranges if `sort=True`
622 if sort:
623 ranges = sorted(
624 zip(paths, starts_i, ends_i),
625 # None end sorts last (covers furthest into the file)
626 key=lambda pse: (pse[0], pse[1], math.inf if pse[2] is None else pse[2]),
627 )
628 paths = [r[0] for r in ranges]
629 starts_i = [r[1] for r in ranges]
630 ends_i = [r[2] for r in ranges]
632 # Loop through the coupled `paths`, `starts`, and
633 # `ends`, and merge adjacent blocks when appropriate
634 new_paths = paths[:1]
635 new_starts = starts_i[:1]
636 new_ends = ends_i[:1]
637 for path, start, end in zip(paths[1:], starts_i[1:], ends_i[1:]):
638 prev_end = new_ends[-1]
639 if path != new_paths[-1]:
640 # Cannot merge with previous block
641 new_paths.append(path)
642 new_starts.append(start)
643 new_ends.append(end)
644 elif start < new_starts[-1]:
645 # Out of order (only possible when `sort=False`). Walking the
646 # current block start backwards would uncover bytes already
647 # attributed to it, so give this range its own block
648 new_paths.append(path)
649 new_starts.append(start)
650 new_ends.append(end)
651 elif prev_end is None:
652 # Previous block already covers the rest of the file
653 continue
654 elif start < prev_end:
655 # Overlap / nested
656 if end is not None and end <= prev_end:
657 # Already covered by the current block
658 continue
659 elif (
660 end is not None
661 and max_block is not None
662 and (end - new_starts[-1]) > max_block
663 ):
664 # Extending would exceed `max_block`. Start a new block,
665 # which overlaps the previous one, rather than letting a
666 # chain of overlaps grow the block without bound
667 new_paths.append(path)
668 new_starts.append(start)
669 new_ends.append(end)
670 else:
671 # An `end` of None extends the block to EOF; a separate
672 # block would subsume the current one anyway
673 new_ends[-1] = end
674 elif (start - prev_end) > max_gap or (
675 max_block is not None
676 and (end is None or (end - new_starts[-1]) > max_block)
677 ):
678 # Gap too large, or merging would exceed `max_block`
679 new_paths.append(path)
680 new_starts.append(start)
681 new_ends.append(end)
682 else:
683 # Merge with the previous block
684 new_ends[-1] = end
686 return new_paths, new_starts, new_ends
689def file_size(filelike: IO[bytes]) -> int:
690 """Find length of any open read-mode file-like"""
691 pos = filelike.tell()
692 try:
693 return filelike.seek(0, 2)
694 finally:
695 filelike.seek(pos)
698@contextlib.contextmanager
699def atomic_write(path: str, mode: str = "wb"):
700 """
701 A context manager that opens a temporary file next to `path` and, on exit,
702 replaces `path` with the temporary file, thereby updating `path`
703 atomically.
704 """
705 fd, fn = tempfile.mkstemp(
706 dir=os.path.dirname(path), prefix=os.path.basename(path) + "-"
707 )
708 try:
709 with open(fd, mode) as fp:
710 yield fp
711 except BaseException:
712 with contextlib.suppress(FileNotFoundError):
713 os.unlink(fn)
714 raise
715 else:
716 os.replace(fn, path)
719def _translate(pat, STAR, QUESTION_MARK):
720 # Copied from: https://github.com/python/cpython/pull/106703.
721 res: list[str] = []
722 add = res.append
723 i, n = 0, len(pat)
724 while i < n:
725 c = pat[i]
726 i = i + 1
727 if c == "*":
728 # compress consecutive `*` into one
729 if (not res) or res[-1] is not STAR:
730 add(STAR)
731 elif c == "?":
732 add(QUESTION_MARK)
733 elif c == "[":
734 j = i
735 if j < n and pat[j] == "!":
736 j = j + 1
737 if j < n and pat[j] == "]":
738 j = j + 1
739 while j < n and pat[j] != "]":
740 j = j + 1
741 if j >= n:
742 add("\\[")
743 else:
744 stuff = pat[i:j]
745 if "-" not in stuff:
746 stuff = stuff.replace("\\", r"\\")
747 else:
748 chunks = []
749 k = i + 2 if pat[i] == "!" else i + 1
750 while True:
751 k = pat.find("-", k, j)
752 if k < 0:
753 break
754 chunks.append(pat[i:k])
755 i = k + 1
756 k = k + 3
757 chunk = pat[i:j]
758 if chunk:
759 chunks.append(chunk)
760 else:
761 chunks[-1] += "-"
762 # Remove empty ranges -- invalid in RE.
763 for k in range(len(chunks) - 1, 0, -1):
764 if chunks[k - 1][-1] > chunks[k][0]:
765 chunks[k - 1] = chunks[k - 1][:-1] + chunks[k][1:]
766 del chunks[k]
767 # Escape backslashes and hyphens for set difference (--).
768 # Hyphens that create ranges shouldn't be escaped.
769 stuff = "-".join(
770 s.replace("\\", r"\\").replace("-", r"\-") for s in chunks
771 )
772 # Escape set operations (&&, ~~ and ||).
773 stuff = re.sub(r"([&~|])", r"\\\1", stuff)
774 i = j + 1
775 if not stuff:
776 # Empty range: never match.
777 add("(?!)")
778 elif stuff == "!":
779 # Negated empty range: match any character.
780 add(".")
781 else:
782 if stuff[0] == "!":
783 stuff = "^" + stuff[1:]
784 elif stuff[0] in ("^", "["):
785 stuff = "\\" + stuff
786 add(f"[{stuff}]")
787 else:
788 add(re.escape(c))
789 assert i == n
790 return res
793def glob_translate(pat):
794 # Copied from: https://github.com/python/cpython/pull/106703.
795 # The keyword parameters' values are fixed to:
796 # recursive=True, include_hidden=True, seps="/"
797 """Translate a pathname with shell wildcards to a regular expression."""
798 # fsspec paths always use "/" as their separator (AbstractFileSystem.sep), on every
799 # platform, and glob() has already put the pattern through _strip_protocol by the
800 # time it reaches here. Taking the separators from os.path instead would make a
801 # backslash a separator on Windows only, so a key containing one, which is an
802 # ordinary character on an object store, would stop matching there.
803 seps = "/"
804 escaped_seps = "".join(map(re.escape, seps))
805 any_sep = f"[{escaped_seps}]" if len(seps) > 1 else escaped_seps
806 not_sep = f"[^{escaped_seps}]"
807 one_last_segment = f"{not_sep}+"
808 one_segment = f"{one_last_segment}{any_sep}"
809 any_segments = f"(?:.+{any_sep})?"
810 any_last_segments = ".*"
811 results = []
812 parts = re.split(any_sep, pat)
813 last_part_idx = len(parts) - 1
814 for idx, part in enumerate(parts):
815 if part == "*":
816 results.append(one_segment if idx < last_part_idx else one_last_segment)
817 continue
818 if part == "**":
819 results.append(any_segments if idx < last_part_idx else any_last_segments)
820 continue
821 elif "**" in part:
822 raise ValueError(
823 "Invalid pattern: '**' can only be an entire path component"
824 )
825 if part:
826 results.extend(_translate(part, f"{not_sep}*", not_sep))
827 if idx < last_part_idx:
828 results.append(any_sep)
829 res = "".join(results)
830 return rf"(?s:{res})\Z"
833try:
834 PyBytes_FromStringAndSize = ctypes.pythonapi.PyBytes_FromStringAndSize
835 PyBytes_FromStringAndSize.argtypes = (ctypes.c_void_p, ctypes.c_ssize_t)
836 PyBytes_FromStringAndSize.restype = ctypes.py_object
838 PyBytes_AsString = ctypes.pythonapi.PyBytes_AsString
839 PyBytes_AsString.argtypes = (ctypes.py_object,)
840 PyBytes_AsString.restype = ctypes.c_void_p
841 HAS_CPYTHON_API = True
842except Exception:
843 PyBytes_FromStringAndSize = None
844 PyBytes_AsString = None
845 HAS_CPYTHON_API = False
848# Please refer to following discussion to understand why this is required at this point
849# Discussion = https://github.com/fsspec/gcsfs/pull/795#discussion_r3032749881
850def _fast_slice(src_bytes: bytes, offset: int, read_size: int) -> bytes:
851 if read_size == 0:
852 return b""
853 if offset < 0 or offset + read_size > len(src_bytes):
854 raise ValueError("Slice indices out of bounds")
856 if HAS_CPYTHON_API:
857 dest_bytes = PyBytes_FromStringAndSize(None, read_size)
858 src_ptr = PyBytes_AsString(src_bytes)
859 dest_ptr = PyBytes_AsString(dest_bytes)
860 # Releases the GIL
861 ctypes.memmove(dest_ptr, src_ptr + offset, read_size)
862 return dest_bytes
863 else:
864 # Standard fallback for PyPy/non-CPython
865 return src_bytes[offset : offset + read_size]