Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/fsspec/utils.py: 14%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1from __future__ import annotations
3import contextlib
4import logging
5import math
6import os
7import re
8import sys
9import tempfile
10from collections.abc import Callable, Iterable, Iterator, Sequence
11from functools import partial
12from hashlib import md5
13from importlib.metadata import version
14from typing import IO, TYPE_CHECKING, Any, TypeVar
15from urllib.parse import urlsplit
17if TYPE_CHECKING:
18 import pathlib
19 from typing import TypeGuard
21 from fsspec.spec import AbstractFileSystem
24DEFAULT_BLOCK_SIZE = 5 * 2**20
26T = TypeVar("T")
29def infer_storage_options(
30 urlpath: str, inherit_storage_options: dict[str, Any] | None = None
31) -> dict[str, Any]:
32 """Infer storage options from URL path and merge it with existing storage
33 options.
35 Parameters
36 ----------
37 urlpath: str or unicode
38 Either local absolute file path or URL (hdfs://namenode:8020/file.csv)
39 inherit_storage_options: dict (optional)
40 Its contents will get merged with the inferred information from the
41 given path
43 Returns
44 -------
45 Storage options dict.
47 Examples
48 --------
49 >>> infer_storage_options('/mnt/datasets/test.csv') # doctest: +SKIP
50 {"protocol": "file", "path", "/mnt/datasets/test.csv"}
51 >>> infer_storage_options(
52 ... 'hdfs://username:pwd@node:123/mnt/datasets/test.csv?q=1',
53 ... inherit_storage_options={'extra': 'value'},
54 ... ) # doctest: +SKIP
55 {"protocol": "hdfs", "username": "username", "password": "pwd",
56 "host": "node", "port": 123, "path": "/mnt/datasets/test.csv",
57 "url_query": "q=1", "extra": "value"}
58 """
60 # Discover Windows paths including disk name in this special case.
61 is_filesystem = re.match(r"^[a-zA-Z]:[\\/]", urlpath)
63 # Discover URI according to RFC 3986: Scheme names consist of a
64 # sequence of characters beginning with a letter and followed by
65 # any combination of letters, digits, plus ("+"), period ("."),
66 # or hyphen ("-").
67 # https://datatracker.ietf.org/doc/html/rfc3986#section-3.1
68 is_uri = re.match(r"^[a-zA-Z0-9+.-]+://", urlpath)
70 if is_filesystem or is_uri is None:
71 return {"protocol": "file", "path": urlpath}
73 parsed_path = urlsplit(urlpath)
74 protocol = parsed_path.scheme or "file"
75 if parsed_path.fragment:
76 path = "#".join([parsed_path.path, parsed_path.fragment])
77 else:
78 path = parsed_path.path
79 if protocol == "file":
80 # Special case parsing file protocol URL on Windows according to:
81 # https://msdn.microsoft.com/en-us/library/jj710207.aspx
82 windows_path = re.match(r"^/([a-zA-Z])[:|]([\\/].*)$", path)
83 if windows_path:
84 drive, path = windows_path.groups()
85 path = f"{drive}:{path}"
87 if protocol in ["http", "https"]:
88 # for HTTP, we don't want to parse, as requests will anyway
89 return {"protocol": protocol, "path": urlpath}
91 options: dict[str, Any] = {"protocol": protocol, "path": path}
93 if parsed_path.netloc:
94 # Parse `hostname` from netloc manually because `parsed_path.hostname`
95 # lowercases the hostname which is not always desirable (e.g. in S3):
96 # https://github.com/dask/dask/issues/1417
97 options["host"] = parsed_path.netloc.rsplit("@", 1)[-1].rsplit(":", 1)[0]
99 if protocol in ("s3", "s3a", "gcs", "gs"):
100 options["path"] = options["host"] + options["path"]
101 else:
102 options["host"] = options["host"]
103 if parsed_path.port:
104 options["port"] = parsed_path.port
105 if parsed_path.username:
106 options["username"] = parsed_path.username
107 if parsed_path.password:
108 options["password"] = parsed_path.password
110 if parsed_path.query:
111 options["url_query"] = parsed_path.query
112 if parsed_path.fragment:
113 options["url_fragment"] = parsed_path.fragment
115 if inherit_storage_options:
116 update_storage_options(options, inherit_storage_options)
118 return options
121def update_storage_options(
122 options: dict[str, Any], inherited: dict[str, Any] | None = None
123) -> None:
124 if not inherited:
125 inherited = {}
126 collisions = set(options) & set(inherited)
127 if collisions:
128 for collision in collisions:
129 if options.get(collision) != inherited.get(collision):
130 raise KeyError(
131 f"Collision between inferred and specified storage "
132 f"option:\n{collision}"
133 )
134 options.update(inherited)
137# Compression extensions registered via fsspec.compression.register_compression
138compressions: dict[str, str] = {}
141def infer_compression(filename: str) -> str | None:
142 """Infer compression, if available, from filename.
144 Infer a named compression type, if registered and available, from filename
145 extension. This includes builtin (gz, bz2, zip) compressions, as well as
146 optional compressions. See fsspec.compression.register_compression.
147 """
148 extension = os.path.splitext(filename)[-1].strip(".").lower()
149 if extension in compressions:
150 return compressions[extension]
151 return None
154def build_name_function(max_int: float) -> Callable[[int], str]:
155 """Returns a function that receives a single integer
156 and returns it as a string padded by enough zero characters
157 to align with maximum possible integer
159 >>> name_f = build_name_function(57)
161 >>> name_f(7)
162 '07'
163 >>> name_f(31)
164 '31'
165 >>> build_name_function(1000)(42)
166 '0042'
167 >>> build_name_function(999)(42)
168 '042'
169 >>> build_name_function(0)(0)
170 '0'
171 """
172 # handle corner cases max_int is 0 or exact power of 10
173 max_int += 1e-8
175 pad_length = int(math.ceil(math.log10(max_int)))
177 def name_function(i: int) -> str:
178 return str(i).zfill(pad_length)
180 return name_function
183def seek_delimiter(file: IO[bytes], delimiter: bytes, blocksize: int) -> bool:
184 r"""Seek current file to file start, file end, or byte after delimiter seq.
186 Seeks file to next chunk delimiter, where chunks are defined on file start,
187 a delimiting sequence, and file end. Use file.tell() to see location afterwards.
188 Note that file start is a valid split, so must be at offset > 0 to seek for
189 delimiter.
191 Parameters
192 ----------
193 file: a file
194 delimiter: bytes
195 a delimiter like ``b'\n'`` or message sentinel, matching file .read() type
196 blocksize: int
197 Number of bytes to read from the file at once.
200 Returns
201 -------
202 Returns True if a delimiter was found, False if at file start or end.
204 """
206 if file.tell() == 0:
207 # beginning-of-file, return without seek
208 return False
210 # Interface is for binary IO, with delimiter as bytes, but initialize last
211 # with result of file.read to preserve compatibility with text IO.
212 last: bytes | None = None
213 while True:
214 current = file.read(blocksize)
215 if not current:
216 # end-of-file without delimiter
217 return False
218 full = last + current if last else current
219 try:
220 if delimiter in full:
221 i = full.index(delimiter)
222 file.seek(file.tell() - (len(full) - i) + len(delimiter))
223 return True
224 elif len(current) < blocksize:
225 # end-of-file without delimiter
226 return False
227 except (OSError, ValueError):
228 pass
229 last = full[-len(delimiter) :]
232def read_block(
233 f: IO[bytes],
234 offset: int,
235 length: int | None,
236 delimiter: bytes | None = None,
237 split_before: bool = False,
238) -> bytes:
239 """Read a block of bytes from a file
241 Parameters
242 ----------
243 f: File
244 Open file
245 offset: int
246 Byte offset to start read
247 length: int
248 Number of bytes to read, read through end of file if None
249 delimiter: bytes (optional)
250 Ensure reading starts and stops at delimiter bytestring
251 split_before: bool (optional)
252 Start/stop read *before* delimiter bytestring.
255 If using the ``delimiter=`` keyword argument we ensure that the read
256 starts and stops at delimiter boundaries that follow the locations
257 ``offset`` and ``offset + length``. If ``offset`` is zero then we
258 start at zero, regardless of delimiter. The bytestring returned WILL
259 include the terminating delimiter string.
261 Examples
262 --------
264 >>> from io import BytesIO # doctest: +SKIP
265 >>> f = BytesIO(b'Alice, 100\\nBob, 200\\nCharlie, 300') # doctest: +SKIP
266 >>> read_block(f, 0, 13) # doctest: +SKIP
267 b'Alice, 100\\nBo'
269 >>> read_block(f, 0, 13, delimiter=b'\\n') # doctest: +SKIP
270 b'Alice, 100\\nBob, 200\\n'
272 >>> read_block(f, 10, 10, delimiter=b'\\n') # doctest: +SKIP
273 b'Bob, 200\\nCharlie, 300'
274 """
275 if delimiter:
276 f.seek(offset)
277 found_start_delim = seek_delimiter(f, delimiter, 2**16)
278 if length is None:
279 return f.read()
280 start = f.tell()
281 length -= start - offset
283 f.seek(start + length)
284 found_end_delim = seek_delimiter(f, delimiter, 2**16)
285 end = f.tell()
287 # Adjust split location to before delimiter if seek found the
288 # delimiter sequence, not start or end of file.
289 if found_start_delim and split_before:
290 start -= len(delimiter)
292 if found_end_delim and split_before:
293 end -= len(delimiter)
295 offset = start
296 length = end - start
298 f.seek(offset)
300 # TODO: allow length to be None and read to the end of the file?
301 assert length is not None
302 b = f.read(length)
303 return b
306def tokenize(*args: Any, **kwargs: Any) -> str:
307 """Deterministic token
309 (modified from dask.base)
311 >>> tokenize([1, 2, '3'])
312 '9d71491b50023b06fc76928e6eddb952'
314 >>> tokenize('Hello') == tokenize('Hello')
315 True
316 """
317 if kwargs:
318 args += (kwargs,)
319 try:
320 h = md5(str(args).encode())
321 except ValueError:
322 # FIPS systems: https://github.com/fsspec/filesystem_spec/issues/380
323 h = md5(str(args).encode(), usedforsecurity=False)
324 return h.hexdigest()
327def stringify_path(filepath: str | os.PathLike[str] | pathlib.Path) -> str:
328 """Attempt to convert a path-like object to a string.
330 Parameters
331 ----------
332 filepath: object to be converted
334 Returns
335 -------
336 filepath_str: maybe a string version of the object
338 Notes
339 -----
340 Objects supporting the fspath protocol are coerced according to its
341 __fspath__ method.
343 For backwards compatibility with older Python version, pathlib.Path
344 objects are specially coerced.
346 Any other object is passed through unchanged, which includes bytes,
347 strings, buffers, or anything else that's not even path-like.
348 """
349 if isinstance(filepath, str):
350 return filepath
351 elif hasattr(filepath, "__fspath__"):
352 return filepath.__fspath__()
353 elif hasattr(filepath, "path"):
354 return filepath.path
355 else:
356 return filepath # type: ignore[return-value]
359def make_instance(
360 cls: Callable[..., T], args: Sequence[Any], kwargs: dict[str, Any]
361) -> T:
362 inst = cls(*args, **kwargs)
363 inst._determine_worker() # type: ignore[attr-defined]
364 return inst
367def common_prefix(paths: Iterable[str]) -> str:
368 """For a list of paths, find the shortest prefix common to all"""
369 parts = [p.split("/") for p in paths]
370 lmax = min(len(p) for p in parts)
371 end = 0
372 for i in range(lmax):
373 end = all(p[i] == parts[0][i] for p in parts)
374 if not end:
375 break
376 i += end
377 return "/".join(parts[0][:i])
380def other_paths(
381 paths: list[str],
382 path2: str | list[str],
383 exists: bool = False,
384 flatten: bool = False,
385) -> list[str]:
386 """In bulk file operations, construct a new file tree from a list of files
388 Parameters
389 ----------
390 paths: list of str
391 The input file tree
392 path2: str or list of str
393 Root to construct the new list in. If this is already a list of str, we just
394 assert it has the right number of elements.
395 exists: bool (optional)
396 For a str destination, it is already exists (and is a dir), files should
397 end up inside.
398 flatten: bool (optional)
399 Whether to flatten the input directory tree structure so that the output files
400 are in the same directory.
402 Returns
403 -------
404 list of str
405 """
407 if isinstance(path2, str):
408 path2 = path2.rstrip("/")
410 if flatten:
411 path2 = ["/".join((path2, p.split("/")[-1])) for p in paths]
412 else:
413 cp = common_prefix(paths)
414 if exists:
415 cp = cp.rsplit("/", 1)[0]
416 if not cp and all(not s.startswith("/") for s in paths):
417 path2 = ["/".join([path2, p]) for p in paths]
418 else:
419 path2 = [p.replace(cp, path2, 1) for p in paths]
420 else:
421 assert len(paths) == len(path2)
422 return path2
425def check_contained(root: str, paths: list[str]) -> None:
426 """Raise if any of ``paths`` lies outside the destination ``root``.
428 Bulk copies build their destination names by joining source names onto a
429 destination root. Those names come from the source listing, so a name
430 holding ".." segments resolves above the root and writes outside the
431 destination the caller asked for.
433 Parameters
434 ----------
435 root: str
436 The destination the caller passed.
437 paths: list of str
438 The destination names built for that root.
439 """
440 root_abs = os.path.abspath(root)
441 # normcase so that a case-insensitive platform does not report a false
442 # escape, while the message keeps the paths as the caller would see them.
443 root_key = os.path.normcase(root_abs)
444 prefix = root_key.rstrip(os.sep) + os.sep
445 for path in paths:
446 path_abs = os.path.abspath(path)
447 path_key = os.path.normcase(path_abs)
448 if path_key != root_key and not path_key.startswith(prefix):
449 raise ValueError(
450 f"path {path!r} would be copied to {path_abs!r}, which is "
451 f"outside the destination {root!r}"
452 )
455def is_exception(obj: Any) -> bool:
456 return isinstance(obj, BaseException)
459def isfilelike(f: Any) -> TypeGuard[IO[bytes]]:
460 return all(hasattr(f, attr) for attr in ["read", "close", "tell"])
463def get_protocol(url: str) -> str:
464 url = stringify_path(url)
465 parts = re.split(r"(\:\:|\://)", url, maxsplit=1)
466 if len(parts) > 1:
467 return parts[0]
468 return "file"
471def get_file_extension(url: str) -> str:
472 url = stringify_path(url)
473 # Only consider the final path component: a "." in a parent directory name
474 # (e.g. "/path/to.dir/file") is not the file's extension.
475 ext_parts = url.rsplit("/", 1)[-1].rsplit(".", 1)
476 if len(ext_parts) > 1:
477 return ext_parts[-1]
478 return ""
481def can_be_local(path: str) -> bool:
482 """Can the given URL be used with open_local?"""
483 from fsspec import get_filesystem_class
485 try:
486 return getattr(get_filesystem_class(get_protocol(path)), "local_file", False)
487 except (ValueError, ImportError):
488 # not in registry or import failed
489 return False
492def get_package_version_without_import(name: str) -> str | None:
493 """For given package name, try to find the version without importing it
495 Import and package.__version__ is still the backup here, so an import
496 *might* happen.
498 Returns either the version string, or None if the package
499 or the version was not readily found.
500 """
501 if name in sys.modules:
502 mod = sys.modules[name]
503 if hasattr(mod, "__version__"):
504 return mod.__version__
505 try:
506 return version(name)
507 except: # noqa: E722
508 pass
509 try:
510 import importlib
512 mod = importlib.import_module(name)
513 return mod.__version__
514 except (ImportError, AttributeError):
515 return None
518def setup_logging(
519 logger: logging.Logger | None = None,
520 logger_name: str | None = None,
521 level: str = "DEBUG",
522 clear: bool = True,
523) -> logging.Logger:
524 if logger is None and logger_name is None:
525 raise ValueError("Provide either logger object or logger name")
526 logger = logger or logging.getLogger(logger_name)
527 handle = logging.StreamHandler()
528 formatter = logging.Formatter(
529 "%(asctime)s - %(name)s - %(levelname)s - %(funcName)s -- %(message)s"
530 )
531 handle.setFormatter(formatter)
532 if clear:
533 logger.handlers.clear()
534 logger.addHandler(handle)
535 logger.setLevel(level)
536 return logger
539def _unstrip_protocol(name: str, fs: AbstractFileSystem) -> str:
540 return fs.unstrip_protocol(name)
543def mirror_from(
544 origin_name: str, methods: Iterable[str]
545) -> Callable[[type[T]], type[T]]:
546 """Mirror attributes and methods from the given
547 origin_name attribute of the instance to the
548 decorated class"""
550 def origin_getter(method: str, self: Any) -> Any:
551 origin = getattr(self, origin_name)
552 return getattr(origin, method)
554 def wrapper(cls: type[T]) -> type[T]:
555 for method in methods:
556 wrapped_method = partial(origin_getter, method)
557 setattr(cls, method, property(wrapped_method))
558 return cls
560 return wrapper
563@contextlib.contextmanager
564def nullcontext(obj: T) -> Iterator[T]:
565 yield obj
568def merge_offset_ranges(
569 paths: list[str],
570 starts: list[int | None] | int | None,
571 ends: list[int | None] | int | None,
572 max_gap: int = 0,
573 max_block: int | None = None,
574 sort: bool = True,
575) -> tuple[list[str], list[int], list[int | None]]:
576 """Merge adjacent byte-offset ranges when the inter-range
577 gap is <= `max_gap`, and when the merged byte range does not
578 exceed `max_block` (if specified). Every input range is covered by
579 at least one returned range. Overlapping input ranges are merged
580 where `max_block` allows it, so returned ranges may overlap once a
581 chain of overlapping inputs reaches `max_block`; a single input
582 range larger than `max_block` is still returned whole.
584 An `end` of `None` means to the end of the file. By default, this
585 function will re-order the input paths and byte ranges to ensure
586 sorted order.
588 Passing `sort=False` skips the re-ordering, which is only worthwhile
589 when the inputs are already grouped by path and ascending by start
590 within each path. Ranges that break that order still appear in the
591 output, as their own range rather than merged, so coverage holds for
592 any input order.
593 """
594 # Check input
595 if not isinstance(paths, list):
596 raise TypeError
597 if not isinstance(starts, list):
598 starts = [starts] * len(paths)
599 if not isinstance(ends, list):
600 ends = [ends] * len(paths)
601 if len(starts) != len(paths) or len(ends) != len(paths):
602 raise ValueError
604 starts_i: list[int] = [s or 0 for s in starts]
605 ends_i: list[int | None] = ends
607 # Early Return
608 if len(starts_i) <= 1:
609 return paths, starts_i, ends_i
611 # Sort by paths and then ranges if `sort=True`
612 if sort:
613 ranges = sorted(
614 zip(paths, starts_i, ends_i),
615 # None end sorts last (covers furthest into the file)
616 key=lambda pse: (pse[0], pse[1], math.inf if pse[2] is None else pse[2]),
617 )
618 paths = [r[0] for r in ranges]
619 starts_i = [r[1] for r in ranges]
620 ends_i = [r[2] for r in ranges]
622 # Loop through the coupled `paths`, `starts`, and
623 # `ends`, and merge adjacent blocks when appropriate
624 new_paths = paths[:1]
625 new_starts = starts_i[:1]
626 new_ends = ends_i[:1]
627 for path, start, end in zip(paths[1:], starts_i[1:], ends_i[1:]):
628 prev_end = new_ends[-1]
629 if path != new_paths[-1]:
630 # Cannot merge with previous block
631 new_paths.append(path)
632 new_starts.append(start)
633 new_ends.append(end)
634 elif start < new_starts[-1]:
635 # Out of order (only possible when `sort=False`). Walking the
636 # current block start backwards would uncover bytes already
637 # attributed to it, so give this range its own block
638 new_paths.append(path)
639 new_starts.append(start)
640 new_ends.append(end)
641 elif prev_end is None:
642 # Previous block already covers the rest of the file
643 continue
644 elif start < prev_end:
645 # Overlap / nested
646 if end is not None and end <= prev_end:
647 # Already covered by the current block
648 continue
649 elif (
650 end is not None
651 and max_block is not None
652 and (end - new_starts[-1]) > max_block
653 ):
654 # Extending would exceed `max_block`. Start a new block,
655 # which overlaps the previous one, rather than letting a
656 # chain of overlaps grow the block without bound
657 new_paths.append(path)
658 new_starts.append(start)
659 new_ends.append(end)
660 else:
661 # An `end` of None extends the block to EOF; a separate
662 # block would subsume the current one anyway
663 new_ends[-1] = end
664 elif (start - prev_end) > max_gap or (
665 max_block is not None
666 and (end is None or (end - new_starts[-1]) > max_block)
667 ):
668 # Gap too large, or merging would exceed `max_block`
669 new_paths.append(path)
670 new_starts.append(start)
671 new_ends.append(end)
672 else:
673 # Merge with the previous block
674 new_ends[-1] = end
676 return new_paths, new_starts, new_ends
679def file_size(filelike: IO[bytes]) -> int:
680 """Find length of any open read-mode file-like"""
681 pos = filelike.tell()
682 try:
683 return filelike.seek(0, 2)
684 finally:
685 filelike.seek(pos)
688@contextlib.contextmanager
689def atomic_write(path: str, mode: str = "wb"):
690 """
691 A context manager that opens a temporary file next to `path` and, on exit,
692 replaces `path` with the temporary file, thereby updating `path`
693 atomically.
694 """
695 fd, fn = tempfile.mkstemp(
696 dir=os.path.dirname(path), prefix=os.path.basename(path) + "-"
697 )
698 try:
699 with open(fd, mode) as fp:
700 yield fp
701 except BaseException:
702 with contextlib.suppress(FileNotFoundError):
703 os.unlink(fn)
704 raise
705 else:
706 os.replace(fn, path)
709def _translate(pat, STAR, QUESTION_MARK):
710 # Copied from: https://github.com/python/cpython/pull/106703.
711 res: list[str] = []
712 add = res.append
713 i, n = 0, len(pat)
714 while i < n:
715 c = pat[i]
716 i = i + 1
717 if c == "*":
718 # compress consecutive `*` into one
719 if (not res) or res[-1] is not STAR:
720 add(STAR)
721 elif c == "?":
722 add(QUESTION_MARK)
723 elif c == "[":
724 j = i
725 if j < n and pat[j] == "!":
726 j = j + 1
727 if j < n and pat[j] == "]":
728 j = j + 1
729 while j < n and pat[j] != "]":
730 j = j + 1
731 if j >= n:
732 add("\\[")
733 else:
734 stuff = pat[i:j]
735 if "-" not in stuff:
736 stuff = stuff.replace("\\", r"\\")
737 else:
738 chunks = []
739 k = i + 2 if pat[i] == "!" else i + 1
740 while True:
741 k = pat.find("-", k, j)
742 if k < 0:
743 break
744 chunks.append(pat[i:k])
745 i = k + 1
746 k = k + 3
747 chunk = pat[i:j]
748 if chunk:
749 chunks.append(chunk)
750 else:
751 chunks[-1] += "-"
752 # Remove empty ranges -- invalid in RE.
753 for k in range(len(chunks) - 1, 0, -1):
754 if chunks[k - 1][-1] > chunks[k][0]:
755 chunks[k - 1] = chunks[k - 1][:-1] + chunks[k][1:]
756 del chunks[k]
757 # Escape backslashes and hyphens for set difference (--).
758 # Hyphens that create ranges shouldn't be escaped.
759 stuff = "-".join(
760 s.replace("\\", r"\\").replace("-", r"\-") for s in chunks
761 )
762 # Escape set operations (&&, ~~ and ||).
763 stuff = re.sub(r"([&~|])", r"\\\1", stuff)
764 i = j + 1
765 if not stuff:
766 # Empty range: never match.
767 add("(?!)")
768 elif stuff == "!":
769 # Negated empty range: match any character.
770 add(".")
771 else:
772 if stuff[0] == "!":
773 stuff = "^" + stuff[1:]
774 elif stuff[0] in ("^", "["):
775 stuff = "\\" + stuff
776 add(f"[{stuff}]")
777 else:
778 add(re.escape(c))
779 assert i == n
780 return res
783def glob_translate(pat):
784 # Copied from: https://github.com/python/cpython/pull/106703.
785 # The keyword parameters' values are fixed to:
786 # recursive=True, include_hidden=True, seps=None
787 """Translate a pathname with shell wildcards to a regular expression."""
788 if os.path.altsep:
789 seps = os.path.sep + os.path.altsep
790 else:
791 seps = os.path.sep
792 escaped_seps = "".join(map(re.escape, seps))
793 any_sep = f"[{escaped_seps}]" if len(seps) > 1 else escaped_seps
794 not_sep = f"[^{escaped_seps}]"
795 one_last_segment = f"{not_sep}+"
796 one_segment = f"{one_last_segment}{any_sep}"
797 any_segments = f"(?:.+{any_sep})?"
798 any_last_segments = ".*"
799 results = []
800 parts = re.split(any_sep, pat)
801 last_part_idx = len(parts) - 1
802 for idx, part in enumerate(parts):
803 if part == "*":
804 results.append(one_segment if idx < last_part_idx else one_last_segment)
805 continue
806 if part == "**":
807 results.append(any_segments if idx < last_part_idx else any_last_segments)
808 continue
809 elif "**" in part:
810 raise ValueError(
811 "Invalid pattern: '**' can only be an entire path component"
812 )
813 if part:
814 results.extend(_translate(part, f"{not_sep}*", not_sep))
815 if idx < last_part_idx:
816 results.append(any_sep)
817 res = "".join(results)
818 return rf"(?s:{res})\Z"