Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/joblib/numpy_pickle.py: 15%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1"""Utilities for fast persistence of big data, with optional compression."""
3# Author: Gael Varoquaux <gael dot varoquaux at normalesup dot org>
4# Copyright (c) 2009 Gael Varoquaux
5# License: BSD Style, 3 clauses.
7import io
8import os
9import pickle
10import warnings
12from .backports import make_memmap
13from .compressor import (
14 _COMPRESSORS,
15 LZ4_NOT_INSTALLED_ERROR,
16 BinaryZlibFile,
17 BZ2CompressorWrapper,
18 GzipCompressorWrapper,
19 LZ4CompressorWrapper,
20 LZMACompressorWrapper,
21 XZCompressorWrapper,
22 ZlibCompressorWrapper,
23 lz4,
24 register_compressor,
25)
27# For compatibility with old versions of joblib, we need ZNDArrayWrapper
28# to be visible in the current namespace.
29from .numpy_pickle_compat import (
30 NDArrayWrapper,
31 ZNDArrayWrapper, # noqa: F401
32 load_compatibility,
33)
34from .numpy_pickle_utils import (
35 BUFFER_SIZE,
36 Pickler,
37 Unpickler,
38 _ensure_native_byte_order,
39 _read_bytes,
40 _reconstruct,
41 _validate_fileobject_and_memmap,
42 _write_fileobject,
43)
45# Register supported compressors
46register_compressor("zlib", ZlibCompressorWrapper())
47register_compressor("gzip", GzipCompressorWrapper())
48register_compressor("bz2", BZ2CompressorWrapper())
49register_compressor("lzma", LZMACompressorWrapper())
50register_compressor("xz", XZCompressorWrapper())
51register_compressor("lz4", LZ4CompressorWrapper())
54###############################################################################
55# Utility objects for persistence.
57# For convenience, 16 bytes are used to be sure to cover all the possible
58# dtypes' alignments. For reference, see:
59# https://numpy.org/devdocs/dev/alignment.html
60NUMPY_ARRAY_ALIGNMENT_BYTES = 16
63class NumpyArrayWrapper(object):
64 """An object to be persisted instead of numpy arrays.
66 This object is used to hack into the pickle machinery and read numpy
67 array data from our custom persistence format.
68 More precisely, this object is used for:
69 * carrying the information of the persisted array: subclass, shape, order,
70 dtype. Those ndarray metadata are used to correctly reconstruct the array
71 with low level numpy functions.
72 * determining if memmap is allowed on the array.
73 * reading the array bytes from a file.
74 * reading the array using memorymap from a file.
75 * writing the array bytes to a file.
77 Attributes
78 ----------
79 subclass: numpy.ndarray subclass
80 Determine the subclass of the wrapped array.
81 shape: numpy.ndarray shape
82 Determine the shape of the wrapped array.
83 order: {'C', 'F'}
84 Determine the order of wrapped array data. 'C' is for C order, 'F' is
85 for fortran order.
86 dtype: numpy.ndarray dtype
87 Determine the data type of the wrapped array.
88 allow_mmap: bool
89 Determine if memory mapping is allowed on the wrapped array.
90 Default: False.
91 """
93 def __init__(
94 self,
95 subclass,
96 shape,
97 order,
98 dtype,
99 allow_mmap=False,
100 numpy_array_alignment_bytes=NUMPY_ARRAY_ALIGNMENT_BYTES,
101 ):
102 """Constructor. Store the useful information for later."""
103 self.subclass = subclass
104 self.shape = shape
105 self.order = order
106 self.dtype = dtype
107 self.allow_mmap = allow_mmap
108 # We make numpy_array_alignment_bytes an instance attribute to allow us
109 # to change our mind about the default alignment and still load the old
110 # pickles (with the previous alignment) correctly
111 self.numpy_array_alignment_bytes = numpy_array_alignment_bytes
113 def safe_get_numpy_array_alignment_bytes(self):
114 # NumpyArrayWrapper instances loaded from joblib <= 1.1 pickles don't
115 # have an numpy_array_alignment_bytes attribute
116 return getattr(self, "numpy_array_alignment_bytes", None)
118 def write_array(self, array, pickler):
119 """Write array bytes to pickler file handle.
121 This function is an adaptation of the numpy write_array function
122 available in version 1.10.1 in numpy/lib/format.py.
123 """
124 # Set buffer size to 16 MiB to hide the Python loop overhead.
125 buffersize = max(16 * 1024**2 // array.itemsize, 1)
126 if array.dtype.hasobject:
127 # We contain Python objects so we cannot write out the data
128 # directly. Instead, we will pickle it out with version 5 of the
129 # pickle protocol.
130 pickle.dump(array, pickler.file_handle, protocol=5)
131 else:
132 numpy_array_alignment_bytes = self.safe_get_numpy_array_alignment_bytes()
133 if numpy_array_alignment_bytes is not None:
134 current_pos = pickler.file_handle.tell()
135 pos_after_padding_byte = current_pos + 1
136 padding_length = numpy_array_alignment_bytes - (
137 pos_after_padding_byte % numpy_array_alignment_bytes
138 )
139 # A single byte is written that contains the padding length in
140 # bytes
141 padding_length_byte = int.to_bytes(
142 padding_length, length=1, byteorder="little"
143 )
144 pickler.file_handle.write(padding_length_byte)
146 if padding_length != 0:
147 padding = b"\xff" * padding_length
148 pickler.file_handle.write(padding)
150 for chunk in pickler.np.nditer(
151 array,
152 flags=["external_loop", "buffered", "zerosize_ok"],
153 buffersize=buffersize,
154 order=self.order,
155 ):
156 pickler.file_handle.write(chunk.tobytes("C"))
158 def read_array(self, unpickler, ensure_native_byte_order):
159 """Read array from unpickler file handle.
161 This function is an adaptation of the numpy read_array function
162 available in version 1.10.1 in numpy/lib/format.py.
163 """
164 if len(self.shape) == 0:
165 count = 1
166 else:
167 # joblib issue #859: we cast the elements of self.shape to int64 to
168 # prevent a potential overflow when computing their product.
169 shape_int64 = [unpickler.np.int64(x) for x in self.shape]
170 count = unpickler.np.multiply.reduce(shape_int64)
171 # Now read the actual data.
172 if self.dtype.hasobject:
173 # The array contained Python objects, serialized as a nested
174 # pickle stream. We read it with a fresh Unpickler (so the nested
175 # stream gets its own opcode stack) but route ``find_class``
176 # through the *outer* unpickler instance to propagate security
177 # harden behavior.
178 # (see https://docs.python.org/3/library/pickle.html#pickle-restrict).
179 inner_unpickler = Unpickler(unpickler.file_handle)
180 inner_unpickler.find_class = unpickler.find_class
181 array = inner_unpickler.load()
182 else:
183 numpy_array_alignment_bytes = self.safe_get_numpy_array_alignment_bytes()
184 if numpy_array_alignment_bytes is not None:
185 padding_byte = unpickler.file_handle.read(1)
186 padding_length = int.from_bytes(padding_byte, byteorder="little")
187 if padding_length != 0:
188 unpickler.file_handle.read(padding_length)
190 # This is not a real file. We have to read it the
191 # memory-intensive way.
192 # crc32 module fails on reads greater than 2 ** 32 bytes,
193 # breaking large reads from gzip streams. Chunk reads to
194 # BUFFER_SIZE bytes to avoid issue and reduce memory overhead
195 # of the read. In non-chunked case count < max_read_count, so
196 # only one read is performed.
197 max_read_count = BUFFER_SIZE // min(BUFFER_SIZE, self.dtype.itemsize)
199 array = unpickler.np.empty(count, dtype=self.dtype)
200 for i in range(0, count, max_read_count):
201 read_count = min(max_read_count, count - i)
202 read_size = int(read_count * self.dtype.itemsize)
203 data = _read_bytes(unpickler.file_handle, read_size, "array data")
204 array[i : i + read_count] = unpickler.np.frombuffer(
205 data, dtype=self.dtype, count=read_count
206 )
207 del data
209 if self.order == "F":
210 array = array.reshape(self.shape[::-1])
211 array = array.transpose()
212 else:
213 array = array.reshape(self.shape)
215 if ensure_native_byte_order:
216 # Detect byte order mismatch and swap as needed.
217 array = _ensure_native_byte_order(array)
219 return array
221 def read_mmap(self, unpickler):
222 """Read an array using numpy memmap."""
223 current_pos = unpickler.file_handle.tell()
224 offset = current_pos
225 numpy_array_alignment_bytes = self.safe_get_numpy_array_alignment_bytes()
227 if numpy_array_alignment_bytes is not None:
228 padding_byte = unpickler.file_handle.read(1)
229 padding_length = int.from_bytes(padding_byte, byteorder="little")
230 # + 1 is for the padding byte
231 offset += padding_length + 1
233 if unpickler.mmap_mode == "w+":
234 unpickler.mmap_mode = "r+"
236 marray = make_memmap(
237 unpickler.filename,
238 dtype=self.dtype,
239 shape=self.shape,
240 order=self.order,
241 mode=unpickler.mmap_mode,
242 offset=offset,
243 )
244 # update the offset so that it corresponds to the end of the read array
245 unpickler.file_handle.seek(offset + marray.nbytes)
247 if (
248 numpy_array_alignment_bytes is None
249 and current_pos % NUMPY_ARRAY_ALIGNMENT_BYTES != 0
250 ):
251 message = (
252 f"The memmapped array {marray} loaded from the file "
253 f"{unpickler.file_handle.name} is not byte aligned. "
254 "This may cause segmentation faults if this memmapped array "
255 "is used in some libraries like BLAS or PyTorch. "
256 "To get rid of this warning, regenerate your pickle file "
257 "with joblib >= 1.2.0. "
258 "See https://github.com/joblib/joblib/issues/563 "
259 "for more details"
260 )
261 warnings.warn(message)
263 return marray
265 def read(self, unpickler, ensure_native_byte_order):
266 """Read the array corresponding to this wrapper.
268 Use the unpickler to get all information to correctly read the array.
270 Parameters
271 ----------
272 unpickler: NumpyUnpickler
273 ensure_native_byte_order: bool
274 If true, coerce the array to use the native endianness of the
275 host system.
277 Returns
278 -------
279 array: numpy.ndarray
281 """
282 # When requested, only use memmap mode if allowed.
283 if unpickler.mmap_mode is not None and self.allow_mmap:
284 assert not ensure_native_byte_order, (
285 "Memmaps cannot be coerced to a given byte order, "
286 "this code path is impossible."
287 )
288 array = self.read_mmap(unpickler)
289 else:
290 array = self.read_array(unpickler, ensure_native_byte_order)
292 # Manage array subclass case
293 if hasattr(array, "__array_prepare__") and self.subclass not in (
294 unpickler.np.ndarray,
295 unpickler.np.memmap,
296 ):
297 # We need to reconstruct another subclass
298 new_array = _reconstruct(self.subclass, (0,), "b")
299 return new_array.__array_prepare__(array)
300 else:
301 return array
304###############################################################################
305# Pickler classes
308class NumpyPickler(Pickler):
309 """A pickler to persist big data efficiently.
311 The main features of this object are:
312 * persistence of numpy arrays in a single file.
313 * optional compression with a special care on avoiding memory copies.
315 Attributes
316 ----------
317 fp: file
318 File object handle used for serializing the input object.
319 protocol: int, optional
320 Pickle protocol used. Default is pickle.DEFAULT_PROTOCOL.
321 """
323 dispatch = Pickler.dispatch.copy()
325 def __init__(self, fp, protocol=None):
326 self.file_handle = fp
327 self.buffered = isinstance(self.file_handle, BinaryZlibFile)
329 # By default we want a pickle protocol that only changes with
330 # the major python version and not the minor one
331 if protocol is None:
332 protocol = pickle.DEFAULT_PROTOCOL
334 Pickler.__init__(self, self.file_handle, protocol=protocol)
335 # delayed import of numpy, to avoid tight coupling
336 try:
337 import numpy as np
338 except ImportError:
339 np = None
340 self.np = np
342 def _create_array_wrapper(self, array):
343 """Create and returns a numpy array wrapper from a numpy array."""
344 order = (
345 "F" if (array.flags.f_contiguous and not array.flags.c_contiguous) else "C"
346 )
347 allow_mmap = not self.buffered and not array.dtype.hasobject
349 kwargs = {}
350 try:
351 self.file_handle.tell()
352 except io.UnsupportedOperation:
353 kwargs = {"numpy_array_alignment_bytes": None}
355 wrapper = NumpyArrayWrapper(
356 type(array),
357 array.shape,
358 order,
359 array.dtype,
360 allow_mmap=allow_mmap,
361 **kwargs,
362 )
364 return wrapper
366 def save(self, obj):
367 """Subclass the Pickler `save` method.
369 This is a total abuse of the Pickler class in order to use the numpy
370 persistence function `save` instead of the default pickle
371 implementation. The numpy array is replaced by a custom wrapper in the
372 pickle persistence stack and the serialized array is written right
373 after in the file. Warning: the file produced does not follow the
374 pickle format. As such it can not be read with `pickle.load`.
375 """
376 if self.np is not None and type(obj) in (
377 self.np.ndarray,
378 self.np.matrix,
379 self.np.memmap,
380 ):
381 if type(obj) is self.np.memmap:
382 # Pickling doesn't work with memmapped arrays
383 obj = self.np.asanyarray(obj)
385 # The array wrapper is pickled instead of the real array.
386 wrapper = self._create_array_wrapper(obj)
387 Pickler.save(self, wrapper)
389 # A framer was introduced with pickle protocol 4 and we want to
390 # ensure the wrapper object is written before the numpy array
391 # buffer in the pickle file.
392 # See https://www.python.org/dev/peps/pep-3154/#framing to get
393 # more information on the framer behavior.
394 if self.proto >= 4:
395 self.framer.commit_frame(force=True)
397 # And then array bytes are written right after the wrapper.
398 wrapper.write_array(obj, self)
399 return
401 return Pickler.save(self, obj)
404class NumpyUnpickler(Unpickler):
405 """A subclass of the Unpickler to unpickle our numpy pickles.
407 Attributes
408 ----------
409 mmap_mode: str
410 The memorymap mode to use for reading numpy arrays.
411 file_handle: file_like
412 File object to unpickle from.
413 ensure_native_byte_order: bool
414 If True, coerce the array to use the native endianness of the
415 host system.
416 filename: str
417 Name of the file to unpickle from. It should correspond to file_handle.
418 This parameter is required when using mmap_mode.
419 np: module
420 Reference to numpy module if numpy is installed else None.
422 """
424 dispatch = Unpickler.dispatch.copy()
426 def __init__(self, filename, file_handle, ensure_native_byte_order, mmap_mode=None):
427 # The next line is for backward compatibility with pickle generated
428 # with joblib versions less than 0.10.
429 self._dirname = os.path.dirname(filename)
431 self.mmap_mode = mmap_mode
432 self.file_handle = file_handle
433 # filename is required for numpy mmap mode.
434 self.filename = filename
435 self.compat_mode = False
436 self.ensure_native_byte_order = ensure_native_byte_order
437 Unpickler.__init__(self, self.file_handle)
438 try:
439 import numpy as np
440 except ImportError:
441 np = None
442 self.np = np
444 def load_build(self):
445 """Called to set the state of a newly created object.
447 We capture it to replace our place-holder objects, NDArrayWrapper or
448 NumpyArrayWrapper, by the array we are interested in. We
449 replace them directly in the stack of pickler.
450 NDArrayWrapper is used for backward compatibility with joblib <= 0.9.
451 """
452 Unpickler.load_build(self)
454 # For backward compatibility, we support NDArrayWrapper objects.
455 if isinstance(self.stack[-1], (NDArrayWrapper, NumpyArrayWrapper)):
456 if self.np is None:
457 raise ImportError(
458 "Trying to unpickle an ndarray, but numpy didn't import correctly"
459 )
460 array_wrapper = self.stack.pop()
461 # If any NDArrayWrapper is found, we switch to compatibility mode,
462 # this will be used to raise a DeprecationWarning to the user at
463 # the end of the unpickling.
464 if isinstance(array_wrapper, NDArrayWrapper):
465 self.compat_mode = True
466 _array_payload = array_wrapper.read(self)
467 else:
468 _array_payload = array_wrapper.read(self, self.ensure_native_byte_order)
470 self.stack.append(_array_payload)
472 # Be careful to register our new method.
473 dispatch[pickle.BUILD[0]] = load_build
476###############################################################################
477# Utility functions
480def dump(value, filename, compress=0, protocol=None):
481 """Persist an arbitrary Python object into one file.
483 Read more in the :ref:`User Guide <persistence>`.
485 Parameters
486 ----------
487 value: any Python object
488 The object to store to disk.
489 filename: str, os.PathLike, or file object.
490 The file object or path of the file in which it is to be stored.
491 The compression method corresponding to one of the supported filename
492 extensions ('.z', '.gz', '.bz2', '.xz' or '.lzma') will be used
493 automatically.
494 compress: int from 0 to 9 or bool or 2-tuple, optional
495 Optional compression level for the data. 0 or False is no compression.
496 Higher value means more compression, but also slower read and
497 write times. Using a value of 3 is often a good compromise.
498 See the notes for more details.
499 If compress is True, the compression level used is 3.
500 If compress is a 2-tuple, the first element must correspond to a string
501 between supported compressors (e.g 'zlib', 'gzip', 'bz2', 'lzma'
502 'xz'), the second element must be an integer from 0 to 9, corresponding
503 to the compression level.
504 protocol: int, optional
505 Pickle protocol, see pickle.dump documentation for more details.
507 Returns
508 -------
509 filenames: list of strings
510 The list of file names in which the data is stored. If
511 compress is false, each array is stored in a different file.
513 See Also
514 --------
515 joblib.load : corresponding loader
517 Notes
518 -----
519 Memmapping on load cannot be used for compressed files. Thus
520 using compression can significantly slow down loading. In
521 addition, compressed files take up extra memory during
522 dump and load.
524 """
526 if isinstance(filename, os.PathLike):
527 filename = os.fspath(filename)
529 is_filename = isinstance(filename, str)
530 is_fileobj = hasattr(filename, "write")
532 compress_method = "zlib" # zlib is the default compression method.
533 if compress is True:
534 # By default, if compress is enabled, we want the default compress
535 # level of the compressor.
536 compress_level = None
537 elif isinstance(compress, tuple):
538 # a 2-tuple was set in compress
539 if len(compress) != 2:
540 raise ValueError(
541 "Compress argument tuple should contain exactly 2 elements: "
542 "(compress method, compress level), you passed {}".format(compress)
543 )
544 compress_method, compress_level = compress
545 elif isinstance(compress, str):
546 compress_method = compress
547 compress_level = None # Use default compress level
548 compress = (compress_method, compress_level)
549 else:
550 compress_level = compress
552 if compress_method == "lz4" and lz4 is None:
553 raise ValueError(LZ4_NOT_INSTALLED_ERROR)
555 if (
556 compress_level is not None
557 and compress_level is not False
558 and compress_level not in range(10)
559 ):
560 # Raising an error if a non valid compress level is given.
561 raise ValueError(
562 'Non valid compress level given: "{}". Possible values are {}.'.format(
563 compress_level, list(range(10))
564 )
565 )
567 if compress_method not in _COMPRESSORS:
568 # Raising an error if an unsupported compression method is given.
569 raise ValueError(
570 'Non valid compression method given: "{}". Possible values are {}.'.format(
571 compress_method, _COMPRESSORS
572 )
573 )
575 if not is_filename and not is_fileobj:
576 # People keep inverting arguments, and the resulting error is
577 # incomprehensible
578 raise ValueError(
579 "Second argument should be a filename or a file-like object, "
580 "%s (type %s) was given." % (filename, type(filename))
581 )
583 if is_filename and not isinstance(compress, tuple):
584 # In case no explicit compression was requested using both compression
585 # method and level in a tuple and the filename has an explicit
586 # extension, we select the corresponding compressor.
588 # unset the variable to be sure no compression level is set afterwards.
589 compress_method = None
590 for name, compressor in _COMPRESSORS.items():
591 if filename.endswith(compressor.extension):
592 compress_method = name
594 if compress_method in _COMPRESSORS and compress_level == 0:
595 # we choose the default compress_level in case it was not given
596 # as an argument (using compress).
597 compress_level = None
599 if compress_level != 0:
600 with _write_fileobject(
601 filename, compress=(compress_method, compress_level)
602 ) as f:
603 NumpyPickler(f, protocol=protocol).dump(value)
604 elif is_filename:
605 with open(filename, "wb") as f:
606 NumpyPickler(f, protocol=protocol).dump(value)
607 else:
608 NumpyPickler(filename, protocol=protocol).dump(value)
610 # If the target container is a file object, nothing is returned.
611 if is_fileobj:
612 return
614 # For compatibility, the list of created filenames (e.g with one element
615 # after 0.10.0) is returned by default.
616 return [filename]
619def _unpickle(fobj, ensure_native_byte_order, filename="", mmap_mode=None):
620 """Internal unpickling function."""
621 # We are careful to open the file handle early and keep it open to
622 # avoid race-conditions on renames.
623 # That said, if data is stored in companion files, which can be
624 # the case with the old persistence format, moving the directory
625 # will create a race when joblib tries to access the companion
626 # files.
627 unpickler = NumpyUnpickler(
628 filename, fobj, ensure_native_byte_order, mmap_mode=mmap_mode
629 )
630 obj = None
631 try:
632 obj = unpickler.load()
633 if unpickler.compat_mode:
634 warnings.warn(
635 "The file '%s' has been generated with a "
636 "joblib version less than 0.10. "
637 "Please regenerate this pickle file." % filename,
638 DeprecationWarning,
639 stacklevel=3,
640 )
641 except UnicodeDecodeError as exc:
642 # More user-friendly error message
643 new_exc = ValueError(
644 "You may be trying to read with "
645 "python 3 a joblib pickle generated with python 2. "
646 "This feature is not supported by joblib."
647 )
648 new_exc.__cause__ = exc
649 raise new_exc
650 return obj
653def load_temporary_memmap(filename, mmap_mode, unlink_on_gc_collect):
654 from ._memmapping_reducer import JOBLIB_MMAPS, add_maybe_unlink_finalizer
656 with open(filename, "rb") as f:
657 with _validate_fileobject_and_memmap(f, filename, mmap_mode) as (
658 fobj,
659 validated_mmap_mode,
660 ):
661 # Memmap are used for interprocess communication, which should
662 # keep the objects untouched. We pass `ensure_native_byte_order=False`
663 # to remain consistent with the loading behavior of non-memmaped arrays
664 # in workers, where the byte order is preserved.
665 # Note that we do not implement endianness change for memmaps, as this
666 # would result in inconsistent behavior.
667 obj = _unpickle(
668 fobj,
669 ensure_native_byte_order=False,
670 filename=filename,
671 mmap_mode=validated_mmap_mode,
672 )
674 JOBLIB_MMAPS.add(obj.filename)
675 if unlink_on_gc_collect:
676 add_maybe_unlink_finalizer(obj)
677 return obj
680def load(filename, mmap_mode=None, ensure_native_byte_order="auto"):
681 """Reconstruct a Python object from a file persisted with joblib.dump.
683 Read more in the :ref:`User Guide <persistence>`.
685 WARNING: joblib.load relies on the pickle module and can therefore
686 execute arbitrary Python code. It should therefore never be used
687 to load files from untrusted sources.
689 Parameters
690 ----------
691 filename: str, os.PathLike, or file object.
692 The file object or path of the file from which to load the object
693 mmap_mode: {None, 'r+', 'r', 'w+', 'c'}, optional
694 If not None, the arrays are memory-mapped from the disk. This
695 mode has no effect for compressed files. Note that in this
696 case the reconstructed object might no longer match exactly
697 the originally pickled object.
698 ensure_native_byte_order: bool, or 'auto', default=='auto'
699 If True, ensures that the byte order of the loaded arrays matches the
700 native byte ordering (or _endianness_) of the host system. This is not
701 compatible with memory-mapped arrays and using non-null `mmap_mode`
702 parameter at the same time will raise an error. The default 'auto'
703 parameter is equivalent to True if `mmap_mode` is None, else False.
705 Returns
706 -------
707 result: any Python object
708 The object stored in the file.
710 See Also
711 --------
712 joblib.dump : function to save an object
714 Notes
715 -----
717 This function can load numpy array files saved separately during the
718 dump. If the mmap_mode argument is given, it is passed to np.load and
719 arrays are loaded as memmaps. As a consequence, the reconstructed
720 object might not match the original pickled object. Note that if the
721 file was saved with compression, the arrays cannot be memmapped.
722 """
723 if ensure_native_byte_order == "auto":
724 ensure_native_byte_order = mmap_mode is None
726 if ensure_native_byte_order and mmap_mode is not None:
727 raise ValueError(
728 "Native byte ordering can only be enforced if 'mmap_mode' parameter "
729 f"is set to None, but got 'mmap_mode={mmap_mode}' instead."
730 )
732 if isinstance(filename, os.PathLike):
733 filename = os.fspath(filename)
735 if hasattr(filename, "read"):
736 fobj = filename
737 filename = getattr(fobj, "name", "")
738 with _validate_fileobject_and_memmap(fobj, filename, mmap_mode) as (fobj, _):
739 obj = _unpickle(fobj, ensure_native_byte_order=ensure_native_byte_order)
740 else:
741 with open(filename, "rb") as f:
742 with _validate_fileobject_and_memmap(f, filename, mmap_mode) as (
743 fobj,
744 validated_mmap_mode,
745 ):
746 if isinstance(fobj, str):
747 # if the returned file object is a string, this means we
748 # try to load a pickle file generated with an version of
749 # Joblib so we load it with joblib compatibility function.
750 return load_compatibility(fobj)
752 # A memory-mapped array has to be mapped with the endianness
753 # it has been written with. Other arrays are coerced to the
754 # native endianness of the host system.
755 obj = _unpickle(
756 fobj,
757 ensure_native_byte_order=ensure_native_byte_order,
758 filename=filename,
759 mmap_mode=validated_mmap_mode,
760 )
762 return obj