Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/joblib/numpy_pickle.py: 15%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

239 statements  

1"""Utilities for fast persistence of big data, with optional compression.""" 

2 

3# Author: Gael Varoquaux <gael dot varoquaux at normalesup dot org> 

4# Copyright (c) 2009 Gael Varoquaux 

5# License: BSD Style, 3 clauses. 

6 

7import io 

8import os 

9import pickle 

10import warnings 

11 

12from .backports import make_memmap 

13from .compressor import ( 

14 _COMPRESSORS, 

15 LZ4_NOT_INSTALLED_ERROR, 

16 BinaryZlibFile, 

17 BZ2CompressorWrapper, 

18 GzipCompressorWrapper, 

19 LZ4CompressorWrapper, 

20 LZMACompressorWrapper, 

21 XZCompressorWrapper, 

22 ZlibCompressorWrapper, 

23 lz4, 

24 register_compressor, 

25) 

26 

27# For compatibility with old versions of joblib, we need ZNDArrayWrapper 

28# to be visible in the current namespace. 

29from .numpy_pickle_compat import ( 

30 NDArrayWrapper, 

31 ZNDArrayWrapper, # noqa: F401 

32 load_compatibility, 

33) 

34from .numpy_pickle_utils import ( 

35 BUFFER_SIZE, 

36 Pickler, 

37 Unpickler, 

38 _ensure_native_byte_order, 

39 _read_bytes, 

40 _reconstruct, 

41 _validate_fileobject_and_memmap, 

42 _write_fileobject, 

43) 

44 

45# Register supported compressors 

46register_compressor("zlib", ZlibCompressorWrapper()) 

47register_compressor("gzip", GzipCompressorWrapper()) 

48register_compressor("bz2", BZ2CompressorWrapper()) 

49register_compressor("lzma", LZMACompressorWrapper()) 

50register_compressor("xz", XZCompressorWrapper()) 

51register_compressor("lz4", LZ4CompressorWrapper()) 

52 

53 

54############################################################################### 

55# Utility objects for persistence. 

56 

57# For convenience, 16 bytes are used to be sure to cover all the possible 

58# dtypes' alignments. For reference, see: 

59# https://numpy.org/devdocs/dev/alignment.html 

60NUMPY_ARRAY_ALIGNMENT_BYTES = 16 

61 

62 

63class NumpyArrayWrapper(object): 

64 """An object to be persisted instead of numpy arrays. 

65 

66 This object is used to hack into the pickle machinery and read numpy 

67 array data from our custom persistence format. 

68 More precisely, this object is used for: 

69 * carrying the information of the persisted array: subclass, shape, order, 

70 dtype. Those ndarray metadata are used to correctly reconstruct the array 

71 with low level numpy functions. 

72 * determining if memmap is allowed on the array. 

73 * reading the array bytes from a file. 

74 * reading the array using memorymap from a file. 

75 * writing the array bytes to a file. 

76 

77 Attributes 

78 ---------- 

79 subclass: numpy.ndarray subclass 

80 Determine the subclass of the wrapped array. 

81 shape: numpy.ndarray shape 

82 Determine the shape of the wrapped array. 

83 order: {'C', 'F'} 

84 Determine the order of wrapped array data. 'C' is for C order, 'F' is 

85 for fortran order. 

86 dtype: numpy.ndarray dtype 

87 Determine the data type of the wrapped array. 

88 allow_mmap: bool 

89 Determine if memory mapping is allowed on the wrapped array. 

90 Default: False. 

91 """ 

92 

93 def __init__( 

94 self, 

95 subclass, 

96 shape, 

97 order, 

98 dtype, 

99 allow_mmap=False, 

100 numpy_array_alignment_bytes=NUMPY_ARRAY_ALIGNMENT_BYTES, 

101 ): 

102 """Constructor. Store the useful information for later.""" 

103 self.subclass = subclass 

104 self.shape = shape 

105 self.order = order 

106 self.dtype = dtype 

107 self.allow_mmap = allow_mmap 

108 # We make numpy_array_alignment_bytes an instance attribute to allow us 

109 # to change our mind about the default alignment and still load the old 

110 # pickles (with the previous alignment) correctly 

111 self.numpy_array_alignment_bytes = numpy_array_alignment_bytes 

112 

113 def safe_get_numpy_array_alignment_bytes(self): 

114 # NumpyArrayWrapper instances loaded from joblib <= 1.1 pickles don't 

115 # have an numpy_array_alignment_bytes attribute 

116 return getattr(self, "numpy_array_alignment_bytes", None) 

117 

118 def write_array(self, array, pickler): 

119 """Write array bytes to pickler file handle. 

120 

121 This function is an adaptation of the numpy write_array function 

122 available in version 1.10.1 in numpy/lib/format.py. 

123 """ 

124 # Set buffer size to 16 MiB to hide the Python loop overhead. 

125 buffersize = max(16 * 1024**2 // array.itemsize, 1) 

126 if array.dtype.hasobject: 

127 # We contain Python objects so we cannot write out the data 

128 # directly. Instead, we will pickle it out with version 5 of the 

129 # pickle protocol. 

130 pickle.dump(array, pickler.file_handle, protocol=5) 

131 else: 

132 numpy_array_alignment_bytes = self.safe_get_numpy_array_alignment_bytes() 

133 if numpy_array_alignment_bytes is not None: 

134 current_pos = pickler.file_handle.tell() 

135 pos_after_padding_byte = current_pos + 1 

136 padding_length = numpy_array_alignment_bytes - ( 

137 pos_after_padding_byte % numpy_array_alignment_bytes 

138 ) 

139 # A single byte is written that contains the padding length in 

140 # bytes 

141 padding_length_byte = int.to_bytes( 

142 padding_length, length=1, byteorder="little" 

143 ) 

144 pickler.file_handle.write(padding_length_byte) 

145 

146 if padding_length != 0: 

147 padding = b"\xff" * padding_length 

148 pickler.file_handle.write(padding) 

149 

150 for chunk in pickler.np.nditer( 

151 array, 

152 flags=["external_loop", "buffered", "zerosize_ok"], 

153 buffersize=buffersize, 

154 order=self.order, 

155 ): 

156 pickler.file_handle.write(chunk.tobytes("C")) 

157 

158 def read_array(self, unpickler, ensure_native_byte_order): 

159 """Read array from unpickler file handle. 

160 

161 This function is an adaptation of the numpy read_array function 

162 available in version 1.10.1 in numpy/lib/format.py. 

163 """ 

164 if len(self.shape) == 0: 

165 count = 1 

166 else: 

167 # joblib issue #859: we cast the elements of self.shape to int64 to 

168 # prevent a potential overflow when computing their product. 

169 shape_int64 = [unpickler.np.int64(x) for x in self.shape] 

170 count = unpickler.np.multiply.reduce(shape_int64) 

171 # Now read the actual data. 

172 if self.dtype.hasobject: 

173 # The array contained Python objects, serialized as a nested 

174 # pickle stream. We read it with a fresh Unpickler (so the nested 

175 # stream gets its own opcode stack) but route ``find_class`` 

176 # through the *outer* unpickler instance to propagate security 

177 # harden behavior. 

178 # (see https://docs.python.org/3/library/pickle.html#pickle-restrict). 

179 inner_unpickler = Unpickler(unpickler.file_handle) 

180 inner_unpickler.find_class = unpickler.find_class 

181 array = inner_unpickler.load() 

182 else: 

183 numpy_array_alignment_bytes = self.safe_get_numpy_array_alignment_bytes() 

184 if numpy_array_alignment_bytes is not None: 

185 padding_byte = unpickler.file_handle.read(1) 

186 padding_length = int.from_bytes(padding_byte, byteorder="little") 

187 if padding_length != 0: 

188 unpickler.file_handle.read(padding_length) 

189 

190 # This is not a real file. We have to read it the 

191 # memory-intensive way. 

192 # crc32 module fails on reads greater than 2 ** 32 bytes, 

193 # breaking large reads from gzip streams. Chunk reads to 

194 # BUFFER_SIZE bytes to avoid issue and reduce memory overhead 

195 # of the read. In non-chunked case count < max_read_count, so 

196 # only one read is performed. 

197 max_read_count = BUFFER_SIZE // min(BUFFER_SIZE, self.dtype.itemsize) 

198 

199 array = unpickler.np.empty(count, dtype=self.dtype) 

200 for i in range(0, count, max_read_count): 

201 read_count = min(max_read_count, count - i) 

202 read_size = int(read_count * self.dtype.itemsize) 

203 data = _read_bytes(unpickler.file_handle, read_size, "array data") 

204 array[i : i + read_count] = unpickler.np.frombuffer( 

205 data, dtype=self.dtype, count=read_count 

206 ) 

207 del data 

208 

209 if self.order == "F": 

210 array = array.reshape(self.shape[::-1]) 

211 array = array.transpose() 

212 else: 

213 array = array.reshape(self.shape) 

214 

215 if ensure_native_byte_order: 

216 # Detect byte order mismatch and swap as needed. 

217 array = _ensure_native_byte_order(array) 

218 

219 return array 

220 

221 def read_mmap(self, unpickler): 

222 """Read an array using numpy memmap.""" 

223 current_pos = unpickler.file_handle.tell() 

224 offset = current_pos 

225 numpy_array_alignment_bytes = self.safe_get_numpy_array_alignment_bytes() 

226 

227 if numpy_array_alignment_bytes is not None: 

228 padding_byte = unpickler.file_handle.read(1) 

229 padding_length = int.from_bytes(padding_byte, byteorder="little") 

230 # + 1 is for the padding byte 

231 offset += padding_length + 1 

232 

233 if unpickler.mmap_mode == "w+": 

234 unpickler.mmap_mode = "r+" 

235 

236 marray = make_memmap( 

237 unpickler.filename, 

238 dtype=self.dtype, 

239 shape=self.shape, 

240 order=self.order, 

241 mode=unpickler.mmap_mode, 

242 offset=offset, 

243 ) 

244 # update the offset so that it corresponds to the end of the read array 

245 unpickler.file_handle.seek(offset + marray.nbytes) 

246 

247 if ( 

248 numpy_array_alignment_bytes is None 

249 and current_pos % NUMPY_ARRAY_ALIGNMENT_BYTES != 0 

250 ): 

251 message = ( 

252 f"The memmapped array {marray} loaded from the file " 

253 f"{unpickler.file_handle.name} is not byte aligned. " 

254 "This may cause segmentation faults if this memmapped array " 

255 "is used in some libraries like BLAS or PyTorch. " 

256 "To get rid of this warning, regenerate your pickle file " 

257 "with joblib >= 1.2.0. " 

258 "See https://github.com/joblib/joblib/issues/563 " 

259 "for more details" 

260 ) 

261 warnings.warn(message) 

262 

263 return marray 

264 

265 def read(self, unpickler, ensure_native_byte_order): 

266 """Read the array corresponding to this wrapper. 

267 

268 Use the unpickler to get all information to correctly read the array. 

269 

270 Parameters 

271 ---------- 

272 unpickler: NumpyUnpickler 

273 ensure_native_byte_order: bool 

274 If true, coerce the array to use the native endianness of the 

275 host system. 

276 

277 Returns 

278 ------- 

279 array: numpy.ndarray 

280 

281 """ 

282 # When requested, only use memmap mode if allowed. 

283 if unpickler.mmap_mode is not None and self.allow_mmap: 

284 assert not ensure_native_byte_order, ( 

285 "Memmaps cannot be coerced to a given byte order, " 

286 "this code path is impossible." 

287 ) 

288 array = self.read_mmap(unpickler) 

289 else: 

290 array = self.read_array(unpickler, ensure_native_byte_order) 

291 

292 # Manage array subclass case 

293 if hasattr(array, "__array_prepare__") and self.subclass not in ( 

294 unpickler.np.ndarray, 

295 unpickler.np.memmap, 

296 ): 

297 # We need to reconstruct another subclass 

298 new_array = _reconstruct(self.subclass, (0,), "b") 

299 return new_array.__array_prepare__(array) 

300 else: 

301 return array 

302 

303 

304############################################################################### 

305# Pickler classes 

306 

307 

308class NumpyPickler(Pickler): 

309 """A pickler to persist big data efficiently. 

310 

311 The main features of this object are: 

312 * persistence of numpy arrays in a single file. 

313 * optional compression with a special care on avoiding memory copies. 

314 

315 Attributes 

316 ---------- 

317 fp: file 

318 File object handle used for serializing the input object. 

319 protocol: int, optional 

320 Pickle protocol used. Default is pickle.DEFAULT_PROTOCOL. 

321 """ 

322 

323 dispatch = Pickler.dispatch.copy() 

324 

325 def __init__(self, fp, protocol=None): 

326 self.file_handle = fp 

327 self.buffered = isinstance(self.file_handle, BinaryZlibFile) 

328 

329 # By default we want a pickle protocol that only changes with 

330 # the major python version and not the minor one 

331 if protocol is None: 

332 protocol = pickle.DEFAULT_PROTOCOL 

333 

334 Pickler.__init__(self, self.file_handle, protocol=protocol) 

335 # delayed import of numpy, to avoid tight coupling 

336 try: 

337 import numpy as np 

338 except ImportError: 

339 np = None 

340 self.np = np 

341 

342 def _create_array_wrapper(self, array): 

343 """Create and returns a numpy array wrapper from a numpy array.""" 

344 order = ( 

345 "F" if (array.flags.f_contiguous and not array.flags.c_contiguous) else "C" 

346 ) 

347 allow_mmap = not self.buffered and not array.dtype.hasobject 

348 

349 kwargs = {} 

350 try: 

351 self.file_handle.tell() 

352 except io.UnsupportedOperation: 

353 kwargs = {"numpy_array_alignment_bytes": None} 

354 

355 wrapper = NumpyArrayWrapper( 

356 type(array), 

357 array.shape, 

358 order, 

359 array.dtype, 

360 allow_mmap=allow_mmap, 

361 **kwargs, 

362 ) 

363 

364 return wrapper 

365 

366 def save(self, obj): 

367 """Subclass the Pickler `save` method. 

368 

369 This is a total abuse of the Pickler class in order to use the numpy 

370 persistence function `save` instead of the default pickle 

371 implementation. The numpy array is replaced by a custom wrapper in the 

372 pickle persistence stack and the serialized array is written right 

373 after in the file. Warning: the file produced does not follow the 

374 pickle format. As such it can not be read with `pickle.load`. 

375 """ 

376 if self.np is not None and type(obj) in ( 

377 self.np.ndarray, 

378 self.np.matrix, 

379 self.np.memmap, 

380 ): 

381 if type(obj) is self.np.memmap: 

382 # Pickling doesn't work with memmapped arrays 

383 obj = self.np.asanyarray(obj) 

384 

385 # The array wrapper is pickled instead of the real array. 

386 wrapper = self._create_array_wrapper(obj) 

387 Pickler.save(self, wrapper) 

388 

389 # A framer was introduced with pickle protocol 4 and we want to 

390 # ensure the wrapper object is written before the numpy array 

391 # buffer in the pickle file. 

392 # See https://www.python.org/dev/peps/pep-3154/#framing to get 

393 # more information on the framer behavior. 

394 if self.proto >= 4: 

395 self.framer.commit_frame(force=True) 

396 

397 # And then array bytes are written right after the wrapper. 

398 wrapper.write_array(obj, self) 

399 return 

400 

401 return Pickler.save(self, obj) 

402 

403 

404class NumpyUnpickler(Unpickler): 

405 """A subclass of the Unpickler to unpickle our numpy pickles. 

406 

407 Attributes 

408 ---------- 

409 mmap_mode: str 

410 The memorymap mode to use for reading numpy arrays. 

411 file_handle: file_like 

412 File object to unpickle from. 

413 ensure_native_byte_order: bool 

414 If True, coerce the array to use the native endianness of the 

415 host system. 

416 filename: str 

417 Name of the file to unpickle from. It should correspond to file_handle. 

418 This parameter is required when using mmap_mode. 

419 np: module 

420 Reference to numpy module if numpy is installed else None. 

421 

422 """ 

423 

424 dispatch = Unpickler.dispatch.copy() 

425 

426 def __init__(self, filename, file_handle, ensure_native_byte_order, mmap_mode=None): 

427 # The next line is for backward compatibility with pickle generated 

428 # with joblib versions less than 0.10. 

429 self._dirname = os.path.dirname(filename) 

430 

431 self.mmap_mode = mmap_mode 

432 self.file_handle = file_handle 

433 # filename is required for numpy mmap mode. 

434 self.filename = filename 

435 self.compat_mode = False 

436 self.ensure_native_byte_order = ensure_native_byte_order 

437 Unpickler.__init__(self, self.file_handle) 

438 try: 

439 import numpy as np 

440 except ImportError: 

441 np = None 

442 self.np = np 

443 

444 def load_build(self): 

445 """Called to set the state of a newly created object. 

446 

447 We capture it to replace our place-holder objects, NDArrayWrapper or 

448 NumpyArrayWrapper, by the array we are interested in. We 

449 replace them directly in the stack of pickler. 

450 NDArrayWrapper is used for backward compatibility with joblib <= 0.9. 

451 """ 

452 Unpickler.load_build(self) 

453 

454 # For backward compatibility, we support NDArrayWrapper objects. 

455 if isinstance(self.stack[-1], (NDArrayWrapper, NumpyArrayWrapper)): 

456 if self.np is None: 

457 raise ImportError( 

458 "Trying to unpickle an ndarray, but numpy didn't import correctly" 

459 ) 

460 array_wrapper = self.stack.pop() 

461 # If any NDArrayWrapper is found, we switch to compatibility mode, 

462 # this will be used to raise a DeprecationWarning to the user at 

463 # the end of the unpickling. 

464 if isinstance(array_wrapper, NDArrayWrapper): 

465 self.compat_mode = True 

466 _array_payload = array_wrapper.read(self) 

467 else: 

468 _array_payload = array_wrapper.read(self, self.ensure_native_byte_order) 

469 

470 self.stack.append(_array_payload) 

471 

472 # Be careful to register our new method. 

473 dispatch[pickle.BUILD[0]] = load_build 

474 

475 

476############################################################################### 

477# Utility functions 

478 

479 

480def dump(value, filename, compress=0, protocol=None): 

481 """Persist an arbitrary Python object into one file. 

482 

483 Read more in the :ref:`User Guide <persistence>`. 

484 

485 Parameters 

486 ---------- 

487 value: any Python object 

488 The object to store to disk. 

489 filename: str, os.PathLike, or file object. 

490 The file object or path of the file in which it is to be stored. 

491 The compression method corresponding to one of the supported filename 

492 extensions ('.z', '.gz', '.bz2', '.xz' or '.lzma') will be used 

493 automatically. 

494 compress: int from 0 to 9 or bool or 2-tuple, optional 

495 Optional compression level for the data. 0 or False is no compression. 

496 Higher value means more compression, but also slower read and 

497 write times. Using a value of 3 is often a good compromise. 

498 See the notes for more details. 

499 If compress is True, the compression level used is 3. 

500 If compress is a 2-tuple, the first element must correspond to a string 

501 between supported compressors (e.g 'zlib', 'gzip', 'bz2', 'lzma' 

502 'xz'), the second element must be an integer from 0 to 9, corresponding 

503 to the compression level. 

504 protocol: int, optional 

505 Pickle protocol, see pickle.dump documentation for more details. 

506 

507 Returns 

508 ------- 

509 filenames: list of strings 

510 The list of file names in which the data is stored. If 

511 compress is false, each array is stored in a different file. 

512 

513 See Also 

514 -------- 

515 joblib.load : corresponding loader 

516 

517 Notes 

518 ----- 

519 Memmapping on load cannot be used for compressed files. Thus 

520 using compression can significantly slow down loading. In 

521 addition, compressed files take up extra memory during 

522 dump and load. 

523 

524 """ 

525 

526 if isinstance(filename, os.PathLike): 

527 filename = os.fspath(filename) 

528 

529 is_filename = isinstance(filename, str) 

530 is_fileobj = hasattr(filename, "write") 

531 

532 compress_method = "zlib" # zlib is the default compression method. 

533 if compress is True: 

534 # By default, if compress is enabled, we want the default compress 

535 # level of the compressor. 

536 compress_level = None 

537 elif isinstance(compress, tuple): 

538 # a 2-tuple was set in compress 

539 if len(compress) != 2: 

540 raise ValueError( 

541 "Compress argument tuple should contain exactly 2 elements: " 

542 "(compress method, compress level), you passed {}".format(compress) 

543 ) 

544 compress_method, compress_level = compress 

545 elif isinstance(compress, str): 

546 compress_method = compress 

547 compress_level = None # Use default compress level 

548 compress = (compress_method, compress_level) 

549 else: 

550 compress_level = compress 

551 

552 if compress_method == "lz4" and lz4 is None: 

553 raise ValueError(LZ4_NOT_INSTALLED_ERROR) 

554 

555 if ( 

556 compress_level is not None 

557 and compress_level is not False 

558 and compress_level not in range(10) 

559 ): 

560 # Raising an error if a non valid compress level is given. 

561 raise ValueError( 

562 'Non valid compress level given: "{}". Possible values are {}.'.format( 

563 compress_level, list(range(10)) 

564 ) 

565 ) 

566 

567 if compress_method not in _COMPRESSORS: 

568 # Raising an error if an unsupported compression method is given. 

569 raise ValueError( 

570 'Non valid compression method given: "{}". Possible values are {}.'.format( 

571 compress_method, _COMPRESSORS 

572 ) 

573 ) 

574 

575 if not is_filename and not is_fileobj: 

576 # People keep inverting arguments, and the resulting error is 

577 # incomprehensible 

578 raise ValueError( 

579 "Second argument should be a filename or a file-like object, " 

580 "%s (type %s) was given." % (filename, type(filename)) 

581 ) 

582 

583 if is_filename and not isinstance(compress, tuple): 

584 # In case no explicit compression was requested using both compression 

585 # method and level in a tuple and the filename has an explicit 

586 # extension, we select the corresponding compressor. 

587 

588 # unset the variable to be sure no compression level is set afterwards. 

589 compress_method = None 

590 for name, compressor in _COMPRESSORS.items(): 

591 if filename.endswith(compressor.extension): 

592 compress_method = name 

593 

594 if compress_method in _COMPRESSORS and compress_level == 0: 

595 # we choose the default compress_level in case it was not given 

596 # as an argument (using compress). 

597 compress_level = None 

598 

599 if compress_level != 0: 

600 with _write_fileobject( 

601 filename, compress=(compress_method, compress_level) 

602 ) as f: 

603 NumpyPickler(f, protocol=protocol).dump(value) 

604 elif is_filename: 

605 with open(filename, "wb") as f: 

606 NumpyPickler(f, protocol=protocol).dump(value) 

607 else: 

608 NumpyPickler(filename, protocol=protocol).dump(value) 

609 

610 # If the target container is a file object, nothing is returned. 

611 if is_fileobj: 

612 return 

613 

614 # For compatibility, the list of created filenames (e.g with one element 

615 # after 0.10.0) is returned by default. 

616 return [filename] 

617 

618 

619def _unpickle(fobj, ensure_native_byte_order, filename="", mmap_mode=None): 

620 """Internal unpickling function.""" 

621 # We are careful to open the file handle early and keep it open to 

622 # avoid race-conditions on renames. 

623 # That said, if data is stored in companion files, which can be 

624 # the case with the old persistence format, moving the directory 

625 # will create a race when joblib tries to access the companion 

626 # files. 

627 unpickler = NumpyUnpickler( 

628 filename, fobj, ensure_native_byte_order, mmap_mode=mmap_mode 

629 ) 

630 obj = None 

631 try: 

632 obj = unpickler.load() 

633 if unpickler.compat_mode: 

634 warnings.warn( 

635 "The file '%s' has been generated with a " 

636 "joblib version less than 0.10. " 

637 "Please regenerate this pickle file." % filename, 

638 DeprecationWarning, 

639 stacklevel=3, 

640 ) 

641 except UnicodeDecodeError as exc: 

642 # More user-friendly error message 

643 new_exc = ValueError( 

644 "You may be trying to read with " 

645 "python 3 a joblib pickle generated with python 2. " 

646 "This feature is not supported by joblib." 

647 ) 

648 new_exc.__cause__ = exc 

649 raise new_exc 

650 return obj 

651 

652 

653def load_temporary_memmap(filename, mmap_mode, unlink_on_gc_collect): 

654 from ._memmapping_reducer import JOBLIB_MMAPS, add_maybe_unlink_finalizer 

655 

656 with open(filename, "rb") as f: 

657 with _validate_fileobject_and_memmap(f, filename, mmap_mode) as ( 

658 fobj, 

659 validated_mmap_mode, 

660 ): 

661 # Memmap are used for interprocess communication, which should 

662 # keep the objects untouched. We pass `ensure_native_byte_order=False` 

663 # to remain consistent with the loading behavior of non-memmaped arrays 

664 # in workers, where the byte order is preserved. 

665 # Note that we do not implement endianness change for memmaps, as this 

666 # would result in inconsistent behavior. 

667 obj = _unpickle( 

668 fobj, 

669 ensure_native_byte_order=False, 

670 filename=filename, 

671 mmap_mode=validated_mmap_mode, 

672 ) 

673 

674 JOBLIB_MMAPS.add(obj.filename) 

675 if unlink_on_gc_collect: 

676 add_maybe_unlink_finalizer(obj) 

677 return obj 

678 

679 

680def load(filename, mmap_mode=None, ensure_native_byte_order="auto"): 

681 """Reconstruct a Python object from a file persisted with joblib.dump. 

682 

683 Read more in the :ref:`User Guide <persistence>`. 

684 

685 WARNING: joblib.load relies on the pickle module and can therefore 

686 execute arbitrary Python code. It should therefore never be used 

687 to load files from untrusted sources. 

688 

689 Parameters 

690 ---------- 

691 filename: str, os.PathLike, or file object. 

692 The file object or path of the file from which to load the object 

693 mmap_mode: {None, 'r+', 'r', 'w+', 'c'}, optional 

694 If not None, the arrays are memory-mapped from the disk. This 

695 mode has no effect for compressed files. Note that in this 

696 case the reconstructed object might no longer match exactly 

697 the originally pickled object. 

698 ensure_native_byte_order: bool, or 'auto', default=='auto' 

699 If True, ensures that the byte order of the loaded arrays matches the 

700 native byte ordering (or _endianness_) of the host system. This is not 

701 compatible with memory-mapped arrays and using non-null `mmap_mode` 

702 parameter at the same time will raise an error. The default 'auto' 

703 parameter is equivalent to True if `mmap_mode` is None, else False. 

704 

705 Returns 

706 ------- 

707 result: any Python object 

708 The object stored in the file. 

709 

710 See Also 

711 -------- 

712 joblib.dump : function to save an object 

713 

714 Notes 

715 ----- 

716 

717 This function can load numpy array files saved separately during the 

718 dump. If the mmap_mode argument is given, it is passed to np.load and 

719 arrays are loaded as memmaps. As a consequence, the reconstructed 

720 object might not match the original pickled object. Note that if the 

721 file was saved with compression, the arrays cannot be memmapped. 

722 """ 

723 if ensure_native_byte_order == "auto": 

724 ensure_native_byte_order = mmap_mode is None 

725 

726 if ensure_native_byte_order and mmap_mode is not None: 

727 raise ValueError( 

728 "Native byte ordering can only be enforced if 'mmap_mode' parameter " 

729 f"is set to None, but got 'mmap_mode={mmap_mode}' instead." 

730 ) 

731 

732 if isinstance(filename, os.PathLike): 

733 filename = os.fspath(filename) 

734 

735 if hasattr(filename, "read"): 

736 fobj = filename 

737 filename = getattr(fobj, "name", "") 

738 with _validate_fileobject_and_memmap(fobj, filename, mmap_mode) as (fobj, _): 

739 obj = _unpickle(fobj, ensure_native_byte_order=ensure_native_byte_order) 

740 else: 

741 with open(filename, "rb") as f: 

742 with _validate_fileobject_and_memmap(f, filename, mmap_mode) as ( 

743 fobj, 

744 validated_mmap_mode, 

745 ): 

746 if isinstance(fobj, str): 

747 # if the returned file object is a string, this means we 

748 # try to load a pickle file generated with an version of 

749 # Joblib so we load it with joblib compatibility function. 

750 return load_compatibility(fobj) 

751 

752 # A memory-mapped array has to be mapped with the endianness 

753 # it has been written with. Other arrays are coerced to the 

754 # native endianness of the host system. 

755 obj = _unpickle( 

756 fobj, 

757 ensure_native_byte_order=ensure_native_byte_order, 

758 filename=filename, 

759 mmap_mode=validated_mmap_mode, 

760 ) 

761 

762 return obj