1# Copyright 2024 The Sigstore Authors
2#
3# Licensed under the Apache License, Version 2.0 (the "License");
4# you may not use this file except in compliance with the License.
5# You may obtain a copy of the License at
6#
7# http://www.apache.org/licenses/LICENSE-2.0
8#
9# Unless required by applicable law or agreed to in writing, software
10# distributed under the License is distributed on an "AS IS" BASIS,
11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12# See the License for the specific language governing permissions and
13# limitations under the License.
14
15"""Model serializers that operate at file shard level of granularity."""
16
17from collections.abc import Callable, Iterable
18import concurrent.futures
19import itertools
20import os
21import pathlib
22
23from typing_extensions import override
24
25from model_signing import manifest
26from model_signing._hashing import io
27from model_signing._serialization import serialization
28
29
30def _endpoints(step: int, end: int) -> Iterable[int]:
31 """Yields numbers from `step` to `end` inclusive, spaced by `step`.
32
33 Last value is always equal to `end`, even when `end` is not a multiple of
34 `step`. There is always a value returned.
35
36 Examples:
37 ```python
38 >>> list(_endpoints(2, 8))
39 [2, 4, 6, 8]
40 >>> list(_endpoints(2, 9))
41 [2, 4, 6, 8, 9]
42 >>> list(_endpoints(2, 2))
43 [2]
44
45 Yields:
46 Values in the range, from `step` and up to `end`.
47 """
48 yield from range(step, end, step)
49 yield end
50
51
52class Serializer(serialization.Serializer):
53 """Model serializer that produces a manifest recording every file shard.
54
55 Traverses the model directory and creates digests for every file found,
56 sharding the file in equal shards and computing the digests in parallel.
57 """
58
59 def __init__(
60 self,
61 sharded_hasher_factory: Callable[
62 [pathlib.Path, int, int], io.ShardedFileHasher
63 ],
64 *,
65 max_workers: int | None = None,
66 allow_symlinks: bool = False,
67 ignore_paths: Iterable[pathlib.Path] = frozenset(),
68 ):
69 """Initializes an instance to serialize a model with this serializer.
70
71 Args:
72 sharded_hasher_factory: A callable to build the hash engine used to
73 hash every shard of the files in the model. Because each shard is
74 processed in parallel, every thread needs to call the factory to
75 start hashing. The arguments are the file, and the endpoints of
76 the shard.
77 max_workers: Maximum number of workers to use in parallel. Default
78 is to defer to the `concurrent.futures` library.
79 allow_symlinks: Controls whether symbolic links are included. If a
80 symlink is present but the flag is `False` (default) the
81 serialization would raise an error.
82 ignore_paths: The paths of files to ignore.
83 """
84 self._hasher_factory = sharded_hasher_factory
85 self._max_workers = max_workers
86 self._allow_symlinks = allow_symlinks
87 self._ignore_paths = ignore_paths
88
89 # Precompute some private values only once by using a mock file hasher.
90 # None of the arguments used to build the hasher are used.
91 hasher = sharded_hasher_factory(pathlib.Path(), 0, 1)
92 self._shard_size = hasher.shard_size
93 self._serialization_description = manifest._ShardSerialization(
94 # Here we need the internal hasher name, not the mangled name.
95 # This name is used when guessing the hashing configuration.
96 hasher._content_hasher.digest_name,
97 self._shard_size,
98 self._allow_symlinks,
99 self._ignore_paths,
100 )
101
102 def set_allow_symlinks(self, allow_symlinks: bool) -> None:
103 """Set whether following symlinks is allowed."""
104 self._allow_symlinks = allow_symlinks
105 hasher = self._hasher_factory(pathlib.Path(), 0, 1)
106 self._serialization_description = manifest._ShardSerialization(
107 hasher._content_hasher.digest_name,
108 self._shard_size,
109 self._allow_symlinks,
110 self._ignore_paths,
111 )
112
113 @override
114 def serialize(
115 self,
116 model_path: pathlib.Path,
117 *,
118 ignore_paths: Iterable[pathlib.Path] = frozenset(),
119 files_to_hash: Iterable[pathlib.Path] | None = None,
120 ) -> manifest.Manifest:
121 """Serializes the model given by the `model_path` argument.
122
123 Args:
124 model_path: The path to the model.
125 ignore_paths: The paths to ignore during serialization. If a
126 provided path is a directory, all children of the directory are
127 ignored.
128 files_to_hash: Optional list of files to hash; ignore all others
129
130 Returns:
131 The model's serialized manifest.
132
133 Raises:
134 ValueError: The model contains a symbolic link, but the serializer
135 was not initialized with `allow_symlinks=True`.
136 """
137 shards = []
138 file_count = 0
139 # TODO: github.com/sigstore/model-transparency/issues/200 - When
140 # Python3.12 is the minimum supported version, the glob can be replaced
141 # with `pathlib.Path.walk` for a clearer interface, and some speed
142 # improvement.
143 if files_to_hash is None:
144 files_to_hash = itertools.chain(
145 (model_path,), model_path.glob("**/*")
146 )
147 for path in files_to_hash:
148 if serialization.should_ignore(path, ignore_paths):
149 continue
150 serialization.check_file_or_directory(
151 path, allow_symlinks=self._allow_symlinks
152 )
153 if path.is_file():
154 file_count += 1
155 shards.extend(self._get_shards(path))
156
157 if file_count == 0:
158 raise ValueError(
159 "Model contains no regular files after applying exclusions"
160 )
161
162 if not shards:
163 raise ValueError(
164 "Model contains only empty files which produce no shards"
165 )
166
167 manifest_items = []
168 with concurrent.futures.ThreadPoolExecutor(
169 max_workers=self._max_workers
170 ) as tpe:
171 futures = [
172 tpe.submit(self._compute_hash, model_path, path, start, end)
173 for path, start, end in shards
174 ]
175 for future in concurrent.futures.as_completed(futures):
176 manifest_items.append(future.result())
177
178 # Recreate serialization_description for new ignore_paths
179 if ignore_paths:
180 rel_ignore_paths = []
181 for p in ignore_paths:
182 rp = os.path.relpath(p, model_path)
183 # rp may start with "../" if it is not relative to model_path
184 if not rp.startswith("../"):
185 rel_ignore_paths.append(pathlib.Path(rp))
186
187 hasher = self._hasher_factory(pathlib.Path(), 0, 1)
188 self._serialization_description = manifest._ShardSerialization(
189 hasher._content_hasher.digest_name,
190 self._shard_size,
191 self._allow_symlinks,
192 frozenset(list(self._ignore_paths) + rel_ignore_paths),
193 )
194
195 model_name = model_path.name
196 if not model_name or model_name == "..":
197 model_name = os.path.basename(model_path.resolve())
198
199 return manifest.Manifest(
200 model_name, manifest_items, self._serialization_description
201 )
202
203 def _get_shards(
204 self, path: pathlib.Path
205 ) -> list[tuple[pathlib.Path, int, int]]:
206 """Determines the shards of a given file path."""
207 shards = []
208 path_size = path.stat().st_size
209 if path_size > 0:
210 start = 0
211 for end in _endpoints(self._shard_size, path_size):
212 shards.append((path, start, end))
213 start = end
214 return shards
215
216 def _compute_hash(
217 self, model_path: pathlib.Path, path: pathlib.Path, start: int, end: int
218 ) -> manifest.ShardedFileManifestItem:
219 """Produces the manifest item of the file given by `path`.
220
221 Args:
222 model_path: The path to the model.
223 path: Path to the file in the model, that is currently transformed
224 to a manifest item.
225 start: The start offset of the shard (included).
226 end: The end offset of the shard (not included).
227
228 Returns:
229 The itemized manifest.
230 """
231 relative_path = path.relative_to(model_path)
232 digest = self._hasher_factory(path, start, end).compute()
233 return manifest.ShardedFileManifestItem(
234 path=relative_path, digest=digest, start=start, end=end
235 )