Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/werkzeug/formparser.py: 25%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1from __future__ import annotations
3import typing as t
4from io import BytesIO
5from urllib.parse import parse_qsl
7from ._internal import _plain_int
8from .datastructures import FileStorage
9from .datastructures import Headers
10from .datastructures import MultiDict
11from .exceptions import RequestEntityTooLarge
12from .http import parse_options_header
13from .sansio.multipart import Data
14from .sansio.multipart import Epilogue
15from .sansio.multipart import Field
16from .sansio.multipart import File
17from .sansio.multipart import MultipartDecoder
18from .sansio.multipart import NeedData
19from .wsgi import get_content_length
20from .wsgi import get_input_stream
22# there are some platforms where SpooledTemporaryFile is not available.
23# In that case we need to provide a fallback.
24try:
25 from tempfile import SpooledTemporaryFile
26except ImportError:
27 from tempfile import TemporaryFile
29 SpooledTemporaryFile = None # type: ignore
31if t.TYPE_CHECKING:
32 import typing as te
34 from _typeshed.wsgi import WSGIEnvironment
36 t_parse_result = tuple[
37 t.IO[bytes], MultiDict[str, str], MultiDict[str, FileStorage]
38 ]
40 class TStreamFactory(te.Protocol):
41 def __call__(
42 self,
43 total_content_length: int | None,
44 content_type: str | None,
45 filename: str | None,
46 content_length: int | None = None,
47 ) -> t.IO[bytes]: ...
50F = t.TypeVar("F", bound=t.Callable[..., t.Any])
53def default_stream_factory(
54 total_content_length: int | None,
55 content_type: str | None,
56 filename: str | None,
57 content_length: int | None = None,
58) -> t.IO[bytes]:
59 max_size = 1024 * 500
61 if SpooledTemporaryFile is not None:
62 return t.cast(t.IO[bytes], SpooledTemporaryFile(max_size=max_size, mode="rb+"))
63 elif total_content_length is None or total_content_length > max_size:
64 return t.cast(t.IO[bytes], TemporaryFile("rb+"))
66 return BytesIO()
69def parse_form_data(
70 environ: WSGIEnvironment,
71 stream_factory: TStreamFactory | None = None,
72 max_form_memory_size: int | None = None,
73 max_content_length: int | None = None,
74 cls: type[MultiDict[str, t.Any]] | None = None,
75 silent: bool = True,
76 *,
77 max_form_parts: int | None = None,
78) -> t_parse_result:
79 """Parse the form data in the environ and return it as tuple in the form
80 ``(stream, form, files)``. You should only call this method if the
81 transport method is `POST`, `PUT`, or `PATCH`.
83 If the mimetype of the data transmitted is `multipart/form-data` the
84 files multidict will be filled with `FileStorage` objects. If the
85 mimetype is unknown the input stream is wrapped and returned as first
86 argument, else the stream is empty.
88 This is a shortcut for the common usage of :class:`FormDataParser`.
90 :param environ: the WSGI environment to be used for parsing.
91 :param stream_factory: An optional callable that returns a new read and
92 writeable file descriptor. This callable works
93 the same as :meth:`Response._get_file_stream`.
94 :param max_content_length: If the data is larger than this many bytes, raise
95 :exc:`.RequestEntityTooLarge`. This is used by :meth:`parse_from_environ`
96 to set up a limited stream. When using :meth:`parse`, you must get a
97 limited stream with :func:`.get_input_stream`.
98 :param max_form_memory_size: If a ``multipart/form-data`` text part is larger
99 than this many bytes, raise :exc:`~exceptions.RequestEntityTooLarge`.
100 File parts are written to disk after this size. This is an additional
101 check, it does not replace using a limited stream and
102 ``max_content_length``.
103 :param max_form_parts: If more than this number of ``multipart/form-data``
104 parts are received, raise :exc:`.RequestEntityTooLarge`. This is an
105 additional check, it does not replace using a limited stream and
106 ``max_content_length``.
107 :param cls: an optional dict class to use. If this is not specified
108 or `None` the default :class:`MultiDict` is used.
109 :param silent: If set to False parsing errors will not be caught.
110 :return: A tuple in the form ``(stream, form, files)``.
112 .. versionchanged:: 3.1.9
113 ``max_form_memory_size`` is not applied to
114 ``application/x-www-form-urlencoded``.
116 .. versionchanged:: 3.0
117 The ``charset`` and ``errors`` parameters were removed.
119 .. versionchanged:: 2.3
120 Added the ``max_form_parts`` parameter.
122 .. versionadded:: 0.5.1
123 Added the ``silent`` parameter.
125 .. versionadded:: 0.5
126 Added the ``max_form_memory_size``, ``max_content_length``, and ``cls``
127 parameters.
128 """
129 return FormDataParser(
130 stream_factory=stream_factory,
131 max_form_memory_size=max_form_memory_size,
132 max_content_length=max_content_length,
133 max_form_parts=max_form_parts,
134 silent=silent,
135 cls=cls,
136 ).parse_from_environ(environ)
139class FormDataParser:
140 """This class implements parsing of form data for Werkzeug. By itself
141 it can parse multipart and url encoded form data. It can be subclassed
142 and extended but for most mimetypes it is a better idea to use the
143 untouched stream and expose it as separate attributes on a request
144 object.
146 :param stream_factory: An optional callable that returns a new read and
147 writeable file descriptor. This callable works
148 the same as :meth:`Response._get_file_stream`.
149 :param max_content_length: If the data is larger than this many bytes, raise
150 :exc:`.RequestEntityTooLarge`. This is used by :meth:`parse_from_environ`
151 to set up a limited stream. When using :meth:`parse`, you must get a
152 limited stream with :func:`.get_input_stream`.
153 :param max_form_memory_size: If a ``multipart/form-data`` text part is larger
154 than this many bytes, raise :exc:`~exceptions.RequestEntityTooLarge`.
155 File parts are written to disk after this size. This is an additional
156 check, it does not replace using a limited stream and
157 ``max_content_length``.
158 :param max_form_parts: If more than this number of ``multipart/form-data``
159 parts are received, raise :exc:`.RequestEntityTooLarge`. This is an
160 additional check, it does not replace using a limited stream and
161 ``max_content_length``.
162 :param cls: an optional dict class to use. If this is not specified
163 or `None` the default :class:`MultiDict` is used.
164 :param silent: If set to False parsing errors will not be caught.
166 .. versionchanged:: 3.0
167 The ``charset`` and ``errors`` parameters were removed.
169 .. versionchanged:: 3.0
170 The ``parse_functions`` attribute and ``get_parse_func`` methods were removed.
172 .. versionchanged:: 2.2.3
173 Added the ``max_form_parts`` parameter.
175 .. versionadded:: 0.8
176 """
178 def __init__(
179 self,
180 stream_factory: TStreamFactory | None = None,
181 max_form_memory_size: int | None = None,
182 max_content_length: int | None = None,
183 cls: type[MultiDict[str, t.Any]] | None = None,
184 silent: bool = True,
185 *,
186 max_form_parts: int | None = None,
187 ) -> None:
188 if stream_factory is None:
189 stream_factory = default_stream_factory
191 self.stream_factory = stream_factory
192 self.max_content_length = max_content_length
193 self.max_form_memory_size = max_form_memory_size
194 self.max_form_parts = max_form_parts
196 if cls is None:
197 cls = t.cast("type[MultiDict[str, t.Any]]", MultiDict)
199 self.cls = cls
200 self.silent = silent
202 def parse_from_environ(self, environ: WSGIEnvironment) -> t_parse_result:
203 """Parses the information from the environment as form data.
205 :param environ: the WSGI environment to be used for parsing.
206 :return: A tuple in the form ``(stream, form, files)``.
207 """
208 stream = get_input_stream(environ, max_content_length=self.max_content_length)
209 content_length = get_content_length(environ)
210 mimetype, options = parse_options_header(environ.get("CONTENT_TYPE"))
211 return self.parse(
212 stream,
213 content_length=content_length,
214 mimetype=mimetype,
215 options=options,
216 )
218 def parse(
219 self,
220 stream: t.IO[bytes],
221 mimetype: str,
222 content_length: int | None,
223 options: dict[str, str] | None = None,
224 ) -> t_parse_result:
225 """Parses the information from the given stream, mimetype,
226 content length and mimetype parameters.
228 :param stream: A limited input stream, from :meth:`.get_input_stream`.
229 :param mimetype: The mimetype used to choose how to parse the form.
230 ``multipart/form-data`` and ``application/x-www-form-urlencoded``
231 are supported.
232 :param content_length: the content length of the incoming data
233 :param options: optional mimetype parameters (used for
234 the multipart boundary for instance)
235 :return: A tuple in the form ``(stream, form, files)``.
237 .. versionchanged:: 3.0
238 The invalid ``application/x-url-encoded`` content type is not
239 treated as ``application/x-www-form-urlencoded``.
240 """
241 if mimetype == "multipart/form-data":
242 parse_func = self._parse_multipart
243 elif mimetype == "application/x-www-form-urlencoded":
244 parse_func = self._parse_urlencoded
245 else:
246 return stream, self.cls(), self.cls()
248 if options is None:
249 options = {}
251 try:
252 return parse_func(stream, mimetype, content_length, options)
253 except ValueError:
254 if not self.silent:
255 raise
257 return stream, self.cls(), self.cls()
259 def _parse_multipart(
260 self,
261 stream: t.IO[bytes],
262 mimetype: str,
263 content_length: int | None,
264 options: dict[str, str],
265 ) -> t_parse_result:
266 parser = MultiPartParser(
267 stream_factory=self.stream_factory,
268 max_form_memory_size=self.max_form_memory_size,
269 max_form_parts=self.max_form_parts,
270 cls=self.cls,
271 )
272 boundary = options.get("boundary", "").encode("ascii")
274 if not boundary:
275 raise ValueError("Missing boundary")
277 form, files = parser.parse(stream, boundary, content_length)
278 return stream, form, files
280 def _parse_urlencoded(
281 self,
282 stream: t.IO[bytes],
283 mimetype: str,
284 content_length: int | None,
285 options: dict[str, str],
286 ) -> t_parse_result:
287 # The stream must already be limited to max_content_length.
288 # max_form_memory_size can't apply, since the entire stream is read at
289 # once instead of incrementally parsed.
290 # max_form_parts could be applied here, but doesn't impose a useful
291 # limit given the previous points and because the parser is much simpler
292 # and faster already.
293 items = parse_qsl(
294 stream.read().decode(),
295 keep_blank_values=True,
296 errors="werkzeug.url_quote",
297 )
298 return stream, self.cls(items), self.cls()
301class MultiPartParser:
302 def __init__(
303 self,
304 stream_factory: TStreamFactory | None = None,
305 max_form_memory_size: int | None = None,
306 cls: type[MultiDict[str, t.Any]] | None = None,
307 buffer_size: int = 64 * 1024,
308 max_form_parts: int | None = None,
309 ) -> None:
310 self.max_form_memory_size = max_form_memory_size
311 self.max_form_parts = max_form_parts
313 if stream_factory is None:
314 stream_factory = default_stream_factory
316 self.stream_factory = stream_factory
318 if cls is None:
319 cls = t.cast("type[MultiDict[str, t.Any]]", MultiDict)
321 self.cls = cls
322 self.buffer_size = buffer_size
324 def fail(self, message: str) -> te.NoReturn:
325 raise ValueError(message)
327 def get_part_charset(self, headers: Headers) -> str:
328 # Figure out input charset for current part
329 content_type = headers.get("content-type")
331 if content_type:
332 parameters = parse_options_header(content_type)[1]
333 ct_charset = parameters.get("charset", "").lower()
335 # A safe list of encodings. Modern clients should only send ASCII or UTF-8.
336 # This list will not be extended further.
337 if ct_charset in {"ascii", "us-ascii", "utf-8", "iso-8859-1"}:
338 return ct_charset
340 return "utf-8"
342 def start_file_streaming(
343 self, event: File, total_content_length: int | None
344 ) -> t.IO[bytes]:
345 content_type = event.headers.get("content-type")
347 try:
348 content_length = _plain_int(event.headers["content-length"])
349 except (KeyError, ValueError):
350 content_length = 0
352 container = self.stream_factory(
353 total_content_length=total_content_length,
354 filename=event.filename,
355 content_type=content_type,
356 content_length=content_length,
357 )
358 return container
360 def parse(
361 self, stream: t.IO[bytes], boundary: bytes, content_length: int | None
362 ) -> tuple[MultiDict[str, str], MultiDict[str, FileStorage]]:
363 current_part: Field | File
364 field_size: int | None = None
365 container: t.IO[bytes] | list[bytes]
366 _write: t.Callable[[bytes], t.Any]
368 parser = MultipartDecoder(
369 boundary,
370 max_form_memory_size=self.max_form_memory_size,
371 max_parts=self.max_form_parts,
372 )
374 fields = []
375 files = []
377 for data in _chunk_iter(stream.read, self.buffer_size):
378 parser.receive_data(data)
379 event = parser.next_event()
380 while not isinstance(event, (Epilogue, NeedData)):
381 if isinstance(event, Field):
382 current_part = event
383 field_size = 0
384 container = []
385 _write = container.append
386 elif isinstance(event, File):
387 current_part = event
388 field_size = None
389 container = self.start_file_streaming(event, content_length)
390 _write = container.write
391 elif isinstance(event, Data):
392 if self.max_form_memory_size is not None and field_size is not None:
393 # Ensure that accumulated data events do not exceed limit.
394 # Also checked within single event in MultipartDecoder.
395 field_size += len(event.data)
397 if field_size > self.max_form_memory_size:
398 raise RequestEntityTooLarge()
400 _write(event.data)
401 if not event.more_data:
402 if isinstance(current_part, Field):
403 value = b"".join(container).decode(
404 self.get_part_charset(current_part.headers), "replace"
405 )
406 fields.append((current_part.name, value))
407 else:
408 container = t.cast(t.IO[bytes], container)
409 container.seek(0)
410 files.append(
411 (
412 current_part.name,
413 FileStorage(
414 container,
415 current_part.filename,
416 current_part.name,
417 headers=current_part.headers,
418 ),
419 )
420 )
422 event = parser.next_event()
424 return self.cls(fields), self.cls(files)
427def _chunk_iter(read: t.Callable[[int], bytes], size: int) -> t.Iterator[bytes | None]:
428 """Read data in chunks for multipart/form-data parsing. Stop if no data is read.
429 Yield ``None`` at the end to signal end of parsing.
430 """
431 while True:
432 data = read(size)
434 if not data:
435 break
437 yield data
439 yield None