Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/werkzeug/formparser.py: 25%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

150 statements  

1from __future__ import annotations 

2 

3import typing as t 

4from io import BytesIO 

5from urllib.parse import parse_qsl 

6 

7from ._internal import _plain_int 

8from .datastructures import FileStorage 

9from .datastructures import Headers 

10from .datastructures import MultiDict 

11from .exceptions import RequestEntityTooLarge 

12from .http import parse_options_header 

13from .sansio.multipart import Data 

14from .sansio.multipart import Epilogue 

15from .sansio.multipart import Field 

16from .sansio.multipart import File 

17from .sansio.multipart import MultipartDecoder 

18from .sansio.multipart import NeedData 

19from .wsgi import get_content_length 

20from .wsgi import get_input_stream 

21 

22# there are some platforms where SpooledTemporaryFile is not available. 

23# In that case we need to provide a fallback. 

24try: 

25 from tempfile import SpooledTemporaryFile 

26except ImportError: 

27 from tempfile import TemporaryFile 

28 

29 SpooledTemporaryFile = None # type: ignore 

30 

31if t.TYPE_CHECKING: 

32 import typing as te 

33 

34 from _typeshed.wsgi import WSGIEnvironment 

35 

36 t_parse_result = tuple[ 

37 t.IO[bytes], MultiDict[str, str], MultiDict[str, FileStorage] 

38 ] 

39 

40 class TStreamFactory(te.Protocol): 

41 def __call__( 

42 self, 

43 total_content_length: int | None, 

44 content_type: str | None, 

45 filename: str | None, 

46 content_length: int | None = None, 

47 ) -> t.IO[bytes]: ... 

48 

49 

50F = t.TypeVar("F", bound=t.Callable[..., t.Any]) 

51 

52 

53def default_stream_factory( 

54 total_content_length: int | None, 

55 content_type: str | None, 

56 filename: str | None, 

57 content_length: int | None = None, 

58) -> t.IO[bytes]: 

59 max_size = 1024 * 500 

60 

61 if SpooledTemporaryFile is not None: 

62 return t.cast(t.IO[bytes], SpooledTemporaryFile(max_size=max_size, mode="rb+")) 

63 elif total_content_length is None or total_content_length > max_size: 

64 return t.cast(t.IO[bytes], TemporaryFile("rb+")) 

65 

66 return BytesIO() 

67 

68 

69def parse_form_data( 

70 environ: WSGIEnvironment, 

71 stream_factory: TStreamFactory | None = None, 

72 max_form_memory_size: int | None = None, 

73 max_content_length: int | None = None, 

74 cls: type[MultiDict[str, t.Any]] | None = None, 

75 silent: bool = True, 

76 *, 

77 max_form_parts: int | None = None, 

78) -> t_parse_result: 

79 """Parse the form data in the environ and return it as tuple in the form 

80 ``(stream, form, files)``. You should only call this method if the 

81 transport method is `POST`, `PUT`, or `PATCH`. 

82 

83 If the mimetype of the data transmitted is `multipart/form-data` the 

84 files multidict will be filled with `FileStorage` objects. If the 

85 mimetype is unknown the input stream is wrapped and returned as first 

86 argument, else the stream is empty. 

87 

88 This is a shortcut for the common usage of :class:`FormDataParser`. 

89 

90 :param environ: the WSGI environment to be used for parsing. 

91 :param stream_factory: An optional callable that returns a new read and 

92 writeable file descriptor. This callable works 

93 the same as :meth:`Response._get_file_stream`. 

94 :param max_content_length: If the data is larger than this many bytes, raise 

95 :exc:`.RequestEntityTooLarge`. This is used by :meth:`parse_from_environ` 

96 to set up a limited stream. When using :meth:`parse`, you must get a 

97 limited stream with :func:`.get_input_stream`. 

98 :param max_form_memory_size: If a ``multipart/form-data`` text part is larger 

99 than this many bytes, raise :exc:`~exceptions.RequestEntityTooLarge`. 

100 File parts are written to disk after this size. This is an additional 

101 check, it does not replace using a limited stream and 

102 ``max_content_length``. 

103 :param max_form_parts: If more than this number of ``multipart/form-data`` 

104 parts are received, raise :exc:`.RequestEntityTooLarge`. This is an 

105 additional check, it does not replace using a limited stream and 

106 ``max_content_length``. 

107 :param cls: an optional dict class to use. If this is not specified 

108 or `None` the default :class:`MultiDict` is used. 

109 :param silent: If set to False parsing errors will not be caught. 

110 :return: A tuple in the form ``(stream, form, files)``. 

111 

112 .. versionchanged:: 3.1.9 

113 ``max_form_memory_size`` is not applied to 

114 ``application/x-www-form-urlencoded``. 

115 

116 .. versionchanged:: 3.0 

117 The ``charset`` and ``errors`` parameters were removed. 

118 

119 .. versionchanged:: 2.3 

120 Added the ``max_form_parts`` parameter. 

121 

122 .. versionadded:: 0.5.1 

123 Added the ``silent`` parameter. 

124 

125 .. versionadded:: 0.5 

126 Added the ``max_form_memory_size``, ``max_content_length``, and ``cls`` 

127 parameters. 

128 """ 

129 return FormDataParser( 

130 stream_factory=stream_factory, 

131 max_form_memory_size=max_form_memory_size, 

132 max_content_length=max_content_length, 

133 max_form_parts=max_form_parts, 

134 silent=silent, 

135 cls=cls, 

136 ).parse_from_environ(environ) 

137 

138 

139class FormDataParser: 

140 """This class implements parsing of form data for Werkzeug. By itself 

141 it can parse multipart and url encoded form data. It can be subclassed 

142 and extended but for most mimetypes it is a better idea to use the 

143 untouched stream and expose it as separate attributes on a request 

144 object. 

145 

146 :param stream_factory: An optional callable that returns a new read and 

147 writeable file descriptor. This callable works 

148 the same as :meth:`Response._get_file_stream`. 

149 :param max_content_length: If the data is larger than this many bytes, raise 

150 :exc:`.RequestEntityTooLarge`. This is used by :meth:`parse_from_environ` 

151 to set up a limited stream. When using :meth:`parse`, you must get a 

152 limited stream with :func:`.get_input_stream`. 

153 :param max_form_memory_size: If a ``multipart/form-data`` text part is larger 

154 than this many bytes, raise :exc:`~exceptions.RequestEntityTooLarge`. 

155 File parts are written to disk after this size. This is an additional 

156 check, it does not replace using a limited stream and 

157 ``max_content_length``. 

158 :param max_form_parts: If more than this number of ``multipart/form-data`` 

159 parts are received, raise :exc:`.RequestEntityTooLarge`. This is an 

160 additional check, it does not replace using a limited stream and 

161 ``max_content_length``. 

162 :param cls: an optional dict class to use. If this is not specified 

163 or `None` the default :class:`MultiDict` is used. 

164 :param silent: If set to False parsing errors will not be caught. 

165 

166 .. versionchanged:: 3.0 

167 The ``charset`` and ``errors`` parameters were removed. 

168 

169 .. versionchanged:: 3.0 

170 The ``parse_functions`` attribute and ``get_parse_func`` methods were removed. 

171 

172 .. versionchanged:: 2.2.3 

173 Added the ``max_form_parts`` parameter. 

174 

175 .. versionadded:: 0.8 

176 """ 

177 

178 def __init__( 

179 self, 

180 stream_factory: TStreamFactory | None = None, 

181 max_form_memory_size: int | None = None, 

182 max_content_length: int | None = None, 

183 cls: type[MultiDict[str, t.Any]] | None = None, 

184 silent: bool = True, 

185 *, 

186 max_form_parts: int | None = None, 

187 ) -> None: 

188 if stream_factory is None: 

189 stream_factory = default_stream_factory 

190 

191 self.stream_factory = stream_factory 

192 self.max_content_length = max_content_length 

193 self.max_form_memory_size = max_form_memory_size 

194 self.max_form_parts = max_form_parts 

195 

196 if cls is None: 

197 cls = t.cast("type[MultiDict[str, t.Any]]", MultiDict) 

198 

199 self.cls = cls 

200 self.silent = silent 

201 

202 def parse_from_environ(self, environ: WSGIEnvironment) -> t_parse_result: 

203 """Parses the information from the environment as form data. 

204 

205 :param environ: the WSGI environment to be used for parsing. 

206 :return: A tuple in the form ``(stream, form, files)``. 

207 """ 

208 stream = get_input_stream(environ, max_content_length=self.max_content_length) 

209 content_length = get_content_length(environ) 

210 mimetype, options = parse_options_header(environ.get("CONTENT_TYPE")) 

211 return self.parse( 

212 stream, 

213 content_length=content_length, 

214 mimetype=mimetype, 

215 options=options, 

216 ) 

217 

218 def parse( 

219 self, 

220 stream: t.IO[bytes], 

221 mimetype: str, 

222 content_length: int | None, 

223 options: dict[str, str] | None = None, 

224 ) -> t_parse_result: 

225 """Parses the information from the given stream, mimetype, 

226 content length and mimetype parameters. 

227 

228 :param stream: A limited input stream, from :meth:`.get_input_stream`. 

229 :param mimetype: The mimetype used to choose how to parse the form. 

230 ``multipart/form-data`` and ``application/x-www-form-urlencoded`` 

231 are supported. 

232 :param content_length: the content length of the incoming data 

233 :param options: optional mimetype parameters (used for 

234 the multipart boundary for instance) 

235 :return: A tuple in the form ``(stream, form, files)``. 

236 

237 .. versionchanged:: 3.0 

238 The invalid ``application/x-url-encoded`` content type is not 

239 treated as ``application/x-www-form-urlencoded``. 

240 """ 

241 if mimetype == "multipart/form-data": 

242 parse_func = self._parse_multipart 

243 elif mimetype == "application/x-www-form-urlencoded": 

244 parse_func = self._parse_urlencoded 

245 else: 

246 return stream, self.cls(), self.cls() 

247 

248 if options is None: 

249 options = {} 

250 

251 try: 

252 return parse_func(stream, mimetype, content_length, options) 

253 except ValueError: 

254 if not self.silent: 

255 raise 

256 

257 return stream, self.cls(), self.cls() 

258 

259 def _parse_multipart( 

260 self, 

261 stream: t.IO[bytes], 

262 mimetype: str, 

263 content_length: int | None, 

264 options: dict[str, str], 

265 ) -> t_parse_result: 

266 parser = MultiPartParser( 

267 stream_factory=self.stream_factory, 

268 max_form_memory_size=self.max_form_memory_size, 

269 max_form_parts=self.max_form_parts, 

270 cls=self.cls, 

271 ) 

272 boundary = options.get("boundary", "").encode("ascii") 

273 

274 if not boundary: 

275 raise ValueError("Missing boundary") 

276 

277 form, files = parser.parse(stream, boundary, content_length) 

278 return stream, form, files 

279 

280 def _parse_urlencoded( 

281 self, 

282 stream: t.IO[bytes], 

283 mimetype: str, 

284 content_length: int | None, 

285 options: dict[str, str], 

286 ) -> t_parse_result: 

287 # The stream must already be limited to max_content_length. 

288 # max_form_memory_size can't apply, since the entire stream is read at 

289 # once instead of incrementally parsed. 

290 # max_form_parts could be applied here, but doesn't impose a useful 

291 # limit given the previous points and because the parser is much simpler 

292 # and faster already. 

293 items = parse_qsl( 

294 stream.read().decode(), 

295 keep_blank_values=True, 

296 errors="werkzeug.url_quote", 

297 ) 

298 return stream, self.cls(items), self.cls() 

299 

300 

301class MultiPartParser: 

302 def __init__( 

303 self, 

304 stream_factory: TStreamFactory | None = None, 

305 max_form_memory_size: int | None = None, 

306 cls: type[MultiDict[str, t.Any]] | None = None, 

307 buffer_size: int = 64 * 1024, 

308 max_form_parts: int | None = None, 

309 ) -> None: 

310 self.max_form_memory_size = max_form_memory_size 

311 self.max_form_parts = max_form_parts 

312 

313 if stream_factory is None: 

314 stream_factory = default_stream_factory 

315 

316 self.stream_factory = stream_factory 

317 

318 if cls is None: 

319 cls = t.cast("type[MultiDict[str, t.Any]]", MultiDict) 

320 

321 self.cls = cls 

322 self.buffer_size = buffer_size 

323 

324 def fail(self, message: str) -> te.NoReturn: 

325 raise ValueError(message) 

326 

327 def get_part_charset(self, headers: Headers) -> str: 

328 # Figure out input charset for current part 

329 content_type = headers.get("content-type") 

330 

331 if content_type: 

332 parameters = parse_options_header(content_type)[1] 

333 ct_charset = parameters.get("charset", "").lower() 

334 

335 # A safe list of encodings. Modern clients should only send ASCII or UTF-8. 

336 # This list will not be extended further. 

337 if ct_charset in {"ascii", "us-ascii", "utf-8", "iso-8859-1"}: 

338 return ct_charset 

339 

340 return "utf-8" 

341 

342 def start_file_streaming( 

343 self, event: File, total_content_length: int | None 

344 ) -> t.IO[bytes]: 

345 content_type = event.headers.get("content-type") 

346 

347 try: 

348 content_length = _plain_int(event.headers["content-length"]) 

349 except (KeyError, ValueError): 

350 content_length = 0 

351 

352 container = self.stream_factory( 

353 total_content_length=total_content_length, 

354 filename=event.filename, 

355 content_type=content_type, 

356 content_length=content_length, 

357 ) 

358 return container 

359 

360 def parse( 

361 self, stream: t.IO[bytes], boundary: bytes, content_length: int | None 

362 ) -> tuple[MultiDict[str, str], MultiDict[str, FileStorage]]: 

363 current_part: Field | File 

364 field_size: int | None = None 

365 container: t.IO[bytes] | list[bytes] 

366 _write: t.Callable[[bytes], t.Any] 

367 

368 parser = MultipartDecoder( 

369 boundary, 

370 max_form_memory_size=self.max_form_memory_size, 

371 max_parts=self.max_form_parts, 

372 ) 

373 

374 fields = [] 

375 files = [] 

376 

377 for data in _chunk_iter(stream.read, self.buffer_size): 

378 parser.receive_data(data) 

379 event = parser.next_event() 

380 while not isinstance(event, (Epilogue, NeedData)): 

381 if isinstance(event, Field): 

382 current_part = event 

383 field_size = 0 

384 container = [] 

385 _write = container.append 

386 elif isinstance(event, File): 

387 current_part = event 

388 field_size = None 

389 container = self.start_file_streaming(event, content_length) 

390 _write = container.write 

391 elif isinstance(event, Data): 

392 if self.max_form_memory_size is not None and field_size is not None: 

393 # Ensure that accumulated data events do not exceed limit. 

394 # Also checked within single event in MultipartDecoder. 

395 field_size += len(event.data) 

396 

397 if field_size > self.max_form_memory_size: 

398 raise RequestEntityTooLarge() 

399 

400 _write(event.data) 

401 if not event.more_data: 

402 if isinstance(current_part, Field): 

403 value = b"".join(container).decode( 

404 self.get_part_charset(current_part.headers), "replace" 

405 ) 

406 fields.append((current_part.name, value)) 

407 else: 

408 container = t.cast(t.IO[bytes], container) 

409 container.seek(0) 

410 files.append( 

411 ( 

412 current_part.name, 

413 FileStorage( 

414 container, 

415 current_part.filename, 

416 current_part.name, 

417 headers=current_part.headers, 

418 ), 

419 ) 

420 ) 

421 

422 event = parser.next_event() 

423 

424 return self.cls(fields), self.cls(files) 

425 

426 

427def _chunk_iter(read: t.Callable[[int], bytes], size: int) -> t.Iterator[bytes | None]: 

428 """Read data in chunks for multipart/form-data parsing. Stop if no data is read. 

429 Yield ``None`` at the end to signal end of parsing. 

430 """ 

431 while True: 

432 data = read(size) 

433 

434 if not data: 

435 break 

436 

437 yield data 

438 

439 yield None