/rust/registry/src/index.crates.io-1949cf8c6b5b557f/flate2-1.1.10/src/gz/mod.rs
Line | Count | Source |
1 | | use crate::io::{BufRead, Error, ErrorKind, Read, Result, Write}; |
2 | | use alloc::boxed::Box; |
3 | | use alloc::ffi::CString; |
4 | | use alloc::vec::Vec; |
5 | | use core::convert::TryFrom; |
6 | | use core::time; |
7 | | |
8 | | use crate::bufreader::BufReader; |
9 | | use crate::{Compression, Crc}; |
10 | | |
11 | | pub static FHCRC: u8 = 1 << 1; |
12 | | pub static FEXTRA: u8 = 1 << 2; |
13 | | pub static FNAME: u8 = 1 << 3; |
14 | | pub static FCOMMENT: u8 = 1 << 4; |
15 | | pub static FRESERVED: u8 = 1 << 5 | 1 << 6 | 1 << 7; |
16 | | |
17 | | pub mod bufread; |
18 | | pub mod read; |
19 | | pub mod write; |
20 | | |
21 | | // The maximum length of the header filename and comment fields. More than |
22 | | // enough for these fields in reasonable use, but prevents possible attacks. |
23 | | const MAX_HEADER_BUF: usize = 65535; |
24 | | |
25 | | /// A structure representing the header of a gzip stream. |
26 | | /// |
27 | | /// The header can contain metadata about the file that was compressed, if |
28 | | /// present. |
29 | | #[derive(PartialEq, Clone, Debug, Default)] |
30 | | pub struct GzHeader { |
31 | | extra: Option<Vec<u8>>, |
32 | | filename: Option<Vec<u8>>, |
33 | | comment: Option<Vec<u8>>, |
34 | | operating_system: u8, |
35 | | mtime: u32, |
36 | | } |
37 | | |
38 | | impl GzHeader { |
39 | | /// Returns the `filename` field of this gzip stream's header, if present. |
40 | 0 | pub fn filename(&self) -> Option<&[u8]> { |
41 | 0 | self.filename.as_ref().map(|s| &s[..]) |
42 | 0 | } |
43 | | |
44 | | /// Returns the `extra` field of this gzip stream's header, if present. |
45 | 0 | pub fn extra(&self) -> Option<&[u8]> { |
46 | 0 | self.extra.as_ref().map(|s| &s[..]) |
47 | 0 | } |
48 | | |
49 | | /// Returns the `comment` field of this gzip stream's header, if present. |
50 | 0 | pub fn comment(&self) -> Option<&[u8]> { |
51 | 0 | self.comment.as_ref().map(|s| &s[..]) |
52 | 0 | } |
53 | | |
54 | | /// Returns the `operating_system` field of this gzip stream's header. |
55 | | /// |
56 | | /// There are predefined values for various operating systems. |
57 | | /// 255 means that the value is unknown. |
58 | 0 | pub fn operating_system(&self) -> u8 { |
59 | 0 | self.operating_system |
60 | 0 | } |
61 | | |
62 | | /// This gives the most recent modification time of the original file being compressed. |
63 | | /// |
64 | | /// The time is in Unix format, i.e., seconds since 00:00:00 GMT, Jan. 1, 1970. |
65 | | /// (Note that this may cause problems for MS-DOS and other systems that use local |
66 | | /// rather than Universal time.) If the compressed data did not come from a file, |
67 | | /// `mtime` is set to the time at which compression started. |
68 | | /// `mtime` = 0 means no time stamp is available. |
69 | | /// |
70 | | /// The usage of `mtime` is discouraged because of Year 2038 problem. |
71 | 0 | pub fn mtime(&self) -> u32 { |
72 | 0 | self.mtime |
73 | 0 | } |
74 | | |
75 | | /// Returns the most recent modification time represented by a date-time type. |
76 | | /// Returns `None` if the value of the underlying counter is 0, |
77 | | /// indicating no time stamp is available. |
78 | | /// |
79 | | /// |
80 | | /// The time is measured as seconds since 00:00:00 GMT, Jan. 1 1970. |
81 | | /// See [`mtime`](#method.mtime) for more detail. |
82 | | #[cfg(not(flate2_unstable_nightly_alloc_io))] |
83 | 0 | pub fn mtime_as_datetime(&self) -> Option<std::time::SystemTime> { |
84 | 0 | self.mtime_as_duration().map(|d| std::time::UNIX_EPOCH + d) |
85 | 0 | } |
86 | | |
87 | | /// Returns the [`Duration`](time::Duration) between the most recent modification |
88 | | /// time and 00:00:00 GMT, Jan. 1 1970, also known as Unix epoch. |
89 | | /// See [`mtime`](#method.mtime) for more detail. |
90 | | /// Returns `None` if the value of the underlying counter is 0, |
91 | | /// indicating no time stamp is available. |
92 | 0 | pub fn mtime_as_duration(&self) -> Option<time::Duration> { |
93 | 0 | if self.mtime == 0 { |
94 | 0 | None |
95 | | } else { |
96 | 0 | let duration = time::Duration::new(u64::from(self.mtime), 0); |
97 | 0 | Some(duration) |
98 | | } |
99 | 0 | } |
100 | | } |
101 | | |
102 | | #[derive(Debug, Default)] |
103 | | pub enum GzHeaderState { |
104 | | Start(u8, [u8; 10]), |
105 | | Xlen(Option<Box<Crc>>, u8, [u8; 2]), |
106 | | Extra(Option<Box<Crc>>, u16), |
107 | | Filename(Option<Box<Crc>>), |
108 | | Comment(Option<Box<Crc>>), |
109 | | Crc(Option<Box<Crc>>, u8, [u8; 2]), |
110 | | #[default] |
111 | | Complete, |
112 | | } |
113 | | |
114 | | #[derive(Debug, Default)] |
115 | | pub struct GzHeaderParser { |
116 | | state: GzHeaderState, |
117 | | flags: u8, |
118 | | header: GzHeader, |
119 | | } |
120 | | |
121 | | impl GzHeaderParser { |
122 | 0 | fn new() -> Self { |
123 | 0 | GzHeaderParser { |
124 | 0 | state: GzHeaderState::Start(0, [0; 10]), |
125 | 0 | flags: 0, |
126 | 0 | header: GzHeader::default(), |
127 | 0 | } |
128 | 0 | } |
129 | | |
130 | 0 | fn parse<R: BufRead>(&mut self, r: &mut R) -> Result<()> { |
131 | | loop { |
132 | 0 | match &mut self.state { |
133 | 0 | GzHeaderState::Start(count, buffer) => { |
134 | 0 | while (*count as usize) < buffer.len() { |
135 | 0 | *count += read_into(r, &mut buffer[*count as usize..])? as u8; |
136 | | } |
137 | | // Gzip identification bytes |
138 | 0 | if buffer[0] != 0x1f || buffer[1] != 0x8b { |
139 | 0 | return Err(bad_header()); |
140 | 0 | } |
141 | | // Gzip compression method (8 = deflate) |
142 | 0 | if buffer[2] != 8 { |
143 | 0 | return Err(bad_header()); |
144 | 0 | } |
145 | 0 | self.flags = buffer[3]; |
146 | | // RFC1952: "must give an error indication if any reserved bit is non-zero" |
147 | 0 | if self.flags & FRESERVED != 0 { |
148 | 0 | return Err(bad_header()); |
149 | 0 | } |
150 | 0 | self.header.mtime = (buffer[4] as u32) |
151 | 0 | | ((buffer[5] as u32) << 8) |
152 | 0 | | ((buffer[6] as u32) << 16) |
153 | 0 | | ((buffer[7] as u32) << 24); |
154 | 0 | let _xfl = buffer[8]; |
155 | 0 | self.header.operating_system = buffer[9]; |
156 | 0 | let crc = if self.flags & FHCRC != 0 { |
157 | 0 | let mut crc = Box::new(Crc::new()); |
158 | 0 | crc.update(buffer); |
159 | 0 | Some(crc) |
160 | | } else { |
161 | 0 | None |
162 | | }; |
163 | 0 | self.state = GzHeaderState::Xlen(crc, 0, [0; 2]); |
164 | | } |
165 | 0 | GzHeaderState::Xlen(crc, count, buffer) => { |
166 | 0 | if self.flags & FEXTRA != 0 { |
167 | 0 | while (*count as usize) < buffer.len() { |
168 | 0 | *count += read_into(r, &mut buffer[*count as usize..])? as u8; |
169 | | } |
170 | 0 | if let Some(crc) = crc { |
171 | 0 | crc.update(buffer); |
172 | 0 | } |
173 | 0 | let xlen = parse_le_u16(buffer); |
174 | 0 | self.header.extra = Some(vec![0; xlen as usize]); |
175 | 0 | self.state = GzHeaderState::Extra(crc.take(), 0); |
176 | 0 | } else { |
177 | 0 | self.state = GzHeaderState::Filename(crc.take()); |
178 | 0 | } |
179 | | } |
180 | 0 | GzHeaderState::Extra(crc, count) => { |
181 | 0 | debug_assert!(self.header.extra.is_some()); |
182 | 0 | let extra = self.header.extra.as_mut().unwrap(); |
183 | 0 | while (*count as usize) < extra.len() { |
184 | 0 | *count += read_into(r, &mut extra[*count as usize..])? as u16; |
185 | | } |
186 | 0 | if let Some(crc) = crc { |
187 | 0 | crc.update(extra); |
188 | 0 | } |
189 | 0 | self.state = GzHeaderState::Filename(crc.take()); |
190 | | } |
191 | 0 | GzHeaderState::Filename(crc) => { |
192 | 0 | if self.flags & FNAME != 0 { |
193 | 0 | let filename = self.header.filename.get_or_insert_with(Vec::new); |
194 | 0 | read_to_nul(r, filename)?; |
195 | 0 | if let Some(crc) = crc { |
196 | 0 | crc.update(filename); |
197 | 0 | crc.update(b"\0"); |
198 | 0 | } |
199 | 0 | } |
200 | 0 | self.state = GzHeaderState::Comment(crc.take()); |
201 | | } |
202 | 0 | GzHeaderState::Comment(crc) => { |
203 | 0 | if self.flags & FCOMMENT != 0 { |
204 | 0 | let comment = self.header.comment.get_or_insert_with(Vec::new); |
205 | 0 | read_to_nul(r, comment)?; |
206 | 0 | if let Some(crc) = crc { |
207 | 0 | crc.update(comment); |
208 | 0 | crc.update(b"\0"); |
209 | 0 | } |
210 | 0 | } |
211 | 0 | self.state = GzHeaderState::Crc(crc.take(), 0, [0; 2]); |
212 | | } |
213 | 0 | GzHeaderState::Crc(crc, count, buffer) => { |
214 | 0 | if let Some(crc) = crc { |
215 | 0 | debug_assert!(self.flags & FHCRC != 0); |
216 | 0 | while (*count as usize) < buffer.len() { |
217 | 0 | *count += read_into(r, &mut buffer[*count as usize..])? as u8; |
218 | | } |
219 | 0 | let stored_crc = parse_le_u16(buffer); |
220 | 0 | let calced_crc = crc.sum() as u16; |
221 | 0 | if stored_crc != calced_crc { |
222 | 0 | return Err(corrupt()); |
223 | 0 | } |
224 | 0 | } |
225 | 0 | self.state = GzHeaderState::Complete; |
226 | | } |
227 | | GzHeaderState::Complete => { |
228 | 0 | return Ok(()); |
229 | | } |
230 | | } |
231 | | } |
232 | 0 | } |
233 | | |
234 | 0 | fn header(&self) -> Option<&GzHeader> { |
235 | 0 | match self.state { |
236 | 0 | GzHeaderState::Complete => Some(&self.header), |
237 | 0 | _ => None, |
238 | | } |
239 | 0 | } |
240 | | } |
241 | | |
242 | | impl From<GzHeaderParser> for GzHeader { |
243 | 0 | fn from(parser: GzHeaderParser) -> Self { |
244 | 0 | debug_assert!(matches!(parser.state, GzHeaderState::Complete)); |
245 | 0 | parser.header |
246 | 0 | } |
247 | | } |
248 | | |
249 | | // Attempt to fill the `buffer` from `r`. Return the number of bytes read. |
250 | | // Return an error if EOF is read before the buffer is full. This differs |
251 | | // from `read` in that Ok(0) means that more data may be available. |
252 | 0 | fn read_into<R: Read>(r: &mut R, buffer: &mut [u8]) -> Result<usize> { |
253 | 0 | debug_assert!(!buffer.is_empty()); |
254 | 0 | match r.read(buffer) { |
255 | 0 | Ok(0) => Err(ErrorKind::UnexpectedEof.into()), |
256 | 0 | Ok(n) => Ok(n), |
257 | 0 | Err(ref e) if e.kind() == ErrorKind::Interrupted => Ok(0), |
258 | 0 | Err(e) => Err(e), |
259 | | } |
260 | 0 | } |
261 | | |
262 | | // Read `r` up to the first nul byte, pushing non-nul bytes to `buffer`. |
263 | 0 | fn read_to_nul<R: BufRead>(r: &mut R, buffer: &mut Vec<u8>) -> Result<()> { |
264 | 0 | let mut bytes = r.bytes(); |
265 | | loop { |
266 | 0 | match bytes.next().transpose()? { |
267 | 0 | Some(0) => return Ok(()), |
268 | 0 | Some(_) if buffer.len() == MAX_HEADER_BUF => { |
269 | 0 | return Err(Error::new( |
270 | 0 | ErrorKind::InvalidInput, |
271 | 0 | "gzip header field too long", |
272 | 0 | )); |
273 | | } |
274 | 0 | Some(byte) => { |
275 | 0 | buffer.push(byte); |
276 | 0 | } |
277 | | None => { |
278 | 0 | return Err(ErrorKind::UnexpectedEof.into()); |
279 | | } |
280 | | } |
281 | | } |
282 | 0 | } |
283 | | |
284 | 0 | fn parse_le_u16(buffer: &[u8; 2]) -> u16 { |
285 | 0 | u16::from_le_bytes(*buffer) |
286 | 0 | } |
287 | | |
288 | 0 | fn bad_header() -> Error { |
289 | 0 | Error::new(ErrorKind::InvalidInput, "invalid gzip header") |
290 | 0 | } |
291 | | |
292 | 0 | fn corrupt() -> Error { |
293 | 0 | Error::new( |
294 | 0 | ErrorKind::InvalidInput, |
295 | | "corrupt gzip stream does not have a matching checksum", |
296 | | ) |
297 | 0 | } |
298 | | |
299 | | /// A builder structure to create a new gzip Encoder. |
300 | | /// |
301 | | /// This structure controls header configuration options such as the filename. |
302 | | /// |
303 | | /// # Examples |
304 | | /// |
305 | | /// ``` |
306 | | /// use std::io::prelude::*; |
307 | | /// # use std::io; |
308 | | /// use std::fs::File; |
309 | | /// use flate2::GzBuilder; |
310 | | /// use flate2::Compression; |
311 | | /// |
312 | | /// // GzBuilder opens a file and writes a sample string using GzBuilder pattern |
313 | | /// |
314 | | /// # fn sample_builder() -> Result<(), io::Error> { |
315 | | /// let f = File::create("examples/hello_world.gz")?; |
316 | | /// let mut gz = GzBuilder::new() |
317 | | /// .filename("hello_world.txt") |
318 | | /// .comment("test file, please delete") |
319 | | /// .write(f, Compression::default()); |
320 | | /// gz.write_all(b"hello world")?; |
321 | | /// gz.finish()?; |
322 | | /// # Ok(()) |
323 | | /// # } |
324 | | /// ``` |
325 | | #[derive(Debug, Default)] |
326 | | pub struct GzBuilder { |
327 | | extra: Option<Vec<u8>>, |
328 | | filename: Option<CString>, |
329 | | comment: Option<CString>, |
330 | | operating_system: Option<u8>, |
331 | | mtime: u32, |
332 | | } |
333 | | |
334 | | impl GzBuilder { |
335 | | /// Create a new blank builder with no header by default. |
336 | 0 | pub fn new() -> GzBuilder { |
337 | 0 | Self::default() |
338 | 0 | } |
339 | | |
340 | | /// Configure the `mtime` field in the gzip header. |
341 | 0 | pub fn mtime(mut self, mtime: u32) -> GzBuilder { |
342 | 0 | self.mtime = mtime; |
343 | 0 | self |
344 | 0 | } |
345 | | |
346 | | /// Configure the `operating_system` field in the gzip header. |
347 | 0 | pub fn operating_system(mut self, os: u8) -> GzBuilder { |
348 | 0 | self.operating_system = Some(os); |
349 | 0 | self |
350 | 0 | } |
351 | | |
352 | | /// Configure the `extra` field in the gzip header. |
353 | | /// |
354 | | /// # Panics |
355 | | /// |
356 | | /// Panics if `extra` is longer than [`u16::MAX`]. |
357 | 0 | pub fn extra<T: Into<Vec<u8>>>(mut self, extra: T) -> GzBuilder { |
358 | 0 | let extra = extra.into(); |
359 | 0 | assert!( |
360 | 0 | extra.len() <= u16::MAX as usize, |
361 | | "gzip extra field length cannot exceed u16::MAX" |
362 | | ); |
363 | 0 | self.extra = Some(extra); |
364 | 0 | self |
365 | 0 | } |
366 | | |
367 | | /// Configure the `filename` field in the gzip header. |
368 | | /// |
369 | | /// # Panics |
370 | | /// |
371 | | /// Panics if the `filename` slice contains a zero. |
372 | 0 | pub fn filename<T: Into<Vec<u8>>>(mut self, filename: T) -> GzBuilder { |
373 | 0 | self.filename = Some(CString::new(filename.into()).unwrap()); |
374 | 0 | self |
375 | 0 | } |
376 | | |
377 | | /// Configure the `comment` field in the gzip header. |
378 | | /// |
379 | | /// # Panics |
380 | | /// |
381 | | /// Panics if the `comment` slice contains a zero. |
382 | 0 | pub fn comment<T: Into<Vec<u8>>>(mut self, comment: T) -> GzBuilder { |
383 | 0 | self.comment = Some(CString::new(comment.into()).unwrap()); |
384 | 0 | self |
385 | 0 | } |
386 | | |
387 | | /// Consume this builder, creating a writer encoder in the process. |
388 | | /// |
389 | | /// The data written to the returned encoder will be compressed and then |
390 | | /// written out to the supplied parameter `w`. |
391 | 0 | pub fn write<W: Write>(self, w: W, lvl: Compression) -> write::GzEncoder<W> { |
392 | 0 | write::gz_encoder(self.into_header(lvl), w, lvl) |
393 | 0 | } |
394 | | |
395 | | /// Consume this builder, creating a reader encoder in the process. |
396 | | /// |
397 | | /// Data read from the returned encoder will be the compressed version of |
398 | | /// the data read from the given reader. |
399 | 0 | pub fn read<R: Read>(self, r: R, lvl: Compression) -> read::GzEncoder<R> { |
400 | 0 | read::gz_encoder(self.buf_read(BufReader::new(r), lvl)) |
401 | 0 | } |
402 | | |
403 | | /// Consume this builder, creating a reader encoder in the process. |
404 | | /// |
405 | | /// Data read from the returned encoder will be the compressed version of |
406 | | /// the data read from the given reader. |
407 | 0 | pub fn buf_read<R>(self, r: R, lvl: Compression) -> bufread::GzEncoder<R> |
408 | 0 | where |
409 | 0 | R: BufRead, |
410 | | { |
411 | 0 | bufread::gz_encoder(self.into_header(lvl), r, lvl) |
412 | 0 | } |
413 | | |
414 | 0 | fn into_header(self, lvl: Compression) -> Vec<u8> { |
415 | | let GzBuilder { |
416 | 0 | extra, |
417 | 0 | filename, |
418 | 0 | comment, |
419 | 0 | operating_system, |
420 | 0 | mtime, |
421 | 0 | } = self; |
422 | 0 | let mut flg = 0; |
423 | 0 | let mut header = vec![0u8; 10]; |
424 | 0 | if let Some(v) = extra { |
425 | 0 | flg |= FEXTRA; |
426 | 0 | header.extend( |
427 | 0 | (u16::try_from(v.len()).expect( |
428 | 0 | "`extra` can only be created from `extra()` which would have panicked on len > u16::MAX", |
429 | 0 | )) |
430 | 0 | .to_le_bytes(), |
431 | 0 | ); |
432 | 0 | header.extend(v); |
433 | 0 | } |
434 | 0 | if let Some(filename) = filename { |
435 | 0 | flg |= FNAME; |
436 | 0 | header.extend(filename.as_bytes_with_nul().iter().copied()); |
437 | 0 | } |
438 | 0 | if let Some(comment) = comment { |
439 | 0 | flg |= FCOMMENT; |
440 | 0 | header.extend(comment.as_bytes_with_nul().iter().copied()); |
441 | 0 | } |
442 | 0 | header[0] = 0x1f; |
443 | 0 | header[1] = 0x8b; |
444 | 0 | header[2] = 8; |
445 | 0 | header[3] = flg; |
446 | 0 | header[4] = mtime as u8; |
447 | 0 | header[5] = (mtime >> 8) as u8; |
448 | 0 | header[6] = (mtime >> 16) as u8; |
449 | 0 | header[7] = (mtime >> 24) as u8; |
450 | 0 | header[8] = if lvl.0 >= Compression::best().0 { |
451 | 0 | 2 |
452 | 0 | } else if lvl.0 <= Compression::fast().0 { |
453 | 0 | 4 |
454 | | } else { |
455 | 0 | 0 |
456 | | }; |
457 | | |
458 | | // Typically this byte indicates what OS the gz stream was created on, |
459 | | // but in an effort to have cross-platform reproducible streams just |
460 | | // default this value to 255. I'm not sure that if we "correctly" set |
461 | | // this it'd do anything anyway... |
462 | 0 | header[9] = operating_system.unwrap_or(255); |
463 | 0 | header |
464 | 0 | } |
465 | | } |
466 | | |
467 | | #[cfg(test)] |
468 | | mod tests { |
469 | | use crate::io::{Read, Write}; |
470 | | use alloc::string::{String, ToString}; |
471 | | use alloc::vec::Vec; |
472 | | |
473 | | use super::{read, write, GzBuilder, GzHeaderParser}; |
474 | | use crate::{Compression, GzHeader}; |
475 | | use rand::{rng, Rng}; |
476 | | |
477 | | #[test] |
478 | | fn roundtrip() { |
479 | | let mut e = write::GzEncoder::new(Vec::new(), Compression::default()); |
480 | | e.write_all(b"foo bar baz").unwrap(); |
481 | | let inner = e.finish().unwrap(); |
482 | | let mut d = read::GzDecoder::new(&inner[..]); |
483 | | let mut s = String::new(); |
484 | | d.read_to_string(&mut s).unwrap(); |
485 | | assert_eq!(s, "foo bar baz"); |
486 | | } |
487 | | |
488 | | #[test] |
489 | | fn roundtrip_zero() { |
490 | | let e = write::GzEncoder::new(Vec::new(), Compression::default()); |
491 | | let inner = e.finish().unwrap(); |
492 | | let mut d = read::GzDecoder::new(&inner[..]); |
493 | | let mut s = String::new(); |
494 | | d.read_to_string(&mut s).unwrap(); |
495 | | assert_eq!(s, ""); |
496 | | } |
497 | | |
498 | | #[test] |
499 | | fn roundtrip_big() { |
500 | | let mut real = Vec::new(); |
501 | | let mut w = write::GzEncoder::new(Vec::new(), Compression::default()); |
502 | | let v = crate::random_bytes().take(1024).collect::<Vec<_>>(); |
503 | | for _ in 0..200 { |
504 | | let to_write = &v[..rng().random_range(0..v.len())]; |
505 | | real.extend(to_write.iter().copied()); |
506 | | w.write_all(to_write).unwrap(); |
507 | | } |
508 | | let result = w.finish().unwrap(); |
509 | | let mut r = read::GzDecoder::new(&result[..]); |
510 | | let mut v = Vec::new(); |
511 | | r.read_to_end(&mut v).unwrap(); |
512 | | assert_eq!(v, real); |
513 | | } |
514 | | |
515 | | #[test] |
516 | | fn roundtrip_big2() { |
517 | | let v = crate::random_bytes().take(1024 * 1024).collect::<Vec<_>>(); |
518 | | let mut r = read::GzDecoder::new(read::GzEncoder::new(&v[..], Compression::default())); |
519 | | let mut res = Vec::new(); |
520 | | r.read_to_end(&mut res).unwrap(); |
521 | | assert_eq!(res, v); |
522 | | } |
523 | | |
524 | | // A Rust implementation of CRC that closely matches the C code in RFC1952. |
525 | | // Only use this to create CRCs for tests. |
526 | | struct Rfc1952Crc { |
527 | | /* Table of CRCs of all 8-bit messages. */ |
528 | | crc_table: [u32; 256], |
529 | | } |
530 | | |
531 | | impl Rfc1952Crc { |
532 | | fn new() -> Self { |
533 | | let mut crc = Rfc1952Crc { |
534 | | crc_table: [0; 256], |
535 | | }; |
536 | | /* Make the table for a fast CRC. */ |
537 | | for n in 0usize..256 { |
538 | | let mut c = n as u32; |
539 | | for _k in 0..8 { |
540 | | if c & 1 != 0 { |
541 | | c = 0xedb88320 ^ (c >> 1); |
542 | | } else { |
543 | | c >>= 1; |
544 | | } |
545 | | } |
546 | | crc.crc_table[n] = c; |
547 | | } |
548 | | crc |
549 | | } |
550 | | |
551 | | /* |
552 | | Update a running crc with the bytes buf and return |
553 | | the updated crc. The crc should be initialized to zero. Pre- and |
554 | | post-conditioning (one's complement) is performed within this |
555 | | function so it shouldn't be done by the caller. |
556 | | */ |
557 | | fn update_crc(&self, crc: u32, buf: &[u8]) -> u32 { |
558 | | let mut c = crc ^ 0xffffffff; |
559 | | |
560 | | for b in buf { |
561 | | c = self.crc_table[(c as u8 ^ *b) as usize] ^ (c >> 8); |
562 | | } |
563 | | c ^ 0xffffffff |
564 | | } |
565 | | |
566 | | /* Return the CRC of the bytes buf. */ |
567 | | fn crc(&self, buf: &[u8]) -> u32 { |
568 | | self.update_crc(0, buf) |
569 | | } |
570 | | } |
571 | | |
572 | | #[test] |
573 | | fn roundtrip_header() { |
574 | | let mut header = GzBuilder::new() |
575 | | .mtime(1234) |
576 | | .operating_system(57) |
577 | | .filename("filename") |
578 | | .comment("comment") |
579 | | .into_header(Compression::fast()); |
580 | | |
581 | | // Add a CRC to the header |
582 | | header[3] ^= super::FHCRC; |
583 | | let rfc1952_crc = Rfc1952Crc::new(); |
584 | | let crc32 = rfc1952_crc.crc(&header); |
585 | | let crc16 = crc32 as u16; |
586 | | header.extend(&crc16.to_le_bytes()); |
587 | | |
588 | | let mut parser = GzHeaderParser::new(); |
589 | | parser.parse(&mut header.as_slice()).unwrap(); |
590 | | let actual = parser.header().unwrap(); |
591 | | assert_eq!( |
592 | | actual, |
593 | | &GzHeader { |
594 | | extra: None, |
595 | | filename: Some("filename".as_bytes().to_vec()), |
596 | | comment: Some("comment".as_bytes().to_vec()), |
597 | | operating_system: 57, |
598 | | mtime: 1234 |
599 | | } |
600 | | ) |
601 | | } |
602 | | |
603 | | #[test] |
604 | | fn gzip_encoder_matches_rfc1952() { |
605 | | /// Extract CRC32 and ISIZE from gzip footer |
606 | | fn extract_zip_footer(compressed: &[u8]) -> (u32, u32) { |
607 | | assert!(compressed.len() >= 8, "Gzip output too short"); |
608 | | let footer_start = compressed.len() - 8; |
609 | | |
610 | | let crc = u32::from_le_bytes([ |
611 | | compressed[footer_start], |
612 | | compressed[footer_start + 1], |
613 | | compressed[footer_start + 2], |
614 | | compressed[footer_start + 3], |
615 | | ]); |
616 | | |
617 | | let size = u32::from_le_bytes([ |
618 | | compressed[footer_start + 4], |
619 | | compressed[footer_start + 5], |
620 | | compressed[footer_start + 6], |
621 | | compressed[footer_start + 7], |
622 | | ]); |
623 | | |
624 | | (crc, size) |
625 | | } |
626 | | |
627 | | #[track_caller] |
628 | | fn test_crc_for_write(data: &[u8], expected_crc: u32, description: &str) { |
629 | | // Compress data using write::GzEncoder |
630 | | let mut encoder = write::GzEncoder::new(Vec::new(), Compression::default()); |
631 | | encoder.write_all(data).unwrap(); |
632 | | let compressed = encoder.finish().unwrap(); |
633 | | |
634 | | let expected_size = data.len() as u32; |
635 | | let (actual_crc, actual_size) = extract_zip_footer(&compressed); |
636 | | |
637 | | assert_eq!( |
638 | | expected_crc, actual_crc, |
639 | | "CRC32 mismatch for write {}: expected {:#08x}, got {:#08x}", |
640 | | description, expected_crc, actual_crc |
641 | | ); |
642 | | assert_eq!( |
643 | | expected_size, actual_size, |
644 | | "Size mismatch for write {}: expected {}, got {}", |
645 | | description, expected_size, actual_size |
646 | | ); |
647 | | } |
648 | | |
649 | | #[track_caller] |
650 | | fn test_crc_for_read(data: &[u8], expected_crc: u32, description: &str) { |
651 | | // Compress data using read::GzEncoder |
652 | | let data_reader = crate::io::Cursor::new(data); |
653 | | let mut encoder = read::GzEncoder::new(data_reader, Compression::default()); |
654 | | let mut compressed = Vec::new(); |
655 | | encoder.read_to_end(&mut compressed).unwrap(); |
656 | | |
657 | | let expected_size = data.len() as u32; |
658 | | let (actual_crc, actual_size) = extract_zip_footer(&compressed); |
659 | | |
660 | | assert_eq!( |
661 | | expected_crc, actual_crc, |
662 | | "CRC32 mismatch for read {}: expected {:#08x}, got {:#08x}", |
663 | | description, expected_crc, actual_crc |
664 | | ); |
665 | | assert_eq!( |
666 | | expected_size, actual_size, |
667 | | "Size mismatch for read {}: expected {}, got {}", |
668 | | description, expected_size, actual_size |
669 | | ); |
670 | | } |
671 | | |
672 | | #[track_caller] |
673 | | fn test_crc_for_data(data: &[u8], description: &str) { |
674 | | let rfc1952_crc = Rfc1952Crc::new(); |
675 | | let expected_crc = rfc1952_crc.crc(data); |
676 | | |
677 | | test_crc_for_write(data, expected_crc, description); |
678 | | test_crc_for_read(data, expected_crc, description); |
679 | | } |
680 | | |
681 | | // Edge cases |
682 | | test_crc_for_data(&[], "empty data"); |
683 | | test_crc_for_data(&[0x00], "single zero byte"); |
684 | | test_crc_for_data(&[0xFF], "single 0xFF byte"); |
685 | | |
686 | | // Simple text patterns |
687 | | test_crc_for_data(b"Hello World", "simple ASCII"); |
688 | | test_crc_for_data(b"AAAAAAA", "repeated 'A'"); |
689 | | test_crc_for_data(b"1234567890", "digits"); |
690 | | |
691 | | // Binary patterns |
692 | | test_crc_for_data(&[0x00, 0x01, 0x02, 0x03, 0x04, 0x05], "sequential bytes"); |
693 | | test_crc_for_data(&[0xAA, 0x55, 0xAA, 0x55, 0xAA, 0x55], "alternating pattern"); |
694 | | test_crc_for_data(&[0x00; 10], "all zeros"); |
695 | | test_crc_for_data(&[0xFF; 10], "all ones"); |
696 | | |
697 | | // Large data |
698 | | let large_data = vec![0x42; 10240]; |
699 | | test_crc_for_data(&large_data, "10 kiB data"); |
700 | | |
701 | | // Test multi-write scenario to ensure CRC accumulation works correctly |
702 | | { |
703 | | let data = b"This is a test of multi-write CRC accumulation"; |
704 | | let rfc1952_crc = Rfc1952Crc::new(); |
705 | | let expected_crc = rfc1952_crc.crc(data); |
706 | | |
707 | | let mut encoder = write::GzEncoder::new(Vec::new(), Compression::default()); |
708 | | // Write in chunks |
709 | | encoder.write_all(&data[..10]).unwrap(); |
710 | | encoder.write_all(&data[10..20]).unwrap(); |
711 | | encoder.write_all(&data[20..]).unwrap(); |
712 | | let compressed = encoder.finish().unwrap(); |
713 | | |
714 | | let expected_size = data.len() as u32; |
715 | | let (actual_crc, actual_size) = extract_zip_footer(&compressed); |
716 | | |
717 | | assert_eq!( |
718 | | expected_crc, actual_crc, |
719 | | "Multi-write CRC mismatch: expected {:#08x}, got {:#08x}", |
720 | | expected_crc, actual_crc |
721 | | ); |
722 | | assert_eq!( |
723 | | expected_size, actual_size, |
724 | | "Size mismatch for multi-write: expected {}, got {}", |
725 | | expected_size, actual_size |
726 | | ); |
727 | | } |
728 | | } |
729 | | |
730 | | fn gzip_corrupted_crc() -> Vec<u8> { |
731 | | let test_data = b"The quick brown fox jumps over the lazy dog"; |
732 | | |
733 | | let mut encoder = write::GzEncoder::new(Vec::new(), Compression::default()); |
734 | | encoder.write_all(test_data).unwrap(); |
735 | | let mut compressed = encoder.finish().unwrap(); |
736 | | |
737 | | // Corrupt the CRC32 in the footer |
738 | | let crc_offset = compressed.len() - 8; |
739 | | compressed[crc_offset] ^= 0xFF; |
740 | | |
741 | | compressed |
742 | | } |
743 | | |
744 | | #[test] |
745 | | fn read_decoder_detects_corrupted_crc() { |
746 | | let compressed = gzip_corrupted_crc(); |
747 | | let mut decoder = read::GzDecoder::new(&compressed[..]); |
748 | | let mut output = Vec::new(); |
749 | | let error = decoder.read_to_end(&mut output).unwrap_err(); |
750 | | assert_eq!(error.kind(), crate::io::ErrorKind::InvalidInput); |
751 | | } |
752 | | |
753 | | #[test] |
754 | | fn read_decoder_rejects_incomplete_deflate_stream() { |
755 | | let mut compressed = gzip_corrupted_crc(); |
756 | | compressed.truncate(11); |
757 | | let error = read::GzDecoder::new(&compressed[..]) |
758 | | .read_to_end(&mut Vec::new()) |
759 | | .unwrap_err(); |
760 | | assert_eq!(error.kind(), crate::io::ErrorKind::UnexpectedEof); |
761 | | assert_eq!(error.to_string(), "incomplete deflate stream"); |
762 | | } |
763 | | |
764 | | #[test] |
765 | | fn write_decoder_detects_corrupted_crc() { |
766 | | let compressed = gzip_corrupted_crc(); |
767 | | let mut decoder = write::GzDecoder::new(Vec::new()); |
768 | | decoder.write_all(&compressed).unwrap(); |
769 | | let error = decoder.finish().unwrap_err(); |
770 | | assert_eq!(error.kind(), crate::io::ErrorKind::InvalidInput); |
771 | | } |
772 | | |
773 | | #[test] |
774 | | fn fields() { |
775 | | let r = [0, 2, 4, 6]; |
776 | | let e = GzBuilder::new() |
777 | | .filename("foo.rs") |
778 | | .comment("bar") |
779 | | .extra(vec![0, 1, 2, 3]) |
780 | | .read(&r[..], Compression::default()); |
781 | | let mut d = read::GzDecoder::new(e); |
782 | | assert_eq!(d.header().unwrap().filename(), Some(&b"foo.rs"[..])); |
783 | | assert_eq!(d.header().unwrap().comment(), Some(&b"bar"[..])); |
784 | | assert_eq!(d.header().unwrap().extra(), Some(&b"\x00\x01\x02\x03"[..])); |
785 | | let mut res = Vec::new(); |
786 | | d.read_to_end(&mut res).unwrap(); |
787 | | assert_eq!(res, vec![0, 2, 4, 6]); |
788 | | } |
789 | | |
790 | | #[test] |
791 | | #[should_panic(expected = "gzip extra field length cannot exceed u16::MAX")] |
792 | | fn extra_too_long() { |
793 | | GzBuilder::new().extra(vec![0; u16::MAX as usize + 1]); |
794 | | } |
795 | | |
796 | | #[test] |
797 | | fn keep_reading_after_end() { |
798 | | let mut e = write::GzEncoder::new(Vec::new(), Compression::default()); |
799 | | e.write_all(b"foo bar baz").unwrap(); |
800 | | let inner = e.finish().unwrap(); |
801 | | let mut d = read::GzDecoder::new(&inner[..]); |
802 | | let mut s = String::new(); |
803 | | d.read_to_string(&mut s).unwrap(); |
804 | | assert_eq!(s, "foo bar baz"); |
805 | | d.read_to_string(&mut s).unwrap(); |
806 | | assert_eq!(s, "foo bar baz"); |
807 | | } |
808 | | |
809 | | #[test] |
810 | | fn qc_reader() { |
811 | | ::quickcheck::quickcheck(test as fn(_) -> _); |
812 | | |
813 | | fn test(v: Vec<u8>) -> bool { |
814 | | let r = read::GzEncoder::new(&v[..], Compression::default()); |
815 | | let mut r = read::GzDecoder::new(r); |
816 | | let mut v2 = Vec::new(); |
817 | | r.read_to_end(&mut v2).unwrap(); |
818 | | v == v2 |
819 | | } |
820 | | } |
821 | | |
822 | | #[test] |
823 | | fn flush_after_write() { |
824 | | let mut f = write::GzEncoder::new(Vec::new(), Compression::default()); |
825 | | write!(f, "Hello world").unwrap(); |
826 | | f.flush().unwrap(); |
827 | | } |
828 | | } |