Skip to main content

flate2/gz/
mod.rs

1use crate::io::{BufRead, Error, ErrorKind, Read, Result, Write};
2use alloc::boxed::Box;
3use alloc::ffi::CString;
4use alloc::vec::Vec;
5use core::convert::TryFrom;
6use core::time;
7
8use crate::bufreader::BufReader;
9use crate::{Compression, Crc};
10
11pub static FHCRC: u8 = 1 << 1;
12pub static FEXTRA: u8 = 1 << 2;
13pub static FNAME: u8 = 1 << 3;
14pub static FCOMMENT: u8 = 1 << 4;
15pub static FRESERVED: u8 = 1 << 5 | 1 << 6 | 1 << 7;
16
17pub mod bufread;
18pub mod read;
19pub mod write;
20
21// The maximum length of the header filename and comment fields. More than
22// enough for these fields in reasonable use, but prevents possible attacks.
23const MAX_HEADER_BUF: usize = 65535;
24
25/// A structure representing the header of a gzip stream.
26///
27/// The header can contain metadata about the file that was compressed, if
28/// present.
29#[derive(PartialEq, Clone, Debug, Default)]
30pub struct GzHeader {
31    extra: Option<Vec<u8>>,
32    filename: Option<Vec<u8>>,
33    comment: Option<Vec<u8>>,
34    operating_system: u8,
35    mtime: u32,
36}
37
38impl GzHeader {
39    /// Returns the `filename` field of this gzip stream's header, if present.
40    pub fn filename(&self) -> Option<&[u8]> {
41        self.filename.as_ref().map(|s| &s[..])
42    }
43
44    /// Returns the `extra` field of this gzip stream's header, if present.
45    pub fn extra(&self) -> Option<&[u8]> {
46        self.extra.as_ref().map(|s| &s[..])
47    }
48
49    /// Returns the `comment` field of this gzip stream's header, if present.
50    pub fn comment(&self) -> Option<&[u8]> {
51        self.comment.as_ref().map(|s| &s[..])
52    }
53
54    /// Returns the `operating_system` field of this gzip stream's header.
55    ///
56    /// There are predefined values for various operating systems.
57    /// 255 means that the value is unknown.
58    pub fn operating_system(&self) -> u8 {
59        self.operating_system
60    }
61
62    /// This gives the most recent modification time of the original file being compressed.
63    ///
64    /// The time is in Unix format, i.e., seconds since 00:00:00 GMT, Jan. 1, 1970.
65    /// (Note that this may cause problems for MS-DOS and other systems that use local
66    /// rather than Universal time.) If the compressed data did not come from a file,
67    /// `mtime` is set to the time at which compression started.
68    /// `mtime` = 0 means no time stamp is available.
69    ///
70    /// The usage of `mtime` is discouraged because of Year 2038 problem.
71    pub fn mtime(&self) -> u32 {
72        self.mtime
73    }
74
75    /// Returns the most recent modification time represented by a date-time type.
76    /// Returns `None` if the value of the underlying counter is 0,
77    /// indicating no time stamp is available.
78    ///
79    ///
80    /// The time is measured as seconds since 00:00:00 GMT, Jan. 1 1970.
81    /// See [`mtime`](#method.mtime) for more detail.
82    #[cfg(not(flate2_unstable_nightly_alloc_io))]
83    pub fn mtime_as_datetime(&self) -> Option<std::time::SystemTime> {
84        self.mtime_as_duration().map(|d| std::time::UNIX_EPOCH + d)
85    }
86
87    /// Returns the [`Duration`](time::Duration) between the most recent modification
88    /// time and 00:00:00 GMT, Jan. 1 1970, also known as Unix epoch.
89    /// See [`mtime`](#method.mtime) for more detail.
90    /// Returns `None` if the value of the underlying counter is 0,
91    /// indicating no time stamp is available.
92    pub fn mtime_as_duration(&self) -> Option<time::Duration> {
93        if self.mtime == 0 {
94            None
95        } else {
96            let duration = time::Duration::new(u64::from(self.mtime), 0);
97            Some(duration)
98        }
99    }
100}
101
102#[derive(Debug, Default)]
103pub enum GzHeaderState {
104    Start(u8, [u8; 10]),
105    Xlen(Option<Box<Crc>>, u8, [u8; 2]),
106    Extra(Option<Box<Crc>>, u16),
107    Filename(Option<Box<Crc>>),
108    Comment(Option<Box<Crc>>),
109    Crc(Option<Box<Crc>>, u8, [u8; 2]),
110    #[default]
111    Complete,
112}
113
114#[derive(Debug, Default)]
115pub struct GzHeaderParser {
116    state: GzHeaderState,
117    flags: u8,
118    header: GzHeader,
119}
120
121impl GzHeaderParser {
122    fn new() -> Self {
123        GzHeaderParser {
124            state: GzHeaderState::Start(0, [0; 10]),
125            flags: 0,
126            header: GzHeader::default(),
127        }
128    }
129
130    fn parse<R: BufRead>(&mut self, r: &mut R) -> Result<()> {
131        loop {
132            match &mut self.state {
133                GzHeaderState::Start(count, buffer) => {
134                    while (*count as usize) < buffer.len() {
135                        *count += read_into(r, &mut buffer[*count as usize..])? as u8;
136                    }
137                    // Gzip identification bytes
138                    if buffer[0] != 0x1f || buffer[1] != 0x8b {
139                        return Err(bad_header());
140                    }
141                    // Gzip compression method (8 = deflate)
142                    if buffer[2] != 8 {
143                        return Err(bad_header());
144                    }
145                    self.flags = buffer[3];
146                    // RFC1952: "must give an error indication if any reserved bit is non-zero"
147                    if self.flags & FRESERVED != 0 {
148                        return Err(bad_header());
149                    }
150                    self.header.mtime = (buffer[4] as u32)
151                        | ((buffer[5] as u32) << 8)
152                        | ((buffer[6] as u32) << 16)
153                        | ((buffer[7] as u32) << 24);
154                    let _xfl = buffer[8];
155                    self.header.operating_system = buffer[9];
156                    let crc = if self.flags & FHCRC != 0 {
157                        let mut crc = Box::new(Crc::new());
158                        crc.update(buffer);
159                        Some(crc)
160                    } else {
161                        None
162                    };
163                    self.state = GzHeaderState::Xlen(crc, 0, [0; 2]);
164                }
165                GzHeaderState::Xlen(crc, count, buffer) => {
166                    if self.flags & FEXTRA != 0 {
167                        while (*count as usize) < buffer.len() {
168                            *count += read_into(r, &mut buffer[*count as usize..])? as u8;
169                        }
170                        if let Some(crc) = crc {
171                            crc.update(buffer);
172                        }
173                        let xlen = parse_le_u16(buffer);
174                        self.header.extra = Some(vec![0; xlen as usize]);
175                        self.state = GzHeaderState::Extra(crc.take(), 0);
176                    } else {
177                        self.state = GzHeaderState::Filename(crc.take());
178                    }
179                }
180                GzHeaderState::Extra(crc, count) => {
181                    debug_assert!(self.header.extra.is_some());
182                    let extra = self.header.extra.as_mut().unwrap();
183                    while (*count as usize) < extra.len() {
184                        *count += read_into(r, &mut extra[*count as usize..])? as u16;
185                    }
186                    if let Some(crc) = crc {
187                        crc.update(extra);
188                    }
189                    self.state = GzHeaderState::Filename(crc.take());
190                }
191                GzHeaderState::Filename(crc) => {
192                    if self.flags & FNAME != 0 {
193                        let filename = self.header.filename.get_or_insert_with(Vec::new);
194                        read_to_nul(r, filename)?;
195                        if let Some(crc) = crc {
196                            crc.update(filename);
197                            crc.update(b"\0");
198                        }
199                    }
200                    self.state = GzHeaderState::Comment(crc.take());
201                }
202                GzHeaderState::Comment(crc) => {
203                    if self.flags & FCOMMENT != 0 {
204                        let comment = self.header.comment.get_or_insert_with(Vec::new);
205                        read_to_nul(r, comment)?;
206                        if let Some(crc) = crc {
207                            crc.update(comment);
208                            crc.update(b"\0");
209                        }
210                    }
211                    self.state = GzHeaderState::Crc(crc.take(), 0, [0; 2]);
212                }
213                GzHeaderState::Crc(crc, count, buffer) => {
214                    if let Some(crc) = crc {
215                        debug_assert!(self.flags & FHCRC != 0);
216                        while (*count as usize) < buffer.len() {
217                            *count += read_into(r, &mut buffer[*count as usize..])? as u8;
218                        }
219                        let stored_crc = parse_le_u16(buffer);
220                        let calced_crc = crc.sum() as u16;
221                        if stored_crc != calced_crc {
222                            return Err(corrupt());
223                        }
224                    }
225                    self.state = GzHeaderState::Complete;
226                }
227                GzHeaderState::Complete => {
228                    return Ok(());
229                }
230            }
231        }
232    }
233
234    fn header(&self) -> Option<&GzHeader> {
235        match self.state {
236            GzHeaderState::Complete => Some(&self.header),
237            _ => None,
238        }
239    }
240}
241
242impl From<GzHeaderParser> for GzHeader {
243    fn from(parser: GzHeaderParser) -> Self {
244        debug_assert!(matches!(parser.state, GzHeaderState::Complete));
245        parser.header
246    }
247}
248
249// Attempt to fill the `buffer` from `r`. Return the number of bytes read.
250// Return an error if EOF is read before the buffer is full.  This differs
251// from `read` in that Ok(0) means that more data may be available.
252fn read_into<R: Read>(r: &mut R, buffer: &mut [u8]) -> Result<usize> {
253    debug_assert!(!buffer.is_empty());
254    match r.read(buffer) {
255        Ok(0) => Err(ErrorKind::UnexpectedEof.into()),
256        Ok(n) => Ok(n),
257        Err(ref e) if e.kind() == ErrorKind::Interrupted => Ok(0),
258        Err(e) => Err(e),
259    }
260}
261
262// Read `r` up to the first nul byte, pushing non-nul bytes to `buffer`.
263fn read_to_nul<R: BufRead>(r: &mut R, buffer: &mut Vec<u8>) -> Result<()> {
264    let mut bytes = r.bytes();
265    loop {
266        match bytes.next().transpose()? {
267            Some(0) => return Ok(()),
268            Some(_) if buffer.len() == MAX_HEADER_BUF => {
269                return Err(Error::new(
270                    ErrorKind::InvalidInput,
271                    "gzip header field too long",
272                ));
273            }
274            Some(byte) => {
275                buffer.push(byte);
276            }
277            None => {
278                return Err(ErrorKind::UnexpectedEof.into());
279            }
280        }
281    }
282}
283
284fn parse_le_u16(buffer: &[u8; 2]) -> u16 {
285    u16::from_le_bytes(*buffer)
286}
287
288fn bad_header() -> Error {
289    Error::new(ErrorKind::InvalidInput, "invalid gzip header")
290}
291
292fn corrupt() -> Error {
293    Error::new(
294        ErrorKind::InvalidInput,
295        "corrupt gzip stream does not have a matching checksum",
296    )
297}
298
299/// A builder structure to create a new gzip Encoder.
300///
301/// This structure controls header configuration options such as the filename.
302///
303/// # Examples
304///
305/// ```
306/// use std::io::prelude::*;
307/// # use std::io;
308/// use std::fs::File;
309/// use flate2::GzBuilder;
310/// use flate2::Compression;
311///
312/// // GzBuilder opens a file and writes a sample string using GzBuilder pattern
313///
314/// # fn sample_builder() -> Result<(), io::Error> {
315/// let f = File::create("examples/hello_world.gz")?;
316/// let mut gz = GzBuilder::new()
317///                 .filename("hello_world.txt")
318///                 .comment("test file, please delete")
319///                 .write(f, Compression::default());
320/// gz.write_all(b"hello world")?;
321/// gz.finish()?;
322/// # Ok(())
323/// # }
324/// ```
325#[derive(Debug, Default)]
326pub struct GzBuilder {
327    extra: Option<Vec<u8>>,
328    filename: Option<CString>,
329    comment: Option<CString>,
330    operating_system: Option<u8>,
331    mtime: u32,
332}
333
334impl GzBuilder {
335    /// Create a new blank builder with no header by default.
336    pub fn new() -> GzBuilder {
337        Self::default()
338    }
339
340    /// Configure the `mtime` field in the gzip header.
341    pub fn mtime(mut self, mtime: u32) -> GzBuilder {
342        self.mtime = mtime;
343        self
344    }
345
346    /// Configure the `operating_system` field in the gzip header.
347    pub fn operating_system(mut self, os: u8) -> GzBuilder {
348        self.operating_system = Some(os);
349        self
350    }
351
352    /// Configure the `extra` field in the gzip header.
353    ///
354    /// # Panics
355    ///
356    /// Panics if `extra` is longer than [`u16::MAX`].
357    pub fn extra<T: Into<Vec<u8>>>(mut self, extra: T) -> GzBuilder {
358        let extra = extra.into();
359        assert!(
360            extra.len() <= u16::MAX as usize,
361            "gzip extra field length cannot exceed u16::MAX"
362        );
363        self.extra = Some(extra);
364        self
365    }
366
367    /// Configure the `filename` field in the gzip header.
368    ///
369    /// # Panics
370    ///
371    /// Panics if the `filename` slice contains a zero.
372    pub fn filename<T: Into<Vec<u8>>>(mut self, filename: T) -> GzBuilder {
373        self.filename = Some(CString::new(filename.into()).unwrap());
374        self
375    }
376
377    /// Configure the `comment` field in the gzip header.
378    ///
379    /// # Panics
380    ///
381    /// Panics if the `comment` slice contains a zero.
382    pub fn comment<T: Into<Vec<u8>>>(mut self, comment: T) -> GzBuilder {
383        self.comment = Some(CString::new(comment.into()).unwrap());
384        self
385    }
386
387    /// Consume this builder, creating a writer encoder in the process.
388    ///
389    /// The data written to the returned encoder will be compressed and then
390    /// written out to the supplied parameter `w`.
391    pub fn write<W: Write>(self, w: W, lvl: Compression) -> write::GzEncoder<W> {
392        write::gz_encoder(self.into_header(lvl), w, lvl)
393    }
394
395    /// Consume this builder, creating a reader encoder in the process.
396    ///
397    /// Data read from the returned encoder will be the compressed version of
398    /// the data read from the given reader.
399    pub fn read<R: Read>(self, r: R, lvl: Compression) -> read::GzEncoder<R> {
400        read::gz_encoder(self.buf_read(BufReader::new(r), lvl))
401    }
402
403    /// Consume this builder, creating a reader encoder in the process.
404    ///
405    /// Data read from the returned encoder will be the compressed version of
406    /// the data read from the given reader.
407    pub fn buf_read<R>(self, r: R, lvl: Compression) -> bufread::GzEncoder<R>
408    where
409        R: BufRead,
410    {
411        bufread::gz_encoder(self.into_header(lvl), r, lvl)
412    }
413
414    fn into_header(self, lvl: Compression) -> Vec<u8> {
415        let GzBuilder {
416            extra,
417            filename,
418            comment,
419            operating_system,
420            mtime,
421        } = self;
422        let mut flg = 0;
423        let mut header = vec![0u8; 10];
424        if let Some(v) = extra {
425            flg |= FEXTRA;
426            header.extend(
427                (u16::try_from(v.len()).expect(
428                    "`extra` can only be created from `extra()` which would have panicked on len > u16::MAX",
429                ))
430                .to_le_bytes(),
431            );
432            header.extend(v);
433        }
434        if let Some(filename) = filename {
435            flg |= FNAME;
436            header.extend(filename.as_bytes_with_nul().iter().copied());
437        }
438        if let Some(comment) = comment {
439            flg |= FCOMMENT;
440            header.extend(comment.as_bytes_with_nul().iter().copied());
441        }
442        header[0] = 0x1f;
443        header[1] = 0x8b;
444        header[2] = 8;
445        header[3] = flg;
446        header[4] = mtime as u8;
447        header[5] = (mtime >> 8) as u8;
448        header[6] = (mtime >> 16) as u8;
449        header[7] = (mtime >> 24) as u8;
450        header[8] = if lvl.0 >= Compression::best().0 {
451            2
452        } else if lvl.0 <= Compression::fast().0 {
453            4
454        } else {
455            0
456        };
457
458        // Typically this byte indicates what OS the gz stream was created on,
459        // but in an effort to have cross-platform reproducible streams just
460        // default this value to 255. I'm not sure that if we "correctly" set
461        // this it'd do anything anyway...
462        header[9] = operating_system.unwrap_or(255);
463        header
464    }
465}
466
467#[cfg(test)]
468mod tests {
469    use crate::io::{Read, Write};
470    use alloc::string::{String, ToString};
471    use alloc::vec::Vec;
472
473    use super::{read, write, GzBuilder, GzHeaderParser};
474    use crate::{Compression, GzHeader};
475    use rand::{rng, Rng};
476
477    #[test]
478    fn roundtrip() {
479        let mut e = write::GzEncoder::new(Vec::new(), Compression::default());
480        e.write_all(b"foo bar baz").unwrap();
481        let inner = e.finish().unwrap();
482        let mut d = read::GzDecoder::new(&inner[..]);
483        let mut s = String::new();
484        d.read_to_string(&mut s).unwrap();
485        assert_eq!(s, "foo bar baz");
486    }
487
488    #[test]
489    fn roundtrip_zero() {
490        let e = write::GzEncoder::new(Vec::new(), Compression::default());
491        let inner = e.finish().unwrap();
492        let mut d = read::GzDecoder::new(&inner[..]);
493        let mut s = String::new();
494        d.read_to_string(&mut s).unwrap();
495        assert_eq!(s, "");
496    }
497
498    #[test]
499    fn roundtrip_big() {
500        let mut real = Vec::new();
501        let mut w = write::GzEncoder::new(Vec::new(), Compression::default());
502        let v = crate::random_bytes().take(1024).collect::<Vec<_>>();
503        for _ in 0..200 {
504            let to_write = &v[..rng().random_range(0..v.len())];
505            real.extend(to_write.iter().copied());
506            w.write_all(to_write).unwrap();
507        }
508        let result = w.finish().unwrap();
509        let mut r = read::GzDecoder::new(&result[..]);
510        let mut v = Vec::new();
511        r.read_to_end(&mut v).unwrap();
512        assert_eq!(v, real);
513    }
514
515    #[test]
516    fn roundtrip_big2() {
517        let v = crate::random_bytes().take(1024 * 1024).collect::<Vec<_>>();
518        let mut r = read::GzDecoder::new(read::GzEncoder::new(&v[..], Compression::default()));
519        let mut res = Vec::new();
520        r.read_to_end(&mut res).unwrap();
521        assert_eq!(res, v);
522    }
523
524    // A Rust implementation of CRC that closely matches the C code in RFC1952.
525    // Only use this to create CRCs for tests.
526    struct Rfc1952Crc {
527        /* Table of CRCs of all 8-bit messages. */
528        crc_table: [u32; 256],
529    }
530
531    impl Rfc1952Crc {
532        fn new() -> Self {
533            let mut crc = Rfc1952Crc {
534                crc_table: [0; 256],
535            };
536            /* Make the table for a fast CRC. */
537            for n in 0usize..256 {
538                let mut c = n as u32;
539                for _k in 0..8 {
540                    if c & 1 != 0 {
541                        c = 0xedb88320 ^ (c >> 1);
542                    } else {
543                        c >>= 1;
544                    }
545                }
546                crc.crc_table[n] = c;
547            }
548            crc
549        }
550
551        /*
552         Update a running crc with the bytes buf and return
553         the updated crc. The crc should be initialized to zero. Pre- and
554         post-conditioning (one's complement) is performed within this
555         function so it shouldn't be done by the caller.
556        */
557        fn update_crc(&self, crc: u32, buf: &[u8]) -> u32 {
558            let mut c = crc ^ 0xffffffff;
559
560            for b in buf {
561                c = self.crc_table[(c as u8 ^ *b) as usize] ^ (c >> 8);
562            }
563            c ^ 0xffffffff
564        }
565
566        /* Return the CRC of the bytes buf. */
567        fn crc(&self, buf: &[u8]) -> u32 {
568            self.update_crc(0, buf)
569        }
570    }
571
572    #[test]
573    fn roundtrip_header() {
574        let mut header = GzBuilder::new()
575            .mtime(1234)
576            .operating_system(57)
577            .filename("filename")
578            .comment("comment")
579            .into_header(Compression::fast());
580
581        // Add a CRC to the header
582        header[3] ^= super::FHCRC;
583        let rfc1952_crc = Rfc1952Crc::new();
584        let crc32 = rfc1952_crc.crc(&header);
585        let crc16 = crc32 as u16;
586        header.extend(&crc16.to_le_bytes());
587
588        let mut parser = GzHeaderParser::new();
589        parser.parse(&mut header.as_slice()).unwrap();
590        let actual = parser.header().unwrap();
591        assert_eq!(
592            actual,
593            &GzHeader {
594                extra: None,
595                filename: Some("filename".as_bytes().to_vec()),
596                comment: Some("comment".as_bytes().to_vec()),
597                operating_system: 57,
598                mtime: 1234
599            }
600        )
601    }
602
603    #[test]
604    fn gzip_encoder_matches_rfc1952() {
605        /// Extract CRC32 and ISIZE from gzip footer
606        fn extract_zip_footer(compressed: &[u8]) -> (u32, u32) {
607            assert!(compressed.len() >= 8, "Gzip output too short");
608            let footer_start = compressed.len() - 8;
609
610            let crc = u32::from_le_bytes([
611                compressed[footer_start],
612                compressed[footer_start + 1],
613                compressed[footer_start + 2],
614                compressed[footer_start + 3],
615            ]);
616
617            let size = u32::from_le_bytes([
618                compressed[footer_start + 4],
619                compressed[footer_start + 5],
620                compressed[footer_start + 6],
621                compressed[footer_start + 7],
622            ]);
623
624            (crc, size)
625        }
626
627        #[track_caller]
628        fn test_crc_for_write(data: &[u8], expected_crc: u32, description: &str) {
629            // Compress data using write::GzEncoder
630            let mut encoder = write::GzEncoder::new(Vec::new(), Compression::default());
631            encoder.write_all(data).unwrap();
632            let compressed = encoder.finish().unwrap();
633
634            let expected_size = data.len() as u32;
635            let (actual_crc, actual_size) = extract_zip_footer(&compressed);
636
637            assert_eq!(
638                expected_crc, actual_crc,
639                "CRC32 mismatch for write {}: expected {:#08x}, got {:#08x}",
640                description, expected_crc, actual_crc
641            );
642            assert_eq!(
643                expected_size, actual_size,
644                "Size mismatch for write {}: expected {}, got {}",
645                description, expected_size, actual_size
646            );
647        }
648
649        #[track_caller]
650        fn test_crc_for_read(data: &[u8], expected_crc: u32, description: &str) {
651            // Compress data using read::GzEncoder
652            let data_reader = crate::io::Cursor::new(data);
653            let mut encoder = read::GzEncoder::new(data_reader, Compression::default());
654            let mut compressed = Vec::new();
655            encoder.read_to_end(&mut compressed).unwrap();
656
657            let expected_size = data.len() as u32;
658            let (actual_crc, actual_size) = extract_zip_footer(&compressed);
659
660            assert_eq!(
661                expected_crc, actual_crc,
662                "CRC32 mismatch for read {}: expected {:#08x}, got {:#08x}",
663                description, expected_crc, actual_crc
664            );
665            assert_eq!(
666                expected_size, actual_size,
667                "Size mismatch for read {}: expected {}, got {}",
668                description, expected_size, actual_size
669            );
670        }
671
672        #[track_caller]
673        fn test_crc_for_data(data: &[u8], description: &str) {
674            let rfc1952_crc = Rfc1952Crc::new();
675            let expected_crc = rfc1952_crc.crc(data);
676
677            test_crc_for_write(data, expected_crc, description);
678            test_crc_for_read(data, expected_crc, description);
679        }
680
681        // Edge cases
682        test_crc_for_data(&[], "empty data");
683        test_crc_for_data(&[0x00], "single zero byte");
684        test_crc_for_data(&[0xFF], "single 0xFF byte");
685
686        // Simple text patterns
687        test_crc_for_data(b"Hello World", "simple ASCII");
688        test_crc_for_data(b"AAAAAAA", "repeated 'A'");
689        test_crc_for_data(b"1234567890", "digits");
690
691        // Binary patterns
692        test_crc_for_data(&[0x00, 0x01, 0x02, 0x03, 0x04, 0x05], "sequential bytes");
693        test_crc_for_data(&[0xAA, 0x55, 0xAA, 0x55, 0xAA, 0x55], "alternating pattern");
694        test_crc_for_data(&[0x00; 10], "all zeros");
695        test_crc_for_data(&[0xFF; 10], "all ones");
696
697        // Large data
698        let large_data = vec![0x42; 10240];
699        test_crc_for_data(&large_data, "10 kiB data");
700
701        // Test multi-write scenario to ensure CRC accumulation works correctly
702        {
703            let data = b"This is a test of multi-write CRC accumulation";
704            let rfc1952_crc = Rfc1952Crc::new();
705            let expected_crc = rfc1952_crc.crc(data);
706
707            let mut encoder = write::GzEncoder::new(Vec::new(), Compression::default());
708            // Write in chunks
709            encoder.write_all(&data[..10]).unwrap();
710            encoder.write_all(&data[10..20]).unwrap();
711            encoder.write_all(&data[20..]).unwrap();
712            let compressed = encoder.finish().unwrap();
713
714            let expected_size = data.len() as u32;
715            let (actual_crc, actual_size) = extract_zip_footer(&compressed);
716
717            assert_eq!(
718                expected_crc, actual_crc,
719                "Multi-write CRC mismatch: expected {:#08x}, got {:#08x}",
720                expected_crc, actual_crc
721            );
722            assert_eq!(
723                expected_size, actual_size,
724                "Size mismatch for multi-write: expected {}, got {}",
725                expected_size, actual_size
726            );
727        }
728    }
729
730    fn gzip_corrupted_crc() -> Vec<u8> {
731        let test_data = b"The quick brown fox jumps over the lazy dog";
732
733        let mut encoder = write::GzEncoder::new(Vec::new(), Compression::default());
734        encoder.write_all(test_data).unwrap();
735        let mut compressed = encoder.finish().unwrap();
736
737        // Corrupt the CRC32 in the footer
738        let crc_offset = compressed.len() - 8;
739        compressed[crc_offset] ^= 0xFF;
740
741        compressed
742    }
743
744    #[test]
745    fn read_decoder_detects_corrupted_crc() {
746        let compressed = gzip_corrupted_crc();
747        let mut decoder = read::GzDecoder::new(&compressed[..]);
748        let mut output = Vec::new();
749        let error = decoder.read_to_end(&mut output).unwrap_err();
750        assert_eq!(error.kind(), crate::io::ErrorKind::InvalidInput);
751    }
752
753    #[test]
754    fn read_decoder_rejects_incomplete_deflate_stream() {
755        let mut compressed = gzip_corrupted_crc();
756        compressed.truncate(11);
757        let error = read::GzDecoder::new(&compressed[..])
758            .read_to_end(&mut Vec::new())
759            .unwrap_err();
760        assert_eq!(error.kind(), crate::io::ErrorKind::UnexpectedEof);
761        assert_eq!(error.to_string(), "incomplete deflate stream");
762    }
763
764    #[test]
765    fn write_decoder_detects_corrupted_crc() {
766        let compressed = gzip_corrupted_crc();
767        let mut decoder = write::GzDecoder::new(Vec::new());
768        decoder.write_all(&compressed).unwrap();
769        let error = decoder.finish().unwrap_err();
770        assert_eq!(error.kind(), crate::io::ErrorKind::InvalidInput);
771    }
772
773    #[test]
774    fn fields() {
775        let r = [0, 2, 4, 6];
776        let e = GzBuilder::new()
777            .filename("foo.rs")
778            .comment("bar")
779            .extra(vec![0, 1, 2, 3])
780            .read(&r[..], Compression::default());
781        let mut d = read::GzDecoder::new(e);
782        assert_eq!(d.header().unwrap().filename(), Some(&b"foo.rs"[..]));
783        assert_eq!(d.header().unwrap().comment(), Some(&b"bar"[..]));
784        assert_eq!(d.header().unwrap().extra(), Some(&b"\x00\x01\x02\x03"[..]));
785        let mut res = Vec::new();
786        d.read_to_end(&mut res).unwrap();
787        assert_eq!(res, vec![0, 2, 4, 6]);
788    }
789
790    #[test]
791    #[should_panic(expected = "gzip extra field length cannot exceed u16::MAX")]
792    fn extra_too_long() {
793        GzBuilder::new().extra(vec![0; u16::MAX as usize + 1]);
794    }
795
796    #[test]
797    fn keep_reading_after_end() {
798        let mut e = write::GzEncoder::new(Vec::new(), Compression::default());
799        e.write_all(b"foo bar baz").unwrap();
800        let inner = e.finish().unwrap();
801        let mut d = read::GzDecoder::new(&inner[..]);
802        let mut s = String::new();
803        d.read_to_string(&mut s).unwrap();
804        assert_eq!(s, "foo bar baz");
805        d.read_to_string(&mut s).unwrap();
806        assert_eq!(s, "foo bar baz");
807    }
808
809    #[test]
810    fn qc_reader() {
811        ::quickcheck::quickcheck(test as fn(_) -> _);
812
813        fn test(v: Vec<u8>) -> bool {
814            let r = read::GzEncoder::new(&v[..], Compression::default());
815            let mut r = read::GzDecoder::new(r);
816            let mut v2 = Vec::new();
817            r.read_to_end(&mut v2).unwrap();
818            v == v2
819        }
820    }
821
822    #[test]
823    fn flush_after_write() {
824        let mut f = write::GzEncoder::new(Vec::new(), Compression::default());
825        write!(f, "Hello world").unwrap();
826        f.flush().unwrap();
827    }
828}