Skip to main content

zip/
read.rs

1//! Types for reading ZIP archives
2
3use crate::compression::CompressionMethod;
4use crate::cp437::FromCp437;
5use crate::datetime::DateTime;
6use crate::extra_fields::AexEncryption;
7use crate::extra_fields::UnicodeExtraField;
8use crate::extra_fields::Zip64ExtendedInformation;
9use crate::extra_fields::{ExtendedTimestamp, ExtraField, Ntfs, UsedExtraField};
10use crate::read::readers::{ZipFileReader, ZipFileSeekReader};
11use crate::result::{ZipError, ZipResult, invalid};
12use crate::spec::is_dir;
13use crate::spec::{
14    CentralDirectoryEndInfo, DataAndPosition, FixedSizeBlock, ZIP64_BYTES_THR,
15    ZipCentralEntryBlock, ZipFlags,
16};
17use crate::types::{SimpleFileOptions, System, ZipFileData, ffi};
18use crate::unstable::LittleEndianReadExt;
19use core::mem::replace;
20use indexmap::IndexMap;
21use std::borrow::Cow;
22use std::ffi::OsStr;
23use std::io::{self, Read, Seek, SeekFrom, Write, copy, sink};
24use std::path::{Component, Path, PathBuf};
25use std::sync::{Arc, OnceLock};
26
27mod config;
28pub use config::{ArchiveOffset, Config};
29
30/// Provides high level API for reading from a stream.
31pub(crate) mod stream;
32pub use stream::{read_zipfile_from_stream, read_zipfile_from_stream_with_compressed_size};
33
34pub(crate) mod magic_finder;
35pub(crate) mod readers;
36
37pub(crate) mod zip_archive;
38pub use zip_archive::{ZipArchive, ZipArchiveMetadata};
39
40#[cfg(feature = "aes-crypto")]
41pub use crate::aes::AesInfo;
42
43/// A struct for reading a zip file
44///
45/// When reading from a `ZipFile` using [`Self::read()`], keep in mind that `read()` **does not guarantee** the buffer will be fully filled in a single call.
46///
47/// If your logic depends on the buffer being completely populated, use [`Self::read_exact()`] instead. It will continue reading until the entire buffer is filled or an error occurs.
48#[derive(Debug)]
49pub struct ZipFile<'a, R: Read + ?Sized> {
50    pub(crate) data: Cow<'a, ZipFileData>,
51    pub(crate) reader: ZipFileReader<'a, R>,
52}
53
54/// A struct for reading and seeking a zip file
55pub struct ZipFileSeek<'a, R> {
56    data: Cow<'a, ZipFileData>,
57    reader: ZipFileSeekReader<'a, R>,
58}
59
60pub(crate) fn make_writable_dir_all<T: AsRef<Path>>(outpath: T) -> Result<(), ZipError> {
61    use std::fs;
62    fs::create_dir_all(outpath.as_ref())?;
63    #[cfg(unix)]
64    {
65        // Dirs must be writable until all normal files are extracted
66        use std::os::unix::fs::PermissionsExt;
67        std::fs::set_permissions(
68            outpath.as_ref(),
69            std::fs::Permissions::from_mode(
70                0o700 | std::fs::metadata(outpath.as_ref())?.permissions().mode(),
71            ),
72        )?;
73    }
74    Ok(())
75}
76
77#[cfg(unix)]
78pub(crate) fn make_symlink_impl<T>(
79    outpath: &Path,
80    target_str: &str,
81    _existing_files: &IndexMap<Box<[u8]>, T>,
82) -> ZipResult<()> {
83    std::os::unix::fs::symlink(Path::new(&target_str), outpath)?;
84    Ok(())
85}
86
87#[cfg(windows)]
88pub(crate) fn make_symlink_impl<T>(
89    outpath: &Path,
90    target_str: &str,
91    existing_files: &IndexMap<Box<[u8]>, T>,
92) -> ZipResult<()> {
93    let target = Path::new(OsStr::new(&target_str));
94    let target_is_dir_from_archive =
95        existing_files.contains_key(target_str.as_bytes()) && is_dir(target_str);
96    let target_is_dir = if target_is_dir_from_archive {
97        true
98    } else if let Ok(meta) = std::fs::metadata(target) {
99        meta.is_dir()
100    } else {
101        false
102    };
103    if target_is_dir {
104        std::os::windows::fs::symlink_dir(target, outpath)?;
105    } else {
106        std::os::windows::fs::symlink_file(target, outpath)?;
107    }
108    Ok(())
109}
110
111#[cfg(any(windows, unix))]
112pub(crate) fn make_symlink<T>(
113    outpath: &Path,
114    target: &[u8],
115    #[cfg_attr(not(any(windows, unix)), allow(unused))] existing_files: &IndexMap<Box<[u8]>, T>,
116) -> ZipResult<()> {
117    let Ok(target_str) = std::str::from_utf8(target) else {
118        return Err(invalid!("Invalid UTF-8 as symlink target"));
119    };
120    make_symlink_impl(outpath, target_str, existing_files)
121}
122
123#[cfg(not(any(windows, unix)))]
124pub(crate) fn make_symlink<T>(
125    outpath: &Path,
126    target: &[u8],
127    #[cfg_attr(not(any(windows, unix)), allow(unused))] existing_files: &IndexMap<Box<[u8]>, T>,
128) -> ZipResult<()> {
129    let Ok(_) = std::str::from_utf8(target) else {
130        return Err(invalid!("Invalid UTF-8 as symlink target"));
131    };
132    use std::fs::File;
133    let output = File::create(outpath);
134    output?.write_all(target)?;
135    Ok(())
136}
137
138#[derive(Debug)]
139pub(crate) struct CentralDirectoryInfo {
140    pub(crate) archive_offset: u64,
141    pub(crate) directory_start: u64,
142    pub(crate) number_of_files: usize,
143    pub(crate) disk_number: u32,
144    pub(crate) disk_with_central_directory: u32,
145}
146
147impl<'a> TryFrom<&'a CentralDirectoryEndInfo> for CentralDirectoryInfo {
148    type Error = ZipError;
149
150    fn try_from(value: &'a CentralDirectoryEndInfo) -> Result<Self, Self::Error> {
151        let (relative_cd_offset, number_of_files, disk_number, disk_with_central_directory) =
152            match &value.eocd64 {
153                Some(DataAndPosition { data: eocd64, .. }) => {
154                    if eocd64.number_of_files_on_this_disk > eocd64.number_of_files {
155                        return Err(invalid!(
156                            "ZIP64 footer indicates more files on this disk than in the whole archive"
157                        ));
158                    }
159                    (
160                        eocd64.central_directory_offset,
161                        eocd64.number_of_files as usize,
162                        eocd64.disk_number,
163                        eocd64.disk_with_central_directory,
164                    )
165                }
166                _ => (
167                    u64::from(value.eocd.data.central_directory_offset),
168                    value.eocd.data.number_of_files_on_this_disk as usize,
169                    u32::from(value.eocd.data.disk_number),
170                    u32::from(value.eocd.data.disk_with_central_directory),
171                ),
172            };
173
174        let directory_start = relative_cd_offset
175            .checked_add(value.archive_offset)
176            .ok_or(invalid!("Invalid central directory size or offset"))?;
177
178        Ok(Self {
179            archive_offset: value.archive_offset,
180            directory_start,
181            number_of_files,
182            disk_number,
183            disk_with_central_directory,
184        })
185    }
186}
187
188/// Store all entries which specify a numeric "mode" which is familiar to POSIX operating systems.
189#[cfg(unix)]
190#[derive(Default, Debug)]
191struct UnixFileModes {
192    map: std::collections::BTreeMap<PathBuf, u32>,
193}
194
195#[cfg(unix)]
196impl UnixFileModes {
197    #[cfg_attr(not(debug_assertions), allow(unused))]
198    pub fn add_mode(&mut self, path: PathBuf, mode: u32) {
199        // We don't print a warning or consider it remotely out of the ordinary to receive two
200        // separate modes for the same path: just take the later one.
201        let old_entry = self.map.insert(path, mode);
202        debug_assert_eq!(old_entry, None);
203    }
204
205    // Child nodes will be sorted later lexicographically, so reversing the order puts them first.
206    pub fn all_perms_with_children_first(
207        self,
208    ) -> impl IntoIterator<Item = (PathBuf, std::fs::Permissions)> {
209        use std::os::unix::fs::PermissionsExt;
210        self.map
211            .into_iter()
212            .rev()
213            .map(|(p, m)| (p, std::fs::Permissions::from_mode(m)))
214    }
215}
216
217impl<R: Read + Seek> ZipArchive<R> {
218    pub(crate) fn merge_contents<W: Write + Seek>(
219        &mut self,
220        mut w: W,
221    ) -> ZipResult<IndexMap<Box<[u8]>, ZipFileData>> {
222        if self.shared.files.is_empty() {
223            return Ok(IndexMap::new());
224        }
225        let mut new_files = self.shared.files.clone();
226        /* The first file header will probably start at the beginning of the file, but zip doesn't
227         * enforce that, and executable zips like PEX files will have a shebang line so will
228         * definitely be greater than 0.
229         *
230         * assert_eq!(0, new_files[0].header_start); // Avoid this.
231         */
232
233        let first_new_file_header_start = w.stream_position()?;
234
235        /* Push back file header starts for all entries in the covered files. */
236        new_files.values_mut().try_for_each(|f| {
237            /* This is probably the only really important thing to change. */
238            f.header_start = f
239                .header_start
240                .checked_add(first_new_file_header_start)
241                .ok_or(invalid!(
242                    "new header start from merge would have been too large"
243                ))?;
244            /* This is only ever used internally to cache metadata lookups (it's not part of the
245             * zip spec), and 0 is the sentinel value. */
246            f.central_header_start = 0;
247            /* This is an atomic variable so it can be updated from another thread in the
248             * implementation (which is good!). */
249            if let Some(old_data_start) = f.data_start.take() {
250                let new_data_start = old_data_start
251                    .checked_add(first_new_file_header_start)
252                    .ok_or(invalid!(
253                        "new data start from merge would have been too large"
254                    ))?;
255                f.data_start.get_or_init(|| new_data_start);
256            }
257            Ok::<_, ZipError>(())
258        })?;
259
260        /* Rewind to the beginning of the file.
261         *
262         * NB: we *could* decide to start copying from new_files[0].header_start instead, which
263         * would avoid copying over e.g. any pex shebangs or other file contents that start before
264         * the first zip file entry. However, zip files actually shouldn't care about garbage data
265         * in *between* real entries, since the central directory header records the correct start
266         * location of each, and keeping track of that math is more complicated logic that will only
267         * rarely be used, since most zips that get merged together are likely to be produced
268         * specifically for that purpose (and therefore are unlikely to have a shebang or other
269         * preface). Finally, this preserves any data that might actually be useful.
270         */
271        self.reader.rewind()?;
272        /* Find the end of the file data. */
273        let length_to_read = self.shared.dir_start;
274        /* Produce a Read that reads bytes up until the start of the central directory header.
275         * This "as &mut dyn Read" trick is used elsewhere to avoid having to clone the underlying
276         * handle, which it really shouldn't need to anyway. */
277        let mut limited_raw = (&mut self.reader as &mut dyn Read).take(length_to_read);
278        /* Copy over file data from source archive directly. */
279        io::copy(&mut limited_raw, &mut w)?;
280
281        /* Return the files we've just written to the data stream. */
282        Ok(new_files)
283    }
284
285    /// Extract a Zip archive into a directory, overwriting files if they
286    /// already exist. Paths are sanitized with [`ZipFile::enclosed_name`]. Symbolic links are only
287    /// created and followed if the target is within the destination directory (this is checked
288    /// conservatively using [`std::fs::canonicalize`]).
289    ///
290    /// Extraction is not atomic. If an error is encountered, some of the files
291    /// may be left on disk. However, on Unix targets, no newly-created directories with part but
292    /// not all of their contents extracted will be readable, writable or usable as process working
293    /// directories by any non-root user except you.
294    ///
295    /// On Unix and Windows, symbolic links are extracted correctly. On other platforms such as
296    /// WebAssembly, symbolic links aren't supported, so they're extracted as normal files
297    /// containing the target path in UTF-8.
298    pub fn extract<P: AsRef<Path>>(&mut self, directory: P) -> ZipResult<()> {
299        self.extract_internal(directory, None::<fn(&Path) -> bool>)
300    }
301
302    /// Extracts a Zip archive into a directory in the same fashion as
303    /// [`ZipArchive::extract`], but detects a "root" directory in the archive
304    /// (a single top-level directory that contains the rest of the archive's
305    /// entries) and extracts its contents directly.
306    ///
307    /// For a sensible default `filter`, you can use [`root_dir_common_filter`].
308    /// For a custom `filter`, see [`RootDirFilter`].
309    ///
310    /// See [`ZipArchive::root_dir`] for more information on how the root
311    /// directory is detected and the meaning of the `filter` parameter.
312    ///
313    /// ## Example
314    ///
315    /// Imagine a Zip archive with the following structure:
316    ///
317    /// ```text
318    /// root/file1.txt
319    /// root/file2.txt
320    /// root/sub/file3.txt
321    /// root/sub/subsub/file4.txt
322    /// ```
323    ///
324    /// If the archive is extracted to `foo` using [`ZipArchive::extract`],
325    /// the resulting directory structure will be:
326    ///
327    /// ```text
328    /// foo/root/file1.txt
329    /// foo/root/file2.txt
330    /// foo/root/sub/file3.txt
331    /// foo/root/sub/subsub/file4.txt
332    /// ```
333    ///
334    /// If the archive is extracted to `foo` using
335    /// [`ZipArchive::extract_unwrapped_root_dir`], the resulting directory
336    /// structure will be:
337    ///
338    /// ```text
339    /// foo/file1.txt
340    /// foo/file2.txt
341    /// foo/sub/file3.txt
342    /// foo/sub/subsub/file4.txt
343    /// ```
344    ///
345    /// ## Example - No Root Directory
346    ///
347    /// Imagine a Zip archive with the following structure:
348    ///
349    /// ```text
350    /// root/file1.txt
351    /// root/file2.txt
352    /// root/sub/file3.txt
353    /// root/sub/subsub/file4.txt
354    /// other/file5.txt
355    /// ```
356    ///
357    /// Due to the presence of the `other` directory,
358    /// [`ZipArchive::extract_unwrapped_root_dir`] will extract this in the same
359    /// fashion as [`ZipArchive::extract`] as there is now no "root directory."
360    pub fn extract_unwrapped_root_dir<P: AsRef<Path>>(
361        &mut self,
362        directory: P,
363        root_dir_filter: impl RootDirFilter,
364    ) -> ZipResult<()> {
365        self.extract_internal(directory, Some(root_dir_filter))
366    }
367
368    fn extract_internal<P: AsRef<Path>>(
369        &mut self,
370        directory: P,
371        root_dir_filter: Option<impl RootDirFilter>,
372    ) -> ZipResult<()> {
373        use std::fs;
374
375        fs::create_dir_all(&directory)?;
376        let directory = directory.as_ref().canonicalize()?;
377
378        let root_dir = root_dir_filter
379            .and_then(|filter| {
380                self.root_dir(&filter)
381                    .transpose()
382                    .map(|root_dir| root_dir.map(|root_dir| (root_dir, filter)))
383            })
384            .transpose()?;
385
386        // If we have a root dir, simplify the path components to be more
387        // appropriate for passing to `safe_prepare_path`
388        let root_dir = root_dir
389            .as_ref()
390            .map(|(root_dir, filter)| {
391                crate::path::simplified_components(root_dir)
392                    .ok_or_else(|| {
393                        // Should be unreachable
394                        debug_assert!(false, "Invalid root dir path");
395
396                        invalid!("Invalid root dir path")
397                    })
398                    .map(|root_dir| (root_dir, filter))
399            })
400            .transpose()?;
401
402        #[cfg(unix)]
403        let mut files_by_unix_mode = UnixFileModes::default();
404
405        for i in 0..self.len() {
406            let mut file = self.by_index(i)?;
407
408            let mut outpath = directory.clone();
409            /* TODO: the control flow of this method call and subsequent expectations about the
410             *       values in this loop is extremely difficult to follow. It also appears to
411             *       perform a nested loop upon extracting every single file entry? Why does it
412             *       accept two arguments that point to the same directory path, one mutable? */
413            file.safe_prepare_path(directory.as_ref(), &mut outpath, root_dir.as_ref())?;
414
415            #[cfg(any(unix, windows))]
416            if file.is_symlink() {
417                let mut target = Vec::with_capacity(file.size() as usize);
418                file.read_to_end(&mut target)?;
419                drop(file);
420                make_symlink(&outpath, &target, &self.shared.files)?;
421                continue;
422            } else if file.is_dir() {
423                crate::read::make_writable_dir_all(&outpath)?;
424                continue;
425            }
426            let mut outfile = fs::File::create(&outpath)?;
427            io::copy(&mut file, &mut outfile)?;
428
429            // Check for real permissions, which we'll set in a second pass.
430            #[cfg(unix)]
431            if let Some(mode) = file.unix_mode() {
432                files_by_unix_mode.add_mode(outpath, mode);
433            }
434
435            // Set original timestamp.
436            #[cfg(feature = "chrono")]
437            if let Some(last_modified) = file.last_modified()
438                && let Some(t) = last_modified.datetime_to_systemtime()
439            {
440                outfile.set_modified(t)?;
441            }
442        }
443
444        // Ensure we update children's permissions before making a parent unwritable.
445        #[cfg(unix)]
446        for (path, perms) in files_by_unix_mode.all_perms_with_children_first() {
447            std::fs::set_permissions(path, perms)?;
448        }
449
450        Ok(())
451    }
452}
453
454/// Parse a central directory entry to collect the information for the file.
455pub(crate) fn central_header_to_zip_file<R: Read + Seek>(
456    reader: &mut R,
457    central_directory: &CentralDirectoryInfo,
458) -> ZipResult<ZipFileData> {
459    let central_header_start = reader.stream_position()?;
460
461    // Parse central header
462    let block = ZipCentralEntryBlock::parse(reader)?;
463
464    let file = central_header_to_zip_file_inner(
465        reader,
466        central_directory.archive_offset,
467        central_header_start,
468        block,
469    )?;
470
471    let central_header_end = reader.stream_position()?;
472
473    reader.seek(SeekFrom::Start(central_header_end))?;
474    Ok(file)
475}
476
477#[inline]
478fn read_variable_length_byte_field<R: Read>(reader: &mut R, len: usize) -> ZipResult<Box<[u8]>> {
479    let mut data = vec![0; len].into_boxed_slice();
480    if let Err(e) = reader.read_exact(&mut data) {
481        if e.kind() == io::ErrorKind::UnexpectedEof {
482            return Err(invalid!(
483                "Variable-length field extends beyond file boundary"
484            ));
485        }
486        return Err(e.into());
487    }
488    Ok(data)
489}
490
491/// Parse a central directory entry to collect the information for the file.
492fn central_header_to_zip_file_inner<R: Read>(
493    reader: &mut R,
494    archive_offset: u64,
495    central_header_start: u64,
496    block: ZipCentralEntryBlock,
497) -> ZipResult<ZipFileData> {
498    let ZipCentralEntryBlock {
499        // magic,
500        version_made_by,
501        // version_to_extract,
502        flags,
503        compression_method,
504        last_mod_time,
505        last_mod_date,
506        crc32,
507        compressed_size,
508        uncompressed_size,
509        file_name_length,
510        extra_field_length,
511        file_comment_length,
512        // disk_number,
513        // internal_file_attributes,
514        external_file_attributes,
515        offset,
516        ..
517    } = block;
518
519    let encrypted = ZipFlags::matching(flags, ZipFlags::Encrypted);
520    let is_utf8 = ZipFlags::matching(flags, ZipFlags::LanguageEncoding);
521    let using_data_descriptor = ZipFlags::matching(flags, ZipFlags::UsingDataDescriptor);
522
523    let file_name_raw = read_variable_length_byte_field(reader, file_name_length as usize)?;
524    let extra_field = read_variable_length_byte_field(reader, extra_field_length as usize)?;
525    let file_comment_raw = read_variable_length_byte_field(reader, file_comment_length as usize)?;
526    let file_name: Box<str> = if is_utf8 {
527        String::from_utf8_lossy(&file_name_raw).into()
528    } else {
529        file_name_raw.from_cp437()?.into()
530    };
531    let file_comment: Box<str> = if is_utf8 {
532        String::from_utf8_lossy(&file_comment_raw).into()
533    } else {
534        file_comment_raw.from_cp437()?.into()
535    };
536
537    let (version_made_by, system) = System::extract_bytes(version_made_by);
538    // Construct the result
539    let mut result = ZipFileData {
540        system,
541        version_made_by,
542        encrypted,
543        using_data_descriptor,
544        is_utf8,
545        compression_method: CompressionMethod::parse_from_u16(compression_method),
546        compression_level: None,
547        last_modified_time: DateTime::try_from_msdos(last_mod_date, last_mod_time).ok(),
548        crc32,
549        compressed_size: compressed_size.into(),
550        uncompressed_size: uncompressed_size.into(),
551        flags,
552        file_name,
553        file_name_raw,
554        extra_field: Some(Arc::from(extra_field)),
555        central_extra_field: None,
556        file_comment,
557        header_start: offset.into(),
558        extra_data_start: None,
559        central_header_start,
560        data_start: OnceLock::new(),
561        external_attributes: external_file_attributes,
562        large_file: false,
563        aes_mode: None,
564        aes_extra_data_start: 0,
565        extra_fields: Vec::new(),
566    };
567    parse_extra_field(&mut result)?;
568
569    let aes_enabled = result.compression_method == CompressionMethod::AES;
570    if aes_enabled && result.aes_mode.is_none() {
571        return Err(invalid!("AES encryption without AES extra data field"));
572    }
573
574    // Account for shifted zip offsets.
575    result.header_start = result
576        .header_start
577        .checked_add(archive_offset)
578        .ok_or(invalid!("Archive header is too large"))?;
579
580    Ok(result)
581}
582
583pub(crate) fn parse_extra_field(file: &mut ZipFileData) -> ZipResult<()> {
584    let mut extra_field = file.extra_field.clone();
585    let mut central_extra_field = file.central_extra_field.clone();
586    for field_group in [&mut extra_field, &mut central_extra_field] {
587        let Some(extra_field) = field_group else {
588            continue;
589        };
590        let mut modified = false;
591        let mut processed_extra_field = vec![];
592        let len = extra_field.len();
593        let mut reader = io::Cursor::new(&**extra_field);
594
595        let mut position = reader.position();
596        while position < len as u64 {
597            let old_position = position;
598            let remove = parse_single_extra_field(file, &mut reader, position, false)?;
599            position = reader.position();
600            if remove {
601                modified = true;
602            } else {
603                let field_len = (position - old_position) as usize;
604                let write_start = processed_extra_field.len();
605                reader.seek(SeekFrom::Start(old_position))?;
606                processed_extra_field.extend_from_slice(&vec![0u8; field_len]);
607                if let Err(e) = reader
608                    .read_exact(&mut processed_extra_field[write_start..(write_start + field_len)])
609                {
610                    if e.kind() == io::ErrorKind::UnexpectedEof {
611                        return Err(invalid!("Extra field content exceeds declared length"));
612                    }
613                    return Err(e.into());
614                }
615            }
616        }
617        if modified {
618            *field_group = Some(Arc::from(processed_extra_field.into_boxed_slice()));
619        }
620    }
621    file.extra_field = extra_field;
622    file.central_extra_field = central_extra_field;
623    Ok(())
624}
625
626pub(crate) fn parse_single_extra_field<R: Read>(
627    file: &mut ZipFileData,
628    reader: &mut R,
629    bytes_already_read: u64,
630    disallow_zip64: bool,
631) -> ZipResult<bool> {
632    let kind = match reader.read_u16_le() {
633        Ok(kind) => kind,
634        Err(e) if e.kind() == io::ErrorKind::UnexpectedEof => return Ok(false),
635        Err(e) => return Err(e.into()),
636    };
637    let decoded_extra_field = UsedExtraField::try_from(kind);
638    let len = match decoded_extra_field {
639        Ok(known_field) => match reader.read_u16_le() {
640            Ok(len) => len,
641            Err(e) if e.kind() == io::ErrorKind::UnexpectedEof => {
642                return Err(invalid!("Extra field {} header truncated", known_field));
643            }
644            Err(e) => return Err(e.into()),
645        },
646        Err(()) => {
647            match reader.read_u16_le() {
648                Ok(len) => len,
649                Err(e) if e.kind() == io::ErrorKind::UnexpectedEof => return Ok(false), // early return, most likely a padding
650                Err(_e) => {
651                    // Consume remaining bytes to avoid infinite loop in caller
652                    let mut buf = Vec::new();
653                    let _ = reader.read_to_end(&mut buf);
654                    return Ok(false);
655                }
656            }
657        }
658    };
659    match decoded_extra_field {
660        // Zip64 extended information extra field
661        Ok(UsedExtraField::Zip64ExtendedInfo) => {
662            if disallow_zip64 {
663                return Err(invalid!("Can't write a custom field using the ZIP64 ID"));
664            }
665            file.large_file = true;
666            Zip64ExtendedInformation::parse(
667                reader,
668                len,
669                &mut file.uncompressed_size,
670                &mut file.compressed_size,
671                &mut file.header_start,
672            )?;
673            return Ok(true);
674        }
675        Ok(UsedExtraField::Ntfs) => {
676            // NTFS extra field
677            file.extra_fields
678                .push(ExtraField::Ntfs(Ntfs::try_from_reader(reader, len)?));
679        }
680        Ok(UsedExtraField::AeXEncryption) => {
681            // AES
682            AexEncryption::parse(
683                reader,
684                len,
685                &mut file.aes_mode,
686                &mut file.compression_method,
687            )?;
688            file.aes_extra_data_start = bytes_already_read;
689        }
690        Ok(UsedExtraField::ExtendedTimestamp) => {
691            file.extra_fields.push(ExtraField::ExtendedTimestamp(
692                ExtendedTimestamp::try_from_reader(reader, len)?,
693            ));
694        }
695        Ok(UsedExtraField::UnicodeComment) => {
696            // Info-ZIP Unicode Comment Extra Field
697            // APPNOTE 4.6.8 and https://libzip.org/specifications/extrafld.txt
698            file.file_comment = String::from_utf8(
699                UnicodeExtraField::try_from_reader(reader, len)?
700                    .unwrap_valid(file.file_comment.as_bytes())?
701                    .into_vec(),
702            )?
703            .into();
704        }
705        Ok(UsedExtraField::UnicodePath) => {
706            // Info-ZIP Unicode Path Extra Field
707            // APPNOTE 4.6.9 and https://libzip.org/specifications/extrafld.txt
708            file.file_name_raw = UnicodeExtraField::try_from_reader(reader, len)?
709                .unwrap_valid(&file.file_name_raw)?;
710            file.file_name =
711                String::from_utf8(file.file_name_raw.clone().into_vec())?.into_boxed_str();
712            file.is_utf8 = true;
713        }
714        _ => {
715            if let Err(e) = reader.read_exact(&mut vec![0u8; len as usize]) {
716                if e.kind() == io::ErrorKind::UnexpectedEof {
717                    return Err(invalid!("Extra field content truncated"));
718                }
719                return Err(e.into());
720            }
721            // Other fields are ignored
722        }
723    }
724    Ok(false)
725}
726
727/// A trait for exposing file metadata inside the zip.
728pub trait HasZipMetadata {
729    /// Get the file metadata
730    fn get_metadata(&self) -> &ZipFileData;
731}
732
733/// Options for reading a file from an archive.
734#[derive(Default)]
735pub struct ZipReadOptions<'a> {
736    /// The password to use when decrypting the file.  This is ignored if not required.
737    password: Option<&'a [u8]>,
738
739    /// Ignore the value of the encryption flag and proceed as if the file were plaintext.
740    ignore_encryption_flag: bool,
741
742    /// Ignore the crc32 of the file
743    ignore_crc: bool,
744}
745
746impl<'a> ZipReadOptions<'a> {
747    /// Create a new set of options with the default values.
748    #[must_use]
749    pub fn new() -> Self {
750        Self::default()
751    }
752
753    /// Set the password, if any, to use.  Return for chaining.
754    #[must_use]
755    pub fn password(mut self, password: Option<&'a [u8]>) -> Self {
756        self.password = password;
757        self
758    }
759
760    /// Set the ignore encryption flag.  Return for chaining.
761    #[must_use]
762    pub fn ignore_encryption_flag(mut self, ignore: bool) -> Self {
763        self.ignore_encryption_flag = ignore;
764        self
765    }
766
767    /// Ignore the CRC32 of the file
768    #[must_use]
769    pub fn ignore_crc32(mut self, should_ignore: bool) -> Self {
770        self.ignore_crc = should_ignore;
771        self
772    }
773}
774
775/// Methods for retrieving information on zip files
776impl<'a, R: Read + ?Sized> ZipFile<'a, R> {
777    pub(crate) fn take_raw_reader(&mut self) -> io::Result<io::Take<&'a mut R>> {
778        replace(&mut self.reader, ZipFileReader::NoReader).into_inner()
779    }
780
781    /// Get the version of the file
782    pub fn version_made_by(&self) -> (u8, u8) {
783        (
784            self.get_metadata().version_made_by / 10,
785            self.get_metadata().version_made_by % 10,
786        )
787    }
788
789    /// Get the name of the file
790    ///
791    /// # Warnings
792    ///
793    /// It is dangerous to use this name directly when extracting an archive.
794    /// It may contain an absolute path (`/etc/shadow`), or break out of the
795    /// current directory (`../runtime`). Carelessly writing to these paths
796    /// allows an attacker to craft a ZIP archive that will overwrite critical
797    /// files.
798    ///
799    /// You can use the [`ZipFile::enclosed_name`] method to validate the name
800    /// as a safe path.
801    pub fn name(&self) -> &str {
802        &self.get_metadata().file_name
803    }
804
805    /// Get the name of the file, in the raw (internal) byte representation.
806    ///
807    /// The encoding of this data is currently undefined.
808    pub fn name_raw(&self) -> &[u8] {
809        &self.get_metadata().file_name_raw
810    }
811
812    /// Get the name of the file in a sanitized form. It truncates the name to the first NULL byte,
813    /// removes a leading '/' and removes '..' parts.
814    #[deprecated(
815        since = "0.5.7",
816        note = "by stripping `..`s from the path, the meaning of paths can change.
817                `mangled_name` can be used if this behaviour is desirable"
818    )]
819    pub fn sanitized_name(&self) -> PathBuf {
820        self.mangled_name()
821    }
822
823    /// Rewrite the path, ignoring any path components with special meaning.
824    ///
825    /// - Absolute paths are made relative
826    /// - [`ParentDir`]s are ignored
827    /// - Truncates the filename at a NULL byte
828    ///
829    /// This is appropriate if you need to be able to extract *something* from
830    /// any archive, but will easily misrepresent trivial paths like
831    /// `foo/../bar` as `foo/bar` (instead of `bar`). Because of this,
832    /// [`ZipFile::enclosed_name`] is the better option in most scenarios.
833    ///
834    /// [`ParentDir`]: `Component::ParentDir`
835    pub fn mangled_name(&self) -> PathBuf {
836        self.get_metadata().file_name_sanitized()
837    }
838
839    /// Ensure the file path is safe to use as a [`Path`].
840    ///
841    /// - It can't contain NULL bytes
842    /// - It can't resolve to a path outside the current directory
843    ///   > `foo/../bar` is fine, `foo/../../bar` is not.
844    /// - It can't be an absolute path
845    ///
846    /// This will read well-formed ZIP files correctly, and is resistant
847    /// to path-based exploits. It is recommended over
848    /// [`ZipFile::mangled_name`].
849    pub fn enclosed_name(&self) -> Option<PathBuf> {
850        self.get_metadata().enclosed_name()
851    }
852
853    pub(crate) fn simplified_components(&self) -> Option<Vec<&OsStr>> {
854        self.get_metadata().simplified_components()
855    }
856
857    /// Prepare the path for extraction by creating necessary missing directories and checking for symlinks to be contained within the base path.
858    ///
859    /// `base_path` parameter is assumed to be canonicalized.
860    pub(crate) fn safe_prepare_path(
861        &self,
862        base_path: &Path,
863        outpath: &mut PathBuf,
864        root_dir: Option<&(Vec<&OsStr>, impl RootDirFilter)>,
865    ) -> ZipResult<()> {
866        let components = self
867            .simplified_components()
868            .ok_or(invalid!("Invalid file path"))?;
869
870        let components = match root_dir {
871            Some((root_dir, filter)) => match components.strip_prefix(&**root_dir) {
872                Some(components) => components,
873
874                // In this case, we expect that the file was not in the root
875                // directory, but was filtered out when searching for the
876                // root directory.
877                None => {
878                    // We could technically find ourselves at this code
879                    // path if the user provides an unstable or
880                    // non-deterministic `filter` function.
881                    //
882                    // If debug assertions are on, we should panic here.
883                    // Otherwise, the safest thing to do here is to just
884                    // extract as-is.
885                    debug_assert!(
886                        !filter(&PathBuf::from_iter(components.iter())),
887                        "Root directory filter should not match at this point"
888                    );
889
890                    // Extract as-is.
891                    &components[..]
892                }
893            },
894
895            None => &components[..],
896        };
897
898        let components_len = components.len();
899
900        for (is_last, component) in components
901            .iter()
902            .copied()
903            .enumerate()
904            .map(|(i, c)| (i == components_len - 1, c))
905        {
906            // we can skip the target directory itself because the base path is assumed to be "trusted" (if the user say extract to a symlink we can follow it)
907            outpath.push(component);
908
909            // check if the path is a symlink, the target must be _inherently_ within the directory
910            for limit in (0..5u8).rev() {
911                let meta = match std::fs::symlink_metadata(&outpath) {
912                    Ok(meta) => meta,
913                    Err(e) if e.kind() == io::ErrorKind::NotFound => {
914                        if !is_last {
915                            crate::read::make_writable_dir_all(&outpath)?;
916                        }
917                        break;
918                    }
919                    Err(e) => return Err(e.into()),
920                };
921
922                if !meta.is_symlink() {
923                    break;
924                }
925
926                if limit == 0 {
927                    return Err(invalid!("Extraction followed a symlink too deep"));
928                }
929
930                // note that we cannot accept links that do not inherently resolve to a path inside the directory to prevent:
931                // - disclosure of unrelated path exists (no check for a path exist and then ../ out)
932                // - issues with file-system specific path resolution (case sensitivity, etc)
933                let target = std::fs::read_link(&outpath)?;
934
935                if !crate::path::simplified_components(&target)
936                    .ok_or(invalid!("Invalid symlink target path"))?
937                    .starts_with(
938                        &crate::path::simplified_components(base_path)
939                            .ok_or(invalid!("Invalid base path"))?,
940                    )
941                {
942                    let is_absolute_enclosed = base_path
943                        .components()
944                        .map(Some)
945                        .chain(std::iter::once(None))
946                        .zip(target.components().map(Some).chain(std::iter::repeat(None)))
947                        .all(|(a, b)| match (a, b) {
948                            // both components are normal
949                            (Some(Component::Normal(a)), Some(Component::Normal(b))) => a == b,
950                            // both components consumed fully
951                            (None, None) => true,
952                            // target consumed fully but base path is not
953                            (Some(_), None) => false,
954                            // base path consumed fully but target is not (and normal)
955                            (None, Some(Component::CurDir | Component::Normal(_))) => true,
956                            _ => false,
957                        });
958
959                    if !is_absolute_enclosed {
960                        return Err(invalid!("Symlink is not inherently safe"));
961                    }
962                }
963
964                outpath.push(target);
965            }
966        }
967        Ok(())
968    }
969
970    /// Get the comment of the file
971    pub fn comment(&self) -> &str {
972        &self.get_metadata().file_comment
973    }
974
975    /// Get the compression method used to store the file
976    pub fn compression(&self) -> CompressionMethod {
977        self.get_metadata().compression_method
978    }
979
980    /// Get if the files is encrypted or not
981    pub fn encrypted(&self) -> bool {
982        self.data.encrypted
983    }
984
985    /// Get the size of the file, in bytes, in the archive
986    pub fn compressed_size(&self) -> u64 {
987        self.get_metadata().compressed_size
988    }
989
990    /// Get the size of the file, in bytes, when uncompressed
991    pub fn size(&self) -> u64 {
992        self.get_metadata().uncompressed_size
993    }
994
995    /// Get the time the file was last modified
996    pub fn last_modified(&self) -> Option<DateTime> {
997        self.data.last_modified_time
998    }
999    /// Returns whether the file is actually a directory
1000    pub fn is_dir(&self) -> bool {
1001        is_dir(self.name())
1002    }
1003
1004    /// Returns whether the file is actually a symbolic link
1005    pub fn is_symlink(&self) -> bool {
1006        self.unix_mode()
1007            .is_some_and(|mode| mode & ffi::S_IFLNK == ffi::S_IFLNK)
1008    }
1009
1010    /// Returns whether the file is a normal file (i.e. not a directory or symlink)
1011    pub fn is_file(&self) -> bool {
1012        !self.is_dir() && !self.is_symlink()
1013    }
1014
1015    /// Get unix mode for the file
1016    pub fn unix_mode(&self) -> Option<u32> {
1017        self.get_metadata().unix_mode()
1018    }
1019
1020    /// Get the CRC32 hash of the original file
1021    pub fn crc32(&self) -> u32 {
1022        self.get_metadata().crc32
1023    }
1024
1025    /// Get the extra data of the zip header for this file
1026    pub fn extra_data(&self) -> Option<&[u8]> {
1027        self.get_metadata().extra_field.as_deref()
1028    }
1029
1030    /// Get the starting offset of the data of the compressed file
1031    pub fn data_start(&self) -> Option<u64> {
1032        self.data.data_start.get().copied()
1033    }
1034
1035    /// Get the starting offset of the zip header for this file
1036    pub fn header_start(&self) -> u64 {
1037        self.get_metadata().header_start
1038    }
1039    /// Get the starting offset of the zip header in the central directory for this file
1040    pub fn central_header_start(&self) -> u64 {
1041        self.get_metadata().central_header_start
1042    }
1043
1044    /// Get the [`SimpleFileOptions`] that would be used to write this file to
1045    /// a new zip archive.
1046    pub fn options(&self) -> SimpleFileOptions {
1047        let mut options = SimpleFileOptions::default()
1048            .large_file(self.compressed_size().max(self.size()) > ZIP64_BYTES_THR)
1049            .compression_method(self.compression())
1050            .unix_permissions(self.unix_mode().unwrap_or(0o644) | ffi::S_IFREG)
1051            .last_modified_time(
1052                self.last_modified()
1053                    .filter(DateTime::is_valid)
1054                    .unwrap_or_else(DateTime::default_for_write),
1055            );
1056
1057        options.normalize();
1058        #[cfg(feature = "aes-crypto")]
1059        if let Some((mode, vendor_version, compression_method)) = self.get_metadata().aes_mode {
1060            // Preserve AES metadata in options for downstream writers.
1061            // This is metadata-only and does not trigger encryption.
1062            options.aes_mode = Some(crate::aes::AesModeOptions::new(
1063                mode,
1064                vendor_version,
1065                compression_method,
1066                None,
1067            ));
1068        }
1069        options
1070    }
1071}
1072
1073/// Methods for retrieving information on zip files
1074impl<R: Read> ZipFile<'_, R> {
1075    /// iterate through all extra fields
1076    pub fn extra_data_fields(&self) -> impl Iterator<Item = &ExtraField> {
1077        self.data.extra_fields.iter()
1078    }
1079}
1080
1081impl<R: Read + ?Sized> HasZipMetadata for ZipFile<'_, R> {
1082    fn get_metadata(&self) -> &ZipFileData {
1083        self.data.as_ref()
1084    }
1085}
1086
1087impl<R: Read + ?Sized> Read for ZipFile<'_, R> {
1088    fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
1089        self.reader.read(buf)
1090    }
1091
1092    fn read_exact(&mut self, buf: &mut [u8]) -> io::Result<()> {
1093        self.reader.read_exact(buf)
1094    }
1095
1096    fn read_to_end(&mut self, buf: &mut Vec<u8>) -> io::Result<usize> {
1097        self.reader.read_to_end(buf)
1098    }
1099
1100    fn read_to_string(&mut self, buf: &mut String) -> io::Result<usize> {
1101        self.reader.read_to_string(buf)
1102    }
1103}
1104
1105impl<R: Read> Read for ZipFileSeek<'_, R> {
1106    fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
1107        match &mut self.reader {
1108            ZipFileSeekReader::Raw(r) => r.read(buf),
1109        }
1110    }
1111}
1112
1113impl<R: Seek> Seek for ZipFileSeek<'_, R> {
1114    fn seek(&mut self, pos: SeekFrom) -> io::Result<u64> {
1115        match &mut self.reader {
1116            ZipFileSeekReader::Raw(r) => r.seek(pos),
1117        }
1118    }
1119}
1120
1121impl<R> HasZipMetadata for ZipFileSeek<'_, R> {
1122    fn get_metadata(&self) -> &ZipFileData {
1123        self.data.as_ref()
1124    }
1125}
1126
1127impl<R: Read + ?Sized> Drop for ZipFile<'_, R> {
1128    fn drop(&mut self) {
1129        // self.data is Owned, this reader is constructed by a streaming reader.
1130        // In this case, we want to exhaust the reader so that the next file is accessible.
1131        if let Cow::Owned(_) = self.data {
1132            // Get the inner `Take` reader so all decryption, decompression and CRC calculation is skipped.
1133            if let Ok(mut inner) = self.take_raw_reader() {
1134                let _ = copy(&mut inner, &mut sink());
1135            }
1136        }
1137    }
1138}
1139
1140/// A filter that determines whether an entry should be ignored when searching
1141/// for the root directory of a Zip archive.
1142///
1143/// Returns `true` if the entry should be considered, and `false` if it should
1144/// be ignored.
1145///
1146/// See [`root_dir_common_filter`] for a sensible default filter.
1147pub trait RootDirFilter: Fn(&Path) -> bool {}
1148impl<F: Fn(&Path) -> bool> RootDirFilter for F {}
1149
1150/// Common filters when finding the root directory of a Zip archive.
1151///
1152/// This filter is a sensible default for most use cases and filters out common
1153/// system files that are usually irrelevant to the contents of the archive.
1154///
1155/// Currently, the filter ignores:
1156/// - `/__MACOSX/`
1157/// - `/.DS_Store`
1158/// - `/Thumbs.db`
1159///
1160/// **This function is not guaranteed to be stable and may change in future versions.**
1161///
1162/// # Example
1163///
1164/// ```rust
1165/// # use std::path::Path;
1166/// assert!(zip::read::root_dir_common_filter(Path::new("foo.txt")));
1167/// assert!(!zip::read::root_dir_common_filter(Path::new(".DS_Store")));
1168/// assert!(!zip::read::root_dir_common_filter(Path::new("Thumbs.db")));
1169/// assert!(!zip::read::root_dir_common_filter(Path::new("__MACOSX")));
1170/// assert!(!zip::read::root_dir_common_filter(Path::new("__MACOSX/foo.txt")));
1171/// ```
1172#[must_use]
1173pub fn root_dir_common_filter(path: &Path) -> bool {
1174    const COMMON_FILTER_ROOT_FILES: &[&str] = &[".DS_Store", "Thumbs.db"];
1175
1176    if path.starts_with("__MACOSX") {
1177        return false;
1178    }
1179
1180    if path.components().count() == 1
1181        && path.file_name().is_some_and(|file_name| {
1182            COMMON_FILTER_ROOT_FILES
1183                .iter()
1184                .map(OsStr::new)
1185                .any(|cmp| cmp == file_name)
1186        })
1187    {
1188        return false;
1189    }
1190
1191    true
1192}
1193
1194#[cfg(test)]
1195mod tests {
1196    use std::io::Cursor;
1197
1198    /// Only on little endian because we cannot use fs with miri CI
1199    #[cfg(all(target_endian = "little", not(miri)))]
1200    #[test]
1201    fn test_is_symlink() -> std::io::Result<()> {
1202        use super::ZipArchive;
1203        use tempfile::TempDir;
1204
1205        let mut reader = ZipArchive::new(Cursor::new(include_bytes!("../tests/data/symlink.zip")))?;
1206        assert!(reader.by_index(0)?.is_symlink());
1207        let tempdir = TempDir::with_prefix("test_is_symlink")?;
1208        reader.extract(&tempdir)?;
1209        assert!(tempdir.path().join("bar").is_symlink());
1210        Ok(())
1211    }
1212
1213    #[test]
1214    #[cfg(feature = "deflate-flate2")]
1215    fn test_utf8_extra_field() {
1216        use super::ZipArchive;
1217
1218        let mut reader =
1219            ZipArchive::new(Cursor::new(include_bytes!("../tests/data/chinese.zip"))).unwrap();
1220        reader.by_name("七个房间.txt").unwrap();
1221    }
1222
1223    #[test]
1224    fn test_utf8() {
1225        use super::ZipArchive;
1226
1227        let mut reader =
1228            ZipArchive::new(Cursor::new(include_bytes!("../tests/data/linux-7z.zip"))).unwrap();
1229        reader.by_name("你好.txt").unwrap();
1230    }
1231
1232    #[test]
1233    fn test_utf8_2() {
1234        use super::ZipArchive;
1235
1236        let mut reader = ZipArchive::new(Cursor::new(include_bytes!(
1237            "../tests/data/windows-7zip.zip"
1238        )))
1239        .unwrap();
1240        reader.by_name("你好.txt").unwrap();
1241    }
1242
1243    /// Only on little endian because it runs too long with Miri CI
1244    #[cfg(all(target_endian = "little", not(miri)))]
1245    #[test]
1246    fn test_64k_files() -> crate::result::ZipResult<()> {
1247        use super::ZipArchive;
1248        use crate::CompressionMethod::Stored;
1249        use crate::ZipWriter;
1250        use crate::types::SimpleFileOptions;
1251        use std::io::{Read, Write};
1252
1253        let mut writer = ZipWriter::new(Cursor::new(Vec::new()));
1254        let options = SimpleFileOptions {
1255            compression_method: Stored,
1256            ..Default::default()
1257        };
1258        for i in 0..=u16::MAX {
1259            let file_name = format!("{i}.txt");
1260            writer.start_file(&*file_name, options)?;
1261            writer.write_all(i.to_string().as_bytes())?;
1262        }
1263
1264        let mut reader = ZipArchive::new(writer.finish()?)?;
1265        for i in 0..=u16::MAX {
1266            let expected_name = format!("{i}.txt");
1267            let expected_contents = i.to_string();
1268            let expected_contents = expected_contents.as_bytes();
1269            let mut file = reader.by_name(&expected_name)?;
1270            let mut contents = Vec::with_capacity(expected_contents.len());
1271            file.read_to_end(&mut contents)?;
1272            assert_eq!(contents, expected_contents);
1273            drop(file);
1274            contents.clear();
1275            let mut file = reader.by_index(i as usize)?;
1276            file.read_to_end(&mut contents)?;
1277            assert_eq!(contents, expected_contents);
1278        }
1279        Ok(())
1280    }
1281
1282    /// Symlinks being extracted shouldn't be followed out of the destination directory.
1283    /// Only on little endian because we cannot use fs with miri CI
1284    #[cfg(all(target_endian = "little", not(miri)))]
1285    #[test]
1286    fn test_cannot_symlink_outside_destination() -> crate::result::ZipResult<()> {
1287        use crate::ZipWriter;
1288        use crate::types::SimpleFileOptions;
1289        use std::fs::create_dir;
1290        use tempfile::TempDir;
1291
1292        let mut writer = ZipWriter::new(Cursor::new(Vec::new()));
1293        writer.add_symlink("symlink/", "../dest-sibling/", SimpleFileOptions::default())?;
1294        writer.start_file("symlink/dest-file", SimpleFileOptions::default())?;
1295        let mut reader = writer.finish_into_readable()?;
1296        let dest_parent = TempDir::with_prefix("read__test_cannot_symlink_outside_destination")?;
1297        let dest_sibling = dest_parent.path().join("dest-sibling");
1298        create_dir(&dest_sibling)?;
1299        let dest = dest_parent.path().join("dest");
1300        create_dir(&dest)?;
1301        assert!(reader.extract(dest).is_err());
1302        assert!(!dest_sibling.join("dest-file").exists());
1303        Ok(())
1304    }
1305
1306    /// Only on little endian because we cannot use fs with miri CI
1307    #[cfg(all(target_endian = "little", not(miri)))]
1308    #[test]
1309    fn test_can_create_destination() -> crate::result::ZipResult<()> {
1310        use super::ZipArchive;
1311        use tempfile::TempDir;
1312
1313        let mut reader =
1314            ZipArchive::new(Cursor::new(include_bytes!("../tests/data/mimetype.zip")))?;
1315        let dest = TempDir::with_prefix("read__test_can_create_destination")?;
1316        reader.extract(&dest)?;
1317        assert!(dest.path().join("mimetype").exists());
1318        Ok(())
1319    }
1320}