diff --git a/Cargo.lock b/Cargo.lock index 2a8476a15..336364b90 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -442,6 +442,12 @@ dependencies = [ "winapi", ] +[[package]] +name = "core_detect" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7f8f80099a98041a3d1622845c271458a2d73e688351bf3cb999266764b81d48" + [[package]] name = "cpubits" version = "0.1.1" @@ -564,6 +570,20 @@ version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" +[[package]] +name = "encoding_rs" +version = "0.8.42" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e985e0451871ad22fb8d2b6b076e2028a502a0d3950998c2c5c0a4f9b5d9679" +dependencies = [ + "cfg-if", + "core_detect", + "multiversion_no_op", + "rustversion", + "scopeguard", + "simdutf8", +] + [[package]] name = "enumflags2" version = "0.7.12" @@ -1039,6 +1059,12 @@ dependencies = [ "simd-adler32", ] +[[package]] +name = "multiversion_no_op" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "743fb55ba31b18fb1ecef6bdc9aa2743314978ac084044301a7eee33fb99a20d" + [[package]] name = "nanorand" version = "0.7.0" @@ -1118,6 +1144,8 @@ dependencies = [ "clap_complete", "clap_complete_nushell", "clap_mangen", + "crc32fast", + "encoding_rs", "file_type_enum", "filetime_creation", "flate2", @@ -1150,6 +1178,7 @@ dependencies = [ "tempfile", "test-strategy", "time", + "typed-path", "unrar-ng", "zip", "zstd 0.14.0", @@ -1548,6 +1577,12 @@ version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + [[package]] name = "similar" version = "2.7.0" diff --git a/Cargo.toml b/Cargo.toml index a6c7f36da..4e9d842f8 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -20,6 +20,7 @@ bstr = { version = "1.12.1", default-features = false, features = ["std"] } bzip2 = "0.6.1" bzip3 = { version = "0.12.0", features = ["bundled"], optional = true } clap = { version = "4.6.1", features = ["derive", "env"] } +encoding_rs = "0.8.35" file_type_enum = "3.0.1" filetime_creation = "0.2" flate2 = { version = "1.1.9", default-features = false } @@ -42,6 +43,7 @@ strum = { version = "0.28.0", features = ["derive"] } tar = "0.4.46" tempfile = "3.27.0" time = { version = "0.3.47", default-features = false, features = ["local-offset"] } +typed-path = "0.12.3" unrar = { package = "unrar-ng", version = "0.7.6", optional = true } zip = { version = "8.6.0", default-features = false, features = [ "time", @@ -56,6 +58,7 @@ landlock = "0.4.5" [dev-dependencies] anyhow = "1.0.102" assert_cmd = "2.2.1" +crc32fast = "1.5.0" glob = "0.3.3" infer = "0.22.0" insta = { version = "1.47.2", features = ["filters"] } diff --git a/src/archive/zip.rs b/src/archive/zip.rs index 1a27665dd..2031c002a 100644 --- a/src/archive/zip.rs +++ b/src/archive/zip.rs @@ -3,16 +3,19 @@ #[cfg(unix)] use std::os::unix::fs::PermissionsExt; use std::{ + borrow::Cow, io::{self, prelude::*}, path::{Path, PathBuf}, }; +use encoding_rs::Encoding; use filetime_creation::{FileTime, set_file_mtime}; use fs_err as fs; use is_executable::is_executable; use same_file::Handle; use time::{OffsetDateTime, PrimitiveDateTime, UtcOffset}; -use zip::{self, DateTime, ZipArchive, read::ZipFile}; +use typed_path::{Utf8WindowsComponent, Utf8WindowsPath}; +use zip::{self, DateTime, HasZipMetadata, ZipArchive, read::ZipFile}; #[cfg(unix)] use crate::utils::sanitize_archive_mode; @@ -32,10 +35,14 @@ use crate::{ /// Unpacks the archive given by `archive` into the folder given by `output_folder`. /// Assumes that output_folder is empty +/// +/// `name_encoding` is the optional charset used to decode entry names whose UTF-8 +/// flag is unset (see `--encoding`). pub fn unpack_archive( reader: R, output_folder: &Path, password: Option<&[u8]>, + name_encoding: Option<&'static Encoding>, question_policy: QuestionPolicy, ) -> Result where @@ -49,10 +56,11 @@ where Some(password) => archive.by_index_decrypt(idx, password)?, None => archive.by_index(idx)?, }; - let relpath = match file.enclosed_name() { - Some(path) => path.to_owned(), + let entry_name = decoded_entry_name(&file, name_encoding); + let relpath = match enclosed_name(&entry_name) { + Some(path) => path, None => { - warning!("skipping entry {} with unsafe name: {}", idx, file.name()); + warning!("skipping entry {} with unsafe name: {}", idx, entry_name); continue; } }; @@ -61,9 +69,9 @@ where validate_dest_inside_root(output_folder, &file_path)?; - display_zip_comment_if_exists(&file); + display_zip_comment_if_exists(&file, &entry_name); - match file.name().ends_with('/') { + match file.is_dir() { _is_dir @ true => { info!("Directory {} created", PathFmt(&file_path)); @@ -149,6 +157,7 @@ where pub fn list_archive( mut archive: ZipArchive, password: Option<&[u8]>, + name_encoding: Option<&'static Encoding>, ) -> impl Iterator> where R: Read + Seek, @@ -166,7 +175,10 @@ where Err(e) => return Err(e.into()), }; - let path = file.enclosed_name().unwrap_or_else(|| file.mangled_name()).to_owned(); + let path = { + let entry_name = decoded_entry_name(&file, name_encoding); + enclosed_name(&entry_name).unwrap_or_else(|| mangled_name(&entry_name)) + }; let size = Some(file.size()); let file_type = if file.is_dir() { @@ -319,7 +331,83 @@ fn symlink_target_from_bytes(bytes: &[u8]) -> PathBuf { } } -fn display_zip_comment_if_exists(file: &ZipFile<'_, R>) { +/// Decode an entry's file name. +/// +/// Names carrying the UTF-8 flag (or fixed up through the Unicode Path extra field) +/// are already decoded by the `zip` crate. For the rest, the raw bytes are decoded +/// with `fallback_encoding` when one is given; otherwise they are taken as UTF-8 if +/// they are valid UTF-8, or decoded as CP437, which is what the ZIP specification +/// mandates when the UTF-8 flag is unset. +fn decoded_entry_name<'a, R: Read + ?Sized>( + file: &'a ZipFile<'a, R>, + fallback_encoding: Option<&'static Encoding>, +) -> Cow<'a, str> { + if file.get_metadata().is_utf8 { + return Cow::Borrowed(file.name()); + } + let raw = file.name_raw(); + if let Some(encoding) = fallback_encoding { + let (decoded, _, had_errors) = encoding.decode(raw); + if had_errors { + warning!( + "Failed to decode entry name {} with {}, some characters were replaced with U+FFFD", + String::from_utf8_lossy(raw), + encoding.name() + ); + } + return decoded; + } + match std::str::from_utf8(raw) { + Ok(name) => Cow::Borrowed(name), + Err(_) => Cow::Borrowed(file.name()), + } +} + +/// Equivalent to the `zip` crate's [`ZipFile::enclosed_name`], but evaluated on an +/// already-decoded entry name instead of `file.name()`. +/// +/// Returns `None` if the name contains a NUL byte, is an absolute path, or could +/// escape the destination directory. +fn enclosed_name(file_name: &str) -> Option { + if file_name.contains('\0') { + return None; + } + let mut depth = 0usize; + let mut out_path = PathBuf::new(); + for component in Utf8WindowsPath::new(file_name).components() { + match component { + Utf8WindowsComponent::Prefix(_) | Utf8WindowsComponent::RootDir => return None, + Utf8WindowsComponent::ParentDir => { + depth = depth.checked_sub(1)?; + out_path.pop(); + } + Utf8WindowsComponent::Normal(s) => { + depth += 1; + out_path.push(s); + } + Utf8WindowsComponent::CurDir => (), + } + } + Some(out_path) +} + +/// Equivalent to the `zip` crate's [`ZipFile::mangled_name`], but evaluated on an +/// already-decoded entry name. +fn mangled_name(file_name: &str) -> PathBuf { + let no_null_filename = match file_name.find('\0') { + Some(index) => &file_name[0..index], + None => file_name, + }; + Utf8WindowsPath::new(no_null_filename) + .components() + .filter_map(|component| match component { + Utf8WindowsComponent::Normal(s) => Some(s), + _ => None, + }) + .collect() +} + +fn display_zip_comment_if_exists(file: &ZipFile<'_, R>, entry_name: &str) { let comment = file.comment(); if !comment.is_empty() { // Zip file comments seem to be pretty rare, but if they are used, @@ -332,7 +420,7 @@ fn display_zip_comment_if_exists(file: &ZipFile<'_, R>) { // the future, maybe asking the user if he wants to display the comment // (informing him of its size) would be sensible for both normal and // accessibility mode.. - info_accessible!("Found comment in {}: {}", file.name(), comment); + info_accessible!("Found comment in {}: {}", entry_name, comment); } } @@ -446,4 +534,177 @@ mod tests { let file_time = file_time_from_local_datetime(local, resolved_offset); assert_eq!(file_time.unix_seconds(), instant.unix_timestamp()); } + + /// Build a minimal stored zip whose entry names are the exact raw bytes given. + /// `flags` is written to the general purpose bit flag; `0` means "not UTF-8". + fn build_zip(entries: &[(&[u8], &[u8], u16)]) -> Vec { + let mut out = Vec::new(); + let mut central = Vec::new(); + for &(name, data, flags) in entries { + let crc = crc32fast::hash(data); + let offset = out.len() as u32; + + // Local file header + out.extend_from_slice(&0x0403_4b50u32.to_le_bytes()); + out.extend_from_slice(&20u16.to_le_bytes()); // version needed + out.extend_from_slice(&flags.to_le_bytes()); + out.extend_from_slice(&0u16.to_le_bytes()); // method: stored + out.extend_from_slice(&0u16.to_le_bytes()); // mod time + out.extend_from_slice(&0u16.to_le_bytes()); // mod date + out.extend_from_slice(&crc.to_le_bytes()); + out.extend_from_slice(&(data.len() as u32).to_le_bytes()); + out.extend_from_slice(&(data.len() as u32).to_le_bytes()); + out.extend_from_slice(&(name.len() as u16).to_le_bytes()); + out.extend_from_slice(&0u16.to_le_bytes()); // extra field length + out.extend_from_slice(name); + out.extend_from_slice(data); + + // Central directory entry + central.extend_from_slice(&0x0201_4b50u32.to_le_bytes()); + central.extend_from_slice(&20u16.to_le_bytes()); // version made by + central.extend_from_slice(&20u16.to_le_bytes()); // version needed + central.extend_from_slice(&flags.to_le_bytes()); + central.extend_from_slice(&0u16.to_le_bytes()); // method: stored + central.extend_from_slice(&0u16.to_le_bytes()); // mod time + central.extend_from_slice(&0u16.to_le_bytes()); // mod date + central.extend_from_slice(&crc.to_le_bytes()); + central.extend_from_slice(&(data.len() as u32).to_le_bytes()); + central.extend_from_slice(&(data.len() as u32).to_le_bytes()); + central.extend_from_slice(&(name.len() as u16).to_le_bytes()); + central.extend_from_slice(&0u16.to_le_bytes()); // extra field length + central.extend_from_slice(&0u16.to_le_bytes()); // comment length + central.extend_from_slice(&0u16.to_le_bytes()); // disk number + central.extend_from_slice(&0u16.to_le_bytes()); // internal attributes + central.extend_from_slice(&0u32.to_le_bytes()); // external attributes + central.extend_from_slice(&offset.to_le_bytes()); + central.extend_from_slice(name); + } + + // End of central directory + let cd_offset = out.len() as u32; + out.extend_from_slice(¢ral); + out.extend_from_slice(&0x0605_4b50u32.to_le_bytes()); + out.extend_from_slice(&0u16.to_le_bytes()); // disk number + out.extend_from_slice(&0u16.to_le_bytes()); // central dir disk + out.extend_from_slice(&(entries.len() as u16).to_le_bytes()); + out.extend_from_slice(&(entries.len() as u16).to_le_bytes()); + out.extend_from_slice(&(central.len() as u32).to_le_bytes()); + out.extend_from_slice(&cd_offset.to_le_bytes()); + out.extend_from_slice(&0u16.to_le_bytes()); // comment length + out + } + + /// Extract `zip_bytes` into a fresh tempdir and return every extracted file name. + fn unpack_and_collect_names(zip_bytes: Vec, encoding: Option<&'static Encoding>) -> Vec { + let dir = tempfile::tempdir().unwrap(); + unpack_archive( + io::Cursor::new(zip_bytes), + dir.path(), + None, + encoding, + QuestionPolicy::AlwaysYes, + ) + .unwrap(); + + fn names(dir: &Path, out: &mut Vec) { + for entry in std::fs::read_dir(dir).unwrap() { + let entry = entry.unwrap(); + if entry.file_type().unwrap().is_dir() { + names(&entry.path(), out); + } else { + out.push(entry.file_name().to_string_lossy().into_owned()); + } + } + } + let mut out = Vec::new(); + names(dir.path(), &mut out); + out + } + + const UTF8_FLAG: u16 = 0x0800; + + /// GBK (CP936) bytes of "这是一个测试文件.txt", as produced by `convmv -t CP936` + /// followed by `zip` on a system without UTF-8 filenames (issue #691). + fn gbk_name_bytes() -> Vec { + let (encoded, _, had_errors) = Encoding::for_label(b"gbk").unwrap().encode("这是一个测试文件.txt"); + assert!(!had_errors); + encoded.into_owned() + } + + #[test] + fn extracts_non_utf8_name_with_encoding_option() { + let zip_bytes = build_zip(&[(&gbk_name_bytes(), b"hello", 0)]); + let names = unpack_and_collect_names(zip_bytes, Encoding::for_label(b"gbk")); + assert_eq!(names, ["这是一个测试文件.txt"]); + } + + #[test] + fn lists_non_utf8_name_with_encoding_option() { + let zip_bytes = build_zip(&[(&gbk_name_bytes(), b"hello", 0)]); + let archive = ZipArchive::new(io::Cursor::new(zip_bytes)).unwrap(); + let paths: Vec = list_archive(archive, None, Encoding::for_label(b"gbk")) + .map(|entry| entry.unwrap().path) + .collect(); + assert_eq!(paths, [PathBuf::from("这是一个测试文件.txt")]); + } + + #[test] + fn unmarked_utf8_name_is_recovered_without_encoding_option() { + // Some tools write UTF-8 names but forget to set the UTF-8 flag (issue #691). + let zip_bytes = build_zip(&[("Schwarz-weiß".as_bytes(), b"hi", 0)]); + let names = unpack_and_collect_names(zip_bytes, None); + assert_eq!(names, ["Schwarz-weiß"]); + } + + #[test] + fn unmarked_non_utf8_name_defaults_to_cp437() { + // Without --encoding, non-UTF-8 names keep the spec-mandated CP437 decoding. + let zip_bytes = build_zip(&[(&gbk_name_bytes(), b"hello", 0)]); + let mut archive = ZipArchive::new(io::Cursor::new(zip_bytes.clone())).unwrap(); + let cp437_name = archive.by_index(0).unwrap().name().to_owned(); + + let names = unpack_and_collect_names(zip_bytes, None); + assert_eq!(names, [cp437_name]); + } + + #[test] + fn utf8_flagged_name_ignores_encoding_option() { + // Names flagged as UTF-8 always decode as UTF-8, per the ZIP spec. + let zip_bytes = build_zip(&[("Schwarz-weiß".as_bytes(), b"hi", UTF8_FLAG)]); + let names = unpack_and_collect_names(zip_bytes, Encoding::for_label(b"gbk")); + assert_eq!(names, ["Schwarz-weiß"]); + } + + #[test] + fn enclosed_name_rejects_unsafe_paths() { + assert_eq!(enclosed_name("a/b/c.txt"), Some(PathBuf::from("a/b/c.txt"))); + assert_eq!(enclosed_name("../evil.txt"), None); + assert_eq!(enclosed_name("a/../../evil.txt"), None); + assert_eq!(enclosed_name("evil\0.txt"), None); + // Backslashes count as separators, like the `zip` crate's own check. + assert_eq!(enclosed_name("..\\evil.txt"), None); + // Absolute paths are rejected, not silently made relative. + assert_eq!(enclosed_name("/abs/evil.txt"), None); + assert_eq!(enclosed_name("C:/abs/evil.txt"), None); + assert_eq!(enclosed_name("C:\\abs\\evil.txt"), None); + } + + /// An entry with an absolute name is skipped instead of being extracted + /// under the output directory. + #[test] + fn skips_entry_with_absolute_name() { + let zip_bytes = build_zip(&[("/abs/evil.txt".as_bytes(), b"evil", UTF8_FLAG)]); + let names = unpack_and_collect_names(zip_bytes, None); + assert!(names.is_empty()); + } + + /// A wrong `--encoding` decodes with replacement chars; the entry is still + /// extracted and a warning is emitted (not observable by this harness). + #[test] + fn wrong_encoding_extracts_with_replacement_chars() { + let zip_bytes = build_zip(&[(&gbk_name_bytes(), b"hello", 0)]); + let names = unpack_and_collect_names(zip_bytes, Encoding::for_label(b"utf-8")); + assert_eq!(names.len(), 1); + assert!(names[0].contains('\u{FFFD}')); + } } diff --git a/src/cli/args.rs b/src/cli/args.rs index 98012a1ae..b5bbdc9c1 100644 --- a/src/cli/args.rs +++ b/src/cli/args.rs @@ -109,6 +109,11 @@ pub enum Subcommand { /// Remove the source file after successful decompression #[arg(short = 'r', long)] remove: bool, + + /// Charset used to decode entry names not marked as UTF-8 (zip only); + /// accepts labels such as "gbk", "big5", "shift_jis", "windows-1251" or "cp936" + #[arg(long, value_name = "ENCODING")] + encoding: Option, }, /// List contents of an archive #[command(visible_aliases = ["l", "ls"])] @@ -127,6 +132,11 @@ pub enum Subcommand { /// Only list entries up to this recursion limit #[arg(long)] depth: Option, + + /// Charset used to decode entry names not marked as UTF-8 (zip only); + /// accepts labels such as "gbk", "big5", "shift_jis", "windows-1251" or "cp936" + #[arg(long, value_name = "ENCODING")] + encoding: Option, }, } @@ -172,6 +182,7 @@ mod tests { output_dir: None, here: false, remove: false, + encoding: None, }, } } @@ -186,6 +197,7 @@ mod tests { output_dir: None, here: false, remove: false, + encoding: None, }, ..mock_cli_args() } @@ -198,6 +210,7 @@ mod tests { output_dir: None, here: false, remove: false, + encoding: None, }, ..mock_cli_args() } @@ -210,6 +223,20 @@ mod tests { output_dir: None, here: false, remove: false, + encoding: None, + }, + ..mock_cli_args() + } + ); + test!( + "ouch d archive.zip --encoding gbk", + CliArgs { + cmd: Subcommand::Decompress { + files: to_paths(["archive.zip"]), + output_dir: None, + here: false, + remove: false, + encoding: Some("gbk".to_string()), }, ..mock_cli_args() } diff --git a/src/commands/decompress.rs b/src/commands/decompress.rs index d7ed63f99..196f8cd1e 100644 --- a/src/commands/decompress.rs +++ b/src/commands/decompress.rs @@ -22,6 +22,7 @@ use crate::{ io::{ReadSeek, lock_and_flush_output_stdio}, is_path_stdin, resolve_path_conflict, user_wants_to_continue, }, + warning, }; pub struct DecompressOptions<'a> { @@ -38,6 +39,9 @@ pub struct DecompressOptions<'a> { pub here: bool, pub question_policy: QuestionPolicy, pub password: Option<&'a [u8]>, + /// `--encoding`: charset for archive entry names not flagged as UTF-8. + /// Currently only used for zip archives. + pub archive_encoding: Option<&'static encoding_rs::Encoding>, pub remove: bool, /// Resolved target prepared upfront (before sandbox is applied). pub prepared: PreparedTarget, @@ -184,6 +188,14 @@ pub fn decompress_file(options: DecompressOptions) -> Result<()> { return Ok(()); }; + // The archive format sits at the front, so --encoding only reaches zip inputs + if options.archive_encoding.is_some() && !matches!(first_extension, Zip) { + warning!( + "The --encoding option only applies to zip archives and is ignored for {}", + PathFmt(options.input_file_path) + ); + } + let control_flow = match first_extension { Gzip | Bzip | Bzip3 | Lz4 | Lzma | Xz | Lzip | Snappy | Zstd | Brotli => { let reader = create_decoder_up_to_first_extension()?; @@ -211,11 +223,20 @@ pub fn decompress_file(options: DecompressOptions) -> Result<()> { dir, )?, Zip | SevenZip => { - let unpack_fn = match first_extension { - Zip => crate::archive::zip::unpack_archive, - SevenZip => crate::archive::sevenz::unpack_archive, - _ => unreachable!(), - }; + let unpack_fn: Box, &Path, Option<&[u8]>, QuestionPolicy) -> Result> = + match first_extension { + Zip => Box::new(|reader, output_dir, password, question_policy| { + crate::archive::zip::unpack_archive( + reader, + output_dir, + password, + options.archive_encoding, + question_policy, + ) + }), + SevenZip => Box::new(crate::archive::sevenz::unpack_archive), + _ => unreachable!(), + }; let should_load_everything_into_memory = input_is_stdin || !extensions.is_empty(); diff --git a/src/commands/list.rs b/src/commands/list.rs index 9de520400..613aa26a1 100644 --- a/src/commands/list.rs +++ b/src/commands/list.rs @@ -12,9 +12,10 @@ use crate::{ list::{self, FileInArchive, ListOptions}, non_archive::lz4::MultiFrameLz4Decoder, utils::{ - LZMA_MEMLIMIT_BYTES, LimitedReader, copy_limited_decompression, io::lock_and_flush_output_stdio, + LZMA_MEMLIMIT_BYTES, LimitedReader, PathFmt, copy_limited_decompression, io::lock_and_flush_output_stdio, user_wants_to_continue, }, + warning, }; /// File at archive_path is opened for reading, example: "archive.tar.gz" @@ -25,6 +26,8 @@ pub fn list_archive_contents( list_options: ListOptions, question_policy: QuestionPolicy, password: Option<&[u8]>, + // `--encoding`: charset for zip entry names not flagged as UTF-8. + archive_encoding: Option<&'static encoding_rs::Encoding>, // Pre-opened tempfile FD for multi-format RAR list under the sandbox. #[allow(unused_variables)] rar_spill_tempfile: Option, ) -> Result<()> { @@ -39,7 +42,7 @@ pub fn list_archive_contents( // Any other Zip decompression done can take up the whole RAM and freeze ouch. if let &[Zip] = formats.as_slice() { let zip_archive = zip::ZipArchive::new(reader)?; - let files = crate::archive::zip::list_archive(zip_archive, password); + let files = crate::archive::zip::list_archive(zip_archive, password, archive_encoding); list::list_files(archive_path, files, list_options)?; return Ok(()); } @@ -87,6 +90,13 @@ pub fn list_archive_contents( } let archive_format = misplaced_archive_format.unwrap_or(formats[0]); + if archive_encoding.is_some() && !matches!(archive_format, Zip) { + warning!( + "The --encoding option only applies to zip archives and is ignored for {}", + PathFmt(archive_path) + ); + } + let files: Box>> = match archive_format { Tar => { let limited = LimitedReader::new(reader); @@ -108,7 +118,11 @@ pub fn list_archive_contents( copy_limited_decompression(&mut reader, &mut vec)?; let zip_archive = zip::ZipArchive::new(io::Cursor::new(vec))?; - Box::new(crate::archive::zip::list_archive(zip_archive, password)) + Box::new(crate::archive::zip::list_archive( + zip_archive, + password, + archive_encoding, + )) } #[cfg(feature = "unrar")] Rar => { diff --git a/src/commands/mod.rs b/src/commands/mod.rs index cd3e6bf9a..4a331eeb3 100644 --- a/src/commands/mod.rs +++ b/src/commands/mod.rs @@ -28,6 +28,32 @@ use crate::{ warning, }; +/// Resolve an `--encoding` label (e.g. "gbk" or "cp936") into its [`encoding_rs::Encoding`]. +/// +/// Accepts WHATWG encoding labels, plus the familiar code page names +/// ("cp936", "windows-1252", ...) that WHATWG does not assign labels to. +fn parse_archive_encoding(label: &str) -> Result<&'static encoding_rs::Encoding> { + let lowercased = label.trim().to_ascii_lowercase(); + let encoding = encoding_rs::Encoding::for_label(lowercased.as_bytes()).or_else(|| { + let number = lowercased + .strip_prefix("cp") + .or_else(|| lowercased.strip_prefix("windows-"))?; + match number { + "932" => Some(encoding_rs::SHIFT_JIS), + "936" => Some(encoding_rs::GBK), + "949" => Some(encoding_rs::EUC_KR), + "950" => Some(encoding_rs::BIG5), + "65001" => Some(encoding_rs::UTF_8), + number => encoding_rs::Encoding::for_label(format!("windows-{number}").as_bytes()), + } + }); + encoding.ok_or_else(|| { + FinalError::with_title(format!("Unknown encoding: \"{label}\"")) + .detail("Expected a WHATWG encoding label or code page name such as \"gbk\", \"big5\", \"shift_jis\", \"windows-1251\" or \"cp936\"") + .into() + }) +} + /// Warn the user that (de)compressing this .zip archive might freeze their system. fn warn_user_about_loading_zip_in_memory() { const ZIP_IN_MEMORY_LIMITATION_WARNING: &str = "\n \ @@ -177,7 +203,9 @@ pub fn run(args: CliArgs, question_policy: QuestionPolicy, file_visibility_polic output_dir, here, remove, + encoding, } => { + let archive_encoding = encoding.as_deref().map(parse_archive_encoding).transpose()?; let mut files_output_paths: Vec<_> = vec![]; let mut files_extensions: Vec> = vec![]; @@ -341,6 +369,7 @@ pub fn run(args: CliArgs, question_policy: QuestionPolicy, file_visibility_polic password: args.password.as_deref().map(|str| { <[u8] as ByteSlice>::from_os_str(str).expect("convert password to bytes failed") }), + archive_encoding, remove, prepared, }) @@ -363,7 +392,9 @@ pub fn run(args: CliArgs, question_policy: QuestionPolicy, file_visibility_polic tree, show_size, depth, + encoding, } => { + let archive_encoding = encoding.as_deref().map(parse_archive_encoding).transpose()?; let mut formats = vec![]; if let Some(format) = args.format { @@ -463,6 +494,7 @@ pub fn run(args: CliArgs, question_policy: QuestionPolicy, file_visibility_polic args.password .as_deref() .map(|str| <[u8] as ByteSlice>::from_os_str(str).expect("convert password to bytes failed")), + archive_encoding, spill.take(), )?; } diff --git a/tests/integration.rs b/tests/integration.rs index 2acafa78d..784ea6e78 100644 --- a/tests/integration.rs +++ b/tests/integration.rs @@ -2217,3 +2217,30 @@ fn merging_a_rar_asks_before_replacing_each_file() { .success(); assert_eq!("Testing 123\n", fs::read_to_string(out.join("testfile.txt")).unwrap()); } + +// `--encoding` only applies to zip archives; on other formats the user must be warned +#[test] +fn decompress_encoding_flag_warns_for_non_zip_archive() { + let (_tempdir, dir) = testdir().unwrap(); + + fs::write(dir.join("file.txt"), b"hello").unwrap(); + crate::utils::cargo_bin() + .current_dir(dir) + .args(["compress", "file.txt", "archive.tar.gz", "--yes"]) + .assert() + .success(); + + let output = crate::utils::cargo_bin() + .current_dir(dir) + .args(["decompress", "archive.tar.gz", "--encoding", "gbk", "--yes"]) + .assert() + .success() + .get_output() + .clone(); + + let stderr = String::from_utf8(output.stderr).unwrap(); + assert!( + stderr.contains("--encoding"), + "expected a warning that --encoding is ignored for non-zip archives" + ); +}