diff --git a/crates/socket-patch-cli/tests/in_process_rollback_hosted.rs b/crates/socket-patch-cli/tests/in_process_rollback_hosted.rs index 070480ecc..d09fce8fa 100644 --- a/crates/socket-patch-cli/tests/in_process_rollback_hosted.rs +++ b/crates/socket-patch-cli/tests/in_process_rollback_hosted.rs @@ -758,6 +758,27 @@ async fn pypi_requirements_hosted_round_trip() { pypi_requirements_round_trip("flask==2.0.1\nrequests==2.31.0\n", &[]).await; } +/// REGRESSION (#1212): the same round trip when the file opens with a +/// PEP 263 coding line naming a codec beyond UTF-8 / ASCII / Latin-1 / +/// cp1252 (a header copied from a template; the bytes are plain ASCII). +/// pip decodes it with that codec, so `get --mode hosted` must wire it, +/// `vex` attest it and `rollback` restore it (they read it as absent: +/// nothing wired, then exit 2 and `manifest_not_found`). +#[tokio::test] +#[serial] +async fn pypi_requirements_hosted_round_trip_with_an_unmodelled_coding_line() { + pypi_requirements_round_trip( + "# -*- coding: iso-8859-15 -*-\nflask==2.0.1\nrequests==2.31.0\n", + &[], + ) + .await; + pypi_requirements_round_trip( + "# deps\n# vim: set fileencoding=cp1250 :\nflask==2.0.1\nrequests==2.31.0\n", + &[], + ) + .await; +} + /// REGRESSION (#1086): the same round trip when an in-root `-r` include /// pins the same `requests==2.31.0` (split base/dev files), with the `-r` /// line before and after the root pin. pip reads the root and its includes diff --git a/crates/socket-patch-cli/tests/scan_requirements_lock_only.rs b/crates/socket-patch-cli/tests/scan_requirements_lock_only.rs index 24953fa2a..96eeaf2fe 100644 --- a/crates/socket-patch-cli/tests/scan_requirements_lock_only.rs +++ b/crates/socket-patch-cli/tests/scan_requirements_lock_only.rs @@ -10,6 +10,8 @@ //! `pip freeze >` output), which pip decodes. //! * #1119: a file pip decodes through a PEP 263 coding line //! (`# -*- coding: latin-1 -*-`), as the root file or an include. +//! * #1212: a plain-ASCII file whose coding line names any other +//! ASCII-compatible codec (`iso-8859-15`, `cp1250`, `mac-roman`, `gbk`). //! * #994: include targets pip unquotes (`-r "dev reqs.txt"`, //! `--requirement="dev.txt"`, `-r dev\ reqs.txt`) or expands //! (`-r ${REQDIR}/dev.txt`); @@ -240,6 +242,37 @@ async fn lock_only_scan_discovers_pep_263_pins() { .await; } +/// #1212: a coding line naming a codec the decoder has no table for +/// (copied from a template; the file itself is plain ASCII) decodes like +/// ASCII under every ASCII-compatible codec, as it does for pip, so the +/// pins are discovered instead of reading as "No packages found". +#[tokio::test] +async fn lock_only_scan_discovers_ascii_pins_under_any_ascii_coding_line() { + for coding in ["iso-8859-15", "latin9", "cp1250", "mac-roman", "gbk"] { + let root = format!("# -*- coding: {coding} -*-\nsp-fixture-six==1.16.0\n"); + assert_lock_only_discovers_bytes( + &[("requirements.txt", root.as_bytes())], + &["pkg:pypi/sp-fixture-six@1.16.0"], + ) + .await; + let dev = format!("# coding={coding}\nsp-fixture-six==1.16.0\n"); + assert_lock_only_discovers_bytes( + &[ + ( + "requirements.txt", + &b"-r dev.txt\nsp-fixture-idna==3.7\n"[..], + ), + ("dev.txt", dev.as_bytes()), + ], + &[ + "pkg:pypi/sp-fixture-idna@3.7", + "pkg:pypi/sp-fixture-six@1.16.0", + ], + ) + .await; + } +} + /// #994: pip `shlex`-splits an include line's options, so a quoted or /// backslash-escaped target names the file without its quotes, and a /// target with a space is one path, not two words. diff --git a/crates/socket-patch-core/src/utils/requirements.rs b/crates/socket-patch-core/src/utils/requirements.rs index ab15ff0ab..1bc0c0029 100644 --- a/crates/socket-patch-core/src/utils/requirements.rs +++ b/crates/socket-patch-core/src/utils/requirements.rs @@ -61,6 +61,9 @@ pub(crate) fn decode(bytes: &[u8]) -> Option { .then(|| String::from_utf8_lossy(bytes).into_owned()), Codec::Latin1 => Some(bytes.iter().map(|&b| char::from(b)).collect()), Codec::Cp1252 => bytes.iter().map(|&b| cp1252_char(b)).collect(), + Codec::AsciiCompatible => bytes + .is_ascii() + .then(|| String::from_utf8_lossy(bytes).into_owned()), }; } } @@ -74,6 +77,9 @@ enum Codec { Ascii, Latin1, Cp1252, + /// A codec with no table here whose low half is ASCII: plain-ASCII + /// bytes decode as ASCII under it, anything else is unreadable. + AsciiCompatible, } /// The encoding name of pip's PEP 263 check: the first of the file's first @@ -104,8 +110,9 @@ fn coding_line(bytes: &[u8]) -> Option { /// Python's codec lookup for the names [`decode`] models: case-folded, /// `-` and ` ` read as `_`, then Python's alias table for UTF-8, ASCII, -/// Latin-1 and cp1252. Any other codec (pip would use it) is `None`, so -/// the file reads as undecodable rather than being guessed at. +/// Latin-1 and cp1252, then [`ASCII_COMPATIBLE_CODECS`] (#1212). Any +/// other codec (pip would use it) is `None`, so the file reads as +/// undecodable rather than being guessed at. fn coding_line_codec(name: &str) -> Option { let name = name.to_ascii_lowercase().replace(['-', ' '], "_"); Some(match name.as_str() { @@ -120,10 +127,70 @@ fn coding_line_codec(name: &str) -> Option { Codec::Latin1 } "cp1252" | "windows_1252" | "1252" => Codec::Cp1252, + _ if ASCII_COMPATIBLE_CODECS + .split_ascii_whitespace() + .any(|known| known == python_codec_key(&name)) => + { + Codec::AsciiCompatible + } _ => return None, }) } +/// A codec name as Python's `encodings.search_function` looks it up: +/// runs of anything but ASCII letters, digits and `.` become one `_` +/// (`normalize_encoding`), then `.` reads as `_`. +fn python_codec_key(name: &str) -> String { + name.split(|c: char| !(c.is_ascii_alphanumeric() || c == '.')) + .filter(|part| !part.is_empty()) + .collect::>() + .join("_") + .replace('.', "_") +} + +/// Every name (module and alias, as [`python_codec_key`] spells it) of +/// the Python 3.11 codecs, beyond those [`coding_line_codec`] models, +/// that decode each ASCII byte as itself: the ISO-8859, Windows, DOS, +/// Mac, KOI8 and CJK multibyte families. A header copied from a template +/// (`# -*- coding: iso-8859-15 -*-`) over a plain-ASCII file is the +/// common case, and pip reads it as ASCII. Left out, so still `None`: +/// codecs that read ASCII bytes differently (EBCDIC, cp864, UTF-16/32, +/// UTF-7, HZ, ISO-2022, Shift_JIS-2004, the escape codecs, idna, +/// punycode) and the bytes-to-bytes codecs pip cannot decode text with. +/// Space-separated, sorted. +const ASCII_COMPATIBLE_CODECS: &str = + "1125 1250 1251 1253 1254 1255 1256 1257 1258 437 775 850 852 855 857 858 860 861 862 863 \ + 865 866 869 932 936 949 950 arabic asmo_708 big5 big5_hkscs big5_tw big5hkscs chinese \ + cp1006 cp1051 cp1125 cp1250 cp1251 cp1253 cp1254 cp1255 cp1256 cp1257 cp1258 cp1361 \ + cp154 cp437 cp720 cp737 cp775 cp850 cp852 cp855 cp856 cp857 cp858 cp860 cp861 cp862 \ + cp863 cp865 cp866 cp866u cp869 cp874 cp932 cp936 cp949 cp950 cp_gr cp_is csbig5 csibm855 \ + csibm857 csibm858 csibm860 csibm861 csibm863 csibm865 csibm866 csibm869 csiso58gb231280 \ + csisolatin2 csisolatin3 csisolatin4 csisolatin5 csisolatin6 csisolatinarabic \ + csisolatincyrillic csisolatingreek csisolatinhebrew cskoi8r cspc775baltic \ + cspc850multilingual cspc862latinhebrew cspc8codepage437 cspcp852 csptcp154 csshiftjis \ + cyrillic cyrillic_asian ecma_114 ecma_118 elot_928 euc_cn euc_jis2004 euc_jis_2004 \ + euc_jisx0213 euc_jp euc_kr euccn eucgb2312_cn eucjis2004 eucjisx0213 eucjp euckr gb18030 \ + gb18030_2000 gb2312 gb2312_1980 gb2312_80 gbk greek greek8 hebrew hkscs hp_roman8 \ + ibm1051 ibm1125 ibm437 ibm775 ibm850 ibm852 ibm855 ibm857 ibm858 ibm860 ibm861 ibm862 \ + ibm863 ibm865 ibm866 ibm869 iso8859_10 iso8859_11 iso8859_13 iso8859_14 iso8859_15 \ + iso8859_16 iso8859_2 iso8859_3 iso8859_4 iso8859_5 iso8859_6 iso8859_7 iso8859_8 \ + iso8859_9 iso_8859_10 iso_8859_10_1992 iso_8859_11 iso_8859_11_2001 iso_8859_13 \ + iso_8859_14 iso_8859_14_1998 iso_8859_15 iso_8859_16 iso_8859_16_2001 iso_8859_2 \ + iso_8859_2_1987 iso_8859_3 iso_8859_3_1988 iso_8859_4 iso_8859_4_1988 iso_8859_5 \ + iso_8859_5_1988 iso_8859_6 iso_8859_6_1987 iso_8859_7 iso_8859_7_1987 iso_8859_8 \ + iso_8859_8_1988 iso_8859_9 iso_8859_9_1989 iso_celtic iso_ir_101 iso_ir_109 iso_ir_110 \ + iso_ir_126 iso_ir_127 iso_ir_138 iso_ir_144 iso_ir_148 iso_ir_157 iso_ir_166 iso_ir_199 \ + iso_ir_226 iso_ir_58 jisx0213 johab koi8_r koi8_t koi8_u korean ks_c_5601 ks_c_5601_1987 \ + ks_x_1001 ksc5601 ksx1001 kz1048 kz_1048 l10 l2 l3 l4 l5 l6 l7 l8 l9 latin10 latin2 \ + latin3 latin4 latin5 latin6 latin7 latin8 latin9 mac_arabic mac_centeuro mac_croatian \ + mac_cyrillic mac_farsi mac_greek mac_iceland mac_latin2 mac_roman mac_romanian \ + mac_turkish maccentraleurope maccyrillic macgreek maciceland macintosh maclatin2 \ + macroman macturkish ms1361 ms932 ms936 ms949 ms950 ms_kanji mskanji palmos pt154 ptcp154 \ + r8 rk1048 roman8 ruscii s_jis shift_jis shiftjis sjis strk1048_2002 thai tis620 tis_620 \ + tis_620_0 tis_620_2529_0 tis_620_2529_1 u_jis uhc ujis windows_1250 windows_1251 \ + windows_1253 windows_1254 windows_1255 windows_1256 windows_1257 windows_1258 \ + x_mac_japanese x_mac_korean x_mac_simp_chinese x_mac_trad_chinese"; + /// One cp1252 byte as Python decodes it: Latin-1 outside `0x80..=0x9F`, /// Windows punctuation inside it, and five bytes Python leaves undefined. fn cp1252_char(b: u8) -> Option { @@ -524,6 +591,71 @@ mod tests { ); } + /// #1212: pip decodes with whatever codec the coding line names, and + /// every ASCII-compatible codec decodes ASCII bytes as ASCII, so a + /// plain-ASCII file under a header naming one this reader has no table + /// for (copied from a template) reads as its text. Its non-ASCII bytes + /// are still unreadable, and so is ASCII under a codec that decodes it + /// differently (EBCDIC, UTF-16/32, UTF-7, HZ, ISO-2022, escape codecs) + /// or a name Python does not know (pip fails on those). + #[test] + fn decode_reads_ascii_under_any_ascii_compatible_coding_line() { + for coding in [ + "iso-8859-15", + "latin9", + "ISO8859_15", + "iso.8859.15", + "l9", + "cp1250", + "windows-1250", + "mac-roman", + "macintosh", + "gbk", + "GB18030", + "shift_jis", + "big5", + "euc-kr", + "koi8-r", + "cp437", + "iso-8859-2", + "tis-620", + ] { + let text = format!("# -*- coding: {coding} -*-\nsix==1.16.0\n"); + assert_eq!( + decode(text.as_bytes()).as_deref(), + Some(&text[..]), + "{coding}" + ); + let mut high = text.into_bytes(); + high.extend_from_slice(b"# Jos\xe9\n"); + assert_eq!(decode(&high), None, "{coding}"); + } + for coding in [ + "cp037", + "cp500", + "cp864", + "utf-16", + "utf-32-le", + "utf-7", + "hz", + "iso2022_jp", + "shift_jis_2004", + "unicode_escape", + "raw_unicode_escape", + "idna", + "punycode", + "rot13", + "hex", + // Not names Python's lookup knows (pip raises LookupError). + "latin-9", + "cp-1250", + "no-such-codec", + ] { + let text = format!("# coding: {coding}\nsix==1.16.0\n"); + assert_eq!(decode(text.as_bytes()), None, "{coding}"); + } + } + #[test] fn requires_hashes_reads_pip_hash_checking_mode() { for hashed in [ diff --git a/crates/socket-patch-core/src/vex/discover/pypi_other.rs b/crates/socket-patch-core/src/vex/discover/pypi_other.rs index 120a11f95..c1a8d3c56 100644 --- a/crates/socket-patch-core/src/vex/discover/pypi_other.rs +++ b/crates/socket-patch-core/src/vex/discover/pypi_other.rs @@ -1078,6 +1078,34 @@ mod tests { ); } + /// #1212: a plain-ASCII requirements file whose coding line names an + /// ASCII-compatible codec the decoder has no table for is read as pip + /// reads it: its pins are evidence, not an unreadable file. + #[tokio::test] + async fn requirements_files_under_an_ascii_compatible_coding_line_are_read() { + for coding in ["iso-8859-15", "cp1250", "mac-roman", "gbk"] { + let p = Project::new(); + p.write( + "requirements.txt", + format!("# -*- coding: {coding} -*-\nsix==1.16.0\n"), + ); + let out = run(&p).await; + assert!( + out.diagnostics.is_empty(), + "{coding}: {:?}", + out.diagnostics + ); + assert_eq!( + out.elsewhere + .iter() + .map(|e| e.purl.as_str()) + .collect::>(), + vec!["pkg:pypi/six@1.16.0"], + "{coding}" + ); + } + } + /// A vendored server sdist is wired like a wheel (requirements line, /// Pipfile.lock `file`) and names its package the same way. #[tokio::test]