oxedyne/fe2o3/fe2o3_file/src/zip/read.rs
9.9 KiB, 21 runs
created by r1870400018:22542, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Reading an archive's directory, and nothing else. |
| 2 | //! |
| 3 | //! Nothing is inflated here and no content is copied. The pass reads the central directory, finds |
| 4 | //! where each member's bytes sit, and records the byte ranges -- so opening a hundred megabyte |
| 5 | //! archive to look at one small part inside it costs the directory and no more. |
| 6 | //! |
| 7 | //! The ranges are what the writer copies from. A member's `whole` range runs from its local header to |
| 8 | //! the start of whatever follows it, so anything the archive holds between members -- padding, an |
| 9 | //! alignment gap, a data descriptor written in either of its two shapes -- travels with the member |
| 10 | //! before it and survives the round trip without this having to understand it. |
| 11 | //! |
| 12 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 13 | //! Anthropic Claude |
| 14 | |
| 15 | use crate::zip::{ |
| 16 | Body, |
| 17 | Member, |
| 18 | Method, |
| 19 | Zip, |
| 20 | u16le, |
| 21 | u32le, |
| 22 | u64le, |
| 23 | }; |
| 24 | |
| 25 | use oxedyne_fe2o3_core::prelude::*; |
| 26 | |
| 27 | /// The end of central directory record. |
| 28 | const SIG_EOCD: u32 = 0x0605_4b50; |
| 29 | /// The ZIP64 end of central directory record. |
| 30 | const SIG_EOCD64: u32 = 0x0606_4b50; |
| 31 | /// The ZIP64 end of central directory locator. |
| 32 | const SIG_LOC64: u32 = 0x0706_4b50; |
| 33 | /// A central directory entry. |
| 34 | const SIG_CEN: u32 = 0x0201_4b50; |
| 35 | /// A local file header. |
| 36 | const SIG_LOC: u32 = 0x0403_4b50; |
| 37 | |
| 38 | /// The fixed part of a central directory entry. |
| 39 | const CEN_LEN: usize = 46; |
| 40 | /// The fixed part of a local file header. |
| 41 | const LOC_LEN: usize = 30; |
| 42 | |
| 43 | /// The most an end record's comment may be, which bounds the search for the record itself. |
| 44 | const MAX_COMMENT: usize = 0xFFFF; |
| 45 | |
| 46 | impl Zip { |
| 47 | |
| 48 | /// Reads an archive from the bytes of one. |
| 49 | /// |
| 50 | /// The bytes are kept, because every member that is not later replaced is written back out of |
| 51 | /// them. See the module's own note on why that is the whole point. |
| 52 | pub fn read(src: Vec<u8>) -> Outcome<Self> { |
| 53 | let eocd = res!(find_eocd(&src)); |
| 54 | let mut count = res!(u16le(&src, eocd + 10)) as u64; |
| 55 | let mut cd_len = res!(u32le(&src, eocd + 12)) as u64; |
| 56 | let mut cd_at = res!(u32le(&src, eocd + 16)) as u64; |
| 57 | let clen = res!(u16le(&src, eocd + 20)) as usize; |
| 58 | let comment = match src.get(eocd + 22..eocd + 22 + clen) { |
| 59 | Some(s) => s.to_vec(), |
| 60 | None => return Err(err!( |
| 61 | "The end record says its comment is {} bytes, which runs past the end of an \ |
| 62 | archive of {} bytes.", clen, src.len(); Invalid, Input)), |
| 63 | }; |
| 64 | let disk = res!(u16le(&src, eocd + 4)); |
| 65 | let cd_disk = res!(u16le(&src, eocd + 6)); |
| 66 | |
| 67 | // A sentinel in any of the three says the real value is in the ZIP64 record, which the locator |
| 68 | // twenty bytes before the end record points at. |
| 69 | let mut zip64 = false; |
| 70 | if count == 0xFFFF || cd_len == 0xFFFF_FFFF || cd_at == 0xFFFF_FFFF || disk == 0xFFFF { |
| 71 | let (n, len, at) = res!(read_eocd64(&src, eocd)); |
| 72 | count = n; |
| 73 | cd_len = len; |
| 74 | cd_at = at; |
| 75 | zip64 = true; |
| 76 | } |
| 77 | if !zip64 && (disk != 0 || cd_disk != 0) { |
| 78 | return Err(err!( |
| 79 | "The archive is split across {} disks. This reads no split archive.", |
| 80 | disk as u32 + 1; Invalid, Input, Unimplemented)); |
| 81 | } |
| 82 | let cd_at = cd_at as usize; |
| 83 | let cd_end = cd_at.saturating_add(cd_len as usize); |
| 84 | if cd_end > src.len() { |
| 85 | return Err(err!( |
| 86 | "The directory is said to run to byte {} of an archive of {} bytes.", |
| 87 | cd_end, src.len(); Invalid, Input, Range)); |
| 88 | } |
| 89 | |
| 90 | // Pass one: the directory, which says what the archive holds and where each member starts. |
| 91 | let mut raw: Vec<Raw> = Vec::with_capacity(count.min(4096) as usize); |
| 92 | let mut i = cd_at; |
| 93 | while i < cd_end { |
| 94 | if res!(u32le(&src, i)) != SIG_CEN { |
| 95 | break; |
| 96 | } |
| 97 | let ent = res!(read_cen(&src, i)); |
| 98 | i = ent.cen_end; |
| 99 | raw.push(ent); |
| 100 | } |
| 101 | if raw.len() as u64 != count { |
| 102 | return Err(err!( |
| 103 | "The end record counts {} members and the directory holds {}.", count, raw.len(); |
| 104 | Invalid, Input, Mismatch)); |
| 105 | } |
| 106 | |
| 107 | // Pass two: where each member's bytes end, which is where the next one's begin. Anything |
| 108 | // between them travels with the member before it, so padding and data descriptors of either |
| 109 | // shape survive without this having to read them. |
| 110 | raw.sort_by_key(|r| r.at); |
| 111 | let mut members = Vec::with_capacity(raw.len()); |
| 112 | for (n, r) in raw.iter().enumerate() { |
| 113 | let ends = match raw.get(n + 1) { |
| 114 | Some(next) => next.at as usize, |
| 115 | None => cd_at, |
| 116 | }; |
| 117 | let at = r.at as usize; |
| 118 | if at >= ends || ends > src.len() { |
| 119 | return Err(err!( |
| 120 | "'{}' is said to start at byte {} and to end at byte {} of an archive of {} \ |
| 121 | bytes.", r.name, at, ends, src.len(); Invalid, Input, Range)); |
| 122 | } |
| 123 | if res!(u32le(&src, at)) != SIG_LOC { |
| 124 | return Err(err!( |
| 125 | "'{}' is said to start at byte {}, where there is no local header.", |
| 126 | r.name, at; Invalid, Input)); |
| 127 | } |
| 128 | let nlen = res!(u16le(&src, at + 26)) as usize; |
| 129 | let elen = res!(u16le(&src, at + 28)) as usize; |
| 130 | let from = at + LOC_LEN + nlen + elen; |
| 131 | let to = from.saturating_add(r.csize as usize); |
| 132 | if to > ends { |
| 133 | return Err(err!( |
| 134 | "'{}' says it holds {} compressed bytes, which runs past the {} bytes the \ |
| 135 | archive gives it.", r.name, r.csize, ends - from; Invalid, Input, Range)); |
| 136 | } |
| 137 | members.push(Member { |
| 138 | name: r.name.clone(), |
| 139 | method: Method::of(r.method), |
| 140 | crc: r.crc, |
| 141 | size: r.size, |
| 142 | csize: r.csize, |
| 143 | flags: r.flags, |
| 144 | body: Body::Held { |
| 145 | whole: at..ends, |
| 146 | data: from..to, |
| 147 | cen: r.cen_at..r.cen_end, |
| 148 | }, |
| 149 | }); |
| 150 | } |
| 151 | Ok(Self { src, members, comment, zip64, touched: false }) |
| 152 | } |
| 153 | } |
| 154 | |
| 155 | /// One central directory entry, as read. |
| 156 | struct Raw { |
| 157 | name: String, |
| 158 | flags: u16, // general purpose bit flag |
| 159 | method: u16, // the compression method's code |
| 160 | crc: u32, // of the uncompressed content |
| 161 | csize: u64, // compressed |
| 162 | size: u64, // uncompressed |
| 163 | at: u64, // where the member's local header sits |
| 164 | cen_at: usize, // where this entry begins in the directory |
| 165 | cen_end: usize, // where it ends |
| 166 | } |
| 167 | |
| 168 | fn read_cen(src: &[u8], i: usize) -> Outcome<Raw> { |
| 169 | let flags = res!(u16le(src, i + 8)); |
| 170 | let method = res!(u16le(src, i + 10)); |
| 171 | let crc = res!(u32le(src, i + 16)); |
| 172 | let csize = res!(u32le(src, i + 20)) as u64; |
| 173 | let size = res!(u32le(src, i + 24)) as u64; |
| 174 | let nlen = res!(u16le(src, i + 28)) as usize; |
| 175 | let elen = res!(u16le(src, i + 30)) as usize; |
| 176 | let clen = res!(u16le(src, i + 32)) as usize; |
| 177 | let at = res!(u32le(src, i + 42)) as u64; |
| 178 | let name_at = i + CEN_LEN; |
| 179 | let name = match src.get(name_at..name_at + nlen) { |
| 180 | Some(s) => String::from_utf8_lossy(s).into_owned(), |
| 181 | None => return Err(err!( |
| 182 | "A directory entry at byte {} names {} bytes, which runs past the end of the archive.", |
| 183 | i, nlen; Invalid, Input, Range)), |
| 184 | }; |
| 185 | let extra_at = name_at + nlen; |
| 186 | let extra = match src.get(extra_at..extra_at + elen) { |
| 187 | Some(s) => s, |
| 188 | None => return Err(err!( |
| 189 | "'{}' carries {} bytes of extra field, which runs past the end of the archive.", |
| 190 | name, elen; Invalid, Input, Range)), |
| 191 | }; |
| 192 | // ZIP64 puts the real sizes and offset in an extra field, in this order, and only for the fields |
| 193 | // whose ordinary slot holds the sentinel. |
| 194 | let (mut size, mut csize, mut at) = (size, csize, at); |
| 195 | if size == 0xFFFF_FFFF || csize == 0xFFFF_FFFF || at == 0xFFFF_FFFF { |
| 196 | match find_extra(extra, 0x0001) { |
| 197 | Some(z) => { |
| 198 | let mut k = 0; |
| 199 | if size == 0xFFFF_FFFF { size = res!(u64le(z, k)); k += 8; } |
| 200 | if csize == 0xFFFF_FFFF { csize = res!(u64le(z, k)); k += 8; } |
| 201 | if at == 0xFFFF_FFFF { at = res!(u64le(z, k)); } |
| 202 | } |
| 203 | None => return Err(err!( |
| 204 | "'{}' has a ZIP64 sentinel in its directory entry and no ZIP64 extra field to \ |
| 205 | read the real value from.", name; Invalid, Input, Missing)), |
| 206 | } |
| 207 | } |
| 208 | Ok(Raw { |
| 209 | name, |
| 210 | flags, |
| 211 | method, |
| 212 | crc, |
| 213 | csize, |
| 214 | size, |
| 215 | at, |
| 216 | cen_at: i, |
| 217 | cen_end: extra_at + elen + clen, |
| 218 | }) |
| 219 | } |
| 220 | |
| 221 | fn find_extra(extra: &[u8], id: u16) -> Option<&[u8]> { |
| 222 | let mut i = 0; |
| 223 | while i + 4 <= extra.len() { |
| 224 | let this = u16::from_le_bytes([extra[i], extra[i + 1]]); |
| 225 | let len = u16::from_le_bytes([extra[i + 2], extra[i + 3]]) as usize; |
| 226 | let from = i + 4; |
| 227 | let to = from.checked_add(len)?; |
| 228 | if to > extra.len() { |
| 229 | return None; |
| 230 | } |
| 231 | if this == id { |
| 232 | return Some(&extra[from..to]); |
| 233 | } |
| 234 | i = to; |
| 235 | } |
| 236 | None |
| 237 | } |
| 238 | |
| 239 | /// Found by searching backwards, because the record is last and carries a comment of unknown |
| 240 | /// length after it. The search is bounded by the largest comment the format allows, so a large file |
| 241 | /// that is not an archive is refused after reading its tail rather than after reading all of it. |
| 242 | fn find_eocd(src: &[u8]) -> Outcome<usize> { |
| 243 | const MIN: usize = 22; |
| 244 | if src.len() < MIN { |
| 245 | return Err(err!( |
| 246 | "{} bytes is too short to be a ZIP archive, which is at least {}.", src.len(), MIN; |
| 247 | Invalid, Input, Size)); |
| 248 | } |
| 249 | let floor = src.len().saturating_sub(MIN + MAX_COMMENT); |
| 250 | let mut i = src.len() - MIN; |
| 251 | loop { |
| 252 | if u32::from_le_bytes([src[i], src[i + 1], src[i + 2], src[i + 3]]) == SIG_EOCD { |
| 253 | // The comment length must account for exactly what follows, or the signature was a |
| 254 | // coincidence inside somebody's data. |
| 255 | let clen = res!(u16le(src, i + 20)) as usize; |
| 256 | if i + MIN + clen == src.len() { |
| 257 | return Ok(i); |
| 258 | } |
| 259 | } |
| 260 | if i == floor { |
| 261 | return Err(err!( |
| 262 | "The bytes end without a ZIP end-of-directory record, so this is not an archive, \ |
| 263 | or it is one that was truncated."; Invalid, Input, Missing)); |
| 264 | } |
| 265 | i -= 1; |
| 266 | } |
| 267 | } |
| 268 | |
| 269 | /// The member count, directory size and directory offset from the ZIP64 end record. |
| 270 | fn read_eocd64(src: &[u8], eocd: usize) -> Outcome<(u64, u64, u64)> { |
| 271 | if eocd < 20 { |
| 272 | return Err(err!( |
| 273 | "The archive has a ZIP64 sentinel and no room before the end record for the locator \ |
| 274 | that would point at the ZIP64 record."; Invalid, Input, Missing)); |
| 275 | } |
| 276 | let loc = eocd - 20; |
| 277 | if res!(u32le(src, loc)) != SIG_LOC64 { |
| 278 | return Err(err!( |
| 279 | "The archive has a ZIP64 sentinel and no ZIP64 locator before its end record."; |
| 280 | Invalid, Input, Missing)); |
| 281 | } |
| 282 | let at = res!(u64le(src, loc + 8)) as usize; |
| 283 | if res!(u32le(src, at)) != SIG_EOCD64 { |
| 284 | return Err(err!( |
| 285 | "The ZIP64 locator points at byte {}, where there is no ZIP64 end record.", at; |
| 286 | Invalid, Input)); |
| 287 | } |
| 288 | let count = res!(u64le(src, at + 32)); |
| 289 | let cd_len = res!(u64le(src, at + 40)); |
| 290 | let cd_at = res!(u64le(src, at + 48)); |
| 291 | Ok((count, cd_len, cd_at)) |
| 292 | } |