oxedyne/fe2o3/fe2o3_stds/src/media.rs
34.2 KiB, 23 runs
created by r1870400018:21409, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! File formats, identified from their leading bytes and from their names, and what each one is. |
| 2 | //! |
| 3 | //! Every program that shows a person a file has to answer the same question first -- what IS this |
| 4 | //! -- and there are only two sources of an answer. The name is what somebody typed and the bytes |
| 5 | //! are what is actually there, so they disagree whenever a file has been renamed, exported wrongly |
| 6 | //! or truncated mid-download. [`identify`] asks both and reports the disagreement rather than |
| 7 | //! quietly choosing, because which of the two is right is the caller's business and the fact that |
| 8 | //! they differ is usually the most interesting thing about the file. |
| 9 | //! |
| 10 | //! # Why this is a standards table and not a heuristic |
| 11 | //! |
| 12 | //! A magic signature is a constant published in a specification: a PNG begins with the eight bytes |
| 13 | //! `89 50 4E 47 0D 0A 1A 0A` and always has, and the second through fourth of those spell `PNG` so |
| 14 | //! that a file transferred through a text-mode channel is visibly corrupt. Nothing here is |
| 15 | //! guessed. What IS a heuristic lives in one function, [`looks_like_text`], and its limitations |
| 16 | //! are written on it. |
| 17 | //! |
| 18 | //! # Checked against something that is not itself |
| 19 | //! |
| 20 | //! The signatures were checked against the `file` command over seventeen real files of seventeen |
| 21 | //! formats -- PDF, PNG, JPEG, SVG, ZIP, gzip, MP4, MP3, WOFF2, TTF, wasm, HEIC, WebP, ICO, DOCX, |
| 22 | //! EPUB, AVI -- and sixteen agreed exactly. The seventeenth is a deliberate divergence: `file` |
| 23 | //! reports a TrueType font as `font/sfnt`, the generic container type, while its own description |
| 24 | //! of the same bytes reads "TrueType Font data". RFC 8081 registers `font/ttf` for a font with |
| 25 | //! TrueType outlines, and the `00 01 00 00` version tag says which those are, so [`Media::Ttf`] |
| 26 | //! reports the specific type. A browser also wants the specific one. |
| 27 | //! |
| 28 | //! # The mistake this crate exists to stop |
| 29 | //! |
| 30 | //! Reading unknown bytes as UTF-8 with a lossy decoder and showing the result. Every byte that is |
| 31 | //! not valid UTF-8 becomes U+FFFD, so the reader is shown a screenful of replacement characters |
| 32 | //! and no indication that anything went wrong -- the program looks broken rather than the format |
| 33 | //! looking unsupported, and those are very different bug reports. A caller that asks |
| 34 | //! [`looks_like_text`] before it decodes cannot make that mistake. |
| 35 | //! |
| 36 | //! # Example |
| 37 | //! |
| 38 | //! ``` |
| 39 | //! use oxedyne_fe2o3_stds::media::{Kind, Media, identify}; |
| 40 | //! |
| 41 | //! let head = b"%PDF-1.7\n%\xE2\xE3\xCF\xD3\n"; |
| 42 | //! let id = identify("notes.txt", head); |
| 43 | //! assert_eq!(id.media, Media::Pdf); // the bytes win |
| 44 | //! assert_eq!(id.by_name, Media::Text); |
| 45 | //! assert!(id.disagree); // and the caller can say so |
| 46 | //! assert_eq!(id.media.kind(), Kind::Document); |
| 47 | //! ``` |
| 48 | |
| 49 | /// The broad class a format belongs to, which is what decides how a viewer shows it. |
| 50 | /// |
| 51 | /// A caller usually wants this before it wants the format: a picture goes in an `img`, a sound in |
| 52 | /// an `audio`, and anything unrecognised goes to a hex dump. Adding a format to [`Media`] without |
| 53 | /// giving it a kind here would leave it silently in [`Kind::Unknown`], so the mapping in |
| 54 | /// [`Media::kind`] is exhaustive by construction rather than by a wildcard arm. |
| 55 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] |
| 56 | pub enum Kind { |
| 57 | /// A still picture. |
| 58 | Image, |
| 59 | /// A moving picture, with or without sound. |
| 60 | Video, |
| 61 | /// Sound alone. |
| 62 | Audio, |
| 63 | /// A paginated or marked-up document. |
| 64 | Document, |
| 65 | /// A container holding other files. |
| 66 | Archive, |
| 67 | /// A typeface. |
| 68 | Font, |
| 69 | /// Characters, meant to be read as characters. |
| 70 | Text, |
| 71 | /// An executable or object file. |
| 72 | Binary, |
| 73 | /// Nothing here recognised it. |
| 74 | Unknown, |
| 75 | } |
| 76 | |
| 77 | /// A file format. |
| 78 | /// |
| 79 | /// Named for the format rather than for its usual extension, because one format wears several |
| 80 | /// extensions (`.jpg`, `.jpeg`, `.jpe`) and one extension is sometimes worn by several formats. |
| 81 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] |
| 82 | pub enum Media { |
| 83 | // ── Pictures ── |
| 84 | /// Portable Network Graphics. |
| 85 | Png, |
| 86 | /// JPEG, in the JFIF or Exif wrapping a camera writes. |
| 87 | Jpeg, |
| 88 | /// Graphics Interchange Format, still or animated. |
| 89 | Gif, |
| 90 | /// WebP, inside a RIFF container. |
| 91 | Webp, |
| 92 | /// AV1 Image File Format, an ISO base media container. |
| 93 | Avif, |
| 94 | /// High Efficiency Image Format, an ISO base media container holding HEVC. |
| 95 | Heic, |
| 96 | /// Windows bitmap. |
| 97 | Bmp, |
| 98 | /// Windows icon. |
| 99 | Ico, |
| 100 | /// Tagged Image File Format, either byte order. |
| 101 | Tiff, |
| 102 | /// Scalable Vector Graphics, which is XML and therefore also text. |
| 103 | Svg, |
| 104 | |
| 105 | // ── Documents ── |
| 106 | /// Portable Document Format. |
| 107 | Pdf, |
| 108 | /// Rich Text Format. |
| 109 | Rtf, |
| 110 | /// PostScript. |
| 111 | PostScript, |
| 112 | /// HTML, which is text and may execute. |
| 113 | Html, |
| 114 | /// XML that is not one of the XML formats named separately here. |
| 115 | Xml, |
| 116 | /// Markdown. |
| 117 | Markdown, |
| 118 | /// JSON. |
| 119 | Json, |
| 120 | /// Comma-separated values. |
| 121 | Csv, |
| 122 | /// Tab-separated values. |
| 123 | Tsv, |
| 124 | |
| 125 | // ── Containers ── |
| 126 | /// A ZIP archive, or one of the formats built on one that was not distinguished. |
| 127 | Zip, |
| 128 | /// A gzip stream. |
| 129 | Gzip, |
| 130 | /// A bzip2 stream. |
| 131 | Bzip2, |
| 132 | /// An xz stream. |
| 133 | Xz, |
| 134 | /// A Zstandard stream. |
| 135 | Zstd, |
| 136 | /// A tar archive. |
| 137 | Tar, |
| 138 | /// A 7-Zip archive. |
| 139 | SevenZip, |
| 140 | /// A RAR archive. |
| 141 | Rar, |
| 142 | /// An Office Open XML word-processing document: a ZIP with a known part inside. |
| 143 | Docx, |
| 144 | /// An Office Open XML spreadsheet. |
| 145 | Xlsx, |
| 146 | /// An Office Open XML presentation. |
| 147 | Pptx, |
| 148 | /// An OpenDocument text document. |
| 149 | Odt, |
| 150 | /// An OpenDocument spreadsheet. |
| 151 | Ods, |
| 152 | /// An OpenDocument presentation. |
| 153 | Odp, |
| 154 | /// An OpenDocument drawing. |
| 155 | Odg, |
| 156 | /// An EPUB book. |
| 157 | Epub, |
| 158 | |
| 159 | // ── Sound ── |
| 160 | /// MPEG-1 Audio Layer III. |
| 161 | Mp3, |
| 162 | /// A RIFF WAVE file. |
| 163 | Wav, |
| 164 | /// FLAC, natively framed. |
| 165 | Flac, |
| 166 | /// An Ogg container, whatever it carries. |
| 167 | Ogg, |
| 168 | /// MPEG-4 audio. |
| 169 | M4a, |
| 170 | |
| 171 | // ── Moving pictures ── |
| 172 | /// MPEG-4 part 14. |
| 173 | Mp4, |
| 174 | /// WebM, which is a Matroska profile. |
| 175 | Webm, |
| 176 | /// Matroska. |
| 177 | Matroska, |
| 178 | /// A RIFF AVI file. |
| 179 | Avi, |
| 180 | /// QuickTime. |
| 181 | QuickTime, |
| 182 | |
| 183 | // ── Typefaces ── |
| 184 | /// TrueType. |
| 185 | Ttf, |
| 186 | /// OpenType with PostScript outlines. |
| 187 | Otf, |
| 188 | /// Web Open Font Format 1. |
| 189 | Woff, |
| 190 | /// Web Open Font Format 2. |
| 191 | Woff2, |
| 192 | |
| 193 | // ── Programs ── |
| 194 | /// An ELF object or executable. |
| 195 | Elf, |
| 196 | /// A DOS, Windows PE or similar `MZ` image. |
| 197 | Exe, |
| 198 | /// A WebAssembly module. |
| 199 | Wasm, |
| 200 | /// A Java class file. |
| 201 | JavaClass, |
| 202 | |
| 203 | // ── Everything else ── |
| 204 | /// Characters, with no more specific format recognised. |
| 205 | Text, |
| 206 | /// Nothing recognised it. |
| 207 | Unknown, |
| 208 | } |
| 209 | |
| 210 | /// What [`identify`] concluded, and from what. |
| 211 | /// |
| 212 | /// Both answers are kept rather than reconciled. A viewer shows [`Self::media`]; a viewer that |
| 213 | /// wants to be trusted also mentions [`Self::disagree`], because a file whose name says `.png` and |
| 214 | /// whose bytes say PDF is how a person finds a broken export, and hiding it helps nobody. |
| 215 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] |
| 216 | pub struct Identified { |
| 217 | /// The format to act on: the bytes when they said anything, else the name. |
| 218 | pub media: Media, |
| 219 | /// What the leading bytes said, or [`Media::Unknown`]. |
| 220 | pub by_magic: Media, |
| 221 | /// What the name said, or [`Media::Unknown`]. |
| 222 | pub by_name: Media, |
| 223 | /// Both spoke, and they said different things. |
| 224 | pub disagree: bool, |
| 225 | } |
| 226 | |
| 227 | /// Identify a file from its name and the front of its bytes. |
| 228 | /// The media type an OpenDocument or EPUB package declares in its first member. |
| 229 | pub const ODF_TEXT: &str = "application/vnd.oasis.opendocument.text"; |
| 230 | /// The media type an OpenDocument spreadsheet declares. |
| 231 | pub const ODF_SHEET: &str = "application/vnd.oasis.opendocument.spreadsheet"; |
| 232 | /// The media type an OpenDocument presentation declares. |
| 233 | pub const ODF_SLIDES: &str = "application/vnd.oasis.opendocument.presentation"; |
| 234 | /// The media type an OpenDocument drawing declares. |
| 235 | pub const ODF_DRAWING: &str = "application/vnd.oasis.opendocument.graphics"; |
| 236 | |
| 237 | /// The format an archive declares in a first member named `mimetype`, where it has one. |
| 238 | /// |
| 239 | /// OpenDocument requires that member to be first in the archive and stored uncompressed, for exactly |
| 240 | /// this purpose: a reader is meant to be able to name the file from its opening bytes rather than |
| 241 | /// from what somebody called it. EPUB borrowed the rule. Nothing else uses it, so an archive without |
| 242 | /// one is simply not one of these and is left as a ZIP. |
| 243 | /// |
| 244 | /// Read from the local header rather than at a fixed offset, because the offset is only fixed when |
| 245 | /// the extra field is empty -- which it usually is, and "usually" is not a thing to encode. |
| 246 | fn odf_mimetype(b: &[u8]) -> Option<Media> { |
| 247 | // A local file header is 30 bytes, then the name, then the extra field, then the data. |
| 248 | if b.len() < 30 || &b[..4] != b"PK\x03\x04" { |
| 249 | return None; |
| 250 | } |
| 251 | // Stored, not deflated. A compressed `mimetype` is a package that has broken the rule, and |
| 252 | // inflating here to be helpful would be reading a format this function is only meant to name. |
| 253 | if u16::from_le_bytes([b[8], b[9]]) != 0 { |
| 254 | return None; |
| 255 | } |
| 256 | let size = u32::from_le_bytes([b[18], b[19], b[20], b[21]]) as usize; |
| 257 | let nlen = u16::from_le_bytes([b[26], b[27]]) as usize; |
| 258 | let elen = u16::from_le_bytes([b[28], b[29]]) as usize; |
| 259 | if nlen != 8 || b.len() < 30 + nlen || &b[30..38] != b"mimetype" { |
| 260 | return None; |
| 261 | } |
| 262 | let from = 30 + nlen + elen; |
| 263 | let to = from.checked_add(size)?; |
| 264 | if to > b.len() || size > 128 { |
| 265 | return None; |
| 266 | } |
| 267 | match core::str::from_utf8(&b[from..to]).ok()?.trim() { |
| 268 | ODF_TEXT => Some(Media::Odt), |
| 269 | ODF_SHEET => Some(Media::Ods), |
| 270 | ODF_SLIDES => Some(Media::Odp), |
| 271 | ODF_DRAWING => Some(Media::Odg), |
| 272 | "application/epub+zip" => Some(Media::Epub), |
| 273 | _ => None, |
| 274 | } |
| 275 | } |
| 276 | |
| 277 | /// |
| 278 | /// The bytes win. A name is a claim and bytes are evidence, and the one case where the name is |
| 279 | /// the better answer -- a format with no magic signature, such as CSV -- is exactly the case where |
| 280 | /// the bytes say nothing and there is no contest. |
| 281 | /// |
| 282 | /// `prefix` may be as short as the caller likes; 512 bytes recognises everything here except a tar |
| 283 | /// archive, whose signature sits at offset 257 and so needs 264. A prefix too short for a |
| 284 | /// signature simply fails to match it, and never misidentifies. |
| 285 | /// |
| 286 | /// # Arguments |
| 287 | /// * `name` - The file name or path; only the extension is read. |
| 288 | /// * `prefix` - The leading bytes of the file. |
| 289 | pub fn identify(name: &str, prefix: &[u8]) -> Identified { |
| 290 | let by_magic = Media::sniff(prefix); |
| 291 | let by_name = Media::from_name(name); |
| 292 | // A magic hit for ZIP is worth refining by name before it is compared: `.docx`, `.epub` and |
| 293 | // `.odt` are all ZIP archives, and reporting a disagreement between "ZIP" and "Word document" |
| 294 | // would be reporting agreement as conflict. |
| 295 | let by_magic = match (by_magic, by_name) { |
| 296 | (Media::Zip, Media::Docx) |
| 297 | | (Media::Zip, Media::Xlsx) |
| 298 | | (Media::Zip, Media::Pptx) |
| 299 | | (Media::Zip, Media::Epub) |
| 300 | | (Media::Zip, Media::Odt) |
| 301 | | (Media::Zip, Media::Ods) |
| 302 | | (Media::Zip, Media::Odp) |
| 303 | | (Media::Zip, Media::Odg) => by_name, |
| 304 | _ => by_magic, |
| 305 | }; |
| 306 | // Text is the weakest possible magic answer -- it means "these bytes are characters", which |
| 307 | // every text format satisfies -- so a name that is more specific refines it rather than |
| 308 | // fighting it. |
| 309 | let (media, disagree) = match (by_magic, by_name) { |
| 310 | (Media::Unknown, n) => (n, false), |
| 311 | (Media::Text, n) if n != Media::Unknown => (n, false), |
| 312 | (m, Media::Unknown) => (m, false), |
| 313 | (m, n) if m == n => (m, false), |
| 314 | (m, _) => (m, true), |
| 315 | }; |
| 316 | Identified { media, by_magic, by_name, disagree } |
| 317 | } |
| 318 | |
| 319 | impl Media { |
| 320 | /// Identify a format from the leading bytes of a file, or [`Self::Unknown`]. |
| 321 | /// |
| 322 | /// Signatures are checked longest-first where two overlap, so that a WebP is not reported as |
| 323 | /// the RIFF container it happens to be carried in. |
| 324 | /// |
| 325 | /// # Arguments |
| 326 | /// * `b` - The leading bytes of the file. |
| 327 | pub fn sniff(b: &[u8]) -> Self { |
| 328 | // ── Fixed signatures at offset zero ── |
| 329 | if starts(b, b"\x89PNG\r\n\x1a\n") { return Self::Png; } |
| 330 | if starts(b, b"\xff\xd8\xff") { return Self::Jpeg; } |
| 331 | if starts(b, b"GIF87a") || starts(b, b"GIF89a") { return Self::Gif; } |
| 332 | if starts(b, b"BM") { return Self::Bmp; } |
| 333 | if starts(b, b"\x00\x00\x01\x00") { return Self::Ico; } |
| 334 | if starts(b, b"II*\x00") || starts(b, b"MM\x00*") { return Self::Tiff; } |
| 335 | if starts(b, b"%PDF-") { return Self::Pdf; } |
| 336 | if starts(b, b"{\\rtf") { return Self::Rtf; } |
| 337 | if starts(b, b"%!PS") { return Self::PostScript; } |
| 338 | if starts(b, b"\x1f\x8b") { return Self::Gzip; } |
| 339 | if starts(b, b"BZh") { return Self::Bzip2; } |
| 340 | if starts(b, b"\xfd7zXZ\x00") { return Self::Xz; } |
| 341 | if starts(b, b"\x28\xb5\x2f\xfd") { return Self::Zstd; } |
| 342 | if starts(b, b"7z\xbc\xaf\x27\x1c") { return Self::SevenZip; } |
| 343 | if starts(b, b"Rar!\x1a\x07") { return Self::Rar; } |
| 344 | if starts(b, b"fLaC") { return Self::Flac; } |
| 345 | if starts(b, b"OggS") { return Self::Ogg; } |
| 346 | if starts(b, b"\x1a\x45\xdf\xa3") { return Self::Matroska; } |
| 347 | if starts(b, b"OTTO") { return Self::Otf; } |
| 348 | if starts(b, b"wOFF") { return Self::Woff; } |
| 349 | if starts(b, b"wOF2") { return Self::Woff2; } |
| 350 | if starts(b, b"\x00\x01\x00\x00") || starts(b, b"true") { return Self::Ttf; } |
| 351 | if starts(b, b"\x7fELF") { return Self::Elf; } |
| 352 | if starts(b, b"\x00asm") { return Self::Wasm; } |
| 353 | if starts(b, b"\xca\xfe\xba\xbe") { return Self::JavaClass; } |
| 354 | if starts(b, b"MZ") { return Self::Exe; } |
| 355 | // Every ZIP-based format begins the same way, and for most of them which one it is lives |
| 356 | // in the central directory at the END of the file, past anything a prefix can see. The |
| 357 | // name settles those, and `identify` does that rather than this. |
| 358 | // |
| 359 | // OpenDocument is the exception, and deliberately so: the format REQUIRES a member named |
| 360 | // `mimetype` to come first and to be stored uncompressed, precisely so that a reader can |
| 361 | // name the file from its opening bytes. So that one is read rather than guessed, and a |
| 362 | // `.odt` somebody renamed is still an `.odt`. EPUB borrowed the same rule. |
| 363 | if starts(b, b"PK\x03\x04") || starts(b, b"PK\x05\x06") || starts(b, b"PK\x07\x08") { |
| 364 | if let Some(m) = odf_mimetype(b) { |
| 365 | return m; |
| 366 | } |
| 367 | return Self::Zip; |
| 368 | } |
| 369 | // An ID3 tag is a wrapper, not a format; what follows it is MPEG audio in practice. |
| 370 | if starts(b, b"ID3") { return Self::Mp3; } |
| 371 | // An MPEG frame sync is eleven set bits. Checked after everything else because two |
| 372 | // bytes of a coincidence is not much evidence, and after ID3 because a tagged file |
| 373 | // does not begin with one. |
| 374 | if b.len() >= 2 && b[0] == 0xff && (b[1] & 0xe0) == 0xe0 { return Self::Mp3; } |
| 375 | |
| 376 | // ── RIFF, which says what it holds in its fifth through eighth bytes ── |
| 377 | if starts(b, b"RIFF") && b.len() >= 12 { |
| 378 | return match &b[8..12] { |
| 379 | b"WEBP" => Self::Webp, |
| 380 | b"WAVE" => Self::Wav, |
| 381 | b"AVI " => Self::Avi, |
| 382 | _ => Self::Unknown, |
| 383 | }; |
| 384 | } |
| 385 | |
| 386 | // ── ISO base media, whose brand sits after the box header ── |
| 387 | // |
| 388 | // The first four bytes are the box length, which varies, so the marker is at four and |
| 389 | // the brand at eight. Compatible brands follow from sixteen, and are not read: the |
| 390 | // major brand is what the file says it IS, and a viewer that prefers a compatible |
| 391 | // brand over the major one is deciding for the file. |
| 392 | if b.len() >= 12 && &b[4..8] == b"ftyp" { |
| 393 | return match &b[8..12] { |
| 394 | b"avif" | b"avis" => Self::Avif, |
| 395 | b"heic" | b"heix" | b"hevc" | b"heim" |
| 396 | | b"heis" | b"mif1" | b"msf1" => Self::Heic, |
| 397 | b"M4A " | b"M4B " | b"m4a " => Self::M4a, |
| 398 | b"qt " => Self::QuickTime, |
| 399 | _ => Self::Mp4, |
| 400 | }; |
| 401 | } |
| 402 | |
| 403 | // ── A tar's signature is 257 bytes in, which is why prefixes should be 512 ── |
| 404 | if b.len() >= 262 && &b[257..262] == b"ustar" { return Self::Tar; } |
| 405 | |
| 406 | // ── Text-shaped formats, recognised only once the bytes are known to be characters ── |
| 407 | if looks_like_text(b) { |
| 408 | let head = lead(b, 512).to_ascii_lowercase(); |
| 409 | let head = head.trim_start(); |
| 410 | if head.starts_with("<?xml") { |
| 411 | // An XML declaration says nothing about the vocabulary; SVG says so in its |
| 412 | // root element, which the declaration precedes. |
| 413 | if head.contains("<svg") { return Self::Svg; } |
| 414 | return Self::Xml; |
| 415 | } |
| 416 | if head.starts_with("<svg") { return Self::Svg; } |
| 417 | if head.starts_with("<!doctype html") |
| 418 | || head.starts_with("<html") { return Self::Html; } |
| 419 | if head.starts_with('<') { return Self::Xml; } |
| 420 | return Self::Text; |
| 421 | } |
| 422 | Self::Unknown |
| 423 | } |
| 424 | |
| 425 | /// Identify a format from a file name or path, or [`Self::Unknown`]. |
| 426 | /// |
| 427 | /// Only the text after the last dot of the last path component is read, so a dot in a |
| 428 | /// directory name cannot be mistaken for an extension. |
| 429 | /// |
| 430 | /// # Arguments |
| 431 | /// * `name` - A file name or path. |
| 432 | pub fn from_name(name: &str) -> Self { |
| 433 | let leaf = name.rsplit(['/', '\\']).next().unwrap_or(name); |
| 434 | let ext = match leaf.rsplit_once('.') { |
| 435 | // A dotfile with no second dot is a name, not an extension. |
| 436 | Some((stem, ext)) if !stem.is_empty() => ext, |
| 437 | _ => return Self::Unknown, |
| 438 | }; |
| 439 | Self::from_extension(ext) |
| 440 | } |
| 441 | |
| 442 | /// Identify a format from an extension, with or without its dot, in any case. |
| 443 | /// |
| 444 | /// # Arguments |
| 445 | /// * `ext` - An extension, such as `png`, `.PNG` or `jpeg`. |
| 446 | pub fn from_extension(ext: &str) -> Self { |
| 447 | let e = ext.trim_start_matches('.').to_ascii_lowercase(); |
| 448 | match e.as_str() { |
| 449 | "png" => Self::Png, |
| 450 | "jpg" | "jpeg" | "jpe" | "jfif" => Self::Jpeg, |
| 451 | "gif" => Self::Gif, |
| 452 | "webp" => Self::Webp, |
| 453 | "avif" => Self::Avif, |
| 454 | "heic" | "heif" => Self::Heic, |
| 455 | "bmp" | "dib" => Self::Bmp, |
| 456 | "ico" => Self::Ico, |
| 457 | "tif" | "tiff" => Self::Tiff, |
| 458 | "svg" => Self::Svg, |
| 459 | |
| 460 | "pdf" => Self::Pdf, |
| 461 | "rtf" => Self::Rtf, |
| 462 | "ps" | "eps" => Self::PostScript, |
| 463 | "html" | "htm" | "xhtml" => Self::Html, |
| 464 | "xml" => Self::Xml, |
| 465 | "md" | "markdown" => Self::Markdown, |
| 466 | "json" | "jsonl" | "ndjson" => Self::Json, |
| 467 | "csv" => Self::Csv, |
| 468 | "tsv" | "tab" => Self::Tsv, |
| 469 | |
| 470 | "zip" => Self::Zip, |
| 471 | "gz" | "tgz" => Self::Gzip, |
| 472 | "bz2" | "tbz2" => Self::Bzip2, |
| 473 | "xz" | "txz" => Self::Xz, |
| 474 | "zst" => Self::Zstd, |
| 475 | "tar" => Self::Tar, |
| 476 | "7z" => Self::SevenZip, |
| 477 | "rar" => Self::Rar, |
| 478 | "docx" | "docm" => Self::Docx, |
| 479 | "xlsx" | "xlsm" => Self::Xlsx, |
| 480 | "pptx" | "pptm" => Self::Pptx, |
| 481 | "odt" | "ott" | "fodt" => Self::Odt, |
| 482 | "ods" | "ots" | "fods" => Self::Ods, |
| 483 | "odp" | "otp" | "fodp" => Self::Odp, |
| 484 | "odg" | "otg" | "fodg" => Self::Odg, |
| 485 | "epub" => Self::Epub, |
| 486 | |
| 487 | "mp3" => Self::Mp3, |
| 488 | "wav" | "wave" => Self::Wav, |
| 489 | "flac" => Self::Flac, |
| 490 | "ogg" | "oga" | "opus" => Self::Ogg, |
| 491 | "m4a" | "m4b" => Self::M4a, |
| 492 | |
| 493 | "mp4" | "m4v" => Self::Mp4, |
| 494 | "webm" => Self::Webm, |
| 495 | "mkv" | "mka" => Self::Matroska, |
| 496 | "avi" => Self::Avi, |
| 497 | "mov" | "qt" => Self::QuickTime, |
| 498 | |
| 499 | "ttf" | "ttc" => Self::Ttf, |
| 500 | "otf" => Self::Otf, |
| 501 | "woff" => Self::Woff, |
| 502 | "woff2" => Self::Woff2, |
| 503 | |
| 504 | "wasm" => Self::Wasm, |
| 505 | "exe" | "dll" => Self::Exe, |
| 506 | "so" | "elf" => Self::Elf, |
| 507 | "class" => Self::JavaClass, |
| 508 | |
| 509 | // Extensions whose files are characters and whose format is "source code". They |
| 510 | // are named rather than defaulted so that a caller may show them with line |
| 511 | // numbers and highlighting without having to sniff first. |
| 512 | "txt" | "text" | "log" | "rs" | "js" | "mjs" | "ts" | "py" | "sh" | "bash" |
| 513 | | "c" | "h" | "cpp" | "hpp" | "go" | "java" | "rb" | "php" | "css" | "scss" |
| 514 | | "toml" | "yaml" | "yml" | "ini" | "conf" | "typ" | "tex" | "sql" | "jdat" |
| 515 | => Self::Text, |
| 516 | _ => Self::Unknown, |
| 517 | } |
| 518 | } |
| 519 | |
| 520 | /// The broad class this format belongs to. |
| 521 | pub fn kind(&self) -> Kind { |
| 522 | match self { |
| 523 | Self::Png | Self::Jpeg | Self::Gif | Self::Webp | Self::Avif | Self::Heic |
| 524 | | Self::Bmp | Self::Ico | Self::Tiff | Self::Svg => Kind::Image, |
| 525 | |
| 526 | Self::Mp4 | Self::Webm | Self::Matroska | Self::Avi |
| 527 | | Self::QuickTime => Kind::Video, |
| 528 | |
| 529 | Self::Mp3 | Self::Wav | Self::Flac | Self::Ogg | Self::M4a => Kind::Audio, |
| 530 | |
| 531 | Self::Pdf | Self::Rtf | Self::PostScript | Self::Html | Self::Xml |
| 532 | | Self::Markdown | Self::Json | Self::Csv | Self::Tsv => Kind::Document, |
| 533 | |
| 534 | Self::Zip | Self::Gzip | Self::Bzip2 | Self::Xz | Self::Zstd | Self::Tar |
| 535 | | Self::SevenZip | Self::Rar | Self::Docx | Self::Xlsx | Self::Pptx |
| 536 | | Self::Odt | Self::Ods | Self::Odp | Self::Odg |
| 537 | | Self::Epub => Kind::Archive, |
| 538 | |
| 539 | Self::Ttf | Self::Otf | Self::Woff | Self::Woff2 => Kind::Font, |
| 540 | |
| 541 | Self::Elf | Self::Exe | Self::Wasm | Self::JavaClass => Kind::Binary, |
| 542 | |
| 543 | Self::Text => Kind::Text, |
| 544 | Self::Unknown => Kind::Unknown, |
| 545 | } |
| 546 | } |
| 547 | |
| 548 | /// The IANA media type, or `application/octet-stream` where there is none. |
| 549 | /// |
| 550 | /// This is what a caller puts on a `Blob` so that a browser hands the bytes to the right |
| 551 | /// decoder; getting it wrong is how a correct picture fails to appear. |
| 552 | pub fn mime(&self) -> &'static str { |
| 553 | match self { |
| 554 | Self::Png => "image/png", |
| 555 | Self::Jpeg => "image/jpeg", |
| 556 | Self::Gif => "image/gif", |
| 557 | Self::Webp => "image/webp", |
| 558 | Self::Avif => "image/avif", |
| 559 | Self::Heic => "image/heic", |
| 560 | Self::Bmp => "image/bmp", |
| 561 | Self::Ico => "image/vnd.microsoft.icon", |
| 562 | Self::Tiff => "image/tiff", |
| 563 | Self::Svg => "image/svg+xml", |
| 564 | |
| 565 | Self::Pdf => "application/pdf", |
| 566 | Self::Rtf => "application/rtf", |
| 567 | Self::PostScript => "application/postscript", |
| 568 | Self::Html => "text/html", |
| 569 | Self::Xml => "application/xml", |
| 570 | Self::Markdown => "text/markdown", |
| 571 | Self::Json => "application/json", |
| 572 | Self::Csv => "text/csv", |
| 573 | Self::Tsv => "text/tab-separated-values", |
| 574 | |
| 575 | Self::Zip => "application/zip", |
| 576 | Self::Gzip => "application/gzip", |
| 577 | Self::Bzip2 => "application/x-bzip2", |
| 578 | Self::Xz => "application/x-xz", |
| 579 | Self::Zstd => "application/zstd", |
| 580 | Self::Tar => "application/x-tar", |
| 581 | Self::SevenZip => "application/x-7z-compressed", |
| 582 | Self::Rar => "application/vnd.rar", |
| 583 | Self::Docx => |
| 584 | "application/vnd.openxmlformats-officedocument.wordprocessingml.document", |
| 585 | Self::Xlsx => |
| 586 | "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", |
| 587 | Self::Pptx => |
| 588 | "application/vnd.openxmlformats-officedocument.presentationml.presentation", |
| 589 | Self::Odt => ODF_TEXT, |
| 590 | Self::Ods => ODF_SHEET, |
| 591 | Self::Odp => ODF_SLIDES, |
| 592 | Self::Odg => ODF_DRAWING, |
| 593 | Self::Epub => "application/epub+zip", |
| 594 | |
| 595 | Self::Mp3 => "audio/mpeg", |
| 596 | Self::Wav => "audio/wav", |
| 597 | Self::Flac => "audio/flac", |
| 598 | Self::Ogg => "audio/ogg", |
| 599 | Self::M4a => "audio/mp4", |
| 600 | |
| 601 | Self::Mp4 => "video/mp4", |
| 602 | Self::Webm => "video/webm", |
| 603 | Self::Matroska => "video/x-matroska", |
| 604 | Self::Avi => "video/x-msvideo", |
| 605 | Self::QuickTime => "video/quicktime", |
| 606 | |
| 607 | Self::Ttf => "font/ttf", |
| 608 | Self::Otf => "font/otf", |
| 609 | Self::Woff => "font/woff", |
| 610 | Self::Woff2 => "font/woff2", |
| 611 | |
| 612 | Self::Wasm => "application/wasm", |
| 613 | Self::Elf | Self::Exe | Self::JavaClass |
| 614 | => "application/octet-stream", |
| 615 | |
| 616 | Self::Text => "text/plain", |
| 617 | Self::Unknown => "application/octet-stream", |
| 618 | } |
| 619 | } |
| 620 | |
| 621 | /// A short human name for the format, in English, for a caller with nowhere to translate. |
| 622 | pub fn label(&self) -> &'static str { |
| 623 | match self { |
| 624 | Self::Png => "PNG image", Self::Jpeg => "JPEG image", |
| 625 | Self::Gif => "GIF image", Self::Webp => "WebP image", |
| 626 | Self::Avif => "AVIF image", Self::Heic => "HEIC image", |
| 627 | Self::Bmp => "Bitmap image", Self::Ico => "Icon", |
| 628 | Self::Tiff => "TIFF image", Self::Svg => "SVG drawing", |
| 629 | Self::Pdf => "PDF document", Self::Rtf => "Rich text", |
| 630 | Self::PostScript => "PostScript", Self::Html => "HTML page", |
| 631 | Self::Xml => "XML", Self::Markdown => "Markdown", |
| 632 | Self::Json => "JSON", Self::Csv => "CSV table", |
| 633 | Self::Tsv => "TSV table", Self::Zip => "ZIP archive", |
| 634 | Self::Gzip => "gzip stream", Self::Bzip2 => "bzip2 stream", |
| 635 | Self::Xz => "xz stream", Self::Zstd => "Zstandard stream", |
| 636 | Self::Tar => "tar archive", Self::SevenZip => "7-Zip archive", |
| 637 | Self::Rar => "RAR archive", Self::Docx => "Word document", |
| 638 | Self::Xlsx => "Excel spreadsheet", Self::Pptx => "PowerPoint presentation", |
| 639 | Self::Odt => "OpenDocument text", Self::Ods => "OpenDocument spreadsheet", |
| 640 | Self::Odp => "OpenDocument presentation", |
| 641 | Self::Odg => "OpenDocument drawing", Self::Epub => "EPUB book", |
| 642 | Self::Mp3 => "MP3 audio", Self::Wav => "WAV audio", |
| 643 | Self::Flac => "FLAC audio", Self::Ogg => "Ogg audio", |
| 644 | Self::M4a => "MPEG-4 audio", Self::Mp4 => "MP4 video", |
| 645 | Self::Webm => "WebM video", Self::Matroska => "Matroska video", |
| 646 | Self::Avi => "AVI video", Self::QuickTime => "QuickTime video", |
| 647 | Self::Ttf => "TrueType font", Self::Otf => "OpenType font", |
| 648 | Self::Woff => "WOFF font", Self::Woff2 => "WOFF2 font", |
| 649 | Self::Elf => "ELF binary", Self::Exe => "Executable", |
| 650 | Self::Wasm => "WebAssembly module", Self::JavaClass => "Java class", |
| 651 | Self::Text => "Text", Self::Unknown => "Unknown", |
| 652 | } |
| 653 | } |
| 654 | |
| 655 | /// Whether a file of this format is characters, and may be decoded as UTF-8 and shown. |
| 656 | /// |
| 657 | /// True of the formats that ARE text, including the ones with markup in them. It is not a |
| 658 | /// claim that the bytes in hand are valid UTF-8 -- that is [`looks_like_text`] -- only that |
| 659 | /// the format is one where trying is the right thing to do. |
| 660 | pub fn is_text(&self) -> bool { |
| 661 | matches!(self, |
| 662 | Self::Text | Self::Markdown | Self::Json | Self::Csv | Self::Tsv |
| 663 | | Self::Html | Self::Xml | Self::Svg | Self::PostScript | Self::Rtf) |
| 664 | } |
| 665 | } |
| 666 | |
| 667 | /// Whether a run of bytes looks like text, and may be decoded and shown as characters. |
| 668 | /// |
| 669 | /// THE ONE HEURISTIC IN THIS MODULE, and the limits of it are these. A NUL byte settles the |
| 670 | /// question -- no text format in use writes one -- and so does a byte sequence that is not valid |
| 671 | /// UTF-8. Beyond that it counts control characters, because a file that is technically valid |
| 672 | /// UTF-8 and is nine tenths control bytes is not something anybody wants shown as characters. |
| 673 | /// |
| 674 | /// A truncated prefix is handled rather than failed: `b` may end in the middle of a multi-byte |
| 675 | /// character, and up to three trailing bytes are therefore ignored before the check. Without that |
| 676 | /// a prefix of a perfectly ordinary UTF-8 file is called binary once every few hundred reads, |
| 677 | /// which is the kind of defect that is never reproduced on demand. |
| 678 | /// |
| 679 | /// An empty run is text. There is nothing in it to show, and calling it binary would put a hex |
| 680 | /// dump in front of somebody who created an empty file a moment ago. |
| 681 | /// |
| 682 | /// # Arguments |
| 683 | /// * `b` - The leading bytes of the file. |
| 684 | pub fn looks_like_text(b: &[u8]) -> bool { |
| 685 | if b.is_empty() { |
| 686 | return true; |
| 687 | } |
| 688 | if b.contains(&0) { |
| 689 | return false; |
| 690 | } |
| 691 | // A character is at most four bytes, so the last one begins at most four bytes from the end. |
| 692 | // Walk back to its leader, work out how long it was MEANT to be, and drop it only if the run |
| 693 | // stops short of that -- a whole character at the end must be kept, or a one-character file |
| 694 | // would be read as an empty one. |
| 695 | let mut end = b.len(); |
| 696 | let mut back = 0; |
| 697 | while back < 4 && end > 0 && (b[end - 1] & 0xc0) == 0x80 { |
| 698 | end -= 1; |
| 699 | back += 1; |
| 700 | } |
| 701 | if end > 0 { |
| 702 | let lead = b[end - 1]; |
| 703 | let want = if lead & 0x80 == 0x00 { 1 } |
| 704 | else if lead & 0xe0 == 0xc0 { 2 } |
| 705 | else if lead & 0xf0 == 0xe0 { 3 } |
| 706 | else if lead & 0xf8 == 0xf0 { 4 } |
| 707 | else { 0 }; // a stray continuation, which is an error |
| 708 | // `back` continuation bytes followed the leader, so the character in hand is |
| 709 | // `1 + back` bytes long. Short of what the leader promised means it was cut. |
| 710 | end = if want > 0 && 1 + back < want { end - 1 } else { b.len() }; |
| 711 | } |
| 712 | let head = &b[..end]; |
| 713 | if core::str::from_utf8(head).is_err() { |
| 714 | return false; |
| 715 | } |
| 716 | // Tab, newline and carriage return are text; the rest of C0, and DEL, are not. One in ten |
| 717 | // is generous -- ordinary prose has none at all -- and it is set there so that a file with a |
| 718 | // stray form feed or a few ANSI escapes is still readable rather than being hidden. |
| 719 | let ctrl = head.iter() |
| 720 | .filter(|&&c| (c < 0x20 && c != b'\t' && c != b'\n' && c != b'\r') || c == 0x7f) |
| 721 | .count(); |
| 722 | ctrl * 10 <= head.len() |
| 723 | } |
| 724 | |
| 725 | /// Whether `b` begins with `sig`. |
| 726 | fn starts(b: &[u8], sig: &[u8]) -> bool { |
| 727 | b.len() >= sig.len() && &b[..sig.len()] == sig |
| 728 | } |
| 729 | |
| 730 | /// The first `n` bytes of `b` as a string, stopping at the last whole character. |
| 731 | fn lead(b: &[u8], n: usize) -> String { |
| 732 | let mut end = b.len().min(n); |
| 733 | while end > 0 && core::str::from_utf8(&b[..end]).is_err() { |
| 734 | end -= 1; |
| 735 | } |
| 736 | String::from_utf8_lossy(&b[..end]).to_string() |
| 737 | } |
| 738 | |
| 739 | #[cfg(test)] |
| 740 | mod tests { |
| 741 | use super::*; |
| 742 | |
| 743 | #[test] |
| 744 | fn test_pictures_are_known_by_their_signatures() { |
| 745 | assert_eq!(Media::sniff(b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR"), Media::Png); |
| 746 | assert_eq!(Media::sniff(b"\xff\xd8\xff\xe0\x00\x10JFIF"), Media::Jpeg); |
| 747 | assert_eq!(Media::sniff(b"GIF89a"), Media::Gif); |
| 748 | assert_eq!(Media::sniff(b"RIFF\x24\x00\x00\x00WEBPVP8 "), Media::Webp); |
| 749 | assert_eq!(Media::sniff(b"II*\x00\x08\x00\x00\x00"), Media::Tiff); |
| 750 | assert_eq!(Media::sniff(b"MM\x00*\x00\x00\x00\x08"), Media::Tiff); |
| 751 | } |
| 752 | |
| 753 | #[test] |
| 754 | fn test_a_riff_container_is_read_for_what_it_holds() { |
| 755 | // The whole point of checking bytes 8..12: three formats share four leading bytes. |
| 756 | assert_eq!(Media::sniff(b"RIFF\x00\x00\x00\x00WAVEfmt "), Media::Wav); |
| 757 | assert_eq!(Media::sniff(b"RIFF\x00\x00\x00\x00AVI LIST"), Media::Avi); |
| 758 | assert_eq!(Media::sniff(b"RIFF\x00\x00\x00\x00WEBPVP8L"), Media::Webp); |
| 759 | // A RIFF of some other kind is not guessed at. |
| 760 | assert_eq!(Media::sniff(b"RIFF\x00\x00\x00\x00ZZZZ"), Media::Unknown); |
| 761 | } |
| 762 | |
| 763 | #[test] |
| 764 | fn test_an_iso_base_media_brand_decides_between_five_formats() { |
| 765 | assert_eq!(Media::sniff(b"\x00\x00\x00\x18ftypavif\x00\x00\x00\x00"), Media::Avif); |
| 766 | assert_eq!(Media::sniff(b"\x00\x00\x00\x18ftypheic\x00\x00\x00\x00"), Media::Heic); |
| 767 | assert_eq!(Media::sniff(b"\x00\x00\x00\x18ftypM4A \x00\x00\x00\x00"), Media::M4a); |
| 768 | assert_eq!(Media::sniff(b"\x00\x00\x00\x14ftypqt \x00\x00\x00\x00"), Media::QuickTime); |
| 769 | // An unrecognised brand in a recognised container is MP4, which is what the container |
| 770 | // is; answering Unknown would hide a playable file. |
| 771 | assert_eq!(Media::sniff(b"\x00\x00\x00\x18ftypisom\x00\x00\x00\x00"), Media::Mp4); |
| 772 | } |
| 773 | |
| 774 | #[test] |
| 775 | fn test_the_pdf_that_started_this() { |
| 776 | // The defect this module was written for: a PDF's first 4 KB carry no NUL byte, so a |
| 777 | // sniff that looks only for NUL calls it text, and a lossy UTF-8 decode then puts a |
| 778 | // screenful of replacement characters in front of the reader. |
| 779 | let head = b"%PDF-1.7\n%\xe2\xe3\xcf\xd3\n1 0 obj\n<< /Type /Catalog >>\n"; |
| 780 | assert_eq!(Media::sniff(head), Media::Pdf); |
| 781 | assert!(!head.contains(&0), "the header this fails on genuinely has no NUL in it"); |
| 782 | } |
| 783 | |
| 784 | #[test] |
| 785 | fn test_text_shaped_markup_is_recognised_only_when_the_bytes_are_characters() { |
| 786 | assert_eq!(Media::sniff(b"<svg xmlns=\"http://www.w3.org/2000/svg\"></svg>"), Media::Svg); |
| 787 | assert_eq!(Media::sniff(b"<?xml version=\"1.0\"?><svg></svg>"), Media::Svg); |
| 788 | assert_eq!(Media::sniff(b"<?xml version=\"1.0\"?><rss></rss>"), Media::Xml); |
| 789 | assert_eq!(Media::sniff(b"<!DOCTYPE html><html>"), Media::Html); |
| 790 | assert_eq!(Media::sniff(b"Just some words.\n"), Media::Text); |
| 791 | } |
| 792 | |
| 793 | #[test] |
| 794 | fn test_bytes_beat_the_name_and_the_disagreement_is_reported() { |
| 795 | let id = identify("photo.png", b"%PDF-1.4\n"); |
| 796 | assert_eq!(id.media, Media::Pdf, "act on the evidence"); |
| 797 | assert_eq!(id.by_name, Media::Png); |
| 798 | assert!(id.disagree, "and the caller must be able to say so"); |
| 799 | } |
| 800 | |
| 801 | #[test] |
| 802 | fn test_a_name_refines_a_weak_magic_answer_rather_than_fighting_it() { |
| 803 | // CSV has no signature. "These bytes are characters" and "this is a CSV" are not in |
| 804 | // conflict, and reporting them as conflict would put a warning on every ordinary file. |
| 805 | let id = identify("rows.csv", b"a,b,c\n1,2,3\n"); |
| 806 | assert_eq!(id.media, Media::Csv); |
| 807 | assert!(!id.disagree); |
| 808 | |
| 809 | // Likewise the ZIP family: a `.docx` IS a ZIP, and saying so as a disagreement would be |
| 810 | // reporting agreement as conflict. |
| 811 | let id = identify("report.docx", b"PK\x03\x04\x14\x00"); |
| 812 | assert_eq!(id.media, Media::Docx); |
| 813 | assert!(!id.disagree); |
| 814 | } |
| 815 | |
| 816 | #[test] |
| 817 | fn test_a_prefix_cut_mid_character_is_still_text() { |
| 818 | // The defect this guards: a read of the first N bytes lands in the middle of a |
| 819 | // multi-byte character perhaps once in a few hundred files, and a naive validity check |
| 820 | // then calls an ordinary UTF-8 document binary. Never reproduced on demand. |
| 821 | let mut s = "Ordinary prose, and then a character that is three bytes: 日".as_bytes().to_vec(); |
| 822 | s.pop(); // cut the last continuation byte |
| 823 | assert!(looks_like_text(&s), "a cut character is not evidence of binary"); |
| 824 | s.pop(); |
| 825 | assert!(looks_like_text(&s)); |
| 826 | s.pop(); // now the leader is gone too, cleanly |
| 827 | assert!(looks_like_text(&s)); |
| 828 | } |
| 829 | |
| 830 | #[test] |
| 831 | fn test_what_is_not_text() { |
| 832 | assert!(!looks_like_text(b"\x00"), "a NUL settles it"); |
| 833 | assert!(!looks_like_text(b"before\x00after")); |
| 834 | assert!(!looks_like_text(&[0xff, 0xfe, 0xfd, 0xfc]), "not valid UTF-8"); |
| 835 | assert!(looks_like_text(b""), "an empty file is text, not a hex dump"); |
| 836 | assert!(looks_like_text("héllo — ok\n".as_bytes())); |
| 837 | // Generous, deliberately: a file with a couple of escapes in it is still readable. |
| 838 | assert!(looks_like_text(b"plain text with one \x1b escape in it, which is fine")); |
| 839 | } |
| 840 | |
| 841 | #[test] |
| 842 | fn test_a_name_is_read_only_where_a_name_is() { |
| 843 | assert_eq!(Media::from_name("/some.dir/file.png"), Media::Png); |
| 844 | assert_eq!(Media::from_name("archive.tar.gz"), Media::Gzip, "the last extension wins"); |
| 845 | assert_eq!(Media::from_name(".gitignore"), Media::Unknown, "a dotfile is a name"); |
| 846 | assert_eq!(Media::from_name("Makefile"), Media::Unknown); |
| 847 | assert_eq!(Media::from_name("IMAGE.PNG"), Media::Png, "case is not information here"); |
| 848 | } |
| 849 | |
| 850 | #[test] |
| 851 | fn test_every_format_has_a_kind_a_mime_and_a_label() { |
| 852 | // `kind`, `mime` and `label` match exhaustively with no wildcard arm, so a format added |
| 853 | // to the enum without being classified will not compile. This asserts the other half: |
| 854 | // that none of them answers with a placeholder. |
| 855 | let all = [ |
| 856 | Media::Png, Media::Jpeg, Media::Gif, Media::Webp, Media::Avif, Media::Heic, |
| 857 | Media::Bmp, Media::Ico, Media::Tiff, Media::Svg, Media::Pdf, Media::Rtf, |
| 858 | Media::PostScript, Media::Html, Media::Xml, Media::Markdown, Media::Json, |
| 859 | Media::Csv, Media::Tsv, Media::Zip, Media::Gzip, Media::Bzip2, Media::Xz, |
| 860 | Media::Zstd, Media::Tar, Media::SevenZip, Media::Rar, Media::Docx, Media::Xlsx, |
| 861 | Media::Pptx, Media::Odt, Media::Ods, Media::Odp, Media::Odg, |
| 862 | Media::Epub, Media::Mp3, Media::Wav, |
| 863 | Media::Flac, Media::Ogg, Media::M4a, Media::Mp4, Media::Webm, Media::Matroska, |
| 864 | Media::Avi, Media::QuickTime, Media::Ttf, Media::Otf, Media::Woff, Media::Woff2, |
| 865 | Media::Elf, Media::Exe, Media::Wasm, Media::JavaClass, Media::Text, |
| 866 | ]; |
| 867 | for m in all.iter() { |
| 868 | assert_ne!(m.kind(), Kind::Unknown, "{:?} has no kind", m); |
| 869 | assert!(!m.label().is_empty(), "{:?} has no label", m); |
| 870 | assert!(m.mime().contains('/'), "{:?} has no media type", m); |
| 871 | } |
| 872 | assert_eq!(Media::Unknown.kind(), Kind::Unknown); |
| 873 | } |
| 874 | } |