oxedyne/fe2o3/fe2o3_graphics/src/matroska.rs
46.8 KiB, 142 runs
created by r1870400018:21768, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Matroska: the size, the running time and the streams of a film in an EBML |
| 2 | //! container. |
| 3 | //! |
| 4 | //! Matroska is what a film collection is mostly written in, and `.mkv` is the |
| 5 | //! extension a catalogue meets far more often than `.avi`. Like [`crate::avi`] |
| 6 | //! this reads the header and nothing else: **no Matroska is decoded here**. What |
| 7 | //! is inside one is H.264, HEVC or AV1, and a caller wanting a frame has a |
| 8 | //! decoder for that stream's own codec or has none. |
| 9 | //! |
| 10 | //! What a catalogue needs is the size of the picture, how long it runs, and |
| 11 | //! **what every stream is coded in** -- the last because a browser will play a |
| 12 | //! film or refuse it on the strength of the codec names alone, and a library |
| 13 | //! that cannot say which of its films will play is a library that offers a |
| 14 | //! black rectangle. So every track is reported, not only the picture: the |
| 15 | //! audio, where AC-3 is common and no browser decodes it, and the subtitles, |
| 16 | //! which are worth offering separately. |
| 17 | //! |
| 18 | //! # The shape of the file |
| 19 | //! |
| 20 | //! EBML is a tree of elements, each an identifier, a length, and that many |
| 21 | //! bytes. Both identifier and length are variable-width integers whose first |
| 22 | //! byte says how wide they are, by the position of its highest set bit: a |
| 23 | //! leading `1` is one byte, `01` two, and so on. An identifier keeps those |
| 24 | //! marker bits -- they are part of it -- while a length drops them, and a |
| 25 | //! length whose value bits are all set means *unknown*, which is how a file |
| 26 | //! still being written says "to the end". |
| 27 | //! |
| 28 | //! Every element that matters here is near the front: |
| 29 | //! |
| 30 | //! - `EBML`, holding `DocType`, which separates Matroska from WebM. |
| 31 | //! - `Segment`, the body of the file, holding |
| 32 | //! - `Info` -- `TimestampScale` and `Duration`, whose product is the running |
| 33 | //! time; the duration is a *float* in scale units, not a count of anything. |
| 34 | //! - `Tracks` -- a `TrackEntry` a stream, each with its `TrackType`, |
| 35 | //! `CodecID`, language, and, for a picture, its pixel size. |
| 36 | //! |
| 37 | //! [`Matroska::read`] skips the `Cluster`s, which are the film itself, by their |
| 38 | //! length without looking at them. [`Clusters`] is the other half, for a caller |
| 39 | //! that wants the coded frames rather than the description -- repackaging the |
| 40 | //! streams into another container, which needs every frame and decodes none of |
| 41 | //! them. |
| 42 | //! |
| 43 | //! # References |
| 44 | //! |
| 45 | //! IETF RFC 8794 for EBML, and the Matroska specification (RFC 9559) for the |
| 46 | //! element identifiers and the meaning of `TimestampScale`. |
| 47 | //! |
| 48 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 49 | //! Anthropic Claude |
| 50 | |
| 51 | use oxedyne_fe2o3_core::prelude::*; |
| 52 | |
| 53 | // the first four bytes of every EBML file: the EBML element's identifier |
| 54 | const MAGIC: [u8; 4] = [0x1A, 0x45, 0xDF, 0xA3]; |
| 55 | |
| 56 | // The most elements walked at one level before a file is given up on. Without a |
| 57 | // bound, a length of nought -- which a truncated or malformed file readily |
| 58 | // supplies -- is an endless walk. A Segment holds thousands of clusters, so this |
| 59 | // must be generous enough that Tracks is still reached in a file that puts |
| 60 | // clusters before it. |
| 61 | const ELEMENT_LIMIT: usize = 4096; |
| 62 | |
| 63 | // The deepest the walk descends. Segment → Tracks → TrackEntry → Video is four, |
| 64 | // and nothing wanted here is deeper. A file claiming an element contains itself |
| 65 | // is one that would otherwise be walked for ever. |
| 66 | const DEPTH_LIMIT: usize = 6; |
| 67 | |
| 68 | // The default TimestampScale, in nanoseconds, where a file states none. A |
| 69 | // million nanoseconds is a millisecond, so a duration in scale units is a |
| 70 | // duration in milliseconds unless the file says otherwise. Nearly every file |
| 71 | // leaves it at this and states it anyway. |
| 72 | const DEFAULT_SCALE: u64 = 1_000_000; |
| 73 | |
| 74 | const ID_EBML: u64 = 0x1A45DFA3; |
| 75 | const ID_DOC_TYPE: u64 = 0x4282; |
| 76 | const ID_SEGMENT: u64 = 0x18538067; |
| 77 | const ID_INFO: u64 = 0x1549A966; |
| 78 | const ID_TIMESTAMP_SCALE: u64 = 0x2AD7B1; |
| 79 | const ID_DURATION: u64 = 0x4489; |
| 80 | const ID_TITLE: u64 = 0x7BA9; |
| 81 | const ID_TRACKS: u64 = 0x1654AE6B; |
| 82 | const ID_TRACK_ENTRY: u64 = 0xAE; |
| 83 | const ID_TRACK_NUMBER: u64 = 0xD7; |
| 84 | const ID_TRACK_TYPE: u64 = 0x83; |
| 85 | const ID_CODEC_ID: u64 = 0x86; |
| 86 | const ID_CODEC_PRIVATE: u64 = 0x63A2; |
| 87 | const ID_NAME: u64 = 0x536E; |
| 88 | const ID_LANGUAGE: u64 = 0x22B59C; |
| 89 | const ID_LANGUAGE_BCP47: u64 = 0x22B59D; |
| 90 | const ID_FLAG_DEFAULT: u64 = 0x88; |
| 91 | const ID_DEFAULT_DURATION: u64 = 0x23E383; |
| 92 | const ID_FLAG_FORCED: u64 = 0x55AA; |
| 93 | const ID_VIDEO: u64 = 0xE0; |
| 94 | const ID_PIXEL_W: u64 = 0xB0; |
| 95 | const ID_PIXEL_H: u64 = 0xBA; |
| 96 | const ID_DISPLAY_W: u64 = 0x54B0; |
| 97 | const ID_DISPLAY_H: u64 = 0x54BA; |
| 98 | const ID_AUDIO: u64 = 0xE1; |
| 99 | const ID_CHANNELS: u64 = 0x9F; |
| 100 | const ID_SAMPLING: u64 = 0xB5; |
| 101 | const ID_CLUSTER: u64 = 0x1F43B675; |
| 102 | const ID_TIMESTAMP: u64 = 0xE7; |
| 103 | const ID_SIMPLE_BLOCK: u64 = 0xA3; |
| 104 | const ID_BLOCK_GROUP: u64 = 0xA0; |
| 105 | const ID_BLOCK: u64 = 0xA1; |
| 106 | const ID_REFERENCE_BLOCK: u64 = 0xFB; |
| 107 | |
| 108 | // The widest element header: a four-byte identifier and an eight-byte length. |
| 109 | // What a caller must have in hand before Clusters::feed can say anything about |
| 110 | // the element in front of it, and therefore the smallest useful window. |
| 111 | const HEADER_MAX: usize = 12; |
| 112 | |
| 113 | /// What a stream is for. |
| 114 | /// |
| 115 | /// The numbers are the file's own. `Other` keeps a value this reader does not |
| 116 | /// name rather than discarding it, because a track it cannot classify is still |
| 117 | /// a track that is there. |
| 118 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 119 | pub enum TrackKind { |
| 120 | Video, |
| 121 | Audio, |
| 122 | Subtitle, |
| 123 | Other(u64), // logos, buttons, control and metadata tracks, and anything later |
| 124 | } |
| 125 | |
| 126 | impl TrackKind { |
| 127 | |
| 128 | /// The kind a `TrackType` value names. |
| 129 | fn of(n: u64) -> Self { |
| 130 | match n { |
| 131 | 1 => Self::Video, |
| 132 | 2 => Self::Audio, |
| 133 | 0x11 => Self::Subtitle, |
| 134 | other => Self::Other(other), |
| 135 | } |
| 136 | } |
| 137 | |
| 138 | /// The short name used in the interface. |
| 139 | pub fn name(&self) -> &'static str { |
| 140 | match self { |
| 141 | Self::Video => "video", |
| 142 | Self::Audio => "audio", |
| 143 | Self::Subtitle => "subtitle", |
| 144 | Self::Other(_) => "other", |
| 145 | } |
| 146 | } |
| 147 | } |
| 148 | |
| 149 | /// One stream, as its `TrackEntry` describes it. |
| 150 | #[derive(Clone, Debug, Default, PartialEq)] |
| 151 | pub struct Track { |
| 152 | number: u64, // the file's own, which its blocks are keyed by |
| 153 | kind: Option<TrackKind>, // absent until the file says |
| 154 | codec: String, // as the file names it: V_MPEG4/ISO/AVC, A_AC3 |
| 155 | // The codec's own configuration -- an avcC or hvcC record for a picture. |
| 156 | // Kept because it is what a caller repackaging the stream into another |
| 157 | // container must copy across, and it is tens of bytes. |
| 158 | private: Vec<u8>, |
| 159 | w: u32, // pixel size, nought if not a picture |
| 160 | h: u32, |
| 161 | dw: u32, // the shape to show it at, if it differs |
| 162 | dh: u32, |
| 163 | channels: u64, // how many channels the audio carries |
| 164 | rate: f64, // samples a second, which a repackager must state |
| 165 | lang: String, // as the file states it, preferring a BCP 47 tag |
| 166 | name: String, // what a player shows for the stream |
| 167 | default: bool, // should a player choose it unasked? |
| 168 | forced: bool, // must a player show it whatever was asked for? |
| 169 | // Nanoseconds one frame of this stream lasts, wanted for one reason and it |
| 170 | // is not decorative: the frames of a laced block are spaced by it. A block |
| 171 | // carrying six frames of sound states one timestamp, and the five after |
| 172 | // the first are that stamp plus one, two, three of these. Without it they |
| 173 | // all appear at the same instant and a repackaged film's sound walks away |
| 174 | // from its picture. |
| 175 | frame_ns: u64, |
| 176 | } |
| 177 | |
| 178 | impl Track { |
| 179 | |
| 180 | pub fn number(&self) -> u64 { self.number } |
| 181 | |
| 182 | pub fn kind(&self) -> Option<TrackKind> { self.kind } |
| 183 | |
| 184 | /// The codec, as the file names it. |
| 185 | /// |
| 186 | /// The name is not interpreted here. A caller deciding whether it can play |
| 187 | /// the stream is the one that knows, and the names are stable strings the |
| 188 | /// specification fixes. |
| 189 | pub fn codec(&self) -> &str { &self.codec } |
| 190 | |
| 191 | /// The codec's own configuration record, empty where the file carried none. |
| 192 | pub fn private(&self) -> &[u8] { &self.private } |
| 193 | |
| 194 | /// The size of the picture in pixels, nought where the stream is not one. |
| 195 | pub fn size(&self) -> (u32, u32) { (self.w, self.h) } |
| 196 | |
| 197 | /// The shape the picture is meant to be shown at. |
| 198 | /// |
| 199 | /// Falls back to the pixel size, which is what a file states nothing means. |
| 200 | pub fn display(&self) -> (u32, u32) { |
| 201 | let w = if self.dw > 0 { self.dw } else { self.w }; |
| 202 | let h = if self.dh > 0 { self.dh } else { self.h }; |
| 203 | (w, h) |
| 204 | } |
| 205 | |
| 206 | /// How many channels the audio carries, nought where the file said none. |
| 207 | pub fn channels(&self) -> u64 { self.channels } |
| 208 | |
| 209 | /// Samples a second, nought where the file said none. |
| 210 | pub fn rate(&self) -> f64 { self.rate } |
| 211 | |
| 212 | /// The language the file states, empty where it states none. |
| 213 | pub fn language(&self) -> &str { &self.lang } |
| 214 | |
| 215 | /// The name a player shows for the stream, empty where it has none. |
| 216 | pub fn name(&self) -> &str { &self.name } |
| 217 | |
| 218 | /// Should a player choose this stream without being asked? |
| 219 | pub fn is_default(&self) -> bool { self.default } |
| 220 | |
| 221 | /// Must a player show it whatever the viewer asked for? |
| 222 | pub fn is_forced(&self) -> bool { self.forced } |
| 223 | |
| 224 | /// Nanoseconds one frame lasts, nought where the file states none. |
| 225 | /// |
| 226 | /// See the field: this is what spaces the frames of a laced block. |
| 227 | pub fn frame_nanos(&self) -> u64 { self.frame_ns } |
| 228 | } |
| 229 | |
| 230 | /// What a film's header says about it. |
| 231 | #[derive(Clone, Debug, Default, PartialEq)] |
| 232 | pub struct Matroska { |
| 233 | doctype: String, // matroska or webm; the two share this structure |
| 234 | title: String, // the title the file carries, where it carries one |
| 235 | scale: u64, // nanoseconds a timestamp unit stands for |
| 236 | duration: f64, // in scale units; a float, and the file's own word |
| 237 | tracks: Vec<Track>, // in the order the Tracks element described them |
| 238 | } |
| 239 | |
| 240 | impl Matroska { |
| 241 | |
| 242 | /// Reads what the header says, from the front of a file. |
| 243 | /// |
| 244 | /// The buffer need not be the whole film. Everything wanted here is at the |
| 245 | /// front, and an element running past the end of what is in hand simply ends |
| 246 | /// the walk -- so a caller holding a sniffing buffer gets the same answer as |
| 247 | /// one holding the file, provided `Tracks` was within it. A file that put its |
| 248 | /// cover art before its tracks may need more than a scanner's usual head, so |
| 249 | /// a caller that finds no tracks and cares should read further and ask again. |
| 250 | pub fn read(bytes: &[u8]) -> Outcome<Self> { |
| 251 | if !is_matroska(bytes) { |
| 252 | return Err(err!( |
| 253 | "Not a Matroska file: the EBML identifier 1A 45 DF A3 was expected."; |
| 254 | Invalid, Input, Format)); |
| 255 | } |
| 256 | let mut out = Self { |
| 257 | scale: DEFAULT_SCALE, |
| 258 | ..Default::default() |
| 259 | }; |
| 260 | res!(out.walk(bytes, Scope::Top, 0)); |
| 261 | if out.doctype.is_empty() { |
| 262 | return Err(err!( |
| 263 | "The EBML header stated no document type."; |
| 264 | Invalid, Input, Missing)); |
| 265 | } |
| 266 | Ok(out) |
| 267 | } |
| 268 | |
| 269 | /// Walks a run of elements, descending only into those that hold something |
| 270 | /// wanted. |
| 271 | /// |
| 272 | /// `scope` is the element this run sits inside, which is what makes the |
| 273 | /// short identifiers unambiguous -- `0xE0` is `Video` inside a `TrackEntry` |
| 274 | /// and something else elsewhere. |
| 275 | fn walk(&mut self, mut b: &[u8], scope: Scope, depth: usize) -> Outcome<()> { |
| 276 | if depth > DEPTH_LIMIT { |
| 277 | return Ok(()); |
| 278 | } |
| 279 | let mut seen = 0usize; |
| 280 | while b.len() >= 2 && seen < ELEMENT_LIMIT { |
| 281 | seen += 1; |
| 282 | let (id, id_len) = match vint_id(b) { |
| 283 | Some(v) => v, |
| 284 | None => break, |
| 285 | }; |
| 286 | let (size, size_len) = match vint_size(&b[id_len..]) { |
| 287 | Some(v) => v, |
| 288 | None => break, |
| 289 | }; |
| 290 | let at = id_len + size_len; |
| 291 | if at > b.len() { |
| 292 | break; |
| 293 | } |
| 294 | let rest = &b[at..]; |
| 295 | // An element longer than what is in hand is not a fault: the caller |
| 296 | // may be holding only the front of the file. An unknown length runs |
| 297 | // to the end of what there is, and nothing can follow it. |
| 298 | let (take, unknown) = match size { |
| 299 | Some(n) => ((n as usize).min(rest.len()), false), |
| 300 | None => (rest.len(), true), |
| 301 | }; |
| 302 | let body = &rest[..take]; |
| 303 | |
| 304 | match (scope, id) { |
| 305 | (Scope::Top, ID_EBML) => res!(self.walk(body, Scope::Head, depth + 1)), |
| 306 | (Scope::Top, ID_SEGMENT) => res!(self.walk(body, Scope::Segment, depth + 1)), |
| 307 | (Scope::Head, ID_DOC_TYPE) => self.doctype = text(body), |
| 308 | (Scope::Segment, ID_INFO) => res!(self.walk(body, Scope::Info, depth + 1)), |
| 309 | (Scope::Segment, ID_TRACKS) => res!(self.walk(body, Scope::Tracks, depth + 1)), |
| 310 | (Scope::Info, ID_TIMESTAMP_SCALE) => { |
| 311 | // A scale of nought would make every running time nought. |
| 312 | let n = uint(body); |
| 313 | if n > 0 { |
| 314 | self.scale = n; |
| 315 | } |
| 316 | }, |
| 317 | (Scope::Info, ID_DURATION) => self.duration = float(body), |
| 318 | (Scope::Info, ID_TITLE) => self.title = text(body), |
| 319 | (Scope::Tracks, ID_TRACK_ENTRY) => { |
| 320 | let mut track = Track::default(); |
| 321 | res!(read_entry(&mut track, body, depth + 1)); |
| 322 | self.tracks.push(track); |
| 323 | }, |
| 324 | _ => {}, |
| 325 | } |
| 326 | |
| 327 | if unknown { |
| 328 | break; |
| 329 | } |
| 330 | let step = at.saturating_add(take); |
| 331 | if step == 0 || step > b.len() { |
| 332 | break; |
| 333 | } |
| 334 | b = &b[step..]; |
| 335 | } |
| 336 | Ok(()) |
| 337 | } |
| 338 | |
| 339 | /// `matroska` or `webm`. |
| 340 | pub fn doctype(&self) -> &str { &self.doctype } |
| 341 | |
| 342 | /// Does the file call itself WebM, the subset a browser plays natively? |
| 343 | pub fn is_webm(&self) -> bool { self.doctype == "webm" } |
| 344 | |
| 345 | /// The title the file carries, empty where it carries none. |
| 346 | pub fn title(&self) -> &str { &self.title } |
| 347 | |
| 348 | /// Every stream, in the order the file listed them. |
| 349 | pub fn tracks(&self) -> &[Track] { &self.tracks } |
| 350 | |
| 351 | /// The first picture stream, which is the one a catalogue shows. |
| 352 | /// |
| 353 | /// A file with two is a file with an alternative take in it, and the |
| 354 | /// catalogue wants the one it will show. A default-flagged picture wins over |
| 355 | /// an earlier one that is not flagged, because that is the file saying which |
| 356 | /// it means. |
| 357 | pub fn video(&self) -> Option<&Track> { |
| 358 | let mut first = None; |
| 359 | for t in &self.tracks { |
| 360 | if t.kind != Some(TrackKind::Video) { |
| 361 | continue; |
| 362 | } |
| 363 | if t.default { |
| 364 | return Some(t); |
| 365 | } |
| 366 | if first.is_none() { |
| 367 | first = Some(t); |
| 368 | } |
| 369 | } |
| 370 | first |
| 371 | } |
| 372 | |
| 373 | pub fn audio(&self) -> Vec<&Track> { |
| 374 | self.tracks.iter().filter(|t| t.kind == Some(TrackKind::Audio)).collect() |
| 375 | } |
| 376 | |
| 377 | pub fn subtitles(&self) -> Vec<&Track> { |
| 378 | self.tracks.iter().filter(|t| t.kind == Some(TrackKind::Subtitle)).collect() |
| 379 | } |
| 380 | |
| 381 | /// The size of the picture, in pixels, or nought where there is none. |
| 382 | pub fn size(&self) -> (u32, u32) { |
| 383 | match self.video() { |
| 384 | Some(t) => t.size(), |
| 385 | None => (0, 0), |
| 386 | } |
| 387 | } |
| 388 | |
| 389 | /// How long the film runs, in milliseconds, where the header says. |
| 390 | /// |
| 391 | /// The duration is stated in timestamp units and the scale says how many |
| 392 | /// nanoseconds a unit is, so the two are needed together; a file still being |
| 393 | /// written states neither and gets `None` rather than nought, which is a |
| 394 | /// different claim. |
| 395 | pub fn millis(&self) -> Option<u64> { |
| 396 | if self.duration <= 0.0 || self.scale == 0 { |
| 397 | return None; |
| 398 | } |
| 399 | let ns = self.duration * self.scale as f64; |
| 400 | if !ns.is_finite() || ns < 0.0 { |
| 401 | return None; |
| 402 | } |
| 403 | Some((ns / 1_000_000.0).round() as u64) |
| 404 | } |
| 405 | } |
| 406 | |
| 407 | // ------------------------------------------------------------- the frames |
| 408 | |
| 409 | /// One coded frame, exactly as a cluster carries it. |
| 410 | /// |
| 411 | /// The bytes are the codec's own and nothing here touches them: a caller |
| 412 | /// repackaging the stream writes them into the new container unchanged, which is |
| 413 | /// what makes a repackaging lossless and what makes it possible at all without a |
| 414 | /// decoder. |
| 415 | #[derive(Clone, Copy, Debug)] |
| 416 | pub struct Frame<'a> { |
| 417 | pub track: u64, // which stream it belongs to, matching Track::number |
| 418 | // When it is shown, in timestamp scale units from the start of the film. |
| 419 | // Signed because a block states its own time as a difference from its |
| 420 | // cluster's, and the difference is signed -- a cluster may carry a frame |
| 421 | // shown fractionally before the cluster's own stamp. |
| 422 | pub time: i64, |
| 423 | pub key: bool, // may decoding begin here? |
| 424 | pub invisible: bool, // decode it but do not show it |
| 425 | pub data: &'a [u8], // the coded bytes |
| 426 | } |
| 427 | |
| 428 | /// What one feed of bytes yielded. |
| 429 | /// |
| 430 | /// The two numbers are what lets a caller hold a window rather than a film. See |
| 431 | /// [`Clusters::feed`] for the loop they are meant to drive. |
| 432 | #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] |
| 433 | pub struct Fed { |
| 434 | pub used: usize, // bytes from the front dealt with, which may now be dropped |
| 435 | // How many bytes must be in hand, counting from the new front, before the |
| 436 | // next element can be read. Nought means the window merely ran out between |
| 437 | // elements and any further bytes at all will make progress. A caller that |
| 438 | // cannot supply want -- because the file ended -- has a truncated file. |
| 439 | pub want: usize, |
| 440 | } |
| 441 | |
| 442 | /// A reader over the coded frames in a film's clusters. |
| 443 | /// |
| 444 | /// # Why this is fed rather than handed the file |
| 445 | /// |
| 446 | /// A film is gigabytes and a repackager must never hold one; the whole reason |
| 447 | /// for repackaging in a photo library is to avoid decoding, so spending the |
| 448 | /// memory a decode would have cost defeats it. So this descends into `Segment` |
| 449 | /// and `Cluster` by consuming their **headers alone** and reading their children |
| 450 | /// as though they sat at the top level. What a caller must keep in hand is |
| 451 | /// therefore one *block* -- a frame, a few hundred kilobytes at worst -- and |
| 452 | /// never one cluster, which is megabytes, and never the film. |
| 453 | /// |
| 454 | /// Elements that are not wanted are passed over by their stated length without |
| 455 | /// ever being in the window at all, which matters more than it sounds: a film |
| 456 | /// with cover art carries an `Attachments` element of some megabytes, and a |
| 457 | /// reader that had to hold what it skips would be holding that. |
| 458 | /// |
| 459 | /// # The loop |
| 460 | /// |
| 461 | /// ```ignore |
| 462 | /// let mkv = res!(Matroska::read(&head)); |
| 463 | /// let mut cl = Clusters::new(&mkv); |
| 464 | /// let mut buf = Vec::new(); |
| 465 | /// loop { |
| 466 | /// // Top the window up, then let the reader take what it can. |
| 467 | /// let fed = res!(cl.feed(&buf, &mut |frame| { … ; Ok(()) })); |
| 468 | /// buf.drain(..fed.used); |
| 469 | /// if fed.want > buf.len() { /* read at least fed.want bytes into buf */ } |
| 470 | /// } |
| 471 | /// ``` |
| 472 | #[derive(Clone, Debug, Default)] |
| 473 | pub struct Clusters { |
| 474 | now: i64, // the timestamp of the cluster being read, in scale units |
| 475 | // Bytes still to be passed over before an element begins again, kept as a |
| 476 | // count rather than by holding the bytes, so that skipping a large element |
| 477 | // costs nothing and needs no window. |
| 478 | skip: u64, |
| 479 | scale: u64, // nanoseconds a timestamp unit stands for |
| 480 | // How long one frame of each stream lasts, in nanoseconds, by track number. |
| 481 | // A short list walked linearly, because a film has a handful of streams and |
| 482 | // a map would cost more than it saved. |
| 483 | frames: Vec<(u64, u64)>, |
| 484 | } |
| 485 | |
| 486 | impl Clusters { |
| 487 | |
| 488 | /// A reader positioned at the start of a file. |
| 489 | /// |
| 490 | /// The header is wanted rather than optional: the timestamp scale and each |
| 491 | /// stream's frame duration are stated there, and without them the frames of |
| 492 | /// a laced block cannot be given their own times. A caller has read the |
| 493 | /// header already, since nothing else says which track numbers mean what. |
| 494 | pub fn new(mkv: &Matroska) -> Self { |
| 495 | Self { |
| 496 | now: 0, |
| 497 | skip: 0, |
| 498 | scale: mkv.scale, |
| 499 | frames: mkv.tracks.iter().map(|t| (t.number, t.frame_ns)).collect(), |
| 500 | } |
| 501 | } |
| 502 | |
| 503 | /// Reads whole elements from the front of `b`, handing every frame to `f`. |
| 504 | /// |
| 505 | /// Stops at the first element it cannot complete, and says in [`Fed`] both |
| 506 | /// what it consumed and what it needs. It never partially reports a frame: |
| 507 | /// a block that is not wholly in the window is left for the next feed. |
| 508 | pub fn feed<F>(&mut self, b: &[u8], f: &mut F) -> Outcome<Fed> |
| 509 | where |
| 510 | F: FnMut(Frame) -> Outcome<()>, |
| 511 | { |
| 512 | let mut i = 0usize; |
| 513 | |
| 514 | // Whatever is left of an element being passed over goes first, and it |
| 515 | // is counted rather than held -- `skip` may be larger than any window. |
| 516 | if self.skip > 0 { |
| 517 | let n = self.skip.min(b.len() as u64) as usize; |
| 518 | i += n; |
| 519 | self.skip -= n as u64; |
| 520 | if self.skip > 0 { |
| 521 | return Ok(Fed { used: i, want: 0 }); |
| 522 | } |
| 523 | } |
| 524 | |
| 525 | while i < b.len() { |
| 526 | let rest = &b[i..]; |
| 527 | let (id, id_len) = match vint_id(rest) { |
| 528 | Some(v) => v, |
| 529 | None => return Ok(Fed { used: i, want: HEADER_MAX }), |
| 530 | }; |
| 531 | let (size, size_len) = match vint_size(&rest[id_len..]) { |
| 532 | Some(v) => v, |
| 533 | None => return Ok(Fed { used: i, want: HEADER_MAX }), |
| 534 | }; |
| 535 | let at = id_len + size_len; |
| 536 | |
| 537 | // A `Segment` states an unknown length in a file still being |
| 538 | // written, and it is descended into regardless -- which is the one |
| 539 | // case where an unknown length is not the end of the walk. |
| 540 | let take = match size { |
| 541 | Some(n) => n, |
| 542 | None => 0, |
| 543 | }; |
| 544 | |
| 545 | match id { |
| 546 | // Descended into by consuming the header alone, so their |
| 547 | // children are read as though they sat at the top. This is what |
| 548 | // keeps the window down to one block. |
| 549 | ID_SEGMENT | ID_CLUSTER => { |
| 550 | i += at; |
| 551 | continue; |
| 552 | }, |
| 553 | // Inside a `Cluster`, and unambiguous *because* nothing else is |
| 554 | // descended into: `Info` holds elements of its own but is passed |
| 555 | // over whole, so a bare `0xE7` here can only be a cluster's. |
| 556 | ID_TIMESTAMP | ID_SIMPLE_BLOCK | ID_BLOCK_GROUP => { |
| 557 | let whole = at.saturating_add(take as usize); |
| 558 | if take > b.len() as u64 || whole > rest.len() { |
| 559 | return Ok(Fed { used: i, want: whole }); |
| 560 | } |
| 561 | let body = &rest[at..whole]; |
| 562 | match id { |
| 563 | ID_TIMESTAMP => self.now = uint(body) as i64, |
| 564 | ID_SIMPLE_BLOCK => res!(self.block(body, None, f)), |
| 565 | _ => res!(self.group(body, f)), |
| 566 | } |
| 567 | i += whole; |
| 568 | }, |
| 569 | // Everything else is passed over by its length, and the bytes |
| 570 | // never enter the window. |
| 571 | _ => { |
| 572 | i += at; |
| 573 | self.skip = take; |
| 574 | let n = self.skip.min((b.len() - i) as u64) as usize; |
| 575 | i += n; |
| 576 | self.skip -= n as u64; |
| 577 | if self.skip > 0 { |
| 578 | return Ok(Fed { used: i, want: 0 }); |
| 579 | } |
| 580 | }, |
| 581 | } |
| 582 | } |
| 583 | Ok(Fed { used: i, want: 0 }) |
| 584 | } |
| 585 | |
| 586 | /// Reads a `BlockGroup`, whose `Block` is a keyframe only if nothing in the |
| 587 | /// group refers to another frame. |
| 588 | /// |
| 589 | /// The reference has to be looked for **before** the block is reported, |
| 590 | /// which is why a group is read whole rather than descended into: a |
| 591 | /// `ReferenceBlock` is a sibling of the `Block` and may follow it, so a |
| 592 | /// reader that reported the block on sight would have to take it back. |
| 593 | fn group<F>(&mut self, b: &[u8], f: &mut F) -> Outcome<()> |
| 594 | where |
| 595 | F: FnMut(Frame) -> Outcome<()>, |
| 596 | { |
| 597 | // Walked here rather than through [`each`], which hands its closure a |
| 598 | // slice of anonymous lifetime and so cannot be used to *keep* one. |
| 599 | let mut block: Option<&[u8]> = None; |
| 600 | let mut refers = false; |
| 601 | let mut i = 0usize; |
| 602 | while i < b.len() { |
| 603 | let rest = &b[i..]; |
| 604 | let (id, id_len) = match vint_id(rest) { |
| 605 | Some(v) => v, |
| 606 | None => break, |
| 607 | }; |
| 608 | let (size, size_len) = match vint_size(&rest[id_len..]) { |
| 609 | Some(v) => v, |
| 610 | None => break, |
| 611 | }; |
| 612 | let at = id_len + size_len; |
| 613 | if at > rest.len() { |
| 614 | break; |
| 615 | } |
| 616 | // An unknown length inside a group ends the walk: nothing may follow |
| 617 | // an element that runs to the end of what there is. |
| 618 | let take = match size { |
| 619 | Some(n) => (n as usize).min(rest.len() - at), |
| 620 | None => break, |
| 621 | }; |
| 622 | let body = &rest[at..at + take]; |
| 623 | match id { |
| 624 | ID_BLOCK => block = Some(body), |
| 625 | ID_REFERENCE_BLOCK => refers = true, |
| 626 | _ => {}, |
| 627 | } |
| 628 | let step = at.saturating_add(take); |
| 629 | if step == 0 { |
| 630 | break; |
| 631 | } |
| 632 | i += step; |
| 633 | } |
| 634 | match block { |
| 635 | Some(body) => self.block(body, Some(!refers), f), |
| 636 | // A group with no block in it is not a fault; it is a file carrying |
| 637 | // something this reader does not want. |
| 638 | None => Ok(()), |
| 639 | } |
| 640 | } |
| 641 | |
| 642 | /// Reads one block, whether laced or not, and reports each frame in it. |
| 643 | /// |
| 644 | /// `key` is the answer a `BlockGroup` worked out from its references; a |
| 645 | /// `SimpleBlock` states its own in its flags, so `None` means read it here. |
| 646 | fn block<F>(&mut self, b: &[u8], key: Option<bool>, f: &mut F) -> Outcome<()> |
| 647 | where |
| 648 | F: FnMut(Frame) -> Outcome<()>, |
| 649 | { |
| 650 | let (track, n) = match vint_size(b) { |
| 651 | Some((Some(t), n)) => (t, n), |
| 652 | // An unknown-length track number is not a thing the format allows. |
| 653 | _ => return Err(err!( |
| 654 | "A block stated no readable track number."; |
| 655 | Invalid, Input, Format)), |
| 656 | }; |
| 657 | if b.len() < n + 3 { |
| 658 | return Err(err!( |
| 659 | "A block of {} bytes is too short to carry a track number, a \ |
| 660 | timestamp and its flags.", b.len(); |
| 661 | Invalid, Input, Format)); |
| 662 | } |
| 663 | // The block's own time is a *difference* from its cluster's, and it is |
| 664 | // signed: two bytes, big-endian, two's complement. |
| 665 | let rel = i16::from_be_bytes([b[n], b[n + 1]]) as i64; |
| 666 | let flags = b[n + 2]; |
| 667 | let body = &b[n + 3..]; |
| 668 | |
| 669 | let key = match key { |
| 670 | Some(k) => k, |
| 671 | None => flags & 0x80 != 0, |
| 672 | }; |
| 673 | let frame = Frame { |
| 674 | track, |
| 675 | time: self.now + rel, |
| 676 | key, |
| 677 | invisible: flags & 0x08 != 0, |
| 678 | data: &[], |
| 679 | }; |
| 680 | |
| 681 | // Lacing packs several frames of sound into one block to save the |
| 682 | // per-block overhead. It is uncommon in a film muxed by a modern tool |
| 683 | // and it is not rare enough to refuse: a reader that ignored it would |
| 684 | // hand a caller one frame made of six glued together, which no decoder |
| 685 | // and no container would accept and nothing would say why. |
| 686 | match (flags >> 1) & 0x03 { |
| 687 | 0 => res!(f(Frame { data: body, ..frame })), |
| 688 | lacing => { |
| 689 | // A laced block states ONE time for all of its frames, and the |
| 690 | // rest follow it a frame duration apart. Giving them all the |
| 691 | // block's stamp is what the first version of this did, and the |
| 692 | // sound of a repackaged film then arrived in bursts: the sizes |
| 693 | // were right, every frame was there, and only the clock was |
| 694 | // wrong -- which is why the oracle compares times and not just |
| 695 | // bytes. |
| 696 | let step = self.frames.iter() |
| 697 | .find(|(n, _)| *n == track) |
| 698 | .map(|(_, d)| *d) |
| 699 | .unwrap_or(0); |
| 700 | let parts = res!(laced(body, lacing)); |
| 701 | let laces = parts.len() as u64; |
| 702 | // The block's whole span first, then that divided among its |
| 703 | // frames -- **not** the frame duration multiplied up. The two |
| 704 | // differ because a timestamp unit cannot represent a frame of |
| 705 | // sound: AAC at 48 kHz lasts 21⅓ milliseconds, and eight of |
| 706 | // them are 170⅔. Multiplying accumulates the third of a |
| 707 | // millisecond until the last frame of a block is stamped later |
| 708 | // than dividing says, and in the limit past the start of the |
| 709 | // block after it. Dividing the span keeps every frame inside |
| 710 | // the block that carries it, which is the property that matters |
| 711 | // to whatever plays the result, and it is what a player does. |
| 712 | let span = if self.scale > 0 { |
| 713 | step.saturating_mul(laces) / self.scale |
| 714 | } else { |
| 715 | 0 |
| 716 | }; |
| 717 | for (i, part) in parts.iter().enumerate() { |
| 718 | let on = if laces > 0 { |
| 719 | (i as u64).saturating_mul(span) / laces |
| 720 | } else { |
| 721 | 0 |
| 722 | }; |
| 723 | res!(f(Frame { |
| 724 | data: part, |
| 725 | time: frame.time.saturating_add(on as i64), |
| 726 | ..frame |
| 727 | })); |
| 728 | } |
| 729 | }, |
| 730 | } |
| 731 | Ok(()) |
| 732 | } |
| 733 | } |
| 734 | |
| 735 | /// Splits a laced block body into its frames. |
| 736 | /// |
| 737 | /// The three schemes differ only in how the sizes are written; the last frame's |
| 738 | /// size is never stated in any of them, because it is whatever is left. |
| 739 | fn laced(b: &[u8], lacing: u8) -> Outcome<Vec<&[u8]>> { |
| 740 | let count = match b.first() { |
| 741 | Some(n) => *n as usize + 1, |
| 742 | None => return Err(err!( |
| 743 | "A laced block stated no frame count."; Invalid, Input, Format)), |
| 744 | }; |
| 745 | let mut at = 1usize; |
| 746 | let mut sizes = Vec::with_capacity(count); |
| 747 | match lacing { |
| 748 | // Fixed: every frame the same size, so nothing is written at all. |
| 749 | 2 => { |
| 750 | let rest = b.len() - at; |
| 751 | if count == 0 || rest % count != 0 { |
| 752 | return Err(err!( |
| 753 | "A fixed-laced block of {} bytes does not divide into {} \ |
| 754 | frames.", rest, count; |
| 755 | Invalid, Input, Format)); |
| 756 | } |
| 757 | for _ in 0..count { |
| 758 | sizes.push(rest / count); |
| 759 | } |
| 760 | }, |
| 761 | // Xiph: each size is a run of bytes, ended by one below 255. |
| 762 | 1 => { |
| 763 | for _ in 0..count - 1 { |
| 764 | let mut n = 0usize; |
| 765 | loop { |
| 766 | let byte = match b.get(at) { |
| 767 | Some(v) => *v, |
| 768 | None => return Err(err!( |
| 769 | "A Xiph-laced block ended inside a frame size."; |
| 770 | Invalid, Input, Format)), |
| 771 | }; |
| 772 | at += 1; |
| 773 | n += byte as usize; |
| 774 | if byte < 255 { |
| 775 | break; |
| 776 | } |
| 777 | } |
| 778 | sizes.push(n); |
| 779 | } |
| 780 | }, |
| 781 | // EBML: the first size is a plain variable-width integer and every one |
| 782 | // after it is a *signed difference* from the one before. |
| 783 | _ => { |
| 784 | let (first, n) = match vint_size(&b[at..]) { |
| 785 | Some((Some(v), n)) => (v as i64, n), |
| 786 | _ => return Err(err!( |
| 787 | "An EBML-laced block stated no readable first frame size."; |
| 788 | Invalid, Input, Format)), |
| 789 | }; |
| 790 | at += n; |
| 791 | sizes.push(first as usize); |
| 792 | let mut prev = first; |
| 793 | for _ in 0..count.saturating_sub(2) { |
| 794 | let (delta, n) = match svint(&b[at..]) { |
| 795 | Some(v) => v, |
| 796 | None => return Err(err!( |
| 797 | "An EBML-laced block ended inside a frame size."; |
| 798 | Invalid, Input, Format)), |
| 799 | }; |
| 800 | at += n; |
| 801 | prev += delta; |
| 802 | if prev < 0 { |
| 803 | return Err(err!( |
| 804 | "An EBML-laced block gave a frame a negative size."; |
| 805 | Invalid, Input, Format)); |
| 806 | } |
| 807 | sizes.push(prev as usize); |
| 808 | } |
| 809 | }, |
| 810 | } |
| 811 | |
| 812 | // The last frame is whatever the stated ones leave, and a file whose stated |
| 813 | // sizes overrun the block is one this must not read past the end of. |
| 814 | let stated: usize = sizes.iter().take(count - 1).sum(); |
| 815 | if at.saturating_add(stated) > b.len() { |
| 816 | return Err(err!( |
| 817 | "A laced block states {} bytes of frames in {} bytes of body.", |
| 818 | stated, b.len() - at.min(b.len()); |
| 819 | Invalid, Input, Format)); |
| 820 | } |
| 821 | if sizes.len() < count { |
| 822 | sizes.push(b.len() - at - stated); |
| 823 | } |
| 824 | |
| 825 | let mut out = Vec::with_capacity(count); |
| 826 | for size in sizes.iter().take(count) { |
| 827 | let end = at.saturating_add(*size); |
| 828 | if end > b.len() { |
| 829 | return Err(err!( |
| 830 | "A laced block's frame runs past the end of it."; |
| 831 | Invalid, Input, Format)); |
| 832 | } |
| 833 | out.push(&b[at..end]); |
| 834 | at = end; |
| 835 | } |
| 836 | Ok(out) |
| 837 | } |
| 838 | |
| 839 | /// A signed variable-width integer, as EBML lacing writes a size difference. |
| 840 | /// |
| 841 | /// The unsigned value is read exactly as a length is, then shifted down by half |
| 842 | /// its range, which is what makes the differences either way round representable. |
| 843 | fn svint(b: &[u8]) -> Option<(i64, usize)> { |
| 844 | let (v, n) = match vint_size(b) { |
| 845 | Some((Some(v), n)) => (v, n), |
| 846 | _ => return None, |
| 847 | }; |
| 848 | let bias = (1i64 << (7 * n as u32 - 1)) - 1; |
| 849 | Some((v as i64 - bias, n)) |
| 850 | } |
| 851 | |
| 852 | /// Which element a run of elements sits inside. |
| 853 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 854 | enum Scope { |
| 855 | Top, |
| 856 | Head, |
| 857 | Segment, |
| 858 | Info, |
| 859 | Tracks, |
| 860 | } |
| 861 | |
| 862 | /// Fills a track from its `TrackEntry`, descending into `Video` and `Audio`. |
| 863 | /// |
| 864 | /// This is walked apart from [`Matroska::walk`] because everything it reads |
| 865 | /// belongs to the one track being built, and threading a half-built track |
| 866 | /// through the file-level walk would mean the file-level walk could put a |
| 867 | /// picture's size on whichever track happened to be open. |
| 868 | fn read_entry(track: &mut Track, mut b: &[u8], depth: usize) -> Outcome<()> { |
| 869 | if depth > DEPTH_LIMIT { |
| 870 | return Ok(()); |
| 871 | } |
| 872 | let mut seen = 0usize; |
| 873 | while b.len() >= 2 && seen < ELEMENT_LIMIT { |
| 874 | seen += 1; |
| 875 | let (id, id_len) = match vint_id(b) { |
| 876 | Some(v) => v, |
| 877 | None => break, |
| 878 | }; |
| 879 | let (size, size_len) = match vint_size(&b[id_len..]) { |
| 880 | Some(v) => v, |
| 881 | None => break, |
| 882 | }; |
| 883 | let at = id_len + size_len; |
| 884 | if at > b.len() { |
| 885 | break; |
| 886 | } |
| 887 | let rest = &b[at..]; |
| 888 | let (take, unknown) = match size { |
| 889 | Some(n) => ((n as usize).min(rest.len()), false), |
| 890 | None => (rest.len(), true), |
| 891 | }; |
| 892 | let body = &rest[..take]; |
| 893 | |
| 894 | match id { |
| 895 | ID_TRACK_NUMBER => track.number = uint(body), |
| 896 | ID_TRACK_TYPE => track.kind = Some(TrackKind::of(uint(body))), |
| 897 | ID_CODEC_ID => track.codec = text(body), |
| 898 | ID_CODEC_PRIVATE => track.private = body.to_vec(), |
| 899 | ID_NAME => track.name = text(body), |
| 900 | // A BCP 47 tag is the later field and the more precise, so it wins; |
| 901 | // the older three-letter field must not overwrite it. |
| 902 | ID_LANGUAGE_BCP47 => track.lang = text(body), |
| 903 | ID_LANGUAGE => if track.lang.is_empty() { track.lang = text(body) }, |
| 904 | ID_FLAG_DEFAULT => track.default = uint(body) != 0, |
| 905 | ID_FLAG_FORCED => track.forced = uint(body) != 0, |
| 906 | ID_DEFAULT_DURATION => track.frame_ns = uint(body), |
| 907 | ID_VIDEO => res!(read_video(track, body, depth + 1)), |
| 908 | ID_AUDIO => res!(read_audio(track, body, depth + 1)), |
| 909 | _ => {}, |
| 910 | } |
| 911 | |
| 912 | if unknown { |
| 913 | break; |
| 914 | } |
| 915 | let step = at.saturating_add(take); |
| 916 | if step == 0 || step > b.len() { |
| 917 | break; |
| 918 | } |
| 919 | b = &b[step..]; |
| 920 | } |
| 921 | Ok(()) |
| 922 | } |
| 923 | |
| 924 | /// Fills a track's picture size from its `Video` element. |
| 925 | fn read_video(track: &mut Track, b: &[u8], depth: usize) -> Outcome<()> { |
| 926 | each(b, depth, &mut |id, body| { |
| 927 | match id { |
| 928 | ID_PIXEL_W => track.w = uint(body) as u32, |
| 929 | ID_PIXEL_H => track.h = uint(body) as u32, |
| 930 | ID_DISPLAY_W => track.dw = uint(body) as u32, |
| 931 | ID_DISPLAY_H => track.dh = uint(body) as u32, |
| 932 | _ => {}, |
| 933 | } |
| 934 | }) |
| 935 | } |
| 936 | |
| 937 | /// Fills a track's channel count and sampling rate from its `Audio` element. |
| 938 | fn read_audio(track: &mut Track, b: &[u8], depth: usize) -> Outcome<()> { |
| 939 | each(b, depth, &mut |id, body| { |
| 940 | match id { |
| 941 | ID_CHANNELS => track.channels = uint(body), |
| 942 | ID_SAMPLING => track.rate = float(body), |
| 943 | _ => {}, |
| 944 | } |
| 945 | }) |
| 946 | } |
| 947 | |
| 948 | /// Walks a run of leaf elements, handing each identifier and body to `f`. |
| 949 | /// |
| 950 | /// The two innermost elements hold nothing but leaves, so they need none of the |
| 951 | /// descent [`Matroska::walk`] carries and share this instead. |
| 952 | fn each(mut b: &[u8], depth: usize, f: &mut dyn FnMut(u64, &[u8])) -> Outcome<()> { |
| 953 | if depth > DEPTH_LIMIT { |
| 954 | return Ok(()); |
| 955 | } |
| 956 | let mut seen = 0usize; |
| 957 | while b.len() >= 2 && seen < ELEMENT_LIMIT { |
| 958 | seen += 1; |
| 959 | let (id, id_len) = match vint_id(b) { |
| 960 | Some(v) => v, |
| 961 | None => break, |
| 962 | }; |
| 963 | let (size, size_len) = match vint_size(&b[id_len..]) { |
| 964 | Some(v) => v, |
| 965 | None => break, |
| 966 | }; |
| 967 | let at = id_len + size_len; |
| 968 | if at > b.len() { |
| 969 | break; |
| 970 | } |
| 971 | let rest = &b[at..]; |
| 972 | let take = match size { |
| 973 | Some(n) => (n as usize).min(rest.len()), |
| 974 | None => break, |
| 975 | }; |
| 976 | f(id, &rest[..take]); |
| 977 | let step = at.saturating_add(take); |
| 978 | if step == 0 || step > b.len() { |
| 979 | break; |
| 980 | } |
| 981 | b = &b[step..]; |
| 982 | } |
| 983 | Ok(()) |
| 984 | } |
| 985 | |
| 986 | /// Does a head begin an EBML file? |
| 987 | /// |
| 988 | /// This does not say the file is Matroska rather than WebM, or that it is |
| 989 | /// either: both, and any other EBML document, open the same way. The `DocType` |
| 990 | /// inside is what separates them, and [`Matroska::read`] reports it. |
| 991 | pub fn is_matroska(head: &[u8]) -> bool { |
| 992 | head.len() >= 4 && head[..4] == MAGIC |
| 993 | } |
| 994 | |
| 995 | /// An element identifier, with its marker bits kept. |
| 996 | /// |
| 997 | /// The marker is part of the identifier -- `0x83` is `TrackType`, not `0x03` -- |
| 998 | /// which is why this differs from [`vint_size`] in exactly that respect. |
| 999 | fn vint_id(b: &[u8]) -> Option<(u64, usize)> { |
| 1000 | let first = match b.first() { |
| 1001 | Some(f) => *f, |
| 1002 | None => return None, |
| 1003 | }; |
| 1004 | if first == 0 { |
| 1005 | // No marker bit in the first byte means an identifier wider than the |
| 1006 | // four bytes the specification allows. |
| 1007 | return None; |
| 1008 | } |
| 1009 | let len = first.leading_zeros() as usize + 1; |
| 1010 | if len > 4 || b.len() < len { |
| 1011 | return None; |
| 1012 | } |
| 1013 | let mut v = 0u64; |
| 1014 | for byte in &b[..len] { |
| 1015 | v = (v << 8) | *byte as u64; |
| 1016 | } |
| 1017 | Some((v, len)) |
| 1018 | } |
| 1019 | |
| 1020 | /// An element length, with its marker bits removed. |
| 1021 | /// |
| 1022 | /// `None` for the length means the file stated *unknown*: every value bit set, |
| 1023 | /// which is how an element that runs to the end of the file is written. |
| 1024 | fn vint_size(b: &[u8]) -> Option<(Option<u64>, usize)> { |
| 1025 | let first = match b.first() { |
| 1026 | Some(f) => *f, |
| 1027 | None => return None, |
| 1028 | }; |
| 1029 | if first == 0 { |
| 1030 | return None; |
| 1031 | } |
| 1032 | let len = first.leading_zeros() as usize + 1; |
| 1033 | if len > 8 || b.len() < len { |
| 1034 | return None; |
| 1035 | } |
| 1036 | // The marker bit and everything above it leave the first byte. At eight |
| 1037 | // bytes the marker is the lowest bit of the first byte, so nothing of the |
| 1038 | // value is in it -- and `0xFF >> 8` is not a shift Rust will perform. |
| 1039 | let mask = if len >= 8 { 0u8 } else { 0xFFu8 >> len }; |
| 1040 | let mut v = (first & mask) as u64; |
| 1041 | for byte in &b[1..len] { |
| 1042 | v = (v << 8) | *byte as u64; |
| 1043 | } |
| 1044 | let bits = 7 * len as u32; |
| 1045 | let ones = if bits >= 64 { u64::MAX } else { (1u64 << bits) - 1 }; |
| 1046 | Some((if v == ones { None } else { Some(v) }, len)) |
| 1047 | } |
| 1048 | |
| 1049 | /// A big-endian unsigned integer of whatever width the element gave it. |
| 1050 | /// |
| 1051 | /// EBML writes the narrowest form that fits, so the same field is one byte in |
| 1052 | /// one file and four in the next; a reader expecting a fixed width is wrong on |
| 1053 | /// half the corpus. An empty body is nought, which is what the specification |
| 1054 | /// says a stated-but-empty integer means. |
| 1055 | fn uint(b: &[u8]) -> u64 { |
| 1056 | let mut v = 0u64; |
| 1057 | for byte in b.iter().take(8) { |
| 1058 | v = (v << 8) | *byte as u64; |
| 1059 | } |
| 1060 | v |
| 1061 | } |
| 1062 | |
| 1063 | /// A big-endian IEEE float, four or eight bytes wide. |
| 1064 | /// |
| 1065 | /// Any other width is nought: the specification allows only these two, and a |
| 1066 | /// duration read out of a body of the wrong width would be a plausible number |
| 1067 | /// rather than an obviously missing one. |
| 1068 | fn float(b: &[u8]) -> f64 { |
| 1069 | match b.len() { |
| 1070 | 4 => { |
| 1071 | let mut a = [0u8; 4]; |
| 1072 | a.copy_from_slice(b); |
| 1073 | f32::from_be_bytes(a) as f64 |
| 1074 | }, |
| 1075 | 8 => { |
| 1076 | let mut a = [0u8; 8]; |
| 1077 | a.copy_from_slice(b); |
| 1078 | f64::from_be_bytes(a) |
| 1079 | }, |
| 1080 | _ => 0.0, |
| 1081 | } |
| 1082 | } |
| 1083 | |
| 1084 | /// A string, with the padding a writer may have left on the end removed. |
| 1085 | /// |
| 1086 | /// Element strings are UTF-8 and may be padded with nulls to a length the |
| 1087 | /// writer reserved. Lossy conversion is deliberate: a mis-encoded title is |
| 1088 | /// worth showing with a replacement character in it, and is not worth refusing |
| 1089 | /// a whole film over. |
| 1090 | fn text(b: &[u8]) -> String { |
| 1091 | let end = b.iter().position(|c| *c == 0).unwrap_or(b.len()); |
| 1092 | String::from_utf8_lossy(&b[..end]).trim().to_string() |
| 1093 | } |
| 1094 | |
| 1095 | #[cfg(test)] |
| 1096 | mod tests { |
| 1097 | use super::*; |
| 1098 | |
| 1099 | /// The narrowest length encoding a value fits in. |
| 1100 | /// |
| 1101 | /// The marker is the `len`-th bit from the top, so it lands in the first |
| 1102 | /// byte at `1 << (8 - len)`, and the value fills the `7 * len` bits below. |
| 1103 | fn size_of(n: u64) -> Vec<u8> { |
| 1104 | for len in 1..=8usize { |
| 1105 | let bits = 7 * len as u32; |
| 1106 | let ones = (1u64 << bits) - 1; |
| 1107 | // The all-ones value means "unknown", so a length that would encode |
| 1108 | // as it needs the next width up. |
| 1109 | if n < ones { |
| 1110 | let mut buf = n.to_be_bytes().to_vec(); |
| 1111 | buf.drain(..8 - len); |
| 1112 | buf[0] |= 1u8 << (8 - len); |
| 1113 | return buf; |
| 1114 | } |
| 1115 | } |
| 1116 | vec![0xFF] |
| 1117 | } |
| 1118 | |
| 1119 | /// An element: its identifier as written, its length, and its body. |
| 1120 | fn el(id: u64, body: &[u8]) -> Vec<u8> { |
| 1121 | let mut out = Vec::new(); |
| 1122 | let idb = id.to_be_bytes(); |
| 1123 | let first = idb.iter().position(|b| *b != 0).unwrap_or(7); |
| 1124 | out.extend_from_slice(&idb[first..]); |
| 1125 | out.extend_from_slice(&size_of(body.len() as u64)); |
| 1126 | out.extend_from_slice(body); |
| 1127 | out |
| 1128 | } |
| 1129 | |
| 1130 | /// An unsigned integer element, written as narrowly as it fits. |
| 1131 | fn u(id: u64, n: u64) -> Vec<u8> { |
| 1132 | let b = n.to_be_bytes(); |
| 1133 | let first = b.iter().position(|x| *x != 0).unwrap_or(7); |
| 1134 | el(id, &b[first..]) |
| 1135 | } |
| 1136 | |
| 1137 | fn s(id: u64, v: &str) -> Vec<u8> { |
| 1138 | el(id, v.as_bytes()) |
| 1139 | } |
| 1140 | |
| 1141 | fn f64_el(id: u64, v: f64) -> Vec<u8> { |
| 1142 | el(id, &v.to_be_bytes()) |
| 1143 | } |
| 1144 | |
| 1145 | fn head(doctype: &str) -> Vec<u8> { |
| 1146 | el(ID_EBML, &s(ID_DOC_TYPE, doctype)) |
| 1147 | } |
| 1148 | |
| 1149 | /// A file: an EBML header and a segment holding whatever is given. |
| 1150 | fn file(doctype: &str, inner: Vec<u8>) -> Vec<u8> { |
| 1151 | let mut out = head(doctype); |
| 1152 | out.extend_from_slice(&el(ID_SEGMENT, &inner)); |
| 1153 | out |
| 1154 | } |
| 1155 | |
| 1156 | fn info(scale: u64, duration: f64) -> Vec<u8> { |
| 1157 | let mut b = u(ID_TIMESTAMP_SCALE, scale); |
| 1158 | b.extend_from_slice(&f64_el(ID_DURATION, duration)); |
| 1159 | el(ID_INFO, &b) |
| 1160 | } |
| 1161 | |
| 1162 | fn video_track(number: u64, codec: &str, w: u64, h: u64) -> Vec<u8> { |
| 1163 | let mut v = u(ID_PIXEL_W, w); |
| 1164 | v.extend_from_slice(&u(ID_PIXEL_H, h)); |
| 1165 | let mut b = u(ID_TRACK_NUMBER, number); |
| 1166 | b.extend_from_slice(&u(ID_TRACK_TYPE, 1)); |
| 1167 | b.extend_from_slice(&s(ID_CODEC_ID, codec)); |
| 1168 | b.extend_from_slice(&el(ID_VIDEO, &v)); |
| 1169 | el(ID_TRACK_ENTRY, &b) |
| 1170 | } |
| 1171 | |
| 1172 | fn audio_track(number: u64, codec: &str, channels: u64, lang: &str) -> Vec<u8> { |
| 1173 | let a = u(ID_CHANNELS, channels); |
| 1174 | let mut b = u(ID_TRACK_NUMBER, number); |
| 1175 | b.extend_from_slice(&u(ID_TRACK_TYPE, 2)); |
| 1176 | b.extend_from_slice(&s(ID_CODEC_ID, codec)); |
| 1177 | b.extend_from_slice(&s(ID_LANGUAGE, lang)); |
| 1178 | b.extend_from_slice(&el(ID_AUDIO, &a)); |
| 1179 | el(ID_TRACK_ENTRY, &b) |
| 1180 | } |
| 1181 | |
| 1182 | #[test] |
| 1183 | fn the_size_and_running_time_are_read() -> Outcome<()> { |
| 1184 | // An hour and a half at the usual scale: 5,400,000 milliseconds. |
| 1185 | let mut inner = info(1_000_000, 5_400_000.0); |
| 1186 | inner.extend_from_slice(&el(ID_TRACKS, &video_track(1, "V_MPEG4/ISO/AVC", 1920, 1080))); |
| 1187 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1188 | |
| 1189 | req!(mkv.doctype(), "matroska"); |
| 1190 | req!(mkv.size(), (1920u32, 1080u32)); |
| 1191 | req!(mkv.millis(), Some(5_400_000u64)); |
| 1192 | Ok(()) |
| 1193 | } |
| 1194 | |
| 1195 | #[test] |
| 1196 | fn the_timestamp_scale_is_applied() -> Outcome<()> { |
| 1197 | // A file whose unit is a microsecond, not a millisecond. The duration |
| 1198 | // reads the same either way, and only the scale tells them apart -- a |
| 1199 | // reader ignoring it says this film is a thousand times too long. |
| 1200 | let mut inner = info(1_000, 5_400_000.0); |
| 1201 | inner.extend_from_slice(&el(ID_TRACKS, &video_track(1, "V_MPEGH/ISO/HEVC", 1280, 720))); |
| 1202 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1203 | |
| 1204 | req!(mkv.millis(), Some(5_400u64), |
| 1205 | "The timestamp scale was ignored, so the running time is out by the \ |
| 1206 | ratio between a microsecond and a millisecond."); |
| 1207 | Ok(()) |
| 1208 | } |
| 1209 | |
| 1210 | #[test] |
| 1211 | fn every_stream_is_reported_not_only_the_picture() -> Outcome<()> { |
| 1212 | // The reason the whole track list matters: this film's picture is one a |
| 1213 | // browser plays and its sound is one no browser decodes, and a reader |
| 1214 | // that looked only at the picture would call the film playable. |
| 1215 | let mut tracks = video_track(1, "V_MPEG4/ISO/AVC", 1920, 800); |
| 1216 | tracks.extend_from_slice(&audio_track(2, "A_AC3", 6, "eng")); |
| 1217 | tracks.extend_from_slice(&audio_track(3, "A_AAC", 2, "fre")); |
| 1218 | let mut inner = info(1_000_000, 1_000.0); |
| 1219 | inner.extend_from_slice(&el(ID_TRACKS, &tracks)); |
| 1220 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1221 | |
| 1222 | req!(mkv.tracks().len(), 3); |
| 1223 | let sound = mkv.audio(); |
| 1224 | req!(sound.len(), 2); |
| 1225 | req!(sound[0].codec(), "A_AC3"); |
| 1226 | req!(sound[0].channels(), 6u64); |
| 1227 | req!(sound[0].language(), "eng"); |
| 1228 | req!(sound[1].codec(), "A_AAC"); |
| 1229 | Ok(()) |
| 1230 | } |
| 1231 | |
| 1232 | #[test] |
| 1233 | fn a_subtitle_track_is_not_taken_for_the_picture() -> Outcome<()> { |
| 1234 | // Subtitles come first in a good many real files, and a reader taking |
| 1235 | // the first track as the picture reports a film with no size at all. |
| 1236 | let mut sub = u(ID_TRACK_NUMBER, 1); |
| 1237 | sub.extend_from_slice(&u(ID_TRACK_TYPE, 0x11)); |
| 1238 | sub.extend_from_slice(&s(ID_CODEC_ID, "S_TEXT/UTF8")); |
| 1239 | sub.extend_from_slice(&s(ID_LANGUAGE, "eng")); |
| 1240 | let mut tracks = el(ID_TRACK_ENTRY, &sub); |
| 1241 | tracks.extend_from_slice(&video_track(2, "V_MPEG4/ISO/AVC", 720, 576)); |
| 1242 | let mut inner = info(1_000_000, 1_000.0); |
| 1243 | inner.extend_from_slice(&el(ID_TRACKS, &tracks)); |
| 1244 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1245 | |
| 1246 | req!(mkv.size(), (720u32, 576u32), |
| 1247 | "The subtitle track was read as the picture."); |
| 1248 | req!(mkv.subtitles().len(), 1); |
| 1249 | req!(mkv.subtitles()[0].language(), "eng"); |
| 1250 | Ok(()) |
| 1251 | } |
| 1252 | |
| 1253 | #[test] |
| 1254 | fn a_segment_of_unknown_length_is_still_walked() -> Outcome<()> { |
| 1255 | // How a file still being written states its segment: every value bit of |
| 1256 | // the length set. The children follow all the same, and a reader that |
| 1257 | // treats the length as a number skips the entire file. |
| 1258 | let mut inner = info(1_000_000, 60_000.0); |
| 1259 | inner.extend_from_slice(&el(ID_TRACKS, &video_track(1, "V_AV1", 3840, 2160))); |
| 1260 | let mut bytes = head("matroska"); |
| 1261 | bytes.extend_from_slice(&ID_SEGMENT.to_be_bytes()[4..]); |
| 1262 | bytes.push(0xFF); |
| 1263 | bytes.extend_from_slice(&inner); |
| 1264 | |
| 1265 | let mkv = res!(Matroska::read(&bytes)); |
| 1266 | req!(mkv.size(), (3840u32, 2160u32)); |
| 1267 | req!(mkv.millis(), Some(60_000u64)); |
| 1268 | Ok(()) |
| 1269 | } |
| 1270 | |
| 1271 | #[test] |
| 1272 | fn a_head_answers_as_well_as_the_whole_file() -> Outcome<()> { |
| 1273 | // The clusters follow the tracks and are not in hand. The walk must |
| 1274 | // answer from what it has rather than refusing. |
| 1275 | let mut inner = info(1_000_000, 120_000.0); |
| 1276 | inner.extend_from_slice(&el(ID_TRACKS, &video_track(1, "V_MPEG4/ISO/AVC", 640, 480))); |
| 1277 | let mut bytes = file("matroska", inner); |
| 1278 | // A cluster claiming a great deal more than is present, as a real file |
| 1279 | // does once only its front has been read. |
| 1280 | bytes.extend_from_slice(&[0x1F, 0x43, 0xB6, 0x75]); |
| 1281 | bytes.extend_from_slice(&[0x08, 0x00, 0x00, 0x00, 0x00, 0x00, 0x40, 0x00]); |
| 1282 | bytes.extend_from_slice(&[0u8; 32]); |
| 1283 | |
| 1284 | let mkv = res!(Matroska::read(&bytes)); |
| 1285 | req!(mkv.size(), (640u32, 480u32)); |
| 1286 | req!(mkv.millis(), Some(120_000u64)); |
| 1287 | Ok(()) |
| 1288 | } |
| 1289 | |
| 1290 | #[test] |
| 1291 | fn webm_names_itself() -> Outcome<()> { |
| 1292 | let mut inner = info(1_000_000, 1_000.0); |
| 1293 | inner.extend_from_slice(&el(ID_TRACKS, &video_track(1, "V_VP9", 854, 480))); |
| 1294 | let mkv = res!(Matroska::read(&file("webm", inner))); |
| 1295 | req!(mkv.is_webm(), true); |
| 1296 | Ok(()) |
| 1297 | } |
| 1298 | |
| 1299 | #[test] |
| 1300 | fn a_flagged_picture_wins_over_an_earlier_one() -> Outcome<()> { |
| 1301 | // Two picture streams, the second flagged default: a film with an |
| 1302 | // alternative take in it, where the file has said which it means. |
| 1303 | let mut tracks = video_track(1, "V_MPEG4/ISO/AVC", 320, 240); |
| 1304 | let mut second = u(ID_TRACK_NUMBER, 2); |
| 1305 | second.extend_from_slice(&u(ID_TRACK_TYPE, 1)); |
| 1306 | second.extend_from_slice(&s(ID_CODEC_ID, "V_MPEGH/ISO/HEVC")); |
| 1307 | second.extend_from_slice(&u(ID_FLAG_DEFAULT, 1)); |
| 1308 | let mut v = u(ID_PIXEL_W, 1920); |
| 1309 | v.extend_from_slice(&u(ID_PIXEL_H, 1080)); |
| 1310 | second.extend_from_slice(&el(ID_VIDEO, &v)); |
| 1311 | tracks.extend_from_slice(&el(ID_TRACK_ENTRY, &second)); |
| 1312 | let mut inner = info(1_000_000, 1_000.0); |
| 1313 | inner.extend_from_slice(&el(ID_TRACKS, &tracks)); |
| 1314 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1315 | |
| 1316 | req!(mkv.size(), (1920u32, 1080u32), |
| 1317 | "The unflagged first picture was preferred to the one the file marked."); |
| 1318 | Ok(()) |
| 1319 | } |
| 1320 | |
| 1321 | #[test] |
| 1322 | fn a_bcp47_tag_beats_the_older_language_field() -> Outcome<()> { |
| 1323 | // Files carry both, and the older field is written for players that do |
| 1324 | // not know the newer one -- so the newer must win whichever comes first. |
| 1325 | let mut b = u(ID_TRACK_NUMBER, 1); |
| 1326 | b.extend_from_slice(&u(ID_TRACK_TYPE, 2)); |
| 1327 | b.extend_from_slice(&s(ID_CODEC_ID, "A_AAC")); |
| 1328 | b.extend_from_slice(&s(ID_LANGUAGE_BCP47, "pt-BR")); |
| 1329 | b.extend_from_slice(&s(ID_LANGUAGE, "por")); |
| 1330 | let mut inner = info(1_000_000, 1_000.0); |
| 1331 | inner.extend_from_slice(&el(ID_TRACKS, &el(ID_TRACK_ENTRY, &b))); |
| 1332 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1333 | |
| 1334 | req!(mkv.audio()[0].language(), "pt-BR"); |
| 1335 | Ok(()) |
| 1336 | } |
| 1337 | |
| 1338 | #[test] |
| 1339 | fn a_file_still_being_written_has_no_running_time() -> Outcome<()> { |
| 1340 | // Nought is not a running time, and reporting it as one puts "0:00" |
| 1341 | // under every film a recorder has not closed. |
| 1342 | let mut inner = el(ID_INFO, &u(ID_TIMESTAMP_SCALE, 1_000_000)); |
| 1343 | inner.extend_from_slice(&el(ID_TRACKS, &video_track(1, "V_MPEG4/ISO/AVC", 640, 480))); |
| 1344 | let mkv = res!(Matroska::read(&file("matroska", inner))); |
| 1345 | |
| 1346 | req!(mkv.millis(), None::<u64>); |
| 1347 | Ok(()) |
| 1348 | } |
| 1349 | |
| 1350 | #[test] |
| 1351 | fn an_mp4_is_not_a_matroska() -> Outcome<()> { |
| 1352 | let bytes = b"\0\0\0\x18ftypmp42\0\0\0\0mp42isom".to_vec(); |
| 1353 | req!(is_matroska(&bytes), false); |
| 1354 | req!(Matroska::read(&bytes).is_err(), true, |
| 1355 | "An MP4 was read as a Matroska file."); |
| 1356 | Ok(()) |
| 1357 | } |
| 1358 | |
| 1359 | #[test] |
| 1360 | fn a_length_of_eight_bytes_does_not_panic() -> Outcome<()> { |
| 1361 | // The widest length the format allows leaves nothing of the value in the |
| 1362 | // first byte, and the mask that clears the marker is a shift of eight. |
| 1363 | let b = [0x01u8, 0, 0, 0, 0, 0, 0x40, 0x00]; |
| 1364 | match vint_size(&b) { |
| 1365 | Some((Some(n), len)) => { |
| 1366 | req!(len, 8usize); |
| 1367 | req!(n, 0x4000u64); |
| 1368 | }, |
| 1369 | other => return Err(err!( |
| 1370 | "An eight-byte length read as {:?}.", other; Invalid)), |
| 1371 | } |
| 1372 | Ok(()) |
| 1373 | } |
| 1374 | |
| 1375 | #[test] |
| 1376 | fn an_unknown_length_is_told_from_a_large_one() -> Outcome<()> { |
| 1377 | // Every value bit set means unknown. A reader taking it as a number gets |
| 1378 | // 2^56-1 and skips the rest of the file. |
| 1379 | match vint_size(&[0xFFu8]) { |
| 1380 | Some((None, 1)) => {}, |
| 1381 | other => return Err(err!( |
| 1382 | "A one-byte unknown length read as {:?}.", other; Invalid)), |
| 1383 | } |
| 1384 | match vint_size(&[0xFEu8]) { |
| 1385 | Some((Some(126), 1)) => {}, |
| 1386 | other => return Err(err!( |
| 1387 | "A one-byte length of 126 read as {:?}.", other; Invalid)), |
| 1388 | } |
| 1389 | Ok(()) |
| 1390 | } |
| 1391 | } |