oxedyne/fe2o3/fe2o3_graphics/tests/mp4_oracle.rs
22.4 KiB, 14 runs
created by r1870400018:20056, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Hand a file we wrote to demuxers that know nothing about us, and check they find the track we |
| 2 | //! put in it. |
| 3 | //! |
| 4 | //! The unit tests in `mp4.rs` assert on the boxes the writer emitted -- their sizes, the run-length |
| 5 | //! coding of the time table, the offsets in the chunk table. That is a fair check of the arithmetic |
| 6 | //! and no check at all of whether the file plays: a sample entry with the wrong four-character |
| 7 | //! code, a duration in the wrong timescale, or a chunk offset measured from the start of `mdat` |
| 8 | //! rather than from the start of the file all produce perfectly well-formed boxes, and every one of |
| 9 | //! those tests still passes. The writer and its unit tests share a hand, so they share any |
| 10 | //! misreading of what the fields mean. |
| 11 | //! |
| 12 | //! # Where the samples come from |
| 13 | //! |
| 14 | //! Nothing here encodes anything, because the crate cannot. FFmpeg encodes a short clip to a raw |
| 15 | //! H.264 elementary stream; the test splits it into NAL units, groups them into access units, |
| 16 | //! builds the decoder configuration record out of the parameter sets it finds, and hands the whole |
| 17 | //! lot to the writer. So the samples are somebody else's, produced by a tool that has no idea this |
| 18 | //! crate exists, and no fixture is checked into the tree. |
| 19 | //! |
| 20 | //! That also makes the strongest check available cheap: the same coded pictures are decoded twice, |
| 21 | //! once out of FFmpeg's own elementary stream and once out of our container, and the raw planes |
| 22 | //! must be identical byte for byte. If the container mistimes, misorders, truncates or misplaces a |
| 23 | //! single sample, the two decodes differ. |
| 24 | //! |
| 25 | //! # The oracles |
| 26 | //! |
| 27 | //! - **FFprobe**, which reports the codec, the dimensions, the timescale, the sample count and the |
| 28 | //! durations from the boxes alone. |
| 29 | //! - **FFmpeg**, decoding to raw planes, compared against the decode of the source stream. |
| 30 | //! - **ExifTool**, a wholly separate implementation in Perl, which walks the box tree itself. |
| 31 | //! - **GStreamer's `gst-discoverer-1.0`**, whose `qtdemux` shares no code with libavformat. |
| 32 | //! - **A box walker in Python**, which re-derives every sample's position from the tables and checks |
| 33 | //! that the bytes at that position are the ones that were pushed. |
| 34 | //! |
| 35 | //! `mp4box` and `mediainfo` are not installed on this machine and `pymp4` is not importable, so |
| 36 | //! neither is used; ExifTool and GStreamer stand in their place, and both are independent of |
| 37 | //! FFmpeg. A missing tool is a failure here rather than a skip: an oracle that quietly does not run |
| 38 | //! is an oracle that is not there. |
| 39 | //! |
| 40 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 41 | //! Anthropic Claude |
| 42 | |
| 43 | use oxedyne_fe2o3_core::prelude::*; |
| 44 | use oxedyne_fe2o3_graphics::mp4::{ |
| 45 | Codec, |
| 46 | Sample, |
| 47 | Track, |
| 48 | }; |
| 49 | |
| 50 | use std::{ |
| 51 | fs, |
| 52 | path::PathBuf, |
| 53 | process::Command, |
| 54 | }; |
| 55 | |
| 56 | // The clip FFmpeg is asked for: a test pattern, small enough to decode in a moment and large |
| 57 | // enough that a plane comparison means something. |
| 58 | const W: u16 = 64; |
| 59 | const H: u16 = 48; |
| 60 | const FPS: u32 = 10; |
| 61 | const FRAMES: usize = 10; |
| 62 | |
| 63 | // The track's timescale, in ticks a second. Ninety thousand is the MPEG transport clock, and is |
| 64 | // chosen here because it is nothing like the frame rate: a writer that quietly used one where it |
| 65 | // meant the other would be caught by it. |
| 66 | const TIMESCALE: u32 = 90_000; |
| 67 | const TICKS: u32 = TIMESCALE / FPS; // one sample's duration in that timescale |
| 68 | |
| 69 | fn tmp(name: &str) -> PathBuf { |
| 70 | PathBuf::from(env!("CARGO_TARGET_TMPDIR")).join(name) |
| 71 | } |
| 72 | |
| 73 | /// Fails unless the named program answers a version query, since an oracle that does not run is not |
| 74 | /// an oracle. |
| 75 | fn require(prog: &str) -> Outcome<()> { |
| 76 | // Three spellings, because FFmpeg takes one, Python and ExifTool another, and GStreamer's |
| 77 | // discoverer has no version flag at all and only answers a request for help. |
| 78 | for flag in ["-version", "--version", "--help"] { |
| 79 | let ok = Command::new(prog) |
| 80 | .arg(flag) |
| 81 | .output() |
| 82 | .map(|o| o.status.success()) |
| 83 | .unwrap_or(false); |
| 84 | if ok { |
| 85 | return Ok(()); |
| 86 | } |
| 87 | } |
| 88 | Err(err!( |
| 89 | "'{}' is not on the path, and the MP4 writer has no oracle without it. Install it rather \ |
| 90 | than skipping the test.", prog; |
| 91 | Missing, Configuration)) |
| 92 | } |
| 93 | |
| 94 | /// Asks FFmpeg for a raw H.264 elementary stream, and gives its bytes. |
| 95 | /// |
| 96 | /// No B-pictures, so decode order and display order agree and the container needs no composition |
| 97 | /// offset table; a keyframe every five frames, so the track has both sync and delta samples and the |
| 98 | /// sync sample table is exercised. |
| 99 | fn source(tag: &str) -> Outcome<(PathBuf, Vec<u8>)> { |
| 100 | res!(require("ffmpeg")); |
| 101 | let path = tmp(&fmt!("mp4_oracle_{}.264", tag)); |
| 102 | let out = res!(Command::new("ffmpeg") |
| 103 | .args(["-y", "-loglevel", "error", "-f", "lavfi", "-i"]) |
| 104 | .arg(fmt!("testsrc=size={}x{}:rate={}:duration=2", W, H, FPS)) |
| 105 | .args(["-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p"]) |
| 106 | .args(["-g", "5", "-bf", "0", "-frames:v"]) |
| 107 | .arg(fmt!("{}", FRAMES)) |
| 108 | .args(["-f", "h264"]) |
| 109 | .arg(&path) |
| 110 | .output()); |
| 111 | if !out.status.success() { |
| 112 | return Err(err!( |
| 113 | "FFmpeg would not encode the source clip: {}", String::from_utf8_lossy(&out.stderr); |
| 114 | Invalid, Input)); |
| 115 | } |
| 116 | let buf = res!(fs::read(&path)); |
| 117 | Ok((path, buf)) |
| 118 | } |
| 119 | |
| 120 | /// Splits an Annex B elementary stream into its NAL units. |
| 121 | /// |
| 122 | /// The units are separated by a three- or four-byte start code, and a unit may be followed by |
| 123 | /// padding zeroes which belong to neither it nor the next start code. Those are trimmed, which is |
| 124 | /// safe because the trailing bits of every NAL end in a set stop bit, so its last meaningful byte |
| 125 | /// is never zero. |
| 126 | fn nal_units(buf: &[u8]) -> Vec<&[u8]> { |
| 127 | let mut starts = Vec::new(); |
| 128 | let mut i = 0usize; |
| 129 | while i + 3 <= buf.len() { |
| 130 | if buf[i] == 0 && buf[i + 1] == 0 && buf[i + 2] == 1 { |
| 131 | starts.push(i + 3); |
| 132 | i += 3; |
| 133 | } else { |
| 134 | i += 1; |
| 135 | } |
| 136 | } |
| 137 | let mut out = Vec::with_capacity(starts.len()); |
| 138 | for (k, s) in starts.iter().enumerate() { |
| 139 | let mut e = if k + 1 < starts.len() { starts[k + 1] - 3 } else { buf.len() }; |
| 140 | while e > *s && buf[e - 1] == 0 { |
| 141 | e -= 1; |
| 142 | } |
| 143 | out.push(&buf[*s..e]); |
| 144 | } |
| 145 | out |
| 146 | } |
| 147 | |
| 148 | /// The parameter sets and the access units of an elementary stream. |
| 149 | /// |
| 150 | /// An access unit is one coded picture. With one slice a picture and no field coding, every video |
| 151 | /// coding layer NAL unit -- types 1 to 5 -- begins one, and the non-video units before it belong to |
| 152 | /// it. Parameter sets and access unit delimiters are dropped from the samples, since the container |
| 153 | /// carries the parameter sets out of band in its decoder configuration record. |
| 154 | fn demux(stream: &[u8]) -> Outcome<(Vec<u8>, Vec<u8>, Vec<(Vec<u8>, bool)>)> { |
| 155 | let mut sps: Option<Vec<u8>> = None; |
| 156 | let mut pps: Option<Vec<u8>> = None; |
| 157 | let mut units: Vec<(Vec<u8>, bool)> = Vec::new(); |
| 158 | let mut pending: Vec<Vec<u8>> = Vec::new(); |
| 159 | for nal in nal_units(stream) { |
| 160 | if nal.is_empty() { |
| 161 | continue; |
| 162 | } |
| 163 | let kind = nal[0] & 0x1F; |
| 164 | match kind { |
| 165 | 7 => if sps.is_none() { sps = Some(nal.to_vec()); }, |
| 166 | 8 => if pps.is_none() { pps = Some(nal.to_vec()); }, |
| 167 | 9 => {}, |
| 168 | 1..=5 => { |
| 169 | let mut au = Vec::new(); |
| 170 | for p in pending.drain(..) { |
| 171 | au.extend_from_slice(&(p.len() as u32).to_be_bytes()); |
| 172 | au.extend_from_slice(&p); |
| 173 | } |
| 174 | au.extend_from_slice(&(nal.len() as u32).to_be_bytes()); |
| 175 | au.extend_from_slice(nal); |
| 176 | units.push((au, kind == 5)); |
| 177 | }, |
| 178 | _ => pending.push(nal.to_vec()), |
| 179 | } |
| 180 | } |
| 181 | let sps = match sps { |
| 182 | Some(v) => v, |
| 183 | None => return Err(err!( |
| 184 | "The encoded stream carries no sequence parameter set."; Invalid, Input, Missing)), |
| 185 | }; |
| 186 | let pps = match pps { |
| 187 | Some(v) => v, |
| 188 | None => return Err(err!( |
| 189 | "The encoded stream carries no picture parameter set."; Invalid, Input, Missing)), |
| 190 | }; |
| 191 | Ok((sps, pps, units)) |
| 192 | } |
| 193 | |
| 194 | /// An `AVCDecoderConfigurationRecord` around one sequence and one picture parameter set, with a |
| 195 | /// four-byte NAL length, as ISO/IEC 14496-15 §5.3.3.1 lays it out. |
| 196 | fn avcc(sps: &[u8], pps: &[u8]) -> Vec<u8> { |
| 197 | let mut rec = vec![1, sps[1], sps[2], sps[3], 0xFF, 0xE1]; |
| 198 | rec.extend_from_slice(&(sps.len() as u16).to_be_bytes()); |
| 199 | rec.extend_from_slice(sps); |
| 200 | rec.push(1); |
| 201 | rec.extend_from_slice(&(pps.len() as u16).to_be_bytes()); |
| 202 | rec.extend_from_slice(pps); |
| 203 | rec |
| 204 | } |
| 205 | |
| 206 | /// Writes the MP4 and gives its path, the path of the elementary stream it came from, and the |
| 207 | /// sample sizes and sync flags that went into it, so a reader's report can be checked against what |
| 208 | /// was pushed rather than against the file. |
| 209 | /// |
| 210 | /// The tag names the pair of files, because the tests run at the same time and two of them writing |
| 211 | /// one path is a race that reads a half-written file. |
| 212 | fn fixture(tag: &str) -> Outcome<(PathBuf, PathBuf, Vec<usize>, Vec<bool>)> { |
| 213 | let (src, stream) = res!(source(tag)); |
| 214 | let (sps, pps, units) = res!(demux(&stream)); |
| 215 | if units.len() != FRAMES { |
| 216 | return Err(err!( |
| 217 | "FFmpeg was asked for {} pictures and its stream demuxes to {} access units.", |
| 218 | FRAMES, units.len(); |
| 219 | Invalid, Input, Mismatch)); |
| 220 | } |
| 221 | |
| 222 | // The dimensions given here are the ones the command line asked FFmpeg for, so the writer |
| 223 | // accepting them is a check of its reading of the sequence parameter set against a third party. |
| 224 | let mut track = res!(Track::new(W, H, TIMESCALE, Codec::Avc(avcc(&sps, &pps)))); |
| 225 | let mut sizes = Vec::with_capacity(units.len()); |
| 226 | let mut syncs = Vec::with_capacity(units.len()); |
| 227 | for (data, sync) in units { |
| 228 | sizes.push(data.len()); |
| 229 | syncs.push(sync); |
| 230 | // No composition offset: the source is encoded with no B-pictures, so decode order and |
| 231 | // display order agree and each sample is shown at its decoding time. |
| 232 | res!(track.push(Sample { data, dur: TICKS, sync, off: 0 })); |
| 233 | } |
| 234 | let buf = res!(track.finish()); |
| 235 | let path = tmp(&fmt!("mp4_oracle_{}.mp4", tag)); |
| 236 | res!(fs::write(&path, &buf)); |
| 237 | Ok((path, src, sizes, syncs)) |
| 238 | } |
| 239 | |
| 240 | /// The value FFprobe printed against a key in its JSON, as text with any quotes stripped. |
| 241 | /// |
| 242 | /// A whole JSON parser is more than this needs: FFprobe's output is one key and one scalar a line, |
| 243 | /// and the keys wanted here appear once. |
| 244 | fn field(json: &str, key: &str) -> Outcome<String> { |
| 245 | let needle = fmt!("\"{}\":", key); |
| 246 | let at = match json.find(&needle) { |
| 247 | Some(i) => i + needle.len(), |
| 248 | None => return Err(err!( |
| 249 | "FFprobe reported no '{}'.", key; Missing, Test)), |
| 250 | }; |
| 251 | let rest = &json[at..]; |
| 252 | // To the end of the line rather than to the next comma: a value may hold commas of its own, |
| 253 | // and `format_name` is a comma-separated list of every format FFmpeg's demuxer answers to. |
| 254 | let end = rest.find('\n').unwrap_or(rest.len()); |
| 255 | Ok(rest[..end].trim().trim_end_matches(',').trim().trim_matches('"').to_string()) |
| 256 | } |
| 257 | |
| 258 | #[test] |
| 259 | fn test_ffprobe_reads_the_track_00() -> Outcome<()> { |
| 260 | res!(require("ffprobe")); |
| 261 | let (path, _, sizes, _) = res!(fixture("ffprobe")); |
| 262 | |
| 263 | let out = res!(Command::new("ffprobe") |
| 264 | .args(["-v", "quiet", "-print_format", "json", "-show_streams", "-show_format"]) |
| 265 | .arg(&path) |
| 266 | .output()); |
| 267 | if !out.status.success() { |
| 268 | return Err(err!( |
| 269 | "FFprobe refused the file: {}", String::from_utf8_lossy(&out.stderr); |
| 270 | Invalid, Input)); |
| 271 | } |
| 272 | let json = String::from_utf8_lossy(&out.stdout).to_string(); |
| 273 | |
| 274 | // What the track is. |
| 275 | req!(res!(field(&json, "codec_name")), "h264".to_string()); |
| 276 | req!(res!(field(&json, "codec_type")), "video".to_string()); |
| 277 | req!(res!(field(&json, "codec_tag_string")), "avc1".to_string()); |
| 278 | req!(res!(field(&json, "nb_streams")), "1".to_string()); |
| 279 | |
| 280 | // How big its pictures are, which came out of the sample entry. |
| 281 | req!(res!(field(&json, "width")), fmt!("{}", W)); |
| 282 | req!(res!(field(&json, "height")), fmt!("{}", H)); |
| 283 | req!(res!(field(&json, "coded_width")), fmt!("{}", W)); |
| 284 | |
| 285 | // How it is timed. The time base is the media header's timescale; the total is the sum of the |
| 286 | // sample durations in the decoding time table; the frame rate is what the two imply. |
| 287 | req!(res!(field(&json, "time_base")), fmt!("1/{}", TIMESCALE)); |
| 288 | req!(res!(field(&json, "duration_ts")), fmt!("{}", TICKS as usize * FRAMES)); |
| 289 | req!(res!(field(&json, "r_frame_rate")), fmt!("{}/1", FPS)); |
| 290 | req!(res!(field(&json, "nb_frames")), fmt!("{}", FRAMES)); |
| 291 | |
| 292 | // The container as a whole: the brand list of the file type box, and the movie header's |
| 293 | // duration, which is the same span restated in milliseconds. |
| 294 | let brands = res!(field(&json, "format_name")); |
| 295 | req!(brands.contains("mp4"), true, "FFprobe does not call the file an MP4: {}", brands); |
| 296 | let secs: f64 = match res!(field(&json, "duration")).parse::<f64>() { |
| 297 | Ok(v) => v, |
| 298 | Err(_) => return Err(err!( |
| 299 | "FFprobe's duration is not a number."; Test, Invalid)), |
| 300 | }; |
| 301 | let want = FRAMES as f64 / FPS as f64; |
| 302 | let close = (secs - want).abs() < 0.002; |
| 303 | req!(close, true, "FFprobe reports {} seconds where {} were written.", secs, want); |
| 304 | |
| 305 | // The size of the file is the size of the index plus the samples, and FFprobe read it. |
| 306 | let bytes: usize = match res!(field(&json, "size")).parse::<usize>() { |
| 307 | Ok(v) => v, |
| 308 | Err(_) => return Err(err!("FFprobe's size is not a number."; Test, Invalid)), |
| 309 | }; |
| 310 | let media: usize = sizes.iter().sum(); |
| 311 | let indexed = bytes > media; |
| 312 | req!(indexed, true, "The file is {} bytes and its media alone is {}.", bytes, media); |
| 313 | |
| 314 | println!("FFprobe: h264 {}x{}, {} frames, {}/{} ticks, {} s, {} bytes.", |
| 315 | W, H, FRAMES, TICKS as usize * FRAMES, TIMESCALE, secs, bytes); |
| 316 | Ok(()) |
| 317 | } |
| 318 | |
| 319 | #[test] |
| 320 | fn test_ffmpeg_decodes_the_same_pixels_01() -> Outcome<()> { |
| 321 | res!(require("ffmpeg")); |
| 322 | let (path, src, _, _) = res!(fixture("decode")); |
| 323 | |
| 324 | // The same decoder, over the same coded pictures, out of two containers -- one FFmpeg's own |
| 325 | // elementary stream and one ours. The planes are taken raw, in the decoder's native format, so |
| 326 | // nothing between the decoder and the comparison can resample, retime or convert. |
| 327 | let decode = |input: &PathBuf| -> Outcome<Vec<u8>> { |
| 328 | let out = res!(Command::new("ffmpeg") |
| 329 | .args(["-loglevel", "error", "-i"]) |
| 330 | .arg(input) |
| 331 | .args(["-fps_mode", "passthrough", "-f", "rawvideo", "-pix_fmt", "yuv420p", "-"]) |
| 332 | .output()); |
| 333 | if !out.status.success() { |
| 334 | return Err(err!( |
| 335 | "FFmpeg would not decode {}: {}", |
| 336 | input.display(), String::from_utf8_lossy(&out.stderr); |
| 337 | Invalid, Input)); |
| 338 | } |
| 339 | Ok(out.stdout) |
| 340 | }; |
| 341 | |
| 342 | let want = res!(decode(&src)); |
| 343 | let got = res!(decode(&path)); |
| 344 | |
| 345 | // A 4:2:0 frame is one luma sample and half a chroma sample a pixel. |
| 346 | let frame = W as usize * H as usize * 3 / 2; |
| 347 | req!(want.len(), frame * FRAMES, "The source stream did not decode to {} frames.", FRAMES); |
| 348 | req!(got.len(), frame * FRAMES, "Our file decoded to {} frames.", got.len() / frame); |
| 349 | |
| 350 | if let Some(i) = want.iter().zip(got.iter()).position(|(a, b)| a != b) { |
| 351 | return Err(err!( |
| 352 | "The decode of our file differs from the decode of the source stream at byte {}, \ |
| 353 | which is frame {}: {} where {} was encoded.", |
| 354 | i, i / frame, got[i], want[i]; |
| 355 | Invalid, Input, Mismatch)); |
| 356 | } |
| 357 | println!("FFmpeg decodes {} frames of {}x{} identically out of both containers.", |
| 358 | FRAMES, W, H); |
| 359 | Ok(()) |
| 360 | } |
| 361 | |
| 362 | #[test] |
| 363 | fn test_exiftool_walks_the_boxes_02() -> Outcome<()> { |
| 364 | res!(require("exiftool")); |
| 365 | let (path, _, _, _) = res!(fixture("exiftool")); |
| 366 | let out = res!(Command::new("exiftool").arg("-s").arg("-G").arg(&path).output()); |
| 367 | if !out.status.success() { |
| 368 | return Err(err!( |
| 369 | "ExifTool refused the file: {}", String::from_utf8_lossy(&out.stderr); |
| 370 | Invalid, Input)); |
| 371 | } |
| 372 | let text = String::from_utf8_lossy(&out.stdout).to_string(); |
| 373 | |
| 374 | // ExifTool is a separate implementation in Perl which walks the box tree itself, so the fields |
| 375 | // below were read out of the boxes rather than out of any FFmpeg parse of them. |
| 376 | let wants = [ |
| 377 | ("MajorBrand", "MP4 Base Media v1 [IS0 14496-12:2003]"), |
| 378 | ("HandlerType", "Video Track"), |
| 379 | ("CompressorID", "avc1"), |
| 380 | ("ImageWidth", "64"), |
| 381 | ("ImageHeight", "48"), |
| 382 | ("MediaTimeScale", "90000"), |
| 383 | ("MediaDuration", "1.00 s"), |
| 384 | ("VideoFrameRate", "10"), |
| 385 | ("Duration", "1.00 s"), |
| 386 | ]; |
| 387 | for (key, want) in wants { |
| 388 | let line = text.lines() |
| 389 | .find(|l| l.split_whitespace().nth(1) == Some(key)) |
| 390 | .map(|l| l.to_string()); |
| 391 | let line = match line { |
| 392 | Some(l) => l, |
| 393 | None => return Err(err!( |
| 394 | "ExifTool reported no '{}'. It read:\n{}", key, text; Missing, Test)), |
| 395 | }; |
| 396 | let got = match line.split_once(": ") { |
| 397 | Some((_, v)) => v.trim().to_string(), |
| 398 | None => return Err(err!( |
| 399 | "ExifTool's line for '{}' has no value: {}", key, line; Test)), |
| 400 | }; |
| 401 | req!(got, want.to_string(), "ExifTool disagrees about '{}'.", key); |
| 402 | } |
| 403 | println!("ExifTool agrees on the brand, the handler, the codec, the size and the timing."); |
| 404 | Ok(()) |
| 405 | } |
| 406 | |
| 407 | #[test] |
| 408 | fn test_gstreamer_demuxes_the_track_03() -> Outcome<()> { |
| 409 | res!(require("gst-discoverer-1.0")); |
| 410 | let (path, _, _, _) = res!(fixture("gst")); |
| 411 | let out = res!(Command::new("gst-discoverer-1.0").arg(&path).output()); |
| 412 | let text = fmt!("{}{}", |
| 413 | String::from_utf8_lossy(&out.stdout), String::from_utf8_lossy(&out.stderr)); |
| 414 | if !out.status.success() { |
| 415 | return Err(err!( |
| 416 | "GStreamer refused the file: {}", text; Invalid, Input)); |
| 417 | } |
| 418 | |
| 419 | // `qtdemux` shares no code with libavformat, so this is a second demuxer's reading of the same |
| 420 | // box tree rather than a second run of the first. |
| 421 | for want in ["H.264", "Width: 64", "Height: 48", "0:00:01.000000000"] { |
| 422 | req!(text.contains(want), true, |
| 423 | "GStreamer's report does not carry '{}'. It read:\n{}", want, text); |
| 424 | } |
| 425 | req!(text.contains("Missing plugins"), false, "GStreamer could not open the track:\n{}", text); |
| 426 | println!("GStreamer's qtdemux finds an H.264 track of 64x48 lasting one second."); |
| 427 | Ok(()) |
| 428 | } |
| 429 | |
| 430 | #[test] |
| 431 | fn test_a_box_walker_finds_every_sample_where_the_tables_say_04() -> Outcome<()> { |
| 432 | res!(require("python3")); |
| 433 | let (path, _, sizes, syncs) = res!(fixture("walker")); |
| 434 | |
| 435 | // A walker that knows only the specification: it descends the box tree, reads the sample size, |
| 436 | // sample to chunk, chunk offset, decoding time and sync sample tables, and derives from them |
| 437 | // where each sample begins and how long it lasts. Nothing in it came from the writer. |
| 438 | let walk = r#" |
| 439 | import struct, sys |
| 440 | |
| 441 | buf = open(sys.argv[1], 'rb').read() |
| 442 | |
| 443 | def children(b, at, end): |
| 444 | out = [] |
| 445 | while at + 8 <= end: |
| 446 | size = struct.unpack('>I', b[at:at+4])[0] |
| 447 | kind = b[at+4:at+8].decode('latin-1') |
| 448 | if size == 1: |
| 449 | size = struct.unpack('>Q', b[at+8:at+16])[0] |
| 450 | body = at + 16 |
| 451 | else: |
| 452 | body = at + 8 |
| 453 | if size < 8 or at + size > end: |
| 454 | raise SystemExit('box %s at %d claims %d bytes' % (kind, at, size)) |
| 455 | out.append((kind, at, body, at + size)) |
| 456 | at += size |
| 457 | if at != end: |
| 458 | raise SystemExit('boxes do not tile: stopped at %d of %d' % (at, end)) |
| 459 | return out |
| 460 | |
| 461 | CONTAINERS = {'moov', 'trak', 'mdia', 'minf', 'stbl'} |
| 462 | index = {} |
| 463 | def descend(at, end, trail): |
| 464 | for kind, start, body, stop in children(buf, at, end): |
| 465 | index.setdefault(trail + '/' + kind, (body, stop)) |
| 466 | if kind in CONTAINERS: |
| 467 | descend(body, stop, trail + '/' + kind) |
| 468 | |
| 469 | top = children(buf, 0, len(buf)) |
| 470 | print('TOP', ','.join(k for k, _, _, _ in top)) |
| 471 | descend(0, len(buf), '') |
| 472 | |
| 473 | def body(pathname): |
| 474 | if pathname not in index: |
| 475 | raise SystemExit('no box at ' + pathname) |
| 476 | return index[pathname] |
| 477 | |
| 478 | # Sample sizes. |
| 479 | b0, b1 = body('/moov/trak/mdia/minf/stbl/stsz') |
| 480 | common, count = struct.unpack('>II', buf[b0+4:b0+12]) |
| 481 | if common != 0: |
| 482 | raise SystemExit('the sample size table declares a common size') |
| 483 | sizes = list(struct.unpack('>%dI' % count, buf[b0+12:b0+12+4*count])) |
| 484 | |
| 485 | # Sample to chunk. |
| 486 | b0, b1 = body('/moov/trak/mdia/minf/stbl/stsc') |
| 487 | n = struct.unpack('>I', buf[b0+4:b0+8])[0] |
| 488 | stsc = [struct.unpack('>III', buf[b0+8+i*12:b0+20+i*12]) for i in range(n)] |
| 489 | |
| 490 | # Chunk offsets. |
| 491 | if '/moov/trak/mdia/minf/stbl/stco' in index: |
| 492 | b0, b1 = body('/moov/trak/mdia/minf/stbl/stco') |
| 493 | nc = struct.unpack('>I', buf[b0+4:b0+8])[0] |
| 494 | offs = list(struct.unpack('>%dI' % nc, buf[b0+8:b0+8+4*nc])) |
| 495 | else: |
| 496 | b0, b1 = body('/moov/trak/mdia/minf/stbl/co64') |
| 497 | nc = struct.unpack('>I', buf[b0+4:b0+8])[0] |
| 498 | offs = list(struct.unpack('>%dQ' % nc, buf[b0+8:b0+8+8*nc])) |
| 499 | |
| 500 | # Decoding times, expanded from their runs. |
| 501 | b0, b1 = body('/moov/trak/mdia/minf/stbl/stts') |
| 502 | n = struct.unpack('>I', buf[b0+4:b0+8])[0] |
| 503 | runs = [struct.unpack('>II', buf[b0+8+i*8:b0+16+i*8]) for i in range(n)] |
| 504 | durs = [] |
| 505 | for c, d in runs: |
| 506 | durs.extend([d] * c) |
| 507 | |
| 508 | # Sync samples, or every sample where the table is absent. |
| 509 | key = '/moov/trak/mdia/minf/stbl/stss' |
| 510 | if key in index: |
| 511 | b0, b1 = body(key) |
| 512 | n = struct.unpack('>I', buf[b0+4:b0+8])[0] |
| 513 | sync = list(struct.unpack('>%dI' % n, buf[b0+8:b0+8+4*n])) |
| 514 | else: |
| 515 | sync = list(range(1, count + 1)) |
| 516 | |
| 517 | # Which chunk each sample is in, and therefore where it begins. |
| 518 | starts = [] |
| 519 | s = 0 |
| 520 | for i, (first, per, _desc) in enumerate(stsc): |
| 521 | last = stsc[i+1][0] - 1 if i + 1 < len(stsc) else len(offs) |
| 522 | for c in range(first, last + 1): |
| 523 | at = offs[c-1] |
| 524 | for _ in range(per): |
| 525 | if s >= count: |
| 526 | break |
| 527 | starts.append(at) |
| 528 | at += sizes[s] |
| 529 | s += 1 |
| 530 | if s != count: |
| 531 | raise SystemExit('the chunk tables place %d of %d samples' % (s, count)) |
| 532 | |
| 533 | # Each sample's first four bytes are its first NAL unit's length, which must fit the sample. |
| 534 | for i, (at, size) in enumerate(zip(starts, sizes)): |
| 535 | if at + size > len(buf): |
| 536 | raise SystemExit('sample %d runs past the end of the file' % i) |
| 537 | nal = struct.unpack('>I', buf[at:at+4])[0] |
| 538 | if nal + 4 > size: |
| 539 | raise SystemExit('sample %d holds a NAL of %d bytes in %d' % (i, nal, size)) |
| 540 | |
| 541 | mdat = [t for t in top if t[0] == 'mdat'][0] |
| 542 | if starts[0] != mdat[2]: |
| 543 | raise SystemExit('the first sample is at %d and mdat begins at %d' % (starts[0], mdat[2])) |
| 544 | if starts[-1] + sizes[-1] != mdat[3]: |
| 545 | raise SystemExit('the last sample ends at %d and mdat ends at %d' |
| 546 | % (starts[-1] + sizes[-1], mdat[3])) |
| 547 | |
| 548 | print('SIZES', ','.join(str(v) for v in sizes)) |
| 549 | print('DURS', ','.join(str(v) for v in durs)) |
| 550 | print('SYNC', ','.join(str(v) for v in sync)) |
| 551 | print('OK') |
| 552 | "#; |
| 553 | let out = res!(Command::new("python3").arg("-c").arg(walk).arg(&path).output()); |
| 554 | let text = String::from_utf8_lossy(&out.stdout).to_string(); |
| 555 | if !out.status.success() { |
| 556 | return Err(err!( |
| 557 | "The box walker rejected the file: {}{}", |
| 558 | text, String::from_utf8_lossy(&out.stderr); |
| 559 | Invalid, Input)); |
| 560 | } |
| 561 | |
| 562 | let line = |k: &str| -> Outcome<String> { |
| 563 | match text.lines().find(|l| l.starts_with(k)) { |
| 564 | Some(l) => Ok(l[k.len()..].trim().to_string()), |
| 565 | None => Err(err!("The box walker printed no '{}' line: {}", k, text; Test)), |
| 566 | } |
| 567 | }; |
| 568 | |
| 569 | req!(res!(line("TOP")), "ftyp,moov,mdat".to_string()); |
| 570 | req!(res!(line("SIZES")), |
| 571 | sizes.iter().map(|n| fmt!("{}", n)).collect::<Vec<_>>().join(",")); |
| 572 | req!(res!(line("DURS")), |
| 573 | (0..FRAMES).map(|_| fmt!("{}", TICKS)).collect::<Vec<_>>().join(",")); |
| 574 | req!(res!(line("SYNC")), |
| 575 | syncs.iter().enumerate().filter(|(_, s)| **s).map(|(i, _)| fmt!("{}", i + 1)) |
| 576 | .collect::<Vec<_>>().join(",")); |
| 577 | req!(text.contains("OK"), true, "The box walker did not finish: {}", text); |
| 578 | |
| 579 | println!("The box walker places all {} samples from the tables alone.", FRAMES); |
| 580 | Ok(()) |
| 581 | } |