oxedyne/fe2o3/fe2o3_infer/tests/guard.rs
4.6 KiB, 1 run
created by r1870400018:19769, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! The regression guard the matrix kernel needs, and cannot do without. |
| 2 | //! |
| 3 | //! Two failures are silent and expensive, and neither shows up in a correctness |
| 4 | //! test because both paths still compute the right answer: |
| 5 | //! |
| 6 | //! - **The register tile falls off its cliff.** Whether the `[[f32; NR]; MR]` |
| 7 | //! accumulator lives in vector registers or spills to the stack is an |
| 8 | //! all-or-nothing decision the code generator makes, and it moves with the |
| 9 | //! tile shape, with the arithmetic form, and with the compiler version. |
| 10 | //! Measured on this workload, adjacent tile heights differ by a factor of |
| 11 | //! thirty-two. A one-character edit, or an upgrade nobody made, can cost |
| 12 | //! thirty times the throughput. |
| 13 | //! - **`mul_add` reaches the path that has no fused multiply-add.** There it |
| 14 | //! becomes a library call, and the same thirty-fold loss appears in the |
| 15 | //! fallback rather than the fast path, where it is even easier to miss. |
| 16 | //! |
| 17 | //! Both are caught here by measurement, on the shape that dominates the |
| 18 | //! embedder: `(m, n, k) = (196, 512, 512)`, five of whose instances are |
| 19 | //! forty-five per cent of the whole network. |
| 20 | //! |
| 21 | //! The floors are deliberately far below what the kernel actually reaches, so |
| 22 | //! that a loaded or thermally throttled machine does not fail the build. They |
| 23 | //! are set to catch a collapse, not a slowdown. |
| 24 | |
| 25 | use std::time::Instant; |
| 26 | |
| 27 | use oxedyne_fe2o3_core::prelude::*; |
| 28 | use oxedyne_fe2o3_infer::kern::{self, Cpu, Scratch, Task, MR, NR}; |
| 29 | |
| 30 | /// The shape that dominates the embedder. |
| 31 | const SHAPE: (usize, usize, usize) = (196, 512, 512); |
| 32 | |
| 33 | /// Below this, on a machine with AVX2 and FMA, the accumulator has spilled. |
| 34 | /// The kernel measures around fifty-eight on an idle machine, and around |
| 35 | /// forty-five with the rest of the suite running beside it; a spill takes it to |
| 36 | /// under two. |
| 37 | const FMA_FLOOR: f64 = 15.0; |
| 38 | |
| 39 | /// Below this the baseline path has reached `mul_add` without the instruction, |
| 40 | /// which costs about thirty times. That path measures around twelve idle and |
| 41 | /// around four under load; the failure it is looking for measures half of one. |
| 42 | const BASELINE_FLOOR: f64 = 1.5; |
| 43 | |
| 44 | /// A cheap reproducible generator, so the guard needs no dependency. |
| 45 | fn fill(n: usize, seed: u64) -> Vec<f32> { |
| 46 | let mut s = seed; |
| 47 | let mut v = Vec::with_capacity(n); |
| 48 | for _ in 0..n { |
| 49 | s = s.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407); |
| 50 | v.push(((s >> 40) as f32 / 8_388_608.0) - 1.0); |
| 51 | } |
| 52 | v |
| 53 | } |
| 54 | |
| 55 | /// Times the kernel on one shape, answering billions of multiply-accumulates |
| 56 | /// per second. |
| 57 | fn rate(cpu: Cpu) -> f64 { |
| 58 | let (m, n, k) = SHAPE; |
| 59 | let a = fill(m * k, 1); |
| 60 | let b = fill(k * n, 2); |
| 61 | let mut c = vec![0.0f32; m * n]; |
| 62 | let mut s = Scratch::new(); |
| 63 | // Warm the buffers and the branch predictor before anything is timed. |
| 64 | kern::run(cpu, Task::Gemm { m, n, k, a: &a, b: &b, c: &mut c, bias: None, scratch: &mut s }); |
| 65 | let macs = (m * n * k) as f64; |
| 66 | let t = Instant::now(); |
| 67 | let mut reps = 0u64; |
| 68 | // Time enough repeats to cover a fifth of a second, so the reading is |
| 69 | // stable without the test being slow. |
| 70 | while t.elapsed().as_secs_f64() < 0.2 { |
| 71 | kern::run(cpu, Task::Gemm { |
| 72 | m, n, k, a: &a, b: &b, c: &mut c, bias: None, scratch: &mut s }); |
| 73 | reps += 1; |
| 74 | } |
| 75 | macs * reps as f64 / t.elapsed().as_secs_f64() / 1e9 |
| 76 | } |
| 77 | |
| 78 | #[test] |
| 79 | fn the_register_tile_is_the_one_that_was_measured() -> Outcome<()> { |
| 80 | // Not a performance claim -- a statement that the shape has not been |
| 81 | // changed without someone reading why it is what it is. |
| 82 | req!(MR, 6usize); |
| 83 | req!(NR, 16usize); |
| 84 | Ok(()) |
| 85 | } |
| 86 | |
| 87 | #[test] |
| 88 | fn the_dispatched_path_has_not_fallen_off_the_cliff() -> Outcome<()> { |
| 89 | let cpu = Cpu::detect(); |
| 90 | if !cpu.has_fma() { |
| 91 | println!("skipped: this machine has no fused multiply-add to dispatch onto"); |
| 92 | return Ok(()); |
| 93 | } |
| 94 | let r = rate(cpu); |
| 95 | println!("{}x{}x{} on {:?}: {:.2} GMAC/s", SHAPE.0, SHAPE.1, SHAPE.2, cpu, r); |
| 96 | if r < FMA_FLOOR { |
| 97 | return Err(err!( |
| 98 | "The matrix kernel reached {:.2} GMAC/s against a floor of {:.1}. Either the \ |
| 99 | register tile now spills -- check {}x{} against the neighbouring shapes -- or \ |
| 100 | the specialised path is no longer reaching the inner loop.", r, FMA_FLOOR, MR, NR; |
| 101 | Invalid, Excessive)); |
| 102 | } |
| 103 | Ok(()) |
| 104 | } |
| 105 | |
| 106 | #[test] |
| 107 | fn the_baseline_path_has_not_reached_mul_add() -> Outcome<()> { |
| 108 | let r = rate(Cpu::Baseline); |
| 109 | println!("{}x{}x{} on baseline: {:.2} GMAC/s", SHAPE.0, SHAPE.1, SHAPE.2, r); |
| 110 | if r < BASELINE_FLOOR { |
| 111 | return Err(err!( |
| 112 | "The baseline kernel reached {:.2} GMAC/s against a floor of {:.1}, which is \ |
| 113 | what happens when `mul_add` is compiled for a target with no fused \ |
| 114 | multiply-add and becomes a library call.", r, BASELINE_FLOOR; |
| 115 | Invalid, Excessive)); |
| 116 | } |
| 117 | Ok(()) |
| 118 | } |