Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_infer/tests/guard.rs

4.6 KiB, 1 run

created by r1870400018:19769, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! The regression guard the matrix kernel needs, and cannot do without.
2//!
3//! Two failures are silent and expensive, and neither shows up in a correctness
4//! test because both paths still compute the right answer:
5//!
6//! - **The register tile falls off its cliff.** Whether the `[[f32; NR]; MR]`
7//! accumulator lives in vector registers or spills to the stack is an
8//! all-or-nothing decision the code generator makes, and it moves with the
9//! tile shape, with the arithmetic form, and with the compiler version.
10//! Measured on this workload, adjacent tile heights differ by a factor of
11//! thirty-two. A one-character edit, or an upgrade nobody made, can cost
12//! thirty times the throughput.
13//! - **`mul_add` reaches the path that has no fused multiply-add.** There it
14//! becomes a library call, and the same thirty-fold loss appears in the
15//! fallback rather than the fast path, where it is even easier to miss.
16//!
17//! Both are caught here by measurement, on the shape that dominates the
18//! embedder: `(m, n, k) = (196, 512, 512)`, five of whose instances are
19//! forty-five per cent of the whole network.
20//!
21//! The floors are deliberately far below what the kernel actually reaches, so
22//! that a loaded or thermally throttled machine does not fail the build. They
23//! are set to catch a collapse, not a slowdown.
24
25use std::time::Instant;
26
27use oxedyne_fe2o3_core::prelude::*;
28use oxedyne_fe2o3_infer::kern::{self, Cpu, Scratch, Task, MR, NR};
29
30/// The shape that dominates the embedder.
31const SHAPE: (usize, usize, usize) = (196, 512, 512);
32
33/// Below this, on a machine with AVX2 and FMA, the accumulator has spilled.
34/// The kernel measures around fifty-eight on an idle machine, and around
35/// forty-five with the rest of the suite running beside it; a spill takes it to
36/// under two.
37const FMA_FLOOR: f64 = 15.0;
38
39/// Below this the baseline path has reached `mul_add` without the instruction,
40/// which costs about thirty times. That path measures around twelve idle and
41/// around four under load; the failure it is looking for measures half of one.
42const BASELINE_FLOOR: f64 = 1.5;
43
44/// A cheap reproducible generator, so the guard needs no dependency.
45fn fill(n: usize, seed: u64) -> Vec<f32> {
46 let mut s = seed;
47 let mut v = Vec::with_capacity(n);
48 for _ in 0..n {
49 s = s.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
50 v.push(((s >> 40) as f32 / 8_388_608.0) - 1.0);
51 }
52 v
53}
54
55/// Times the kernel on one shape, answering billions of multiply-accumulates
56/// per second.
57fn rate(cpu: Cpu) -> f64 {
58 let (m, n, k) = SHAPE;
59 let a = fill(m * k, 1);
60 let b = fill(k * n, 2);
61 let mut c = vec![0.0f32; m * n];
62 let mut s = Scratch::new();
63 // Warm the buffers and the branch predictor before anything is timed.
64 kern::run(cpu, Task::Gemm { m, n, k, a: &a, b: &b, c: &mut c, bias: None, scratch: &mut s });
65 let macs = (m * n * k) as f64;
66 let t = Instant::now();
67 let mut reps = 0u64;
68 // Time enough repeats to cover a fifth of a second, so the reading is
69 // stable without the test being slow.
70 while t.elapsed().as_secs_f64() < 0.2 {
71 kern::run(cpu, Task::Gemm {
72 m, n, k, a: &a, b: &b, c: &mut c, bias: None, scratch: &mut s });
73 reps += 1;
74 }
75 macs * reps as f64 / t.elapsed().as_secs_f64() / 1e9
76}
77
78#[test]
79fn the_register_tile_is_the_one_that_was_measured() -> Outcome<()> {
80 // Not a performance claim -- a statement that the shape has not been
81 // changed without someone reading why it is what it is.
82 req!(MR, 6usize);
83 req!(NR, 16usize);
84 Ok(())
85}
86
87#[test]
88fn the_dispatched_path_has_not_fallen_off_the_cliff() -> Outcome<()> {
89 let cpu = Cpu::detect();
90 if !cpu.has_fma() {
91 println!("skipped: this machine has no fused multiply-add to dispatch onto");
92 return Ok(());
93 }
94 let r = rate(cpu);
95 println!("{}x{}x{} on {:?}: {:.2} GMAC/s", SHAPE.0, SHAPE.1, SHAPE.2, cpu, r);
96 if r < FMA_FLOOR {
97 return Err(err!(
98 "The matrix kernel reached {:.2} GMAC/s against a floor of {:.1}. Either the \
99 register tile now spills -- check {}x{} against the neighbouring shapes -- or \
100 the specialised path is no longer reaching the inner loop.", r, FMA_FLOOR, MR, NR;
101 Invalid, Excessive));
102 }
103 Ok(())
104}
105
106#[test]
107fn the_baseline_path_has_not_reached_mul_add() -> Outcome<()> {
108 let r = rate(Cpu::Baseline);
109 println!("{}x{}x{} on baseline: {:.2} GMAC/s", SHAPE.0, SHAPE.1, SHAPE.2, r);
110 if r < BASELINE_FLOOR {
111 return Err(err!(
112 "The baseline kernel reached {:.2} GMAC/s against a floor of {:.1}, which is \
113 what happens when `mul_add` is compiled for a target with no fused \
114 multiply-add and becomes a library call.", r, BASELINE_FLOOR;
115 Invalid, Excessive));
116 }
117 Ok(())
118}