oxedyne/fe2o3/fe2o3_text/tests/annealer_semantic.rs
11.1 KiB, 56 runs
created by r1870400018:14867, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Semantic-preservation regression for the annealer. |
| 2 | //! |
| 3 | //! A formatter may only change trivia (whitespace, comment layout); |
| 4 | //! it must never alter the significant token stream. This test lexes |
| 5 | //! every `.rs` file in the workspace, formats it, lexes the output, |
| 6 | //! and asserts the two significant-token sequences are identical. |
| 7 | //! |
| 8 | //! This is the strongest correctness bar for a formatter: if the |
| 9 | //! token stream survives, the code cannot have changed meaning. |
| 10 | |
| 11 | use oxedyne_fe2o3_text::fmt::{ |
| 12 | format_rust, |
| 13 | lex::{ |
| 14 | lex, |
| 15 | rust_tokens, |
| 16 | }, |
| 17 | cst::{ |
| 18 | Token, |
| 19 | TokenKind, |
| 20 | Trivia, |
| 21 | }, |
| 22 | spec::FormatSpec, |
| 23 | }; |
| 24 | |
| 25 | /// A token reduced to its semantically-significant identity. |
| 26 | /// |
| 27 | /// Plain comments are not tokens -- the lexer attaches them to the |
| 28 | /// following token as `Trivia` -- so they are lifted into this stream |
| 29 | /// as `Comment` entries. Without that, a formatter could delete every |
| 30 | /// `//` line in the file and this test would still pass. |
| 31 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 32 | enum Sig { |
| 33 | /// A significant token. |
| 34 | Tok { |
| 35 | kind: TokenKind, |
| 36 | text: String, |
| 37 | }, |
| 38 | /// A plain `//` or `/* */` comment, right-trimmed. |
| 39 | Comment(String), |
| 40 | } |
| 41 | |
| 42 | impl Sig { |
| 43 | /// The text, for divergence reporting. |
| 44 | fn text(&self) -> &str { |
| 45 | match self { |
| 46 | Self::Tok { text, .. } => text, |
| 47 | Self::Comment(s) => s, |
| 48 | } |
| 49 | } |
| 50 | } |
| 51 | |
| 52 | /// Lift a token's leading comments into the comparable stream. |
| 53 | /// |
| 54 | /// Whitespace and newlines are skipped: they are exactly what a |
| 55 | /// formatter exists to change. Comment text is right-trimmed, since |
| 56 | /// trailing spaces inside a comment carry no meaning. |
| 57 | fn push_comments(out: &mut Vec<Sig>, tok: &Token) { |
| 58 | for t in &tok.leading_trivia { |
| 59 | match t { |
| 60 | Trivia::LineComment(s) | Trivia::BlockComment(s) => { |
| 61 | out.push(Sig::Comment(s.trim_end().to_string())); |
| 62 | } |
| 63 | Trivia::Whitespace(_) | Trivia::Newline => {} |
| 64 | } |
| 65 | } |
| 66 | } |
| 67 | |
| 68 | /// Reduce a token stream to a semantically-comparable sequence. |
| 69 | /// |
| 70 | /// Two normalisations remove changes that a formatter is entitled to |
| 71 | /// make without altering meaning, so that what remains is genuine |
| 72 | /// corruption: |
| 73 | /// |
| 74 | /// - **Comment trailing whitespace** — a doc comment's trailing spaces |
| 75 | /// carry no meaning, so comment text is right-trimmed. |
| 76 | /// - **Trailing commas** — a comma immediately before a closing |
| 77 | /// delimiter (`)`, `]`, `}`) is insignificant in Rust, so it is |
| 78 | /// dropped from both sequences. |
| 79 | fn sig(tokens: &[Token]) -> Vec<Sig> { |
| 80 | // First pass: drop EOF, right-trim comment text, and split |
| 81 | // angle-bracket-run operators (`>>`, `<<`) into individual |
| 82 | // brackets. The latter makes `Foo<Bar<T> >` and `Foo<Bar<T>>` |
| 83 | // compare equal: removing the space between two closing generics |
| 84 | // is a legitimate, meaning-preserving formatting change that only |
| 85 | // shifts a token boundary (`>` `>` becomes the single `>>`). |
| 86 | let mut raw: Vec<Sig> = Vec::with_capacity(tokens.len()); |
| 87 | for t in tokens.iter() { |
| 88 | // Comments precede the token they are attached to, including |
| 89 | // the EOF token -- a comment at the end of a file hangs off |
| 90 | // EOF and would otherwise be dropped here. |
| 91 | push_comments(&mut raw, t); |
| 92 | if matches!(t.kind, TokenKind::Eof) { |
| 93 | continue; |
| 94 | } |
| 95 | match &t.kind { |
| 96 | TokenKind::DocComment(s) => raw.push(Sig::Tok { |
| 97 | kind: TokenKind::DocComment(s.trim_end().to_string()), |
| 98 | text: t.text.trim_end().to_string(), |
| 99 | }), |
| 100 | TokenKind::Operator(op) |
| 101 | if !op.is_empty() |
| 102 | && (op.bytes().all(|b| b == b'>') || op.bytes().all(|b| b == b'<')) => |
| 103 | { |
| 104 | let ch = op.as_bytes()[0] as char; |
| 105 | for _ in 0..op.len() { |
| 106 | raw.push(Sig::Tok { kind: TokenKind::Punct(ch), text: ch.to_string() }); |
| 107 | } |
| 108 | } |
| 109 | // A number token with a trailing dot (e.g. `0.`) arises when |
| 110 | // the formatter puts a space after a tuple-index dot |
| 111 | // (`self.0.buf` rendered as `self. 0. buf`): the `0.` then |
| 112 | // lexes as a float. Split it back into the digits plus a `.` |
| 113 | // Punct so it matches the source's `0` `.` sequence. This |
| 114 | // is a *split*, not a strip, so a genuine float-to-integer |
| 115 | // change (`0.` becoming `0`) would still be caught. |
| 116 | TokenKind::Number if t.text.len() > 1 && t.text.ends_with('.') => { |
| 117 | raw.push(Sig::Tok { |
| 118 | kind: TokenKind::Number, |
| 119 | text: t.text[..t.text.len() - 1].to_string(), |
| 120 | }); |
| 121 | raw.push(Sig::Tok { kind: TokenKind::Punct('.'), text: ".".to_string() }); |
| 122 | } |
| 123 | _ => raw.push(Sig::Tok { kind: t.kind.clone(), text: t.text.clone() }), |
| 124 | } |
| 125 | } |
| 126 | |
| 127 | raw |
| 128 | } |
| 129 | |
| 130 | /// The significant tokens alone, with insignificant trailing commas |
| 131 | /// removed. |
| 132 | /// |
| 133 | /// The comma drop runs here, on the comment-free sequence, so that a |
| 134 | /// comment sitting between a comma and its closing delimiter cannot |
| 135 | /// change whether that comma is judged insignificant. |
| 136 | fn code_sig(tokens: &[Token]) -> Vec<Sig> { |
| 137 | let raw: Vec<Sig> = sig(tokens) |
| 138 | .into_iter() |
| 139 | .filter(|s| !matches!(s, Sig::Comment(_))) |
| 140 | .collect(); |
| 141 | let mut out: Vec<Sig> = Vec::with_capacity(raw.len()); |
| 142 | for i in 0..raw.len() { |
| 143 | if raw[i] == (Sig::Tok { kind: TokenKind::Punct(','), text: ",".to_string() }) { |
| 144 | if let Some(Sig::Tok { kind, .. }) = raw.get(i + 1) { |
| 145 | if matches!( |
| 146 | kind, |
| 147 | TokenKind::Punct(')') | TokenKind::Punct(']') | TokenKind::Punct('}') |
| 148 | ) { |
| 149 | continue; |
| 150 | } |
| 151 | } |
| 152 | } |
| 153 | out.push(raw[i].clone()); |
| 154 | } |
| 155 | out |
| 156 | } |
| 157 | |
| 158 | /// The comments alone, in order. |
| 159 | fn comment_sig(tokens: &[Token]) -> Vec<Sig> { |
| 160 | sig(tokens) |
| 161 | .into_iter() |
| 162 | .filter(|s| matches!(s, Sig::Comment(_))) |
| 163 | .collect() |
| 164 | } |
| 165 | |
| 166 | /// Report a single file's semantic divergence, if any. |
| 167 | struct Divergence { |
| 168 | file: String, |
| 169 | index: usize, |
| 170 | before: Option<Sig>, |
| 171 | after: Option<Sig>, |
| 172 | context: String, |
| 173 | } |
| 174 | |
| 175 | #[test] |
| 176 | fn test_workspace_semantic_preservation() { |
| 177 | use std::fs; |
| 178 | use std::path::PathBuf; |
| 179 | |
| 180 | let root = PathBuf::from(env!("CARGO_MANIFEST_DIR")) |
| 181 | .parent() |
| 182 | .expect("workspace root") |
| 183 | .to_path_buf(); |
| 184 | let spec = FormatSpec::fe2o3(); |
| 185 | let lang = rust_tokens(); |
| 186 | |
| 187 | let mut checked = 0usize; |
| 188 | let mut lex_errs = 0usize; |
| 189 | let mut fmt_errs = 0usize; |
| 190 | let mut code_diffs = 0usize; |
| 191 | let mut comment_diffs = 0usize; |
| 192 | let mut diffs: Vec<Divergence> = Vec::new(); |
| 193 | |
| 194 | let mut dirs = vec![root.clone()]; |
| 195 | while let Some(dir) = dirs.pop() { |
| 196 | let entries = match fs::read_dir(&dir) { |
| 197 | Ok(e) => e, |
| 198 | Err(_) => continue, |
| 199 | }; |
| 200 | for entry in entries.flatten() { |
| 201 | let path = entry.path(); |
| 202 | if path.is_dir() { |
| 203 | let name = path.file_name().unwrap_or_default(); |
| 204 | // `.claude` holds agent worktrees: whole duplicate |
| 205 | // copies of this tree, which would be checked again |
| 206 | // under a second name. |
| 207 | if name == "target" || name == ".git" || name == ".claude" { |
| 208 | continue; |
| 209 | } |
| 210 | dirs.push(path); |
| 211 | continue; |
| 212 | } |
| 213 | if path.extension().and_then(|e| e.to_str()) != Some("rs") { |
| 214 | continue; |
| 215 | } |
| 216 | let source = match fs::read_to_string(&path) { |
| 217 | Ok(s) => s, |
| 218 | Err(_) => continue, |
| 219 | }; |
| 220 | if source.trim().is_empty() { |
| 221 | continue; |
| 222 | } |
| 223 | let rel = path.strip_prefix(&root).unwrap_or(&path) |
| 224 | .display().to_string(); |
| 225 | |
| 226 | let src_toks = match lex(&source, &lang) { |
| 227 | Ok(t) => t, |
| 228 | Err(_) => { lex_errs += 1; continue; } |
| 229 | }; |
| 230 | let formatted = match format_rust(&source, &spec) { |
| 231 | Ok(s) => s, |
| 232 | Err(_) => { fmt_errs += 1; continue; } |
| 233 | }; |
| 234 | let out_toks = match lex(&formatted, &lang) { |
| 235 | Ok(t) => t, |
| 236 | Err(_) => { lex_errs += 1; continue; } |
| 237 | }; |
| 238 | |
| 239 | let a_code = code_sig(&src_toks); |
| 240 | let b_code = code_sig(&out_toks); |
| 241 | let a_com = comment_sig(&src_toks); |
| 242 | let b_com = comment_sig(&out_toks); |
| 243 | checked += 1; |
| 244 | |
| 245 | // Two separate guarantees, because they carry different |
| 246 | // weight. Losing a comment, or reordering one, is |
| 247 | // corruption. Moving one between the end of a line and the |
| 248 | // line above is layout, which is what a formatter is for. |
| 249 | if a_code != b_code { |
| 250 | code_diffs += 1; |
| 251 | println!(" CODE DIVERGE {}", rel); |
| 252 | } |
| 253 | if a_com != b_com { |
| 254 | comment_diffs += 1; |
| 255 | println!( |
| 256 | " COMMENT DIVERGE {} ({} before, {} after)", |
| 257 | rel, a_com.len(), b_com.len(), |
| 258 | ); |
| 259 | } |
| 260 | let (a, b) = (a_code, b_code); |
| 261 | if a != b { |
| 262 | // Locate the first divergence. |
| 263 | let mut idx = 0; |
| 264 | while idx < a.len() && idx < b.len() && a[idx] == b[idx] { |
| 265 | idx += 1; |
| 266 | } |
| 267 | // A little context: the three significant tokens before. |
| 268 | let lo = idx.saturating_sub(3); |
| 269 | let ctx: Vec<String> = a[lo..idx] |
| 270 | .iter() |
| 271 | .map(|s| s.text().to_string()) |
| 272 | .collect(); |
| 273 | diffs.push(Divergence { |
| 274 | file: rel, |
| 275 | index: idx, |
| 276 | before: a.get(idx).cloned(), |
| 277 | after: b.get(idx).cloned(), |
| 278 | context: ctx.join(" "), |
| 279 | }); |
| 280 | } |
| 281 | } |
| 282 | } |
| 283 | |
| 284 | println!( |
| 285 | "Semantic check: {} files, {} code-divergences, {} comment-divergences, \ |
| 286 | {} layout moves, {} lex-errs, {} refused.", |
| 287 | checked, code_diffs, comment_diffs, |
| 288 | diffs.len().saturating_sub(code_diffs + comment_diffs), |
| 289 | lex_errs, fmt_errs, |
| 290 | ); |
| 291 | for d in diffs.iter().take(20) { |
| 292 | println!( |
| 293 | " DIVERGE {} @tok {} after [{}]:\n source: {:?}\n output: {:?}", |
| 294 | d.file, d.index, d.context, d.before, d.after, |
| 295 | ); |
| 296 | } |
| 297 | |
| 298 | // A formatter may move a comment between the end of a line and the |
| 299 | // line above it -- that is layout. It may never lose one, duplicate |
| 300 | // one, reorder them, or change a single significant token. Those |
| 301 | // are the two assertions worth making, and the reason the earlier |
| 302 | // token-only check passed while comments were being deleted. |
| 303 | assert_eq!( |
| 304 | code_diffs, 0, |
| 305 | "{} file(s) had their significant token stream altered", |
| 306 | code_diffs, |
| 307 | ); |
| 308 | assert_eq!( |
| 309 | comment_diffs, 0, |
| 310 | "{} file(s) had a comment lost, added or reordered", |
| 311 | comment_diffs, |
| 312 | ); |
| 313 | } |