Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/tests/annealer_semantic.rs

11.1 KiB, 56 runs

created by r1870400018:14867, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Semantic-preservation regression for the annealer.
2//!
3//! A formatter may only change trivia (whitespace, comment layout);
4//! it must never alter the significant token stream. This test lexes
5//! every `.rs` file in the workspace, formats it, lexes the output,
6//! and asserts the two significant-token sequences are identical.
7//!
8//! This is the strongest correctness bar for a formatter: if the
9//! token stream survives, the code cannot have changed meaning.
10
11use oxedyne_fe2o3_text::fmt::{
12 format_rust,
13 lex::{
14 lex,
15 rust_tokens,
16 },
17 cst::{
18 Token,
19 TokenKind,
20 Trivia,
21 },
22 spec::FormatSpec,
23};
24
25/// A token reduced to its semantically-significant identity.
26///
27/// Plain comments are not tokens -- the lexer attaches them to the
28/// following token as `Trivia` -- so they are lifted into this stream
29/// as `Comment` entries. Without that, a formatter could delete every
30/// `//` line in the file and this test would still pass.
31#[derive(Debug, Clone, PartialEq, Eq)]
32enum Sig {
33 /// A significant token.
34 Tok {
35 kind: TokenKind,
36 text: String,
37 },
38 /// A plain `//` or `/* */` comment, right-trimmed.
39 Comment(String),
40}
41
42impl Sig {
43 /// The text, for divergence reporting.
44 fn text(&self) -> &str {
45 match self {
46 Self::Tok { text, .. } => text,
47 Self::Comment(s) => s,
48 }
49 }
50}
51
52/// Lift a token's leading comments into the comparable stream.
53///
54/// Whitespace and newlines are skipped: they are exactly what a
55/// formatter exists to change. Comment text is right-trimmed, since
56/// trailing spaces inside a comment carry no meaning.
57fn push_comments(out: &mut Vec<Sig>, tok: &Token) {
58 for t in &tok.leading_trivia {
59 match t {
60 Trivia::LineComment(s) | Trivia::BlockComment(s) => {
61 out.push(Sig::Comment(s.trim_end().to_string()));
62 }
63 Trivia::Whitespace(_) | Trivia::Newline => {}
64 }
65 }
66}
67
68/// Reduce a token stream to a semantically-comparable sequence.
69///
70/// Two normalisations remove changes that a formatter is entitled to
71/// make without altering meaning, so that what remains is genuine
72/// corruption:
73///
74/// - **Comment trailing whitespace** — a doc comment's trailing spaces
75/// carry no meaning, so comment text is right-trimmed.
76/// - **Trailing commas** — a comma immediately before a closing
77/// delimiter (`)`, `]`, `}`) is insignificant in Rust, so it is
78/// dropped from both sequences.
79fn sig(tokens: &[Token]) -> Vec<Sig> {
80 // First pass: drop EOF, right-trim comment text, and split
81 // angle-bracket-run operators (`>>`, `<<`) into individual
82 // brackets. The latter makes `Foo<Bar<T> >` and `Foo<Bar<T>>`
83 // compare equal: removing the space between two closing generics
84 // is a legitimate, meaning-preserving formatting change that only
85 // shifts a token boundary (`>` `>` becomes the single `>>`).
86 let mut raw: Vec<Sig> = Vec::with_capacity(tokens.len());
87 for t in tokens.iter() {
88 // Comments precede the token they are attached to, including
89 // the EOF token -- a comment at the end of a file hangs off
90 // EOF and would otherwise be dropped here.
91 push_comments(&mut raw, t);
92 if matches!(t.kind, TokenKind::Eof) {
93 continue;
94 }
95 match &t.kind {
96 TokenKind::DocComment(s) => raw.push(Sig::Tok {
97 kind: TokenKind::DocComment(s.trim_end().to_string()),
98 text: t.text.trim_end().to_string(),
99 }),
100 TokenKind::Operator(op)
101 if !op.is_empty()
102 && (op.bytes().all(|b| b == b'>') || op.bytes().all(|b| b == b'<')) =>
103 {
104 let ch = op.as_bytes()[0] as char;
105 for _ in 0..op.len() {
106 raw.push(Sig::Tok { kind: TokenKind::Punct(ch), text: ch.to_string() });
107 }
108 }
109 // A number token with a trailing dot (e.g. `0.`) arises when
110 // the formatter puts a space after a tuple-index dot
111 // (`self.0.buf` rendered as `self. 0. buf`): the `0.` then
112 // lexes as a float. Split it back into the digits plus a `.`
113 // Punct so it matches the source's `0` `.` sequence. This
114 // is a *split*, not a strip, so a genuine float-to-integer
115 // change (`0.` becoming `0`) would still be caught.
116 TokenKind::Number if t.text.len() > 1 && t.text.ends_with('.') => {
117 raw.push(Sig::Tok {
118 kind: TokenKind::Number,
119 text: t.text[..t.text.len() - 1].to_string(),
120 });
121 raw.push(Sig::Tok { kind: TokenKind::Punct('.'), text: ".".to_string() });
122 }
123 _ => raw.push(Sig::Tok { kind: t.kind.clone(), text: t.text.clone() }),
124 }
125 }
126
127 raw
128}
129
130/// The significant tokens alone, with insignificant trailing commas
131/// removed.
132///
133/// The comma drop runs here, on the comment-free sequence, so that a
134/// comment sitting between a comma and its closing delimiter cannot
135/// change whether that comma is judged insignificant.
136fn code_sig(tokens: &[Token]) -> Vec<Sig> {
137 let raw: Vec<Sig> = sig(tokens)
138 .into_iter()
139 .filter(|s| !matches!(s, Sig::Comment(_)))
140 .collect();
141 let mut out: Vec<Sig> = Vec::with_capacity(raw.len());
142 for i in 0..raw.len() {
143 if raw[i] == (Sig::Tok { kind: TokenKind::Punct(','), text: ",".to_string() }) {
144 if let Some(Sig::Tok { kind, .. }) = raw.get(i + 1) {
145 if matches!(
146 kind,
147 TokenKind::Punct(')') | TokenKind::Punct(']') | TokenKind::Punct('}')
148 ) {
149 continue;
150 }
151 }
152 }
153 out.push(raw[i].clone());
154 }
155 out
156}
157
158/// The comments alone, in order.
159fn comment_sig(tokens: &[Token]) -> Vec<Sig> {
160 sig(tokens)
161 .into_iter()
162 .filter(|s| matches!(s, Sig::Comment(_)))
163 .collect()
164}
165
166/// Report a single file's semantic divergence, if any.
167struct Divergence {
168 file: String,
169 index: usize,
170 before: Option<Sig>,
171 after: Option<Sig>,
172 context: String,
173}
174
175#[test]
176fn test_workspace_semantic_preservation() {
177 use std::fs;
178 use std::path::PathBuf;
179
180 let root = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
181 .parent()
182 .expect("workspace root")
183 .to_path_buf();
184 let spec = FormatSpec::fe2o3();
185 let lang = rust_tokens();
186
187 let mut checked = 0usize;
188 let mut lex_errs = 0usize;
189 let mut fmt_errs = 0usize;
190 let mut code_diffs = 0usize;
191 let mut comment_diffs = 0usize;
192 let mut diffs: Vec<Divergence> = Vec::new();
193
194 let mut dirs = vec![root.clone()];
195 while let Some(dir) = dirs.pop() {
196 let entries = match fs::read_dir(&dir) {
197 Ok(e) => e,
198 Err(_) => continue,
199 };
200 for entry in entries.flatten() {
201 let path = entry.path();
202 if path.is_dir() {
203 let name = path.file_name().unwrap_or_default();
204 // `.claude` holds agent worktrees: whole duplicate
205 // copies of this tree, which would be checked again
206 // under a second name.
207 if name == "target" || name == ".git" || name == ".claude" {
208 continue;
209 }
210 dirs.push(path);
211 continue;
212 }
213 if path.extension().and_then(|e| e.to_str()) != Some("rs") {
214 continue;
215 }
216 let source = match fs::read_to_string(&path) {
217 Ok(s) => s,
218 Err(_) => continue,
219 };
220 if source.trim().is_empty() {
221 continue;
222 }
223 let rel = path.strip_prefix(&root).unwrap_or(&path)
224 .display().to_string();
225
226 let src_toks = match lex(&source, &lang) {
227 Ok(t) => t,
228 Err(_) => { lex_errs += 1; continue; }
229 };
230 let formatted = match format_rust(&source, &spec) {
231 Ok(s) => s,
232 Err(_) => { fmt_errs += 1; continue; }
233 };
234 let out_toks = match lex(&formatted, &lang) {
235 Ok(t) => t,
236 Err(_) => { lex_errs += 1; continue; }
237 };
238
239 let a_code = code_sig(&src_toks);
240 let b_code = code_sig(&out_toks);
241 let a_com = comment_sig(&src_toks);
242 let b_com = comment_sig(&out_toks);
243 checked += 1;
244
245 // Two separate guarantees, because they carry different
246 // weight. Losing a comment, or reordering one, is
247 // corruption. Moving one between the end of a line and the
248 // line above is layout, which is what a formatter is for.
249 if a_code != b_code {
250 code_diffs += 1;
251 println!(" CODE DIVERGE {}", rel);
252 }
253 if a_com != b_com {
254 comment_diffs += 1;
255 println!(
256 " COMMENT DIVERGE {} ({} before, {} after)",
257 rel, a_com.len(), b_com.len(),
258 );
259 }
260 let (a, b) = (a_code, b_code);
261 if a != b {
262 // Locate the first divergence.
263 let mut idx = 0;
264 while idx < a.len() && idx < b.len() && a[idx] == b[idx] {
265 idx += 1;
266 }
267 // A little context: the three significant tokens before.
268 let lo = idx.saturating_sub(3);
269 let ctx: Vec<String> = a[lo..idx]
270 .iter()
271 .map(|s| s.text().to_string())
272 .collect();
273 diffs.push(Divergence {
274 file: rel,
275 index: idx,
276 before: a.get(idx).cloned(),
277 after: b.get(idx).cloned(),
278 context: ctx.join(" "),
279 });
280 }
281 }
282 }
283
284 println!(
285 "Semantic check: {} files, {} code-divergences, {} comment-divergences, \
286{} layout moves, {} lex-errs, {} refused.",
287 checked, code_diffs, comment_diffs,
288 diffs.len().saturating_sub(code_diffs + comment_diffs),
289 lex_errs, fmt_errs,
290 );
291 for d in diffs.iter().take(20) {
292 println!(
293 " DIVERGE {} @tok {} after [{}]:\n source: {:?}\n output: {:?}",
294 d.file, d.index, d.context, d.before, d.after,
295 );
296 }
297
298 // A formatter may move a comment between the end of a line and the
299 // line above it -- that is layout. It may never lose one, duplicate
300 // one, reorder them, or change a single significant token. Those
301 // are the two assertions worth making, and the reason the earlier
302 // token-only check passed while comments were being deleted.
303 assert_eq!(
304 code_diffs, 0,
305 "{} file(s) had their significant token stream altered",
306 code_diffs,
307 );
308 assert_eq!(
309 comment_diffs, 0,
310 "{} file(s) had a comment lost, added or reordered",
311 comment_diffs,
312 );
313}