oxedyne/fe2o3/fe2o3_text/src/fmt/format.rs
68.5 KiB, 223 runs
created by r1870400018:11626, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! CST to Doc transformation. |
| 2 | //! |
| 3 | //! Walks the concrete syntax tree produced by the parser and converts |
| 4 | //! each node into a layout document (`Doc`) according to the format |
| 5 | //! specification. The layout algebra then handles line-breaking |
| 6 | //! decisions during rendering. |
| 7 | //! |
| 8 | |
| 9 | use crate::fmt::{ |
| 10 | cst::{ |
| 11 | CstChild, |
| 12 | CstNode, |
| 13 | Token, |
| 14 | TokenKind, |
| 15 | Trivia, |
| 16 | }, |
| 17 | doc::{self, Doc}, |
| 18 | lex, |
| 19 | parse, |
| 20 | render, |
| 21 | spec::{ |
| 22 | BinopBreak, |
| 23 | FormatSpec, |
| 24 | ReturnTypePlacement, |
| 25 | }, |
| 26 | }; |
| 27 | |
| 28 | use oxedyne_fe2o3_core::prelude::*; |
| 29 | |
| 30 | |
| 31 | /// Format Rust source code. |
| 32 | pub fn format_rust(source: &str, spec: &FormatSpec) -> Outcome<String> { |
| 33 | let lang = lex::rust_tokens(); |
| 34 | let tokens = res!(lex::lex(source, &lang)); |
| 35 | // The parser reads the EOF token as a stop condition and does not |
| 36 | // put it in the tree, so a comment on the last line -- which the |
| 37 | // lexer attaches to EOF as leading trivia -- never reaches the |
| 38 | // formatter. Carry it across the parse. |
| 39 | let trailing = tokens |
| 40 | .last() |
| 41 | .filter(|t| matches!(t.kind, TokenKind::Eof) |
| 42 | && t.leading_trivia.iter().any(is_comment_trivia)) |
| 43 | .cloned(); |
| 44 | let before = comment_texts(&tokens); |
| 45 | let cst = res!(parse::parse_rust(tokens)); |
| 46 | let document = format_source_file(&cst, trailing, spec); |
| 47 | let indent_str = spec.indent_str(); |
| 48 | let formatted = end_with_newline(render::render(spec.max_width, &indent_str, &document)); |
| 49 | |
| 50 | // Refuse rather than corrupt. Not every layout path carries a |
| 51 | // comment through yet, and a formatter that silently drops one -- |
| 52 | // or moves it onto the next item, where it comes to describe the |
| 53 | // wrong code -- does more harm than one that declines the file. |
| 54 | // Checked on the output, so it holds however the layout changes. |
| 55 | let after = comment_texts(&res!(lex::lex(&formatted, &lang))); |
| 56 | if before != after { |
| 57 | return Err(err!( |
| 58 | "Formatting this file would alter its comments ({} before, {} after). \ |
| 59 | Refusing, rather than dropping one.", |
| 60 | before.len(), after.len(); |
| 61 | Format, Mismatch)); |
| 62 | } |
| 63 | Ok(formatted) |
| 64 | } |
| 65 | |
| 66 | /// Every comment in a token stream, in order. |
| 67 | /// |
| 68 | /// Plain comments are trivia attached to the following token; doc |
| 69 | /// comments are tokens in their own right. Both are compared, so a |
| 70 | /// formatting pass cannot quietly lose either. Right-trimmed, because |
| 71 | /// trailing spaces inside a comment carry no meaning. |
| 72 | fn comment_texts(tokens: &[Token]) -> Vec<String> { |
| 73 | let mut out = Vec::new(); |
| 74 | for tok in tokens { |
| 75 | for t in &tok.leading_trivia { |
| 76 | if let Trivia::LineComment(c) | Trivia::BlockComment(c) = t { |
| 77 | out.push(c.trim_end().to_string()); |
| 78 | } |
| 79 | } |
| 80 | if let TokenKind::DocComment(c) = &tok.kind { |
| 81 | out.push(c.trim_end().to_string()); |
| 82 | } |
| 83 | } |
| 84 | out |
| 85 | } |
| 86 | |
| 87 | /// Give the output the single trailing newline a text file should end |
| 88 | /// with, so a formatted file does not read as "\ No newline at end of |
| 89 | /// file" in every diff. |
| 90 | fn end_with_newline(mut s: String) -> String { |
| 91 | while s.ends_with('\n') { |
| 92 | s.pop(); |
| 93 | } |
| 94 | if !s.is_empty() { |
| 95 | s.push('\n'); |
| 96 | } |
| 97 | s |
| 98 | } |
| 99 | |
| 100 | /// Format source code using a pre-built language token definition. |
| 101 | /// |
| 102 | /// For Rust, this goes through the structural parser for |
| 103 | /// keyword-aware formatting. For other languages, it uses the |
| 104 | /// generic token-stream pipeline directly. |
| 105 | pub fn format_with_lang( |
| 106 | source: &str, |
| 107 | lang: &lex::LangTokens, |
| 108 | spec: &FormatSpec, |
| 109 | ) -> Outcome<String> { |
| 110 | let tokens = res!(lex::lex(source, lang)); |
| 111 | let document = tokens_to_formatted_doc(&tokens, spec); |
| 112 | let indent_str = spec.indent_str(); |
| 113 | let formatted = render::render(spec.max_width, &indent_str, &document); |
| 114 | Ok(end_with_newline(formatted)) |
| 115 | } |
| 116 | |
| 117 | // ── Source file ────────────────────────────────────────────────── |
| 118 | |
| 119 | /// Flatten a CST into tokens and format via the token-stream pipeline. |
| 120 | /// |
| 121 | /// `trailing` is the lexer's EOF token where it carries a comment the |
| 122 | /// parse would otherwise have dropped. |
| 123 | fn format_source_file(node: &CstNode, trailing: Option<Token>, spec: &FormatSpec) -> Doc { |
| 124 | let mut tokens: Vec<Token> = Vec::new(); |
| 125 | collect_tokens(node, &mut tokens); |
| 126 | if let Some(eof) = trailing { |
| 127 | tokens.push(eof); |
| 128 | } |
| 129 | tokens_to_formatted_doc(&tokens, spec) |
| 130 | } |
| 131 | |
| 132 | /// Core formatting logic: convert a token stream to a Doc with |
| 133 | /// proper spacing, newlines, and brace-based indentation. |
| 134 | /// |
| 135 | /// Recursively processes `{ ... }` blocks, wrapping their contents |
| 136 | /// in `Nest` so the renderer applies the correct indentation. |
| 137 | fn tokens_to_formatted_doc(tokens: &[Token], spec: &FormatSpec) -> Doc { |
| 138 | let mut tokens = tokens.to_vec(); |
| 139 | // Apply import reordering before formatting. |
| 140 | if spec.import_reorder { |
| 141 | apply_import_reorder(&mut tokens); |
| 142 | } |
| 143 | let mut pos = 0; |
| 144 | let result = format_token_range(&tokens, &mut pos, spec, false, false); |
| 145 | doc::concat(result) |
| 146 | } |
| 147 | |
| 148 | /// Find `use ... { items }` patterns in the token stream and |
| 149 | /// reorder the items alphabetically. |
| 150 | fn apply_import_reorder(tokens: &mut Vec<Token>) { |
| 151 | // Find each `use` keyword and look for a subsequent `{ ... }`. |
| 152 | let mut i = 0; |
| 153 | while i < tokens.len() { |
| 154 | if matches!(tokens[i].kind, TokenKind::Keyword(ref k) if k == "use") { |
| 155 | // Find the `{` and `}` for this use statement. |
| 156 | let mut j = i + 1; |
| 157 | while j < tokens.len() && !matches!(tokens[j].kind, TokenKind::Punct('{') | TokenKind::Punct(';')) { |
| 158 | j += 1; |
| 159 | } |
| 160 | if j < tokens.len() && matches!(tokens[j].kind, TokenKind::Punct('{')) { |
| 161 | // Extract the range from `use` to the `;` after `}`. |
| 162 | let open = j; |
| 163 | let mut depth = 1usize; |
| 164 | let mut k = open + 1; |
| 165 | while k < tokens.len() && depth > 0 { |
| 166 | match tokens[k].kind { |
| 167 | TokenKind::Punct('{') => depth += 1, |
| 168 | TokenKind::Punct('}') => depth -= 1, |
| 169 | _ => {} |
| 170 | } |
| 171 | k += 1; |
| 172 | } |
| 173 | let close = k - 1; // position of `}` |
| 174 | // Extract and reorder just this use block's tokens. |
| 175 | let mut use_tokens: Vec<Token> = tokens[i..=close].to_vec(); |
| 176 | // Find `;` after `}`. |
| 177 | if close + 1 < tokens.len() && matches!(tokens[close + 1].kind, TokenKind::Punct(';')) { |
| 178 | use_tokens.push(tokens[close + 1].clone()); |
| 179 | } |
| 180 | reorder_use_items(&mut use_tokens); |
| 181 | // Replace in the main token list. |
| 182 | let end = if close + 1 < tokens.len() && matches!(tokens[close + 1].kind, TokenKind::Punct(';')) { |
| 183 | close + 2 |
| 184 | } else { |
| 185 | close + 1 |
| 186 | }; |
| 187 | tokens.splice(i..end, use_tokens); |
| 188 | } |
| 189 | } |
| 190 | i += 1; |
| 191 | } |
| 192 | } |
| 193 | |
| 194 | /// Format tokens from `pos` to the end (or the matching `}`). |
| 195 | /// When `inside_brace` is true, we stop at the matching `}`. |
| 196 | fn format_token_range( |
| 197 | tokens: &[Token], |
| 198 | pos: &mut usize, |
| 199 | spec: &FormatSpec, |
| 200 | inside_brace: bool, |
| 201 | inside_macro: bool, |
| 202 | ) -> Vec<Doc> { |
| 203 | let indent = spec.indent_width; |
| 204 | let mut docs = Vec::new(); |
| 205 | |
| 206 | while *pos < tokens.len() { |
| 207 | let i = *pos; |
| 208 | let tok = &tokens[i]; |
| 209 | |
| 210 | // Emit leading trivia (newlines, comments). |
| 211 | let mut had_newline = false; |
| 212 | let mut newline_count = 0usize; |
| 213 | for t in &tok.leading_trivia { |
| 214 | match t { |
| 215 | Trivia::Newline => { |
| 216 | had_newline = true; |
| 217 | newline_count += 1; |
| 218 | } |
| 219 | Trivia::LineComment(c) => { |
| 220 | flush_newlines(&mut docs, &mut newline_count, &mut had_newline, spec); |
| 221 | docs.push(doc::text(c.as_str())); |
| 222 | } |
| 223 | Trivia::BlockComment(c) => { |
| 224 | flush_newlines(&mut docs, &mut newline_count, &mut had_newline, spec); |
| 225 | docs.push(doc::text(c.as_str())); |
| 226 | } |
| 227 | Trivia::Whitespace(_) => { |
| 228 | if !had_newline && i > 0 { |
| 229 | docs.push(doc::text(" ")); |
| 230 | } |
| 231 | } |
| 232 | } |
| 233 | } |
| 234 | flush_newlines(&mut docs, &mut newline_count, &mut had_newline, spec); |
| 235 | |
| 236 | // Space before this token. |
| 237 | if !had_newline && i > 0 { |
| 238 | let prev = &tokens[i - 1]; |
| 239 | if needs_separator_space(prev, tok) |
| 240 | || (!suppress_space_before(tok) |
| 241 | && !suppress_space_after(prev) |
| 242 | && tok.leading_trivia.is_empty()) |
| 243 | { |
| 244 | docs.push(doc::text(" ")); |
| 245 | } |
| 246 | } |
| 247 | |
| 248 | match tok.kind { |
| 249 | // ── fn signature: Oxedyne layout ───────────────── |
| 250 | TokenKind::Keyword(ref kw) if kw == "fn" => { |
| 251 | let fn_doc = format_fn_from_tokens(tokens, pos, spec); |
| 252 | docs.push(fn_doc); |
| 253 | } |
| 254 | |
| 255 | // ── match expression ───────────────────────────── |
| 256 | TokenKind::Keyword(ref kw) if kw == "match" => { |
| 257 | docs.push(doc::text("match")); |
| 258 | *pos += 1; |
| 259 | // Consume the scrutinee (tokens until `{`). |
| 260 | while *pos < tokens.len() && !matches!(tokens[*pos].kind, TokenKind::Punct('{')) { |
| 261 | let mt = &tokens[*pos]; |
| 262 | if !suppress_space_before(mt) { |
| 263 | docs.push(doc::text(" ")); |
| 264 | } |
| 265 | docs.push(doc::text(mt.text.as_str())); |
| 266 | *pos += 1; |
| 267 | } |
| 268 | // The `{` will be handled by the brace handler on |
| 269 | // the next iteration — but we want match-arm-aware |
| 270 | // formatting. Consume the block ourselves. |
| 271 | if *pos < tokens.len() && matches!(tokens[*pos].kind, TokenKind::Punct('{')) { |
| 272 | docs.push(doc::text(" {")); |
| 273 | *pos += 1; |
| 274 | // Collect arm tokens. |
| 275 | let mut arm_tokens: Vec<Token> = Vec::new(); |
| 276 | let mut depth = 1usize; |
| 277 | while *pos < tokens.len() && depth > 0 { |
| 278 | let at = &tokens[*pos]; |
| 279 | match at.kind { |
| 280 | TokenKind::Punct('{') => { depth += 1; arm_tokens.push(at.clone()); } |
| 281 | TokenKind::Punct('}') => { |
| 282 | depth -= 1; |
| 283 | if depth > 0 { |
| 284 | arm_tokens.push(at.clone()); |
| 285 | } else { |
| 286 | carry_close_comments(&mut arm_tokens, at); |
| 287 | } |
| 288 | } |
| 289 | _ => { arm_tokens.push(at.clone()); } |
| 290 | } |
| 291 | *pos += 1; |
| 292 | } |
| 293 | let arms_doc = format_match_arms(&arm_tokens, spec); |
| 294 | if !arm_tokens.is_empty() { |
| 295 | docs.push(doc::nest(indent, doc::concat(vec![ |
| 296 | doc::hardline(), |
| 297 | arms_doc, |
| 298 | ]))); |
| 299 | } |
| 300 | if depth == 0 { |
| 301 | docs.push(doc::hardline()); |
| 302 | docs.push(doc::text("}")); |
| 303 | } |
| 304 | } |
| 305 | } |
| 306 | |
| 307 | TokenKind::Punct('{') => { |
| 308 | // Check if this is a struct/enum field block. |
| 309 | // Macro bodies and their nested blocks are never |
| 310 | // treated as field blocks. |
| 311 | let is_macro_entry = i > 0 |
| 312 | && matches!(tokens[i - 1].kind, TokenKind::Punct('!')); |
| 313 | let in_macro = inside_macro || is_macro_entry; |
| 314 | let is_field_block = !in_macro |
| 315 | && is_preceded_by_struct_or_enum(tokens, i); |
| 316 | |
| 317 | docs.push(doc::text("{")); |
| 318 | *pos += 1; |
| 319 | |
| 320 | if is_field_block && spec.field_align_threshold > 0 { |
| 321 | // Collect the raw tokens inside the block. |
| 322 | let mut field_tokens: Vec<Token> = Vec::new(); |
| 323 | let mut depth = 1usize; |
| 324 | while *pos < tokens.len() && depth > 0 { |
| 325 | let bt = &tokens[*pos]; |
| 326 | match bt.kind { |
| 327 | TokenKind::Punct('{') => { depth += 1; field_tokens.push(bt.clone()); } |
| 328 | TokenKind::Punct('}') => { |
| 329 | depth -= 1; |
| 330 | if depth > 0 { |
| 331 | field_tokens.push(bt.clone()); |
| 332 | } else { |
| 333 | carry_close_comments(&mut field_tokens, bt); |
| 334 | } |
| 335 | } |
| 336 | _ => { field_tokens.push(bt.clone()); } |
| 337 | } |
| 338 | *pos += 1; |
| 339 | } |
| 340 | // Format as aligned fields. |
| 341 | let field_doc = format_aligned_fields(&field_tokens, spec); |
| 342 | if !field_tokens.is_empty() { |
| 343 | docs.push(doc::nest(indent, doc::concat(vec![ |
| 344 | doc::hardline(), |
| 345 | field_doc, |
| 346 | ]))); |
| 347 | } |
| 348 | if depth == 0 { |
| 349 | docs.push(doc::hardline()); |
| 350 | docs.push(doc::text("}")); |
| 351 | } |
| 352 | } else { |
| 353 | // Generic block: recurse. |
| 354 | let inner = format_token_range(tokens, pos, spec, true, in_macro); |
| 355 | let (content, closing) = split_off_closing_brace(inner); |
| 356 | if !content.is_empty() { |
| 357 | docs.push(doc::nest(indent, doc::concat(content))); |
| 358 | } |
| 359 | docs.extend(closing); |
| 360 | } |
| 361 | } |
| 362 | TokenKind::Punct('}') if inside_brace => { |
| 363 | docs.push(doc::text("}")); |
| 364 | *pos += 1; |
| 365 | return docs; |
| 366 | } |
| 367 | // `where` keyword: optionally indent the predicates. |
| 368 | TokenKind::Keyword(ref kw) if kw == "where" && spec.where_indent => { |
| 369 | docs.push(doc::text("where")); |
| 370 | *pos += 1; |
| 371 | // Collect tokens until `{` or `;`. |
| 372 | let mut pred_docs = Vec::new(); |
| 373 | while *pos < tokens.len() { |
| 374 | let pt = &tokens[*pos]; |
| 375 | if matches!(pt.kind, TokenKind::Punct('{') | TokenKind::Punct(';')) { |
| 376 | break; |
| 377 | } |
| 378 | // Emit trivia + token with smart spacing. |
| 379 | let mut ph = false; |
| 380 | let mut pn = 0usize; |
| 381 | for t in &pt.leading_trivia { |
| 382 | match t { |
| 383 | Trivia::Newline => { ph = true; pn += 1; } |
| 384 | Trivia::LineComment(c) => { |
| 385 | flush_newlines(&mut pred_docs, &mut pn, &mut ph, spec); |
| 386 | pred_docs.push(doc::text(c.as_str())); |
| 387 | } |
| 388 | Trivia::BlockComment(c) => { |
| 389 | flush_newlines(&mut pred_docs, &mut pn, &mut ph, spec); |
| 390 | pred_docs.push(doc::text(c.as_str())); |
| 391 | } |
| 392 | Trivia::Whitespace(_) => { |
| 393 | if !ph { |
| 394 | pred_docs.push(doc::text(" ")); |
| 395 | } |
| 396 | } |
| 397 | } |
| 398 | } |
| 399 | flush_newlines(&mut pred_docs, &mut pn, &mut ph, spec); |
| 400 | if !ph && !pred_docs.is_empty() { |
| 401 | let prev_tok = &tokens[*pos - 1]; |
| 402 | if !suppress_space_before_ctx(pt, Some(prev_tok)) |
| 403 | && !suppress_space_after(prev_tok) |
| 404 | && pt.leading_trivia.is_empty() |
| 405 | { |
| 406 | pred_docs.push(doc::text(" ")); |
| 407 | } |
| 408 | } |
| 409 | if !pt.text.is_empty() { |
| 410 | pred_docs.push(doc::text(pt.text.as_str())); |
| 411 | } |
| 412 | *pos += 1; |
| 413 | } |
| 414 | if !pred_docs.is_empty() { |
| 415 | docs.push(doc::nest(indent, doc::concat(pred_docs))); |
| 416 | } |
| 417 | } |
| 418 | // ── let / return: binop-aware statement ──── |
| 419 | // Skip `let` inside `if let` / `while let` patterns. |
| 420 | TokenKind::Keyword(ref kw) if (kw == "let" || kw == "return") |
| 421 | && !(kw == "let" && i > 0 && matches!( |
| 422 | tokens[i - 1].kind, |
| 423 | TokenKind::Keyword(ref pk) if pk == "if" || pk == "while" |
| 424 | )) => |
| 425 | { |
| 426 | docs.push(doc::text(tok.text.as_str())); |
| 427 | *pos += 1; |
| 428 | // Collect remaining tokens through `;`. |
| 429 | let start = *pos; |
| 430 | let mut end = *pos; |
| 431 | let mut depth = 0i32; |
| 432 | while end < tokens.len() { |
| 433 | match &tokens[end].kind { |
| 434 | TokenKind::Punct('(') | TokenKind::Punct('[') |
| 435 | | TokenKind::Punct('{') => depth += 1, |
| 436 | TokenKind::Punct(')') | TokenKind::Punct(']') => depth -= 1, |
| 437 | TokenKind::Punct('}') => { |
| 438 | depth -= 1; |
| 439 | if depth < 0 { break; } |
| 440 | } |
| 441 | TokenKind::Punct(';') if depth == 0 => { |
| 442 | end += 1; |
| 443 | break; |
| 444 | } |
| 445 | _ => {} |
| 446 | } |
| 447 | end += 1; |
| 448 | } |
| 449 | if start < end { |
| 450 | let mut stmt_rest: Vec<Token> = tokens[start..end].to_vec(); |
| 451 | // First token's trivia already handled by the outer loop. |
| 452 | if let Some(first) = stmt_rest.first_mut() { |
| 453 | first.leading_trivia.clear(); |
| 454 | } |
| 455 | // Space between keyword and first token (unless `;`). |
| 456 | if !matches!(stmt_rest[0].kind, TokenKind::Punct(';')) { |
| 457 | docs.push(doc::text(" ")); |
| 458 | } |
| 459 | if stmt_rest.iter().any(|t| is_unambiguous_binary_op(t)) { |
| 460 | docs.push(format_binop_expr(&stmt_rest, spec)); |
| 461 | } else { |
| 462 | // No chain breaks — method chain wiring is separate. |
| 463 | docs.push(doc::concat(smart_spaced_tokens_inner(&stmt_rest, false))); |
| 464 | } |
| 465 | } |
| 466 | *pos = end; |
| 467 | } |
| 468 | TokenKind::Eof => { |
| 469 | *pos += 1; |
| 470 | break; |
| 471 | } |
| 472 | _ => { |
| 473 | if !tok.text.is_empty() { |
| 474 | docs.push(doc::text(tok.text.as_str())); |
| 475 | } |
| 476 | *pos += 1; |
| 477 | } |
| 478 | } |
| 479 | } |
| 480 | docs |
| 481 | } |
| 482 | |
| 483 | /// Split a doc list into (content, closing). The closing part is |
| 484 | /// the trailing HardLine + `}` that should sit outside the Nest. |
| 485 | fn split_off_closing_brace(mut docs: Vec<Doc>) -> (Vec<Doc>, Vec<Doc>) { |
| 486 | // Walk backwards to find the `}` text and any preceding hardlines. |
| 487 | let mut closing = Vec::new(); |
| 488 | // Pop the `}`. |
| 489 | if let Some(last) = docs.pop() { |
| 490 | closing.push(last); |
| 491 | } |
| 492 | // Pop any hardlines immediately before the `}`. |
| 493 | while let Some(d) = docs.pop() { |
| 494 | if matches!(d, Doc::HardLine) { |
| 495 | closing.push(d); |
| 496 | } else { |
| 497 | // Not a hardline: restore it and stop. |
| 498 | docs.push(d); |
| 499 | break; |
| 500 | } |
| 501 | } |
| 502 | closing.reverse(); |
| 503 | (docs, closing) |
| 504 | } |
| 505 | |
| 506 | /// Format an `fn` signature directly from the token stream. |
| 507 | /// |
| 508 | /// Reads tokens starting at `fn` through to `{` or `;`, and produces |
| 509 | /// the Oxedyne layout: |
| 510 | /// - Flat: `fn foo(x: u32, y: u64) -> bool` |
| 511 | /// - Broken: `fn foo(\n x: u32,\n y: u64,\n)\n -> bool` |
| 512 | /// |
| 513 | /// Does NOT consume the `{` or `;` — leaves that for the caller. |
| 514 | fn format_fn_from_tokens( |
| 515 | tokens: &[Token], |
| 516 | pos: &mut usize, |
| 517 | spec: &FormatSpec, |
| 518 | ) -> Doc { |
| 519 | let indent = spec.indent_width; |
| 520 | |
| 521 | // Phase 1: collect signature tokens into segments. |
| 522 | let mut pre_paren: Vec<Token> = Vec::new(); // fn name<T> |
| 523 | let mut params: Vec<Vec<Token>> = Vec::new(); // param groups |
| 524 | let mut ret_tokens: Vec<Token> = Vec::new(); // -> Type |
| 525 | let mut where_tokens: Vec<Token> = Vec::new();// where T: Foo |
| 526 | let mut has_parens = false; |
| 527 | |
| 528 | // Consume `fn`. Strip its leading trivia since the caller |
| 529 | // already emitted it. |
| 530 | let mut fn_tok = tokens[*pos].clone(); |
| 531 | fn_tok.leading_trivia.clear(); |
| 532 | pre_paren.push(fn_tok); |
| 533 | *pos += 1; |
| 534 | |
| 535 | // Consume name and optional generics (until `(`). |
| 536 | while *pos < tokens.len() { |
| 537 | let t = &tokens[*pos]; |
| 538 | if matches!(t.kind, TokenKind::Punct('(')) { |
| 539 | has_parens = true; |
| 540 | break; |
| 541 | } |
| 542 | if matches!(t.kind, TokenKind::Punct('{') | TokenKind::Punct(';')) { |
| 543 | break; |
| 544 | } |
| 545 | pre_paren.push(t.clone()); |
| 546 | *pos += 1; |
| 547 | } |
| 548 | |
| 549 | // Consume parameter list `(...)`. |
| 550 | if has_parens && *pos < tokens.len() { |
| 551 | *pos += 1; // skip `(` |
| 552 | let mut depth = 1usize; |
| 553 | let mut current_param: Vec<Token> = Vec::new(); |
| 554 | while *pos < tokens.len() && depth > 0 { |
| 555 | let t = &tokens[*pos]; |
| 556 | match t.kind { |
| 557 | TokenKind::Punct('(') => { depth += 1; current_param.push(t.clone()); } |
| 558 | TokenKind::Punct(')') => { |
| 559 | depth -= 1; |
| 560 | if depth == 0 { |
| 561 | // A comment on the parameter list's last line is |
| 562 | // attached to this paren, and is dropped with it. |
| 563 | // Carrying it here does not help -- parameters |
| 564 | // render through smart_spaced_tokens, which emits |
| 565 | // no trivia -- and an empty group would earn a |
| 566 | // comma of its own. The comment guard in |
| 567 | // format_rust refuses such a file instead. |
| 568 | if !current_param.is_empty() { |
| 569 | params.push(current_param); |
| 570 | current_param = Vec::new(); |
| 571 | } |
| 572 | } else { |
| 573 | current_param.push(t.clone()); |
| 574 | } |
| 575 | } |
| 576 | TokenKind::Punct(',') if depth == 1 => { |
| 577 | if !current_param.is_empty() { |
| 578 | params.push(current_param); |
| 579 | current_param = Vec::new(); |
| 580 | } |
| 581 | } |
| 582 | _ => { current_param.push(t.clone()); } |
| 583 | } |
| 584 | *pos += 1; |
| 585 | } |
| 586 | } |
| 587 | |
| 588 | // Consume return type `-> Type` and/or where clause. |
| 589 | let mut in_where = false; |
| 590 | while *pos < tokens.len() { |
| 591 | let t = &tokens[*pos]; |
| 592 | if matches!(t.kind, TokenKind::Punct('{') | TokenKind::Punct(';')) { |
| 593 | break; |
| 594 | } |
| 595 | if matches!(t.kind, TokenKind::Keyword(ref k) if k == "where") { |
| 596 | in_where = true; |
| 597 | } |
| 598 | if in_where { |
| 599 | where_tokens.push(t.clone()); |
| 600 | } else { |
| 601 | ret_tokens.push(t.clone()); |
| 602 | } |
| 603 | *pos += 1; |
| 604 | } |
| 605 | |
| 606 | // Phase 2: build the Doc. |
| 607 | let mut sig = Vec::new(); |
| 608 | |
| 609 | // fn name<T> |
| 610 | sig.push(doc::concat(smart_spaced_tokens(&pre_paren))); |
| 611 | |
| 612 | // Parameters + return type wrapped in ONE group. |
| 613 | // The group's flat/broken state controls both the param layout |
| 614 | // AND the return type placement (Oxedyne rule). |
| 615 | let mut grouped = Vec::new(); |
| 616 | |
| 617 | if has_parens { |
| 618 | if params.is_empty() { |
| 619 | grouped.push(doc::text("()")); |
| 620 | } else { |
| 621 | // Build param docs. When vertical and there are 2+ |
| 622 | // params, align the type column. |
| 623 | let align = params.len() >= 2 && spec.field_align_threshold > 0; |
| 624 | let param_docs = format_param_list(¶ms, align, spec); |
| 625 | |
| 626 | let sep = doc::concat(vec![doc::text(","), doc::line()]); |
| 627 | let body = doc::join(sep, param_docs); |
| 628 | let trailing = if spec.trailing_comma { |
| 629 | doc::trailing_comma() |
| 630 | } else { |
| 631 | doc::empty() |
| 632 | }; |
| 633 | grouped.push(doc::text("(")); |
| 634 | grouped.push(doc::nest(indent, doc::concat(vec![ |
| 635 | doc::softline(), |
| 636 | body, |
| 637 | trailing, |
| 638 | ]))); |
| 639 | grouped.push(doc::softline()); |
| 640 | grouped.push(doc::text(")")); |
| 641 | } |
| 642 | } |
| 643 | |
| 644 | // Return type: inline when flat, own line when broken. |
| 645 | if !ret_tokens.is_empty() { |
| 646 | let ret_doc = doc::concat(smart_spaced_tokens(&ret_tokens)); |
| 647 | match spec.fn_return_type { |
| 648 | ReturnTypePlacement::OwnLine => { |
| 649 | grouped.push(doc::if_break( |
| 650 | // Flat: ` -> Type` |
| 651 | doc::concat(vec![doc::text(" "), ret_doc.clone()]), |
| 652 | // Broken: `\n -> Type` |
| 653 | doc::nest(indent, doc::concat(vec![ |
| 654 | doc::hardline(), |
| 655 | ret_doc, |
| 656 | ])), |
| 657 | )); |
| 658 | } |
| 659 | ReturnTypePlacement::SameLine => { |
| 660 | grouped.push(doc::text(" ")); |
| 661 | grouped.push(ret_doc); |
| 662 | } |
| 663 | } |
| 664 | } |
| 665 | |
| 666 | // Where clause — part of the grouped section so that its |
| 667 | // presence forces a break (which triggers the Oxedyne layout). |
| 668 | if !where_tokens.is_empty() { |
| 669 | grouped.push(doc::hardline()); |
| 670 | if spec.where_indent { |
| 671 | grouped.push(doc::text("where")); |
| 672 | let pred_tokens: Vec<Token> = where_tokens.into_iter().skip(1).collect(); |
| 673 | if !pred_tokens.is_empty() { |
| 674 | grouped.push(doc::nest(indent, doc::concat(vec![ |
| 675 | doc::hardline(), |
| 676 | doc::concat(smart_spaced_tokens(&pred_tokens)), |
| 677 | ]))); |
| 678 | } |
| 679 | } else { |
| 680 | grouped.push(doc::concat(smart_spaced_tokens(&where_tokens))); |
| 681 | } |
| 682 | } |
| 683 | |
| 684 | // Brace + body. |
| 685 | // Oxedyne rule: when the signature is flat, body stays inline |
| 686 | // fn foo(x: u32) -> bool { expr } |
| 687 | // When broken, brace on own line, body indented: |
| 688 | // fn foo( |
| 689 | // x: u32, |
| 690 | // ) |
| 691 | // -> bool |
| 692 | // { |
| 693 | // expr |
| 694 | // } |
| 695 | if *pos < tokens.len() && matches!(tokens[*pos].kind, TokenKind::Punct('{')) { |
| 696 | // Consume `{` and collect body tokens until matching `}`. |
| 697 | *pos += 1; |
| 698 | let mut body_tokens: Vec<Token> = Vec::new(); |
| 699 | let mut depth = 1usize; |
| 700 | while *pos < tokens.len() && depth > 0 { |
| 701 | let bt = &tokens[*pos]; |
| 702 | match bt.kind { |
| 703 | TokenKind::Punct('{') => { depth += 1; body_tokens.push(bt.clone()); } |
| 704 | TokenKind::Punct('}') => { |
| 705 | depth -= 1; |
| 706 | if depth > 0 { |
| 707 | body_tokens.push(bt.clone()); |
| 708 | } else { |
| 709 | carry_close_comments(&mut body_tokens, bt); |
| 710 | } |
| 711 | } |
| 712 | _ => { body_tokens.push(bt.clone()); } |
| 713 | } |
| 714 | *pos += 1; |
| 715 | } |
| 716 | |
| 717 | // Decide: single-expression body (can go on one line) or |
| 718 | // multi-statement body (always vertical). |
| 719 | let is_short = is_short_body(&body_tokens); |
| 720 | |
| 721 | let found_close = depth == 0; |
| 722 | |
| 723 | if is_short && spec.fn_single_line { |
| 724 | // Short body: use smart_spaced_tokens (no trivia hardlines) |
| 725 | // so the group can choose flat. |
| 726 | let body_doc = doc::concat(smart_spaced_tokens(&body_tokens)); |
| 727 | |
| 728 | grouped.push(doc::if_break( |
| 729 | doc::text(" {"), |
| 730 | doc::concat(vec![doc::hardline(), doc::text("{")]), |
| 731 | )); |
| 732 | grouped.push(doc::if_break( |
| 733 | doc::concat(vec![doc::text(" "), body_doc.clone(), doc::text(" ")]), |
| 734 | doc::concat(vec![ |
| 735 | doc::nest(indent, doc::concat(vec![doc::hardline(), body_doc])), |
| 736 | doc::hardline(), |
| 737 | ]), |
| 738 | )); |
| 739 | if found_close { |
| 740 | grouped.push(doc::text("}")); |
| 741 | } |
| 742 | } else { |
| 743 | // Multi-statement body: always vertical, outside the group. |
| 744 | sig.push(doc::group(doc::concat(grouped))); |
| 745 | sig.push(doc::text(" {")); |
| 746 | // Strip leading newlines/whitespace from the first body |
| 747 | // token to avoid doubling the newline after `{`. Only |
| 748 | // strip the prefix before the first comment or content, |
| 749 | // not newlines between consecutive comment lines. |
| 750 | if let Some(first) = body_tokens.first_mut() { |
| 751 | let mut past_lead = false; |
| 752 | first.leading_trivia.retain(|t| { |
| 753 | if past_lead { return true; } |
| 754 | match t { |
| 755 | Trivia::Newline | Trivia::Whitespace(_) => false, |
| 756 | _ => { past_lead = true; true } |
| 757 | } |
| 758 | }); |
| 759 | } |
| 760 | let body_doc = tokens_to_formatted_doc(&body_tokens, spec); |
| 761 | sig.push(doc::nest(indent, doc::concat(vec![ |
| 762 | doc::hardline(), |
| 763 | body_doc, |
| 764 | ]))); |
| 765 | if found_close { |
| 766 | sig.push(doc::hardline()); |
| 767 | sig.push(doc::text("}")); |
| 768 | } |
| 769 | return doc::concat(sig); |
| 770 | } |
| 771 | } else if *pos < tokens.len() && matches!(tokens[*pos].kind, TokenKind::Punct(';')) { |
| 772 | grouped.push(doc::text(";")); |
| 773 | *pos += 1; |
| 774 | } |
| 775 | |
| 776 | sig.push(doc::group(doc::concat(grouped))); |
| 777 | doc::concat(sig) |
| 778 | } |
| 779 | |
| 780 | /// Whether a piece of trivia is a comment rather than whitespace. |
| 781 | fn is_comment_trivia(t: &Trivia) -> bool { |
| 782 | matches!(t, Trivia::LineComment(_) | Trivia::BlockComment(_)) |
| 783 | } |
| 784 | |
| 785 | /// Keep a closing brace's comments when the brace itself is dropped. |
| 786 | /// |
| 787 | /// A scan that gathers a block's contents stops at the closing brace |
| 788 | /// and does not keep it, because the brace is emitted separately. But |
| 789 | /// the lexer attaches any comment on the block's last line to that |
| 790 | /// brace as leading trivia, so the comment would go with it. Push a |
| 791 | /// token that renders nothing and carries the comments instead. The |
| 792 | /// trivia is cut after the last comment, so the newline that ran on to |
| 793 | /// the brace does not become a blank line. |
| 794 | /// Emit comments carried off a block's closing delimiter. |
| 795 | /// |
| 796 | /// `lead_break` puts a line break before the first one, which is wanted |
| 797 | /// when content precedes it and not when the block was otherwise empty. |
| 798 | fn emit_carried_comments(docs: &mut Vec<Doc>, tail: &[Token], lead_break: bool) { |
| 799 | let mut first = true; |
| 800 | for tok in tail { |
| 801 | for t in &tok.leading_trivia { |
| 802 | if let Trivia::LineComment(c) | Trivia::BlockComment(c) = t { |
| 803 | if !first || lead_break { |
| 804 | docs.push(doc::hardline()); |
| 805 | } |
| 806 | docs.push(doc::text(c.as_str())); |
| 807 | first = false; |
| 808 | } |
| 809 | } |
| 810 | } |
| 811 | } |
| 812 | |
| 813 | fn carry_close_comments(out: &mut Vec<Token>, close: &Token) { |
| 814 | if let Some(last) = close.leading_trivia.iter().rposition(is_comment_trivia) { |
| 815 | out.push(Token { |
| 816 | kind: TokenKind::Ident, |
| 817 | text: String::new(), |
| 818 | leading_trivia: close.leading_trivia[..=last].to_vec(), |
| 819 | span: close.span, |
| 820 | }); |
| 821 | } |
| 822 | } |
| 823 | |
| 824 | /// Flush pending newlines as hardlines, capped to max_blank_lines. |
| 825 | fn flush_newlines( |
| 826 | docs: &mut Vec<Doc>, |
| 827 | newline_count: &mut usize, |
| 828 | had_newline: &mut bool, |
| 829 | spec: &FormatSpec, |
| 830 | ) { |
| 831 | if *had_newline { |
| 832 | let blanks = (*newline_count).min(spec.max_blank_lines + 1); |
| 833 | for _ in 0..blanks { |
| 834 | docs.push(doc::hardline()); |
| 835 | } |
| 836 | *newline_count = 0; |
| 837 | *had_newline = false; |
| 838 | } |
| 839 | } |
| 840 | |
| 841 | /// Reorder items inside `use { ... }` blocks alphabetically. |
| 842 | /// Finds the `{` and `}` in the token stream, extracts the |
| 843 | /// comma-separated items between them, sorts by text, and |
| 844 | /// reassembles. |
| 845 | fn reorder_use_items(tokens: &mut Vec<Token>) { |
| 846 | // Find the `{` and `}` positions. |
| 847 | let open = tokens.iter().position(|t| matches!(t.kind, TokenKind::Punct('{'))); |
| 848 | let close = tokens.iter().rposition(|t| matches!(t.kind, TokenKind::Punct('}'))); |
| 849 | let (open_idx, close_idx) = match (open, close) { |
| 850 | (Some(o), Some(c)) if c > o + 1 => (o, c), |
| 851 | _ => return, // No braced block or empty. |
| 852 | }; |
| 853 | |
| 854 | // Extract inner tokens and split by comma. |
| 855 | let inner: Vec<Token> = tokens[open_idx + 1..close_idx].to_vec(); |
| 856 | let mut items = split_by_comma(&inner); |
| 857 | if items.len() < 2 { |
| 858 | return; // Nothing to sort. |
| 859 | } |
| 860 | |
| 861 | // Sort by the concatenated text of each item's tokens. |
| 862 | items.sort_by(|a, b| { |
| 863 | let a_text: String = a.iter().map(|t| t.text.as_str()).collect(); |
| 864 | let b_text: String = b.iter().map(|t| t.text.as_str()).collect(); |
| 865 | a_text.cmp(&b_text) |
| 866 | }); |
| 867 | |
| 868 | // Reassemble: preserve the trivia of the first original inner |
| 869 | // token on the first sorted item. |
| 870 | let first_trivia = if !inner.is_empty() { |
| 871 | inner[0].leading_trivia.clone() |
| 872 | } else { |
| 873 | Vec::new() |
| 874 | }; |
| 875 | |
| 876 | // Build new inner token list from sorted items with commas. |
| 877 | let mut new_inner: Vec<Token> = Vec::new(); |
| 878 | for (i, item) in items.iter().enumerate() { |
| 879 | for (j, tok) in item.iter().enumerate() { |
| 880 | let mut tok = tok.clone(); |
| 881 | // Give the first token of the first item the original |
| 882 | // leading trivia (newline + whitespace for indentation). |
| 883 | if i == 0 && j == 0 { |
| 884 | tok.leading_trivia = first_trivia.clone(); |
| 885 | } |
| 886 | new_inner.push(tok); |
| 887 | } |
| 888 | if i < items.len() - 1 { |
| 889 | // Add comma between items. |
| 890 | new_inner.push(Token { |
| 891 | kind: TokenKind::Punct(','), |
| 892 | text: ",".to_string(), |
| 893 | leading_trivia: Vec::new(), |
| 894 | span: crate::fmt::cst::Span::default(), |
| 895 | }); |
| 896 | } |
| 897 | } |
| 898 | // Trailing comma. |
| 899 | new_inner.push(Token { |
| 900 | kind: TokenKind::Punct(','), |
| 901 | text: ",".to_string(), |
| 902 | leading_trivia: Vec::new(), |
| 903 | span: crate::fmt::cst::Span::default(), |
| 904 | }); |
| 905 | |
| 906 | // Replace the inner tokens. |
| 907 | let mut result: Vec<Token> = Vec::new(); |
| 908 | result.extend_from_slice(&tokens[..open_idx + 1]); // up to and including `{` |
| 909 | result.extend(new_inner); |
| 910 | result.extend_from_slice(&tokens[close_idx..]); // `}` and beyond |
| 911 | *tokens = result; |
| 912 | } |
| 913 | |
| 914 | // ── Helpers ────────────────────────────────────────────────────── |
| 915 | |
| 916 | /// Recursively collect all leaf tokens from a CST node. |
| 917 | fn collect_tokens(node: &CstNode, out: &mut Vec<Token>) { |
| 918 | for child in &node.children { |
| 919 | match child { |
| 920 | CstChild::Token(tok) => out.push(tok.clone()), |
| 921 | CstChild::Node { node: inner, .. } => collect_tokens(inner, out), |
| 922 | } |
| 923 | } |
| 924 | } |
| 925 | |
| 926 | /// Emit just the token text (no trivia-based whitespace). |
| 927 | /// Comments from trivia are preserved; whitespace/newlines are |
| 928 | /// discarded because the Doc controls all spacing. |
| 929 | fn token_text(tok: &Token) -> Doc { |
| 930 | let mut docs = Vec::new(); |
| 931 | |
| 932 | // Preserve comments from trivia. |
| 933 | for t in &tok.leading_trivia { |
| 934 | match t { |
| 935 | Trivia::LineComment(c) => { |
| 936 | docs.push(doc::hardline()); |
| 937 | docs.push(doc::text(c.as_str())); |
| 938 | docs.push(doc::hardline()); |
| 939 | } |
| 940 | Trivia::BlockComment(c) => { |
| 941 | docs.push(doc::text(" ")); |
| 942 | docs.push(doc::text(c.as_str())); |
| 943 | } |
| 944 | // Whitespace and newlines are handled by the Doc, |
| 945 | // not by trivia. |
| 946 | Trivia::Whitespace(_) | Trivia::Newline => {} |
| 947 | } |
| 948 | } |
| 949 | |
| 950 | if !tok.text.is_empty() { |
| 951 | // Doc comments must stay on their own line — if placed |
| 952 | // inline they swallow everything to the end of the line |
| 953 | // when re-lexed. |
| 954 | if matches!(tok.kind, TokenKind::DocComment(_)) { |
| 955 | docs.push(doc::hardline()); |
| 956 | docs.push(doc::text(tok.text.as_str())); |
| 957 | docs.push(doc::hardline()); |
| 958 | } else { |
| 959 | docs.push(doc::text(tok.text.as_str())); |
| 960 | } |
| 961 | } |
| 962 | |
| 963 | doc::concat(docs) |
| 964 | } |
| 965 | |
| 966 | /// Split a token sequence at commas. |
| 967 | fn split_by_comma(tokens: &[Token]) -> Vec<Vec<Token>> { |
| 968 | let mut groups: Vec<Vec<Token>> = Vec::new(); |
| 969 | let mut current: Vec<Token> = Vec::new(); |
| 970 | for tok in tokens { |
| 971 | if tok.kind == TokenKind::Punct(',') { |
| 972 | if !current.is_empty() { |
| 973 | groups.push(current); |
| 974 | current = Vec::new(); |
| 975 | } |
| 976 | } else { |
| 977 | current.push(tok.clone()); |
| 978 | } |
| 979 | } |
| 980 | if !current.is_empty() { |
| 981 | groups.push(current); |
| 982 | } |
| 983 | groups |
| 984 | } |
| 985 | |
| 986 | /// Insert spaces between tokens with language-aware rules. |
| 987 | /// No space before `:`, `;`, `,`, `.`, `)`, `]`, `>`. |
| 988 | /// No space after `(`, `[`, `<`, `.`, `::`, `&`, `*`. |
| 989 | /// Binary operators get a `line()` before or after (soft break |
| 990 | /// point) so long expressions can wrap at operator boundaries. |
| 991 | fn smart_spaced_tokens(tokens: &[Token]) -> Vec<Doc> { |
| 992 | smart_spaced_tokens_inner(tokens, true) |
| 993 | } |
| 994 | |
| 995 | fn smart_spaced_tokens_inner(tokens: &[Token], chain_breaks: bool) -> Vec<Doc> { |
| 996 | let mut docs = Vec::new(); |
| 997 | for (i, tok) in tokens.iter().enumerate() { |
| 998 | if i > 0 { |
| 999 | let prev = &tokens[i - 1]; |
| 1000 | // Method chain break point: after `)` before `.`. |
| 1001 | if chain_breaks |
| 1002 | && matches!(prev.kind, TokenKind::Punct(')')) |
| 1003 | && matches!(tok.kind, TokenKind::Punct('.')) |
| 1004 | { |
| 1005 | docs.push(doc::line()); |
| 1006 | docs.push(token_text(tok)); |
| 1007 | continue; |
| 1008 | } |
| 1009 | let need_space = needs_separator_space(prev, tok) |
| 1010 | || (!suppress_space_after(prev) && !suppress_space_before(tok)); |
| 1011 | if need_space { |
| 1012 | docs.push(doc::text(" ")); |
| 1013 | } |
| 1014 | } |
| 1015 | docs.push(token_text(tok)); |
| 1016 | } |
| 1017 | docs |
| 1018 | } |
| 1019 | |
| 1020 | /// Precedence level of a binary operator. Higher means tighter |
| 1021 | /// binding. Returns 0 for non-operators. |
| 1022 | /// |
| 1023 | /// Oxedyne indentation rule: when a line breaks at an operator, |
| 1024 | /// each lower-precedence operator indents one level deeper. So |
| 1025 | /// `&&` at level 4 sits at the base, `||` at level 3 indents one |
| 1026 | /// level relative to `&&`. |
| 1027 | fn binop_precedence(tok: &Token) -> u8 { |
| 1028 | match &tok.kind { |
| 1029 | TokenKind::Operator(op) => match op.as_str() { |
| 1030 | "||" => 3, |
| 1031 | "&&" => 4, |
| 1032 | "==" | "!=" | "<=" | ">=" => 5, |
| 1033 | "|" => 6, |
| 1034 | "^" => 7, |
| 1035 | "<<" | ">>" => 9, |
| 1036 | "+=" | "-=" | "*=" | "/=" |
| 1037 | | "%=" | "&=" | "|=" | "^=" |
| 1038 | | "<<=" | ">>=" => 1, |
| 1039 | _ => 0, |
| 1040 | }, |
| 1041 | TokenKind::Punct(c) => match c { |
| 1042 | '|' => 6, |
| 1043 | '^' => 7, |
| 1044 | '&' => 8, |
| 1045 | '+' | '-' => 10, |
| 1046 | '*' | '/' | '%' => 11, |
| 1047 | _ => 0, |
| 1048 | }, |
| 1049 | _ => 0, |
| 1050 | } |
| 1051 | } |
| 1052 | |
| 1053 | /// Check whether a token is a binary operator. |
| 1054 | fn is_binary_op(tok: &Token) -> bool { |
| 1055 | binop_precedence(tok) > 0 |
| 1056 | } |
| 1057 | |
| 1058 | /// Check whether a token is an unambiguously binary operator. |
| 1059 | /// |
| 1060 | /// Only matches multi-character Operator tokens (`&&`, `||`, `==`, |
| 1061 | /// `+=`, etc.), not single-character Punct that could be unary |
| 1062 | /// (`&`, `*`, `+`, `-`). |
| 1063 | fn is_unambiguous_binary_op(tok: &Token) -> bool { |
| 1064 | matches!(&tok.kind, TokenKind::Operator(_)) && binop_precedence(tok) > 0 |
| 1065 | } |
| 1066 | |
| 1067 | /// Format a token sequence containing binary operators with |
| 1068 | /// precedence-based indentation. |
| 1069 | /// |
| 1070 | /// Oxedyne rule: each lower-precedence operator indents one level |
| 1071 | /// deeper than the previous. For example: |
| 1072 | /// |
| 1073 | /// ```text |
| 1074 | /// let valid = name.len() > 0 |
| 1075 | /// && age >= 18 |
| 1076 | /// && country == "NZ" |
| 1077 | /// || special_override; |
| 1078 | /// ``` |
| 1079 | /// |
| 1080 | /// `&&` (precedence 4) is at indent+1, `||` (precedence 3) is at |
| 1081 | /// indent+2 because it's a lower-precedence level. |
| 1082 | fn format_binop_expr(tokens: &[Token], spec: &FormatSpec) -> Doc { |
| 1083 | let indent = spec.indent_width; |
| 1084 | |
| 1085 | // Find the distinct precedence levels used (sorted descending). |
| 1086 | let mut prec_levels: Vec<u8> = tokens.iter() |
| 1087 | .map(|t| binop_precedence(t)) |
| 1088 | .filter(|p| *p > 0) |
| 1089 | .collect(); |
| 1090 | prec_levels.sort(); |
| 1091 | prec_levels.dedup(); |
| 1092 | prec_levels.reverse(); // Highest first. |
| 1093 | |
| 1094 | if prec_levels.is_empty() { |
| 1095 | // No operators — emit normally. |
| 1096 | return doc::concat(smart_spaced_tokens(tokens)); |
| 1097 | } |
| 1098 | |
| 1099 | // Map each precedence level to an indent depth. |
| 1100 | // Highest precedence = 1 indent, next = 2 indents, etc. |
| 1101 | let prec_to_depth = |p: u8| -> u16 { |
| 1102 | let idx = prec_levels.iter().position(|&x| x == p).unwrap_or(0); |
| 1103 | (idx as u16 + 1) * indent |
| 1104 | }; |
| 1105 | |
| 1106 | // Build the doc: break before each operator (BinopBreak::Before), |
| 1107 | // nesting to the operator's depth. |
| 1108 | let mut docs = Vec::new(); |
| 1109 | let mut i = 0; |
| 1110 | while i < tokens.len() { |
| 1111 | let tok = &tokens[i]; |
| 1112 | let prec = binop_precedence(tok); |
| 1113 | if prec > 0 && i > 0 { |
| 1114 | // Binary operator — insert break point. |
| 1115 | let depth = prec_to_depth(prec); |
| 1116 | match spec.binop_break { |
| 1117 | BinopBreak::Before => { |
| 1118 | // Break, then indent, then operator, then space. |
| 1119 | docs.push(doc::nest(depth, doc::concat(vec![ |
| 1120 | doc::line(), |
| 1121 | doc::text(tok.text.as_str()), |
| 1122 | ]))); |
| 1123 | } |
| 1124 | BinopBreak::After => { |
| 1125 | // Space, then operator, then break + indent. |
| 1126 | docs.push(doc::text(" ")); |
| 1127 | docs.push(doc::text(tok.text.as_str())); |
| 1128 | docs.push(doc::nest(depth, doc::line())); |
| 1129 | } |
| 1130 | } |
| 1131 | } else { |
| 1132 | if i > 0 { |
| 1133 | let prev = &tokens[i - 1]; |
| 1134 | let need_space = !suppress_space_after(prev) && !suppress_space_before_ctx(tok, Some(prev)); |
| 1135 | if need_space { |
| 1136 | docs.push(doc::text(" ")); |
| 1137 | } |
| 1138 | } |
| 1139 | docs.push(token_text(tok)); |
| 1140 | } |
| 1141 | i += 1; |
| 1142 | } |
| 1143 | |
| 1144 | doc::group(doc::concat(docs)) |
| 1145 | } |
| 1146 | |
| 1147 | /// Adjacent tokens that would re-tokenize differently if the |
| 1148 | /// separating space were removed. Prevents ambiguous sequences |
| 1149 | /// like `:::` (from `: ::` or `:: :`). |
| 1150 | fn needs_separator_space(prev: &Token, next: &Token) -> bool { |
| 1151 | (matches!(prev.kind, TokenKind::Punct(':')) |
| 1152 | && matches!(next.kind, TokenKind::Operator(ref op) if op == "::")) |
| 1153 | || (matches!(prev.kind, TokenKind::Operator(ref op) if op == "::") |
| 1154 | && matches!(next.kind, TokenKind::Punct(':'))) |
| 1155 | } |
| 1156 | |
| 1157 | /// Tokens after which no trailing space should be added. |
| 1158 | fn suppress_space_after(tok: &Token) -> bool { |
| 1159 | match &tok.kind { |
| 1160 | TokenKind::Punct('(') | TokenKind::Punct('[') | TokenKind::Punct('<') |
| 1161 | | TokenKind::Punct('.') | TokenKind::Punct('&') | TokenKind::Punct('*') |
| 1162 | | TokenKind::Punct('!') | TokenKind::Punct('#') |
| 1163 | => true, |
| 1164 | TokenKind::Operator(op) |
| 1165 | if op == "::" || op == ".." || op == "..=" |
| 1166 | => true, |
| 1167 | _ => false, |
| 1168 | } |
| 1169 | } |
| 1170 | |
| 1171 | /// Context-free version of `suppress_space_before_ctx`. |
| 1172 | fn suppress_space_before(tok: &Token) -> bool { |
| 1173 | suppress_space_before_ctx(tok, None) |
| 1174 | } |
| 1175 | |
| 1176 | /// Tokens before which no leading space should be added. |
| 1177 | /// When `prev` is provided, context-sensitive rules apply. |
| 1178 | fn suppress_space_before_ctx(tok: &Token, prev: Option<&Token>) -> bool { |
| 1179 | match &tok.kind { |
| 1180 | TokenKind::Punct(':') | TokenKind::Punct(';') | TokenKind::Punct(',') |
| 1181 | | TokenKind::Punct('.') | TokenKind::Punct(')') | TokenKind::Punct(']') |
| 1182 | | TokenKind::Punct('>') | TokenKind::Punct('!') |
| 1183 | | TokenKind::Punct('(') | TokenKind::Punct('[') |
| 1184 | | TokenKind::Punct('<') |
| 1185 | => true, |
| 1186 | TokenKind::Operator(op) if op == "::" => { |
| 1187 | // Keep space after `:` bound separator: `E: ::std`. |
| 1188 | !matches!(prev.map(|p| &p.kind), Some(TokenKind::Punct(':'))) |
| 1189 | } |
| 1190 | TokenKind::Operator(op) |
| 1191 | if op == ".." || op == "..=" |
| 1192 | || op == ">>" |
| 1193 | => true, |
| 1194 | _ => false, |
| 1195 | } |
| 1196 | } |
| 1197 | |
| 1198 | /// Check whether a body token sequence is short enough to potentially |
| 1199 | /// go on one line. A body is short if it has no semicolons (except |
| 1200 | /// possibly at the end) and has few tokens. |
| 1201 | /// Format a parameter list with optional type-column alignment. |
| 1202 | /// |
| 1203 | /// When `align` is true, the type column is aligned across all |
| 1204 | /// parameters by padding shorter names with spaces. The alignment |
| 1205 | /// only appears in the broken (vertical) layout; the flat layout |
| 1206 | /// uses normal spacing. |
| 1207 | /// |
| 1208 | /// Flat: `(len: usize, charset: &str)` |
| 1209 | /// Broken: `(\n len: usize,\n charset: &str,\n)` |
| 1210 | fn format_param_list( |
| 1211 | params: &[Vec<Token>], |
| 1212 | align: bool, |
| 1213 | _spec: &FormatSpec, |
| 1214 | ) -> Vec<Doc> { |
| 1215 | if !align { |
| 1216 | return params.iter() |
| 1217 | .map(|toks| doc::concat(smart_spaced_tokens(toks))) |
| 1218 | .collect(); |
| 1219 | } |
| 1220 | |
| 1221 | // Split each param at `:` into name-tokens and type-tokens. |
| 1222 | let mut parsed: Vec<(Vec<Token>, Vec<Token>)> = Vec::new(); |
| 1223 | for param_toks in params { |
| 1224 | let colon_pos = param_toks.iter().position(|t| matches!(t.kind, TokenKind::Punct(':'))); |
| 1225 | match colon_pos { |
| 1226 | Some(cp) => { |
| 1227 | let name_toks = param_toks[..cp].to_vec(); |
| 1228 | let type_toks = param_toks[cp + 1..].to_vec(); |
| 1229 | parsed.push((name_toks, type_toks)); |
| 1230 | } |
| 1231 | None => { |
| 1232 | // No colon (e.g. `self`) — no alignment for this param. |
| 1233 | parsed.push((param_toks.clone(), Vec::new())); |
| 1234 | } |
| 1235 | } |
| 1236 | } |
| 1237 | |
| 1238 | // Widths must be what the renderer will actually emit; see |
| 1239 | // rendered_width. |
| 1240 | let name_widths: Vec<usize> = parsed.iter() |
| 1241 | .map(|(name_toks, _)| rendered_width(name_toks)) |
| 1242 | .collect(); |
| 1243 | let max_name = name_widths.iter().copied().max().unwrap_or(0); |
| 1244 | |
| 1245 | // Build docs: flat uses normal spacing, broken uses padded names. |
| 1246 | parsed.iter().zip(name_widths.iter()).map(|((name_toks, type_toks), &name_w)| { |
| 1247 | if type_toks.is_empty() { |
| 1248 | // No colon (self, etc.) — just emit as-is. |
| 1249 | return doc::concat(smart_spaced_tokens(name_toks)); |
| 1250 | } |
| 1251 | |
| 1252 | let name_doc = doc::concat(smart_spaced_tokens(name_toks)); |
| 1253 | let type_doc = doc::concat(smart_spaced_tokens(type_toks)); |
| 1254 | |
| 1255 | // Flat: `name: type` |
| 1256 | let flat = doc::concat(vec![ |
| 1257 | name_doc.clone(), |
| 1258 | doc::text(": "), |
| 1259 | type_doc.clone(), |
| 1260 | ]); |
| 1261 | |
| 1262 | // Broken: `name: type` (padded to align type column). |
| 1263 | let pad_len = max_name - name_w; |
| 1264 | let pad = " ".repeat(pad_len + 1); // +1 for the space after `:` |
| 1265 | let broken = doc::concat(vec![ |
| 1266 | name_doc, |
| 1267 | doc::text(":"), |
| 1268 | doc::text(pad.as_str()), |
| 1269 | type_doc, |
| 1270 | ]); |
| 1271 | |
| 1272 | doc::if_break(flat, broken) |
| 1273 | }).collect() |
| 1274 | } |
| 1275 | |
| 1276 | /// Check whether the `{` at position `brace_pos` is preceded by |
| 1277 | /// a `struct` or `enum` keyword (possibly with intervening ident, |
| 1278 | /// generics, where clause, etc.). |
| 1279 | /// Format match arms with `=>` column alignment. |
| 1280 | /// |
| 1281 | /// Splits the arm tokens at commas (respecting brace depth), then |
| 1282 | /// aligns the `=>` across all simple (single-line) arms. |
| 1283 | fn format_match_arms(tokens: &[Token], spec: &FormatSpec) -> Doc { |
| 1284 | // Split into individual arms at top-level commas. |
| 1285 | let mut arms: Vec<Vec<Token>> = split_by_comma_depth(tokens) |
| 1286 | .into_iter().filter(|a| !a.is_empty()).collect(); |
| 1287 | |
| 1288 | // A trailing group of carried comments is the block's last comment, |
| 1289 | // not an arm. Left in, it would be given an arm's comma, which is a |
| 1290 | // bare `,` on a line of its own and does not compile. |
| 1291 | let mut tail: Vec<Token> = Vec::new(); |
| 1292 | while arms.last().map(|a| is_comment_only(a)).unwrap_or(false) { |
| 1293 | if let Some(a) = arms.pop() { |
| 1294 | tail.splice(0..0, a); |
| 1295 | } |
| 1296 | } |
| 1297 | let arms = arms; |
| 1298 | |
| 1299 | if arms.is_empty() { |
| 1300 | let mut docs = Vec::new(); |
| 1301 | emit_carried_comments(&mut docs, &tail, false); |
| 1302 | return doc::concat(docs); |
| 1303 | } |
| 1304 | |
| 1305 | // Parse each arm: split prefix, find `=>`, split pattern/body. |
| 1306 | struct ArmParts { |
| 1307 | prefix: Vec<Token>, |
| 1308 | pattern: Vec<Token>, |
| 1309 | body: Vec<Token>, |
| 1310 | has_arrow: bool, |
| 1311 | } |
| 1312 | let mut parsed: Vec<ArmParts> = Vec::new(); |
| 1313 | for arm_toks in &arms { |
| 1314 | let (prefix, content) = split_field_prefix(arm_toks); |
| 1315 | // Find `=>` at top-level depth. |
| 1316 | let mut arrow_pos = None; |
| 1317 | let mut depth = 0usize; |
| 1318 | for (j, t) in content.iter().enumerate() { |
| 1319 | match t.kind { |
| 1320 | TokenKind::Punct('(') | TokenKind::Punct('{') | TokenKind::Punct('[') => depth += 1, |
| 1321 | TokenKind::Punct(')') | TokenKind::Punct('}') | TokenKind::Punct(']') => { |
| 1322 | depth = depth.saturating_sub(1); |
| 1323 | } |
| 1324 | _ => {} |
| 1325 | } |
| 1326 | if depth == 0 && matches!(t.kind, TokenKind::Operator(ref op) if op == "=>") { |
| 1327 | arrow_pos = Some(j); |
| 1328 | break; |
| 1329 | } |
| 1330 | } |
| 1331 | match arrow_pos { |
| 1332 | Some(ap) => { |
| 1333 | parsed.push(ArmParts { |
| 1334 | prefix, |
| 1335 | pattern: content[..ap].to_vec(), |
| 1336 | body: content[ap + 1..].to_vec(), |
| 1337 | has_arrow: true, |
| 1338 | }); |
| 1339 | } |
| 1340 | None => { |
| 1341 | parsed.push(ArmParts { |
| 1342 | prefix, |
| 1343 | pattern: content, |
| 1344 | body: Vec::new(), |
| 1345 | has_arrow: false, |
| 1346 | }); |
| 1347 | } |
| 1348 | } |
| 1349 | } |
| 1350 | |
| 1351 | // Pattern widths for alignment (prefix excluded). Measured as the |
| 1352 | // renderer will emit them: `Some(v)` is seven columns, not ten. |
| 1353 | let pat_widths: Vec<usize> = parsed.iter() |
| 1354 | .map(|a| rendered_width(&a.pattern)) |
| 1355 | .collect(); |
| 1356 | let max_pat = pat_widths.iter().copied().max().unwrap_or(0); |
| 1357 | let min_pat = pat_widths.iter().copied().min().unwrap_or(0); |
| 1358 | let align = (max_pat - min_pat) <= spec.field_align_threshold |
| 1359 | && parsed.iter().all(|a| a.has_arrow); |
| 1360 | |
| 1361 | // Build docs. |
| 1362 | let mut docs = Vec::new(); |
| 1363 | for (i, (arm, &pat_w)) in parsed.iter().zip(pat_widths.iter()).enumerate() { |
| 1364 | if i > 0 { |
| 1365 | docs.push(doc::text(",")); |
| 1366 | docs.push(doc::hardline()); |
| 1367 | } |
| 1368 | |
| 1369 | emit_prefix(&mut docs, &arm.prefix); |
| 1370 | let pat_doc = doc::concat(smart_spaced_tokens(&arm.pattern)); |
| 1371 | |
| 1372 | if !arm.has_arrow { |
| 1373 | docs.push(pat_doc); |
| 1374 | continue; |
| 1375 | } |
| 1376 | |
| 1377 | let body_doc = doc::concat(smart_spaced_tokens(&arm.body)); |
| 1378 | |
| 1379 | if align { |
| 1380 | let pad = " ".repeat(max_pat - pat_w); |
| 1381 | docs.push(doc::concat(vec![ |
| 1382 | pat_doc, |
| 1383 | doc::text(pad.as_str()), |
| 1384 | doc::text(" => "), |
| 1385 | body_doc, |
| 1386 | ])); |
| 1387 | } else { |
| 1388 | docs.push(doc::concat(vec![ |
| 1389 | pat_doc, |
| 1390 | doc::text(" => "), |
| 1391 | body_doc, |
| 1392 | ])); |
| 1393 | } |
| 1394 | } |
| 1395 | // Trailing comma on last arm. |
| 1396 | if spec.match_trailing_comma && !parsed.is_empty() { |
| 1397 | docs.push(doc::text(",")); |
| 1398 | } |
| 1399 | emit_carried_comments(&mut docs, &tail, true); |
| 1400 | doc::concat(docs) |
| 1401 | } |
| 1402 | |
| 1403 | /// Split a token sequence at commas, respecting bracket depth. |
| 1404 | /// Unlike `split_by_comma`, this tracks `{`, `(`, `[` depth so |
| 1405 | /// commas inside braced blocks don't split arms. |
| 1406 | /// The column width a token run occupies once rendered. |
| 1407 | /// |
| 1408 | /// Counting `sum(len) + n - 1` assumes a space between every pair of |
| 1409 | /// tokens, which the renderer does not emit: `Some(v)` is four tokens |
| 1410 | /// but seven columns, not ten. Alignment computed from the wrong width |
| 1411 | /// puts the separators in different columns, which reads worse than no |
| 1412 | /// alignment at all. |
| 1413 | fn rendered_width(tokens: &[Token]) -> usize { |
| 1414 | let mut w = 0usize; |
| 1415 | for (i, tok) in tokens.iter().enumerate() { |
| 1416 | if i > 0 { |
| 1417 | let prev = &tokens[i - 1]; |
| 1418 | if needs_separator_space(prev, tok) |
| 1419 | || (!suppress_space_before(tok) && !suppress_space_after(prev)) |
| 1420 | { |
| 1421 | w += 1; |
| 1422 | } |
| 1423 | } |
| 1424 | w += tok.text.chars().count(); |
| 1425 | } |
| 1426 | w |
| 1427 | } |
| 1428 | |
| 1429 | /// Whether a field group carries no code, only carried comments. |
| 1430 | /// |
| 1431 | /// `carry_close_comments` puts a block's last comment on a token that |
| 1432 | /// renders nothing. Split by comma, that becomes a group of its own, |
| 1433 | /// and treating it as a field earns it a comma it must not have. |
| 1434 | fn is_comment_only(field: &[Token]) -> bool { |
| 1435 | field.iter().all(|t| t.text.is_empty()) |
| 1436 | } |
| 1437 | |
| 1438 | /// Give each field back the comment written after its comma. |
| 1439 | /// |
| 1440 | /// A comment at the end of a field's line comes after the comma, so |
| 1441 | /// the lexer attaches it to the *next* field as leading trivia. Left |
| 1442 | /// there, the formatter prints it above that next field, where it |
| 1443 | /// silently comes to describe the wrong one. Comments before the first |
| 1444 | /// newline belong to the field just closed; everything from the |
| 1445 | /// newline on is genuinely the next field's. |
| 1446 | fn lift_trailing_comments(fields: &mut [Vec<Token>]) -> Vec<Vec<String>> { |
| 1447 | let mut trailing: Vec<Vec<String>> = vec![Vec::new(); fields.len()]; |
| 1448 | for i in 1..fields.len() { |
| 1449 | let mut lifted: Vec<String> = Vec::new(); |
| 1450 | if let Some(first) = fields[i].first_mut() { |
| 1451 | let mut keep: Vec<Trivia> = Vec::new(); |
| 1452 | let mut seen_newline = false; |
| 1453 | for t in std::mem::take(&mut first.leading_trivia) { |
| 1454 | if seen_newline { |
| 1455 | keep.push(t); |
| 1456 | continue; |
| 1457 | } |
| 1458 | match &t { |
| 1459 | Trivia::Newline => { |
| 1460 | seen_newline = true; |
| 1461 | keep.push(t); |
| 1462 | } |
| 1463 | Trivia::LineComment(c) | Trivia::BlockComment(c) => { |
| 1464 | lifted.push(c.clone()); |
| 1465 | } |
| 1466 | Trivia::Whitespace(_) => keep.push(t), |
| 1467 | } |
| 1468 | } |
| 1469 | first.leading_trivia = keep; |
| 1470 | } |
| 1471 | trailing[i - 1] = lifted; |
| 1472 | } |
| 1473 | trailing |
| 1474 | } |
| 1475 | |
| 1476 | fn split_by_comma_depth(tokens: &[Token]) -> Vec<Vec<Token>> { |
| 1477 | let mut groups: Vec<Vec<Token>> = Vec::new(); |
| 1478 | let mut current: Vec<Token> = Vec::new(); |
| 1479 | let mut depth = 0usize; |
| 1480 | for tok in tokens { |
| 1481 | match tok.kind { |
| 1482 | TokenKind::Punct('(') | TokenKind::Punct('{') | TokenKind::Punct('[') => { |
| 1483 | depth += 1; |
| 1484 | current.push(tok.clone()); |
| 1485 | } |
| 1486 | TokenKind::Punct(')') | TokenKind::Punct('}') | TokenKind::Punct(']') => { |
| 1487 | depth = depth.saturating_sub(1); |
| 1488 | current.push(tok.clone()); |
| 1489 | } |
| 1490 | TokenKind::Punct(',') if depth == 0 => { |
| 1491 | if !current.is_empty() { |
| 1492 | groups.push(current); |
| 1493 | current = Vec::new(); |
| 1494 | } |
| 1495 | } |
| 1496 | _ => { |
| 1497 | current.push(tok.clone()); |
| 1498 | } |
| 1499 | } |
| 1500 | } |
| 1501 | if !current.is_empty() { |
| 1502 | groups.push(current); |
| 1503 | } |
| 1504 | groups |
| 1505 | } |
| 1506 | |
| 1507 | fn is_preceded_by_struct_or_enum(tokens: &[Token], brace_pos: usize) -> bool { |
| 1508 | // Scan backwards for the nearest keyword. |
| 1509 | for j in (0..brace_pos).rev() { |
| 1510 | match &tokens[j].kind { |
| 1511 | TokenKind::Keyword(k) if k == "struct" || k == "enum" => return true, |
| 1512 | TokenKind::Keyword(_) => return false, |
| 1513 | _ => continue, |
| 1514 | } |
| 1515 | } |
| 1516 | false |
| 1517 | } |
| 1518 | |
| 1519 | /// Format struct/enum fields with type-column alignment. |
| 1520 | /// |
| 1521 | /// Splits the token sequence at commas, parses each field as |
| 1522 | /// `name: Type`, and aligns the type column across all fields. |
| 1523 | /// Format struct/enum fields with alignment. |
| 1524 | /// |
| 1525 | /// Auto-detects the separator to align: |
| 1526 | /// - `:` for struct fields (`name: Type`). |
| 1527 | /// - `=` for enum discriminants (`Name = value`). |
| 1528 | /// Falls back to no alignment if no separator is found. |
| 1529 | /// Split a field's tokens into prefix (doc comments, attributes) |
| 1530 | /// and content. Prefix items are emitted on separate lines. |
| 1531 | fn split_field_prefix(tokens: &[Token]) -> (Vec<Token>, Vec<Token>) { |
| 1532 | let mut prefix = Vec::new(); |
| 1533 | let mut content = Vec::new(); |
| 1534 | let mut past = false; |
| 1535 | for tok in tokens { |
| 1536 | if !past && matches!(tok.kind, TokenKind::DocComment(_) | TokenKind::Attribute) { |
| 1537 | prefix.push(tok.clone()); |
| 1538 | } else { |
| 1539 | past = true; |
| 1540 | content.push(tok.clone()); |
| 1541 | } |
| 1542 | } |
| 1543 | (prefix, content) |
| 1544 | } |
| 1545 | |
| 1546 | /// Emit doc comments and attributes on their own lines. |
| 1547 | fn emit_prefix(docs: &mut Vec<Doc>, prefix: &[Token]) { |
| 1548 | for tok in prefix { |
| 1549 | // A plain comment written above a doc comment or an attribute |
| 1550 | // is attached to that token as leading trivia. Emitting only |
| 1551 | // the token's own text drops it -- and a `// ---- section ----` |
| 1552 | // banner above a documented field is exactly that shape. |
| 1553 | for t in &tok.leading_trivia { |
| 1554 | if let Trivia::LineComment(c) | Trivia::BlockComment(c) = t { |
| 1555 | docs.push(doc::text(c.as_str())); |
| 1556 | docs.push(doc::hardline()); |
| 1557 | } |
| 1558 | } |
| 1559 | docs.push(doc::text(tok.text.as_str())); |
| 1560 | docs.push(doc::hardline()); |
| 1561 | } |
| 1562 | } |
| 1563 | |
| 1564 | /// Find the first occurrence of `ch` as a `Punct` at bracket depth 0. |
| 1565 | fn find_at_depth0(tokens: &[Token], ch: char) -> Option<usize> { |
| 1566 | let mut depth = 0usize; |
| 1567 | for (i, tok) in tokens.iter().enumerate() { |
| 1568 | match tok.kind { |
| 1569 | TokenKind::Punct('(') | TokenKind::Punct('{') | TokenKind::Punct('[') => depth += 1, |
| 1570 | TokenKind::Punct(')') | TokenKind::Punct('}') | TokenKind::Punct(']') => { |
| 1571 | depth = depth.saturating_sub(1); |
| 1572 | } |
| 1573 | TokenKind::Punct(c) if c == ch && depth == 0 => return Some(i), |
| 1574 | _ => {} |
| 1575 | } |
| 1576 | } |
| 1577 | None |
| 1578 | } |
| 1579 | |
| 1580 | /// Find the first `{` at bracket depth 0. |
| 1581 | /// |
| 1582 | /// Unlike `find_at_depth0`, this checks the depth *before* the |
| 1583 | /// open-brace increments it, so it actually finds a `{` that is |
| 1584 | /// not nested inside other brackets. |
| 1585 | fn find_brace_at_depth0(tokens: &[Token]) -> Option<usize> { |
| 1586 | let mut depth = 0usize; |
| 1587 | for (i, tok) in tokens.iter().enumerate() { |
| 1588 | match tok.kind { |
| 1589 | TokenKind::Punct('{') => { |
| 1590 | if depth == 0 { return Some(i); } |
| 1591 | depth += 1; |
| 1592 | } |
| 1593 | TokenKind::Punct('(') | TokenKind::Punct('[') => depth += 1, |
| 1594 | TokenKind::Punct('}') | TokenKind::Punct(')') | TokenKind::Punct(']') => { |
| 1595 | depth = depth.saturating_sub(1); |
| 1596 | } |
| 1597 | _ => {} |
| 1598 | } |
| 1599 | } |
| 1600 | None |
| 1601 | } |
| 1602 | |
| 1603 | /// Format field or variant content, handling struct-like bodies. |
| 1604 | /// |
| 1605 | /// When the content contains a `{` at bracket depth 0 (a struct-like |
| 1606 | /// enum variant body), the part before the brace is emitted as the |
| 1607 | /// variant header and the braced body is recursively formatted with |
| 1608 | /// field alignment. Without a brace, falls back to smart spacing. |
| 1609 | fn format_field_content(tokens: &[Token], spec: &FormatSpec) -> Doc { |
| 1610 | let brace_pos = find_brace_at_depth0(tokens); |
| 1611 | match brace_pos { |
| 1612 | Some(bp) if bp > 0 => { |
| 1613 | let header = &tokens[..bp]; |
| 1614 | let header_doc = doc::concat(smart_spaced_tokens(header)); |
| 1615 | |
| 1616 | // Find the matching `}`. |
| 1617 | let mut depth = 0usize; |
| 1618 | let mut close = tokens.len(); |
| 1619 | for j in bp..tokens.len() { |
| 1620 | match tokens[j].kind { |
| 1621 | TokenKind::Punct('{') => depth += 1, |
| 1622 | TokenKind::Punct('}') => { |
| 1623 | depth -= 1; |
| 1624 | if depth == 0 { |
| 1625 | close = j; |
| 1626 | break; |
| 1627 | } |
| 1628 | } |
| 1629 | _ => {} |
| 1630 | } |
| 1631 | } |
| 1632 | |
| 1633 | let inner = &tokens[bp + 1..close]; |
| 1634 | let inner_doc = format_aligned_fields(inner, spec); |
| 1635 | |
| 1636 | let mut docs = vec![ |
| 1637 | header_doc, |
| 1638 | doc::text(" {"), |
| 1639 | doc::nest(spec.indent_width, doc::concat(vec![ |
| 1640 | doc::hardline(), |
| 1641 | inner_doc, |
| 1642 | ])), |
| 1643 | doc::hardline(), |
| 1644 | doc::text("}"), |
| 1645 | ]; |
| 1646 | |
| 1647 | // Tokens after `}` (rare but handle gracefully). |
| 1648 | if close + 1 < tokens.len() { |
| 1649 | docs.push(doc::concat(smart_spaced_tokens(&tokens[close + 1..]))); |
| 1650 | } |
| 1651 | |
| 1652 | doc::concat(docs) |
| 1653 | } |
| 1654 | _ => doc::concat(smart_spaced_tokens(tokens)), |
| 1655 | } |
| 1656 | } |
| 1657 | |
| 1658 | fn format_aligned_fields(tokens: &[Token], spec: &FormatSpec) -> Doc { |
| 1659 | let mut fields: Vec<Vec<Token>> = split_by_comma_depth(tokens) |
| 1660 | .into_iter().filter(|f| !f.is_empty()).collect(); |
| 1661 | |
| 1662 | // A trailing group of carried comments is the block's last comment, |
| 1663 | // not a field, so it is emitted after the fields and earns no comma. |
| 1664 | let mut tail: Vec<Token> = Vec::new(); |
| 1665 | while fields.last().map(|f| is_comment_only(f)).unwrap_or(false) { |
| 1666 | if let Some(f) = fields.pop() { |
| 1667 | tail.splice(0..0, f); |
| 1668 | } |
| 1669 | } |
| 1670 | |
| 1671 | let trailing = lift_trailing_comments(&mut fields); |
| 1672 | |
| 1673 | // Split each field into prefix (doc comments, attributes) and content. |
| 1674 | let split: Vec<(Vec<Token>, Vec<Token>)> = fields.iter() |
| 1675 | .map(|f| split_field_prefix(f)) |
| 1676 | .collect(); |
| 1677 | |
| 1678 | // Detect separator at depth 0 only (ignore `:` inside `{...}`). |
| 1679 | let has_colon = split.iter().any(|(_, c)| find_at_depth0(c, ':').is_some()); |
| 1680 | let has_eq = split.iter().any(|(_, c)| find_at_depth0(c, '=').is_some()); |
| 1681 | let sep_char = if has_colon { |
| 1682 | Some(':') |
| 1683 | } else if has_eq { |
| 1684 | Some('=') |
| 1685 | } else { |
| 1686 | None |
| 1687 | }; |
| 1688 | |
| 1689 | // Split each field's content at the depth-0 separator, where there |
| 1690 | // is one. Without a separator there is nothing to align on, and the |
| 1691 | // field is emitted whole. |
| 1692 | let parsed: Vec<(Vec<Token>, Vec<Token>, Vec<Token>)> = split.iter() |
| 1693 | .map(|(prefix, content)| match sep_char.and_then(|c| find_at_depth0(content, c)) { |
| 1694 | Some(sp) => ( |
| 1695 | prefix.clone(), |
| 1696 | content[..sp].to_vec(), |
| 1697 | content[sp + 1..].to_vec(), |
| 1698 | ), |
| 1699 | None => (prefix.clone(), content.clone(), Vec::new()), |
| 1700 | }) |
| 1701 | .collect(); |
| 1702 | |
| 1703 | // Alignment is only worth doing when the names are of comparable |
| 1704 | // length; one long name would otherwise push every other value out. |
| 1705 | let name_widths: Vec<usize> = parsed.iter() |
| 1706 | .map(|(_, name_toks, _)| rendered_width(name_toks)) |
| 1707 | .collect(); |
| 1708 | let max_name = name_widths.iter().copied().max().unwrap_or(0); |
| 1709 | let min_name = name_widths.iter().copied().min().unwrap_or(0); |
| 1710 | let align = sep_char.is_some() |
| 1711 | && parsed.len() > 1 |
| 1712 | && max_name.saturating_sub(min_name) <= spec.field_align_threshold; |
| 1713 | |
| 1714 | let mut docs = Vec::new(); |
| 1715 | for (i, ((prefix, name_toks, val_toks), &name_w)) in |
| 1716 | parsed.iter().zip(name_widths.iter()).enumerate() |
| 1717 | { |
| 1718 | emit_prefix(&mut docs, prefix); |
| 1719 | if val_toks.is_empty() { |
| 1720 | docs.push(format_field_content(name_toks, spec)); |
| 1721 | } else { |
| 1722 | let name_doc = doc::concat(smart_spaced_tokens(name_toks)); |
| 1723 | let val_doc = doc::concat(smart_spaced_tokens(val_toks)); |
| 1724 | let pad = if align { " ".repeat(max_name - name_w) } else { String::new() }; |
| 1725 | match sep_char { |
| 1726 | Some('=') => docs.push(doc::concat(vec![ |
| 1727 | name_doc, |
| 1728 | doc::text(pad.as_str()), |
| 1729 | doc::text(" = "), |
| 1730 | val_doc, |
| 1731 | ])), |
| 1732 | _ => docs.push(doc::concat(vec![ |
| 1733 | name_doc, |
| 1734 | doc::text(": "), |
| 1735 | doc::text(pad.as_str()), |
| 1736 | val_doc, |
| 1737 | ])), |
| 1738 | } |
| 1739 | } |
| 1740 | // The comma belongs to the field, and the field's own trailing |
| 1741 | // comment follows it on the same line. |
| 1742 | docs.push(doc::text(",")); |
| 1743 | for c in &trailing[i] { |
| 1744 | docs.push(doc::text(" ")); |
| 1745 | docs.push(doc::text(c.as_str())); |
| 1746 | } |
| 1747 | if i + 1 < parsed.len() { |
| 1748 | docs.push(doc::hardline()); |
| 1749 | } |
| 1750 | } |
| 1751 | |
| 1752 | // The block's last comment, which sat above the closing brace. |
| 1753 | for tok in &tail { |
| 1754 | for t in &tok.leading_trivia { |
| 1755 | match t { |
| 1756 | Trivia::LineComment(c) | Trivia::BlockComment(c) => { |
| 1757 | docs.push(doc::hardline()); |
| 1758 | docs.push(doc::text(c.as_str())); |
| 1759 | } |
| 1760 | Trivia::Newline | Trivia::Whitespace(_) => {} |
| 1761 | } |
| 1762 | } |
| 1763 | } |
| 1764 | |
| 1765 | doc::concat(docs) |
| 1766 | } |
| 1767 | |
| 1768 | /// Check whether a token slice ends with a comma. |
| 1769 | fn ends_with_comma(tokens: &[Token]) -> bool { |
| 1770 | tokens.last().map_or(false, |t| matches!(t.kind, TokenKind::Punct(','))) |
| 1771 | } |
| 1772 | |
| 1773 | fn is_short_body(tokens: &[Token]) -> bool { |
| 1774 | // A body carrying a comment cannot be collapsed onto one line: |
| 1775 | // the single-line path renders through smart_spaced_tokens, which |
| 1776 | // emits no trivia, so the comment would be dropped. |
| 1777 | let has_comment = tokens |
| 1778 | .iter() |
| 1779 | .any(|t| t.leading_trivia.iter().any(is_comment_trivia)); |
| 1780 | if has_comment { return false; } |
| 1781 | // Must be a single expression — no semicolons, no let, no braces. |
| 1782 | let has_semi = tokens.iter().any(|t| matches!(t.kind, TokenKind::Punct(';'))); |
| 1783 | if has_semi { return false; } |
| 1784 | let has_let = tokens.iter().any(|t| matches!(t.kind, TokenKind::Keyword(ref k) if k == "let")); |
| 1785 | if has_let { return false; } |
| 1786 | let has_brace = tokens.iter().any(|t| matches!(t.kind, TokenKind::Punct('{'))); |
| 1787 | if has_brace { return false; } |
| 1788 | // Short enough (heuristic: total text width). |
| 1789 | let width: usize = tokens.iter().map(|t| t.text.len() + 1).sum(); |
| 1790 | width < 40 |
| 1791 | } |
| 1792 | |
| 1793 | |
| 1794 | #[cfg(test)] |
| 1795 | mod tests { |
| 1796 | use super::*; |
| 1797 | use crate::fmt::spec::FormatSpec; |
| 1798 | |
| 1799 | fn fmt(source: &str) -> String { |
| 1800 | let spec = FormatSpec::fe2o3(); |
| 1801 | format_rust(source, &spec).expect("format failed") |
| 1802 | } |
| 1803 | |
| 1804 | fn fmt_with(source: &str, spec: &FormatSpec) -> String { |
| 1805 | format_rust(source, spec).expect("format failed") |
| 1806 | } |
| 1807 | |
| 1808 | #[test] |
| 1809 | fn test_format_empty_fn() { |
| 1810 | let out = fmt("fn main() {}"); |
| 1811 | assert!(out.contains("fn"), "output: {}", out); |
| 1812 | assert!(out.contains("main"), "output: {}", out); |
| 1813 | } |
| 1814 | |
| 1815 | #[test] |
| 1816 | fn test_format_fn_with_body() { |
| 1817 | let out = fmt("fn main() { let x = 42; }"); |
| 1818 | assert!(out.contains("fn main"), "output: {}", out); |
| 1819 | } |
| 1820 | |
| 1821 | #[test] |
| 1822 | fn test_format_struct() { |
| 1823 | let out = fmt("struct Foo { x: u32, y: u64, }"); |
| 1824 | assert!(out.contains("struct Foo"), "output: {}", out); |
| 1825 | } |
| 1826 | |
| 1827 | #[test] |
| 1828 | fn test_format_use() { |
| 1829 | let out = fmt("use std::collections::HashMap;"); |
| 1830 | assert!(out.contains("use"), "output: {}", out); |
| 1831 | assert!(out.contains("HashMap"), "output: {}", out); |
| 1832 | } |
| 1833 | |
| 1834 | #[test] |
| 1835 | fn test_format_impl() { |
| 1836 | let out = fmt("impl Foo { fn bar(&self) {} }"); |
| 1837 | assert!(out.contains("impl Foo"), "output: {}", out); |
| 1838 | } |
| 1839 | |
| 1840 | #[test] |
| 1841 | fn test_format_preserves_comments() { |
| 1842 | let out = fmt("// A comment\nfn foo() {}"); |
| 1843 | assert!(out.contains("// A comment"), "output: {:?}", out); |
| 1844 | } |
| 1845 | |
| 1846 | #[test] |
| 1847 | fn test_format_preserves_doc_comments() { |
| 1848 | let out = fmt("/// A doc comment.\nfn foo() {}"); |
| 1849 | assert!(out.contains("/// A doc comment."), "output: {:?}", out); |
| 1850 | } |
| 1851 | |
| 1852 | #[test] |
| 1853 | fn test_format_multiple_items() { |
| 1854 | let out = fmt("use std::fmt;\n\nfn main() {}\n\nstruct Foo;"); |
| 1855 | assert!(out.contains("use"), "output: {:?}", out); |
| 1856 | assert!(out.contains("fn main"), "output: {:?}", out); |
| 1857 | assert!(out.contains("struct Foo"), "output: {:?}", out); |
| 1858 | } |
| 1859 | |
| 1860 | #[test] |
| 1861 | fn test_format_fn_with_params() { |
| 1862 | let out = fmt("fn foo(x: u32, y: u64) -> bool { true }"); |
| 1863 | assert!(out.contains("fn foo"), "output: {:?}", out); |
| 1864 | assert!(out.contains("x: u32"), "output: {:?}", out); |
| 1865 | } |
| 1866 | |
| 1867 | #[test] |
| 1868 | fn test_roundtrip_simple() { |
| 1869 | // Format twice — should be idempotent. |
| 1870 | let source = "fn main() {\n let x = 42;\n}"; |
| 1871 | let first = fmt(source); |
| 1872 | let second = fmt(&first); |
| 1873 | assert_eq!(first, second, "not idempotent:\nfirst: {:?}\nsecond: {:?}", first, second); |
| 1874 | } |
| 1875 | } |