Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/fmt/format.rs

68.5 KiB, 223 runs

created by r1870400018:11626, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! CST to Doc transformation.
2//!
3//! Walks the concrete syntax tree produced by the parser and converts
4//! each node into a layout document (`Doc`) according to the format
5//! specification. The layout algebra then handles line-breaking
6//! decisions during rendering.
7//!
8
9use crate::fmt::{
10 cst::{
11 CstChild,
12 CstNode,
13 Token,
14 TokenKind,
15 Trivia,
16 },
17 doc::{self, Doc},
18 lex,
19 parse,
20 render,
21 spec::{
22 BinopBreak,
23 FormatSpec,
24 ReturnTypePlacement,
25 },
26};
27
28use oxedyne_fe2o3_core::prelude::*;
29
30
31/// Format Rust source code.
32pub fn format_rust(source: &str, spec: &FormatSpec) -> Outcome<String> {
33 let lang = lex::rust_tokens();
34 let tokens = res!(lex::lex(source, &lang));
35 // The parser reads the EOF token as a stop condition and does not
36 // put it in the tree, so a comment on the last line -- which the
37 // lexer attaches to EOF as leading trivia -- never reaches the
38 // formatter. Carry it across the parse.
39 let trailing = tokens
40 .last()
41 .filter(|t| matches!(t.kind, TokenKind::Eof)
42 && t.leading_trivia.iter().any(is_comment_trivia))
43 .cloned();
44 let before = comment_texts(&tokens);
45 let cst = res!(parse::parse_rust(tokens));
46 let document = format_source_file(&cst, trailing, spec);
47 let indent_str = spec.indent_str();
48 let formatted = end_with_newline(render::render(spec.max_width, &indent_str, &document));
49
50 // Refuse rather than corrupt. Not every layout path carries a
51 // comment through yet, and a formatter that silently drops one --
52 // or moves it onto the next item, where it comes to describe the
53 // wrong code -- does more harm than one that declines the file.
54 // Checked on the output, so it holds however the layout changes.
55 let after = comment_texts(&res!(lex::lex(&formatted, &lang)));
56 if before != after {
57 return Err(err!(
58 "Formatting this file would alter its comments ({} before, {} after). \
59 Refusing, rather than dropping one.",
60 before.len(), after.len();
61 Format, Mismatch));
62 }
63 Ok(formatted)
64}
65
66/// Every comment in a token stream, in order.
67///
68/// Plain comments are trivia attached to the following token; doc
69/// comments are tokens in their own right. Both are compared, so a
70/// formatting pass cannot quietly lose either. Right-trimmed, because
71/// trailing spaces inside a comment carry no meaning.
72fn comment_texts(tokens: &[Token]) -> Vec<String> {
73 let mut out = Vec::new();
74 for tok in tokens {
75 for t in &tok.leading_trivia {
76 if let Trivia::LineComment(c) | Trivia::BlockComment(c) = t {
77 out.push(c.trim_end().to_string());
78 }
79 }
80 if let TokenKind::DocComment(c) = &tok.kind {
81 out.push(c.trim_end().to_string());
82 }
83 }
84 out
85}
86
87/// Give the output the single trailing newline a text file should end
88/// with, so a formatted file does not read as "\ No newline at end of
89/// file" in every diff.
90fn end_with_newline(mut s: String) -> String {
91 while s.ends_with('\n') {
92 s.pop();
93 }
94 if !s.is_empty() {
95 s.push('\n');
96 }
97 s
98}
99
100/// Format source code using a pre-built language token definition.
101///
102/// For Rust, this goes through the structural parser for
103/// keyword-aware formatting. For other languages, it uses the
104/// generic token-stream pipeline directly.
105pub fn format_with_lang(
106 source: &str,
107 lang: &lex::LangTokens,
108 spec: &FormatSpec,
109) -> Outcome<String> {
110 let tokens = res!(lex::lex(source, lang));
111 let document = tokens_to_formatted_doc(&tokens, spec);
112 let indent_str = spec.indent_str();
113 let formatted = render::render(spec.max_width, &indent_str, &document);
114 Ok(end_with_newline(formatted))
115}
116
117// ── Source file ──────────────────────────────────────────────────
118
119/// Flatten a CST into tokens and format via the token-stream pipeline.
120///
121/// `trailing` is the lexer's EOF token where it carries a comment the
122/// parse would otherwise have dropped.
123fn format_source_file(node: &CstNode, trailing: Option<Token>, spec: &FormatSpec) -> Doc {
124 let mut tokens: Vec<Token> = Vec::new();
125 collect_tokens(node, &mut tokens);
126 if let Some(eof) = trailing {
127 tokens.push(eof);
128 }
129 tokens_to_formatted_doc(&tokens, spec)
130}
131
132/// Core formatting logic: convert a token stream to a Doc with
133/// proper spacing, newlines, and brace-based indentation.
134///
135/// Recursively processes `{ ... }` blocks, wrapping their contents
136/// in `Nest` so the renderer applies the correct indentation.
137fn tokens_to_formatted_doc(tokens: &[Token], spec: &FormatSpec) -> Doc {
138 let mut tokens = tokens.to_vec();
139 // Apply import reordering before formatting.
140 if spec.import_reorder {
141 apply_import_reorder(&mut tokens);
142 }
143 let mut pos = 0;
144 let result = format_token_range(&tokens, &mut pos, spec, false, false);
145 doc::concat(result)
146}
147
148/// Find `use ... { items }` patterns in the token stream and
149/// reorder the items alphabetically.
150fn apply_import_reorder(tokens: &mut Vec<Token>) {
151 // Find each `use` keyword and look for a subsequent `{ ... }`.
152 let mut i = 0;
153 while i < tokens.len() {
154 if matches!(tokens[i].kind, TokenKind::Keyword(ref k) if k == "use") {
155 // Find the `{` and `}` for this use statement.
156 let mut j = i + 1;
157 while j < tokens.len() && !matches!(tokens[j].kind, TokenKind::Punct('{') | TokenKind::Punct(';')) {
158 j += 1;
159 }
160 if j < tokens.len() && matches!(tokens[j].kind, TokenKind::Punct('{')) {
161 // Extract the range from `use` to the `;` after `}`.
162 let open = j;
163 let mut depth = 1usize;
164 let mut k = open + 1;
165 while k < tokens.len() && depth > 0 {
166 match tokens[k].kind {
167 TokenKind::Punct('{') => depth += 1,
168 TokenKind::Punct('}') => depth -= 1,
169 _ => {}
170 }
171 k += 1;
172 }
173 let close = k - 1; // position of `}`
174 // Extract and reorder just this use block's tokens.
175 let mut use_tokens: Vec<Token> = tokens[i..=close].to_vec();
176 // Find `;` after `}`.
177 if close + 1 < tokens.len() && matches!(tokens[close + 1].kind, TokenKind::Punct(';')) {
178 use_tokens.push(tokens[close + 1].clone());
179 }
180 reorder_use_items(&mut use_tokens);
181 // Replace in the main token list.
182 let end = if close + 1 < tokens.len() && matches!(tokens[close + 1].kind, TokenKind::Punct(';')) {
183 close + 2
184 } else {
185 close + 1
186 };
187 tokens.splice(i..end, use_tokens);
188 }
189 }
190 i += 1;
191 }
192}
193
194/// Format tokens from `pos` to the end (or the matching `}`).
195/// When `inside_brace` is true, we stop at the matching `}`.
196fn format_token_range(
197 tokens: &[Token],
198 pos: &mut usize,
199 spec: &FormatSpec,
200 inside_brace: bool,
201 inside_macro: bool,
202) -> Vec<Doc> {
203 let indent = spec.indent_width;
204 let mut docs = Vec::new();
205
206 while *pos < tokens.len() {
207 let i = *pos;
208 let tok = &tokens[i];
209
210 // Emit leading trivia (newlines, comments).
211 let mut had_newline = false;
212 let mut newline_count = 0usize;
213 for t in &tok.leading_trivia {
214 match t {
215 Trivia::Newline => {
216 had_newline = true;
217 newline_count += 1;
218 }
219 Trivia::LineComment(c) => {
220 flush_newlines(&mut docs, &mut newline_count, &mut had_newline, spec);
221 docs.push(doc::text(c.as_str()));
222 }
223 Trivia::BlockComment(c) => {
224 flush_newlines(&mut docs, &mut newline_count, &mut had_newline, spec);
225 docs.push(doc::text(c.as_str()));
226 }
227 Trivia::Whitespace(_) => {
228 if !had_newline && i > 0 {
229 docs.push(doc::text(" "));
230 }
231 }
232 }
233 }
234 flush_newlines(&mut docs, &mut newline_count, &mut had_newline, spec);
235
236 // Space before this token.
237 if !had_newline && i > 0 {
238 let prev = &tokens[i - 1];
239 if needs_separator_space(prev, tok)
240 || (!suppress_space_before(tok)
241 && !suppress_space_after(prev)
242 && tok.leading_trivia.is_empty())
243 {
244 docs.push(doc::text(" "));
245 }
246 }
247
248 match tok.kind {
249 // ── fn signature: Oxedyne layout ─────────────────
250 TokenKind::Keyword(ref kw) if kw == "fn" => {
251 let fn_doc = format_fn_from_tokens(tokens, pos, spec);
252 docs.push(fn_doc);
253 }
254
255 // ── match expression ─────────────────────────────
256 TokenKind::Keyword(ref kw) if kw == "match" => {
257 docs.push(doc::text("match"));
258 *pos += 1;
259 // Consume the scrutinee (tokens until `{`).
260 while *pos < tokens.len() && !matches!(tokens[*pos].kind, TokenKind::Punct('{')) {
261 let mt = &tokens[*pos];
262 if !suppress_space_before(mt) {
263 docs.push(doc::text(" "));
264 }
265 docs.push(doc::text(mt.text.as_str()));
266 *pos += 1;
267 }
268 // The `{` will be handled by the brace handler on
269 // the next iteration — but we want match-arm-aware
270 // formatting. Consume the block ourselves.
271 if *pos < tokens.len() && matches!(tokens[*pos].kind, TokenKind::Punct('{')) {
272 docs.push(doc::text(" {"));
273 *pos += 1;
274 // Collect arm tokens.
275 let mut arm_tokens: Vec<Token> = Vec::new();
276 let mut depth = 1usize;
277 while *pos < tokens.len() && depth > 0 {
278 let at = &tokens[*pos];
279 match at.kind {
280 TokenKind::Punct('{') => { depth += 1; arm_tokens.push(at.clone()); }
281 TokenKind::Punct('}') => {
282 depth -= 1;
283 if depth > 0 {
284 arm_tokens.push(at.clone());
285 } else {
286 carry_close_comments(&mut arm_tokens, at);
287 }
288 }
289 _ => { arm_tokens.push(at.clone()); }
290 }
291 *pos += 1;
292 }
293 let arms_doc = format_match_arms(&arm_tokens, spec);
294 if !arm_tokens.is_empty() {
295 docs.push(doc::nest(indent, doc::concat(vec![
296 doc::hardline(),
297 arms_doc,
298 ])));
299 }
300 if depth == 0 {
301 docs.push(doc::hardline());
302 docs.push(doc::text("}"));
303 }
304 }
305 }
306
307 TokenKind::Punct('{') => {
308 // Check if this is a struct/enum field block.
309 // Macro bodies and their nested blocks are never
310 // treated as field blocks.
311 let is_macro_entry = i > 0
312 && matches!(tokens[i - 1].kind, TokenKind::Punct('!'));
313 let in_macro = inside_macro || is_macro_entry;
314 let is_field_block = !in_macro
315 && is_preceded_by_struct_or_enum(tokens, i);
316
317 docs.push(doc::text("{"));
318 *pos += 1;
319
320 if is_field_block && spec.field_align_threshold > 0 {
321 // Collect the raw tokens inside the block.
322 let mut field_tokens: Vec<Token> = Vec::new();
323 let mut depth = 1usize;
324 while *pos < tokens.len() && depth > 0 {
325 let bt = &tokens[*pos];
326 match bt.kind {
327 TokenKind::Punct('{') => { depth += 1; field_tokens.push(bt.clone()); }
328 TokenKind::Punct('}') => {
329 depth -= 1;
330 if depth > 0 {
331 field_tokens.push(bt.clone());
332 } else {
333 carry_close_comments(&mut field_tokens, bt);
334 }
335 }
336 _ => { field_tokens.push(bt.clone()); }
337 }
338 *pos += 1;
339 }
340 // Format as aligned fields.
341 let field_doc = format_aligned_fields(&field_tokens, spec);
342 if !field_tokens.is_empty() {
343 docs.push(doc::nest(indent, doc::concat(vec![
344 doc::hardline(),
345 field_doc,
346 ])));
347 }
348 if depth == 0 {
349 docs.push(doc::hardline());
350 docs.push(doc::text("}"));
351 }
352 } else {
353 // Generic block: recurse.
354 let inner = format_token_range(tokens, pos, spec, true, in_macro);
355 let (content, closing) = split_off_closing_brace(inner);
356 if !content.is_empty() {
357 docs.push(doc::nest(indent, doc::concat(content)));
358 }
359 docs.extend(closing);
360 }
361 }
362 TokenKind::Punct('}') if inside_brace => {
363 docs.push(doc::text("}"));
364 *pos += 1;
365 return docs;
366 }
367 // `where` keyword: optionally indent the predicates.
368 TokenKind::Keyword(ref kw) if kw == "where" && spec.where_indent => {
369 docs.push(doc::text("where"));
370 *pos += 1;
371 // Collect tokens until `{` or `;`.
372 let mut pred_docs = Vec::new();
373 while *pos < tokens.len() {
374 let pt = &tokens[*pos];
375 if matches!(pt.kind, TokenKind::Punct('{') | TokenKind::Punct(';')) {
376 break;
377 }
378 // Emit trivia + token with smart spacing.
379 let mut ph = false;
380 let mut pn = 0usize;
381 for t in &pt.leading_trivia {
382 match t {
383 Trivia::Newline => { ph = true; pn += 1; }
384 Trivia::LineComment(c) => {
385 flush_newlines(&mut pred_docs, &mut pn, &mut ph, spec);
386 pred_docs.push(doc::text(c.as_str()));
387 }
388 Trivia::BlockComment(c) => {
389 flush_newlines(&mut pred_docs, &mut pn, &mut ph, spec);
390 pred_docs.push(doc::text(c.as_str()));
391 }
392 Trivia::Whitespace(_) => {
393 if !ph {
394 pred_docs.push(doc::text(" "));
395 }
396 }
397 }
398 }
399 flush_newlines(&mut pred_docs, &mut pn, &mut ph, spec);
400 if !ph && !pred_docs.is_empty() {
401 let prev_tok = &tokens[*pos - 1];
402 if !suppress_space_before_ctx(pt, Some(prev_tok))
403 && !suppress_space_after(prev_tok)
404 && pt.leading_trivia.is_empty()
405 {
406 pred_docs.push(doc::text(" "));
407 }
408 }
409 if !pt.text.is_empty() {
410 pred_docs.push(doc::text(pt.text.as_str()));
411 }
412 *pos += 1;
413 }
414 if !pred_docs.is_empty() {
415 docs.push(doc::nest(indent, doc::concat(pred_docs)));
416 }
417 }
418 // ── let / return: binop-aware statement ────
419 // Skip `let` inside `if let` / `while let` patterns.
420 TokenKind::Keyword(ref kw) if (kw == "let" || kw == "return")
421 && !(kw == "let" && i > 0 && matches!(
422 tokens[i - 1].kind,
423 TokenKind::Keyword(ref pk) if pk == "if" || pk == "while"
424 )) =>
425 {
426 docs.push(doc::text(tok.text.as_str()));
427 *pos += 1;
428 // Collect remaining tokens through `;`.
429 let start = *pos;
430 let mut end = *pos;
431 let mut depth = 0i32;
432 while end < tokens.len() {
433 match &tokens[end].kind {
434 TokenKind::Punct('(') | TokenKind::Punct('[')
435 | TokenKind::Punct('{') => depth += 1,
436 TokenKind::Punct(')') | TokenKind::Punct(']') => depth -= 1,
437 TokenKind::Punct('}') => {
438 depth -= 1;
439 if depth < 0 { break; }
440 }
441 TokenKind::Punct(';') if depth == 0 => {
442 end += 1;
443 break;
444 }
445 _ => {}
446 }
447 end += 1;
448 }
449 if start < end {
450 let mut stmt_rest: Vec<Token> = tokens[start..end].to_vec();
451 // First token's trivia already handled by the outer loop.
452 if let Some(first) = stmt_rest.first_mut() {
453 first.leading_trivia.clear();
454 }
455 // Space between keyword and first token (unless `;`).
456 if !matches!(stmt_rest[0].kind, TokenKind::Punct(';')) {
457 docs.push(doc::text(" "));
458 }
459 if stmt_rest.iter().any(|t| is_unambiguous_binary_op(t)) {
460 docs.push(format_binop_expr(&stmt_rest, spec));
461 } else {
462 // No chain breaks — method chain wiring is separate.
463 docs.push(doc::concat(smart_spaced_tokens_inner(&stmt_rest, false)));
464 }
465 }
466 *pos = end;
467 }
468 TokenKind::Eof => {
469 *pos += 1;
470 break;
471 }
472 _ => {
473 if !tok.text.is_empty() {
474 docs.push(doc::text(tok.text.as_str()));
475 }
476 *pos += 1;
477 }
478 }
479 }
480 docs
481}
482
483/// Split a doc list into (content, closing). The closing part is
484/// the trailing HardLine + `}` that should sit outside the Nest.
485fn split_off_closing_brace(mut docs: Vec<Doc>) -> (Vec<Doc>, Vec<Doc>) {
486 // Walk backwards to find the `}` text and any preceding hardlines.
487 let mut closing = Vec::new();
488 // Pop the `}`.
489 if let Some(last) = docs.pop() {
490 closing.push(last);
491 }
492 // Pop any hardlines immediately before the `}`.
493 while let Some(d) = docs.pop() {
494 if matches!(d, Doc::HardLine) {
495 closing.push(d);
496 } else {
497 // Not a hardline: restore it and stop.
498 docs.push(d);
499 break;
500 }
501 }
502 closing.reverse();
503 (docs, closing)
504}
505
506/// Format an `fn` signature directly from the token stream.
507///
508/// Reads tokens starting at `fn` through to `{` or `;`, and produces
509/// the Oxedyne layout:
510/// - Flat: `fn foo(x: u32, y: u64) -> bool`
511/// - Broken: `fn foo(\n x: u32,\n y: u64,\n)\n -> bool`
512///
513/// Does NOT consume the `{` or `;` — leaves that for the caller.
514fn format_fn_from_tokens(
515 tokens: &[Token],
516 pos: &mut usize,
517 spec: &FormatSpec,
518) -> Doc {
519 let indent = spec.indent_width;
520
521 // Phase 1: collect signature tokens into segments.
522 let mut pre_paren: Vec<Token> = Vec::new(); // fn name<T>
523 let mut params: Vec<Vec<Token>> = Vec::new(); // param groups
524 let mut ret_tokens: Vec<Token> = Vec::new(); // -> Type
525 let mut where_tokens: Vec<Token> = Vec::new();// where T: Foo
526 let mut has_parens = false;
527
528 // Consume `fn`. Strip its leading trivia since the caller
529 // already emitted it.
530 let mut fn_tok = tokens[*pos].clone();
531 fn_tok.leading_trivia.clear();
532 pre_paren.push(fn_tok);
533 *pos += 1;
534
535 // Consume name and optional generics (until `(`).
536 while *pos < tokens.len() {
537 let t = &tokens[*pos];
538 if matches!(t.kind, TokenKind::Punct('(')) {
539 has_parens = true;
540 break;
541 }
542 if matches!(t.kind, TokenKind::Punct('{') | TokenKind::Punct(';')) {
543 break;
544 }
545 pre_paren.push(t.clone());
546 *pos += 1;
547 }
548
549 // Consume parameter list `(...)`.
550 if has_parens && *pos < tokens.len() {
551 *pos += 1; // skip `(`
552 let mut depth = 1usize;
553 let mut current_param: Vec<Token> = Vec::new();
554 while *pos < tokens.len() && depth > 0 {
555 let t = &tokens[*pos];
556 match t.kind {
557 TokenKind::Punct('(') => { depth += 1; current_param.push(t.clone()); }
558 TokenKind::Punct(')') => {
559 depth -= 1;
560 if depth == 0 {
561 // A comment on the parameter list's last line is
562 // attached to this paren, and is dropped with it.
563 // Carrying it here does not help -- parameters
564 // render through smart_spaced_tokens, which emits
565 // no trivia -- and an empty group would earn a
566 // comma of its own. The comment guard in
567 // format_rust refuses such a file instead.
568 if !current_param.is_empty() {
569 params.push(current_param);
570 current_param = Vec::new();
571 }
572 } else {
573 current_param.push(t.clone());
574 }
575 }
576 TokenKind::Punct(',') if depth == 1 => {
577 if !current_param.is_empty() {
578 params.push(current_param);
579 current_param = Vec::new();
580 }
581 }
582 _ => { current_param.push(t.clone()); }
583 }
584 *pos += 1;
585 }
586 }
587
588 // Consume return type `-> Type` and/or where clause.
589 let mut in_where = false;
590 while *pos < tokens.len() {
591 let t = &tokens[*pos];
592 if matches!(t.kind, TokenKind::Punct('{') | TokenKind::Punct(';')) {
593 break;
594 }
595 if matches!(t.kind, TokenKind::Keyword(ref k) if k == "where") {
596 in_where = true;
597 }
598 if in_where {
599 where_tokens.push(t.clone());
600 } else {
601 ret_tokens.push(t.clone());
602 }
603 *pos += 1;
604 }
605
606 // Phase 2: build the Doc.
607 let mut sig = Vec::new();
608
609 // fn name<T>
610 sig.push(doc::concat(smart_spaced_tokens(&pre_paren)));
611
612 // Parameters + return type wrapped in ONE group.
613 // The group's flat/broken state controls both the param layout
614 // AND the return type placement (Oxedyne rule).
615 let mut grouped = Vec::new();
616
617 if has_parens {
618 if params.is_empty() {
619 grouped.push(doc::text("()"));
620 } else {
621 // Build param docs. When vertical and there are 2+
622 // params, align the type column.
623 let align = params.len() >= 2 && spec.field_align_threshold > 0;
624 let param_docs = format_param_list(&params, align, spec);
625
626 let sep = doc::concat(vec![doc::text(","), doc::line()]);
627 let body = doc::join(sep, param_docs);
628 let trailing = if spec.trailing_comma {
629 doc::trailing_comma()
630 } else {
631 doc::empty()
632 };
633 grouped.push(doc::text("("));
634 grouped.push(doc::nest(indent, doc::concat(vec![
635 doc::softline(),
636 body,
637 trailing,
638 ])));
639 grouped.push(doc::softline());
640 grouped.push(doc::text(")"));
641 }
642 }
643
644 // Return type: inline when flat, own line when broken.
645 if !ret_tokens.is_empty() {
646 let ret_doc = doc::concat(smart_spaced_tokens(&ret_tokens));
647 match spec.fn_return_type {
648 ReturnTypePlacement::OwnLine => {
649 grouped.push(doc::if_break(
650 // Flat: ` -> Type`
651 doc::concat(vec![doc::text(" "), ret_doc.clone()]),
652 // Broken: `\n -> Type`
653 doc::nest(indent, doc::concat(vec![
654 doc::hardline(),
655 ret_doc,
656 ])),
657 ));
658 }
659 ReturnTypePlacement::SameLine => {
660 grouped.push(doc::text(" "));
661 grouped.push(ret_doc);
662 }
663 }
664 }
665
666 // Where clause — part of the grouped section so that its
667 // presence forces a break (which triggers the Oxedyne layout).
668 if !where_tokens.is_empty() {
669 grouped.push(doc::hardline());
670 if spec.where_indent {
671 grouped.push(doc::text("where"));
672 let pred_tokens: Vec<Token> = where_tokens.into_iter().skip(1).collect();
673 if !pred_tokens.is_empty() {
674 grouped.push(doc::nest(indent, doc::concat(vec![
675 doc::hardline(),
676 doc::concat(smart_spaced_tokens(&pred_tokens)),
677 ])));
678 }
679 } else {
680 grouped.push(doc::concat(smart_spaced_tokens(&where_tokens)));
681 }
682 }
683
684 // Brace + body.
685 // Oxedyne rule: when the signature is flat, body stays inline
686 // fn foo(x: u32) -> bool { expr }
687 // When broken, brace on own line, body indented:
688 // fn foo(
689 // x: u32,
690 // )
691 // -> bool
692 // {
693 // expr
694 // }
695 if *pos < tokens.len() && matches!(tokens[*pos].kind, TokenKind::Punct('{')) {
696 // Consume `{` and collect body tokens until matching `}`.
697 *pos += 1;
698 let mut body_tokens: Vec<Token> = Vec::new();
699 let mut depth = 1usize;
700 while *pos < tokens.len() && depth > 0 {
701 let bt = &tokens[*pos];
702 match bt.kind {
703 TokenKind::Punct('{') => { depth += 1; body_tokens.push(bt.clone()); }
704 TokenKind::Punct('}') => {
705 depth -= 1;
706 if depth > 0 {
707 body_tokens.push(bt.clone());
708 } else {
709 carry_close_comments(&mut body_tokens, bt);
710 }
711 }
712 _ => { body_tokens.push(bt.clone()); }
713 }
714 *pos += 1;
715 }
716
717 // Decide: single-expression body (can go on one line) or
718 // multi-statement body (always vertical).
719 let is_short = is_short_body(&body_tokens);
720
721 let found_close = depth == 0;
722
723 if is_short && spec.fn_single_line {
724 // Short body: use smart_spaced_tokens (no trivia hardlines)
725 // so the group can choose flat.
726 let body_doc = doc::concat(smart_spaced_tokens(&body_tokens));
727
728 grouped.push(doc::if_break(
729 doc::text(" {"),
730 doc::concat(vec![doc::hardline(), doc::text("{")]),
731 ));
732 grouped.push(doc::if_break(
733 doc::concat(vec![doc::text(" "), body_doc.clone(), doc::text(" ")]),
734 doc::concat(vec![
735 doc::nest(indent, doc::concat(vec![doc::hardline(), body_doc])),
736 doc::hardline(),
737 ]),
738 ));
739 if found_close {
740 grouped.push(doc::text("}"));
741 }
742 } else {
743 // Multi-statement body: always vertical, outside the group.
744 sig.push(doc::group(doc::concat(grouped)));
745 sig.push(doc::text(" {"));
746 // Strip leading newlines/whitespace from the first body
747 // token to avoid doubling the newline after `{`. Only
748 // strip the prefix before the first comment or content,
749 // not newlines between consecutive comment lines.
750 if let Some(first) = body_tokens.first_mut() {
751 let mut past_lead = false;
752 first.leading_trivia.retain(|t| {
753 if past_lead { return true; }
754 match t {
755 Trivia::Newline | Trivia::Whitespace(_) => false,
756 _ => { past_lead = true; true }
757 }
758 });
759 }
760 let body_doc = tokens_to_formatted_doc(&body_tokens, spec);
761 sig.push(doc::nest(indent, doc::concat(vec![
762 doc::hardline(),
763 body_doc,
764 ])));
765 if found_close {
766 sig.push(doc::hardline());
767 sig.push(doc::text("}"));
768 }
769 return doc::concat(sig);
770 }
771 } else if *pos < tokens.len() && matches!(tokens[*pos].kind, TokenKind::Punct(';')) {
772 grouped.push(doc::text(";"));
773 *pos += 1;
774 }
775
776 sig.push(doc::group(doc::concat(grouped)));
777 doc::concat(sig)
778}
779
780/// Whether a piece of trivia is a comment rather than whitespace.
781fn is_comment_trivia(t: &Trivia) -> bool {
782 matches!(t, Trivia::LineComment(_) | Trivia::BlockComment(_))
783}
784
785/// Keep a closing brace's comments when the brace itself is dropped.
786///
787/// A scan that gathers a block's contents stops at the closing brace
788/// and does not keep it, because the brace is emitted separately. But
789/// the lexer attaches any comment on the block's last line to that
790/// brace as leading trivia, so the comment would go with it. Push a
791/// token that renders nothing and carries the comments instead. The
792/// trivia is cut after the last comment, so the newline that ran on to
793/// the brace does not become a blank line.
794/// Emit comments carried off a block's closing delimiter.
795///
796/// `lead_break` puts a line break before the first one, which is wanted
797/// when content precedes it and not when the block was otherwise empty.
798fn emit_carried_comments(docs: &mut Vec<Doc>, tail: &[Token], lead_break: bool) {
799 let mut first = true;
800 for tok in tail {
801 for t in &tok.leading_trivia {
802 if let Trivia::LineComment(c) | Trivia::BlockComment(c) = t {
803 if !first || lead_break {
804 docs.push(doc::hardline());
805 }
806 docs.push(doc::text(c.as_str()));
807 first = false;
808 }
809 }
810 }
811}
812
813fn carry_close_comments(out: &mut Vec<Token>, close: &Token) {
814 if let Some(last) = close.leading_trivia.iter().rposition(is_comment_trivia) {
815 out.push(Token {
816 kind: TokenKind::Ident,
817 text: String::new(),
818 leading_trivia: close.leading_trivia[..=last].to_vec(),
819 span: close.span,
820 });
821 }
822}
823
824/// Flush pending newlines as hardlines, capped to max_blank_lines.
825fn flush_newlines(
826 docs: &mut Vec<Doc>,
827 newline_count: &mut usize,
828 had_newline: &mut bool,
829 spec: &FormatSpec,
830) {
831 if *had_newline {
832 let blanks = (*newline_count).min(spec.max_blank_lines + 1);
833 for _ in 0..blanks {
834 docs.push(doc::hardline());
835 }
836 *newline_count = 0;
837 *had_newline = false;
838 }
839}
840
841/// Reorder items inside `use { ... }` blocks alphabetically.
842/// Finds the `{` and `}` in the token stream, extracts the
843/// comma-separated items between them, sorts by text, and
844/// reassembles.
845fn reorder_use_items(tokens: &mut Vec<Token>) {
846 // Find the `{` and `}` positions.
847 let open = tokens.iter().position(|t| matches!(t.kind, TokenKind::Punct('{')));
848 let close = tokens.iter().rposition(|t| matches!(t.kind, TokenKind::Punct('}')));
849 let (open_idx, close_idx) = match (open, close) {
850 (Some(o), Some(c)) if c > o + 1 => (o, c),
851 _ => return, // No braced block or empty.
852 };
853
854 // Extract inner tokens and split by comma.
855 let inner: Vec<Token> = tokens[open_idx + 1..close_idx].to_vec();
856 let mut items = split_by_comma(&inner);
857 if items.len() < 2 {
858 return; // Nothing to sort.
859 }
860
861 // Sort by the concatenated text of each item's tokens.
862 items.sort_by(|a, b| {
863 let a_text: String = a.iter().map(|t| t.text.as_str()).collect();
864 let b_text: String = b.iter().map(|t| t.text.as_str()).collect();
865 a_text.cmp(&b_text)
866 });
867
868 // Reassemble: preserve the trivia of the first original inner
869 // token on the first sorted item.
870 let first_trivia = if !inner.is_empty() {
871 inner[0].leading_trivia.clone()
872 } else {
873 Vec::new()
874 };
875
876 // Build new inner token list from sorted items with commas.
877 let mut new_inner: Vec<Token> = Vec::new();
878 for (i, item) in items.iter().enumerate() {
879 for (j, tok) in item.iter().enumerate() {
880 let mut tok = tok.clone();
881 // Give the first token of the first item the original
882 // leading trivia (newline + whitespace for indentation).
883 if i == 0 && j == 0 {
884 tok.leading_trivia = first_trivia.clone();
885 }
886 new_inner.push(tok);
887 }
888 if i < items.len() - 1 {
889 // Add comma between items.
890 new_inner.push(Token {
891 kind: TokenKind::Punct(','),
892 text: ",".to_string(),
893 leading_trivia: Vec::new(),
894 span: crate::fmt::cst::Span::default(),
895 });
896 }
897 }
898 // Trailing comma.
899 new_inner.push(Token {
900 kind: TokenKind::Punct(','),
901 text: ",".to_string(),
902 leading_trivia: Vec::new(),
903 span: crate::fmt::cst::Span::default(),
904 });
905
906 // Replace the inner tokens.
907 let mut result: Vec<Token> = Vec::new();
908 result.extend_from_slice(&tokens[..open_idx + 1]); // up to and including `{`
909 result.extend(new_inner);
910 result.extend_from_slice(&tokens[close_idx..]); // `}` and beyond
911 *tokens = result;
912}
913
914// ── Helpers ──────────────────────────────────────────────────────
915
916/// Recursively collect all leaf tokens from a CST node.
917fn collect_tokens(node: &CstNode, out: &mut Vec<Token>) {
918 for child in &node.children {
919 match child {
920 CstChild::Token(tok) => out.push(tok.clone()),
921 CstChild::Node { node: inner, .. } => collect_tokens(inner, out),
922 }
923 }
924}
925
926/// Emit just the token text (no trivia-based whitespace).
927/// Comments from trivia are preserved; whitespace/newlines are
928/// discarded because the Doc controls all spacing.
929fn token_text(tok: &Token) -> Doc {
930 let mut docs = Vec::new();
931
932 // Preserve comments from trivia.
933 for t in &tok.leading_trivia {
934 match t {
935 Trivia::LineComment(c) => {
936 docs.push(doc::hardline());
937 docs.push(doc::text(c.as_str()));
938 docs.push(doc::hardline());
939 }
940 Trivia::BlockComment(c) => {
941 docs.push(doc::text(" "));
942 docs.push(doc::text(c.as_str()));
943 }
944 // Whitespace and newlines are handled by the Doc,
945 // not by trivia.
946 Trivia::Whitespace(_) | Trivia::Newline => {}
947 }
948 }
949
950 if !tok.text.is_empty() {
951 // Doc comments must stay on their own line — if placed
952 // inline they swallow everything to the end of the line
953 // when re-lexed.
954 if matches!(tok.kind, TokenKind::DocComment(_)) {
955 docs.push(doc::hardline());
956 docs.push(doc::text(tok.text.as_str()));
957 docs.push(doc::hardline());
958 } else {
959 docs.push(doc::text(tok.text.as_str()));
960 }
961 }
962
963 doc::concat(docs)
964}
965
966/// Split a token sequence at commas.
967fn split_by_comma(tokens: &[Token]) -> Vec<Vec<Token>> {
968 let mut groups: Vec<Vec<Token>> = Vec::new();
969 let mut current: Vec<Token> = Vec::new();
970 for tok in tokens {
971 if tok.kind == TokenKind::Punct(',') {
972 if !current.is_empty() {
973 groups.push(current);
974 current = Vec::new();
975 }
976 } else {
977 current.push(tok.clone());
978 }
979 }
980 if !current.is_empty() {
981 groups.push(current);
982 }
983 groups
984}
985
986/// Insert spaces between tokens with language-aware rules.
987/// No space before `:`, `;`, `,`, `.`, `)`, `]`, `>`.
988/// No space after `(`, `[`, `<`, `.`, `::`, `&`, `*`.
989/// Binary operators get a `line()` before or after (soft break
990/// point) so long expressions can wrap at operator boundaries.
991fn smart_spaced_tokens(tokens: &[Token]) -> Vec<Doc> {
992 smart_spaced_tokens_inner(tokens, true)
993}
994
995fn smart_spaced_tokens_inner(tokens: &[Token], chain_breaks: bool) -> Vec<Doc> {
996 let mut docs = Vec::new();
997 for (i, tok) in tokens.iter().enumerate() {
998 if i > 0 {
999 let prev = &tokens[i - 1];
1000 // Method chain break point: after `)` before `.`.
1001 if chain_breaks
1002 && matches!(prev.kind, TokenKind::Punct(')'))
1003 && matches!(tok.kind, TokenKind::Punct('.'))
1004 {
1005 docs.push(doc::line());
1006 docs.push(token_text(tok));
1007 continue;
1008 }
1009 let need_space = needs_separator_space(prev, tok)
1010 || (!suppress_space_after(prev) && !suppress_space_before(tok));
1011 if need_space {
1012 docs.push(doc::text(" "));
1013 }
1014 }
1015 docs.push(token_text(tok));
1016 }
1017 docs
1018}
1019
1020/// Precedence level of a binary operator. Higher means tighter
1021/// binding. Returns 0 for non-operators.
1022///
1023/// Oxedyne indentation rule: when a line breaks at an operator,
1024/// each lower-precedence operator indents one level deeper. So
1025/// `&&` at level 4 sits at the base, `||` at level 3 indents one
1026/// level relative to `&&`.
1027fn binop_precedence(tok: &Token) -> u8 {
1028 match &tok.kind {
1029 TokenKind::Operator(op) => match op.as_str() {
1030 "||" => 3,
1031 "&&" => 4,
1032 "==" | "!=" | "<=" | ">=" => 5,
1033 "|" => 6,
1034 "^" => 7,
1035 "<<" | ">>" => 9,
1036 "+=" | "-=" | "*=" | "/="
1037 | "%=" | "&=" | "|=" | "^="
1038 | "<<=" | ">>=" => 1,
1039 _ => 0,
1040 },
1041 TokenKind::Punct(c) => match c {
1042 '|' => 6,
1043 '^' => 7,
1044 '&' => 8,
1045 '+' | '-' => 10,
1046 '*' | '/' | '%' => 11,
1047 _ => 0,
1048 },
1049 _ => 0,
1050 }
1051}
1052
1053/// Check whether a token is a binary operator.
1054fn is_binary_op(tok: &Token) -> bool {
1055 binop_precedence(tok) > 0
1056}
1057
1058/// Check whether a token is an unambiguously binary operator.
1059///
1060/// Only matches multi-character Operator tokens (`&&`, `||`, `==`,
1061/// `+=`, etc.), not single-character Punct that could be unary
1062/// (`&`, `*`, `+`, `-`).
1063fn is_unambiguous_binary_op(tok: &Token) -> bool {
1064 matches!(&tok.kind, TokenKind::Operator(_)) && binop_precedence(tok) > 0
1065}
1066
1067/// Format a token sequence containing binary operators with
1068/// precedence-based indentation.
1069///
1070/// Oxedyne rule: each lower-precedence operator indents one level
1071/// deeper than the previous. For example:
1072///
1073/// ```text
1074/// let valid = name.len() > 0
1075/// && age >= 18
1076/// && country == "NZ"
1077/// || special_override;
1078/// ```
1079///
1080/// `&&` (precedence 4) is at indent+1, `||` (precedence 3) is at
1081/// indent+2 because it's a lower-precedence level.
1082fn format_binop_expr(tokens: &[Token], spec: &FormatSpec) -> Doc {
1083 let indent = spec.indent_width;
1084
1085 // Find the distinct precedence levels used (sorted descending).
1086 let mut prec_levels: Vec<u8> = tokens.iter()
1087 .map(|t| binop_precedence(t))
1088 .filter(|p| *p > 0)
1089 .collect();
1090 prec_levels.sort();
1091 prec_levels.dedup();
1092 prec_levels.reverse(); // Highest first.
1093
1094 if prec_levels.is_empty() {
1095 // No operators — emit normally.
1096 return doc::concat(smart_spaced_tokens(tokens));
1097 }
1098
1099 // Map each precedence level to an indent depth.
1100 // Highest precedence = 1 indent, next = 2 indents, etc.
1101 let prec_to_depth = |p: u8| -> u16 {
1102 let idx = prec_levels.iter().position(|&x| x == p).unwrap_or(0);
1103 (idx as u16 + 1) * indent
1104 };
1105
1106 // Build the doc: break before each operator (BinopBreak::Before),
1107 // nesting to the operator's depth.
1108 let mut docs = Vec::new();
1109 let mut i = 0;
1110 while i < tokens.len() {
1111 let tok = &tokens[i];
1112 let prec = binop_precedence(tok);
1113 if prec > 0 && i > 0 {
1114 // Binary operator — insert break point.
1115 let depth = prec_to_depth(prec);
1116 match spec.binop_break {
1117 BinopBreak::Before => {
1118 // Break, then indent, then operator, then space.
1119 docs.push(doc::nest(depth, doc::concat(vec![
1120 doc::line(),
1121 doc::text(tok.text.as_str()),
1122 ])));
1123 }
1124 BinopBreak::After => {
1125 // Space, then operator, then break + indent.
1126 docs.push(doc::text(" "));
1127 docs.push(doc::text(tok.text.as_str()));
1128 docs.push(doc::nest(depth, doc::line()));
1129 }
1130 }
1131 } else {
1132 if i > 0 {
1133 let prev = &tokens[i - 1];
1134 let need_space = !suppress_space_after(prev) && !suppress_space_before_ctx(tok, Some(prev));
1135 if need_space {
1136 docs.push(doc::text(" "));
1137 }
1138 }
1139 docs.push(token_text(tok));
1140 }
1141 i += 1;
1142 }
1143
1144 doc::group(doc::concat(docs))
1145}
1146
1147/// Adjacent tokens that would re-tokenize differently if the
1148/// separating space were removed. Prevents ambiguous sequences
1149/// like `:::` (from `: ::` or `:: :`).
1150fn needs_separator_space(prev: &Token, next: &Token) -> bool {
1151 (matches!(prev.kind, TokenKind::Punct(':'))
1152 && matches!(next.kind, TokenKind::Operator(ref op) if op == "::"))
1153 || (matches!(prev.kind, TokenKind::Operator(ref op) if op == "::")
1154 && matches!(next.kind, TokenKind::Punct(':')))
1155}
1156
1157/// Tokens after which no trailing space should be added.
1158fn suppress_space_after(tok: &Token) -> bool {
1159 match &tok.kind {
1160 TokenKind::Punct('(') | TokenKind::Punct('[') | TokenKind::Punct('<')
1161 | TokenKind::Punct('.') | TokenKind::Punct('&') | TokenKind::Punct('*')
1162 | TokenKind::Punct('!') | TokenKind::Punct('#')
1163 => true,
1164 TokenKind::Operator(op)
1165 if op == "::" || op == ".." || op == "..="
1166 => true,
1167 _ => false,
1168 }
1169}
1170
1171/// Context-free version of `suppress_space_before_ctx`.
1172fn suppress_space_before(tok: &Token) -> bool {
1173 suppress_space_before_ctx(tok, None)
1174}
1175
1176/// Tokens before which no leading space should be added.
1177/// When `prev` is provided, context-sensitive rules apply.
1178fn suppress_space_before_ctx(tok: &Token, prev: Option<&Token>) -> bool {
1179 match &tok.kind {
1180 TokenKind::Punct(':') | TokenKind::Punct(';') | TokenKind::Punct(',')
1181 | TokenKind::Punct('.') | TokenKind::Punct(')') | TokenKind::Punct(']')
1182 | TokenKind::Punct('>') | TokenKind::Punct('!')
1183 | TokenKind::Punct('(') | TokenKind::Punct('[')
1184 | TokenKind::Punct('<')
1185 => true,
1186 TokenKind::Operator(op) if op == "::" => {
1187 // Keep space after `:` bound separator: `E: ::std`.
1188 !matches!(prev.map(|p| &p.kind), Some(TokenKind::Punct(':')))
1189 }
1190 TokenKind::Operator(op)
1191 if op == ".." || op == "..="
1192 || op == ">>"
1193 => true,
1194 _ => false,
1195 }
1196}
1197
1198/// Check whether a body token sequence is short enough to potentially
1199/// go on one line. A body is short if it has no semicolons (except
1200/// possibly at the end) and has few tokens.
1201/// Format a parameter list with optional type-column alignment.
1202///
1203/// When `align` is true, the type column is aligned across all
1204/// parameters by padding shorter names with spaces. The alignment
1205/// only appears in the broken (vertical) layout; the flat layout
1206/// uses normal spacing.
1207///
1208/// Flat: `(len: usize, charset: &str)`
1209/// Broken: `(\n len: usize,\n charset: &str,\n)`
1210fn format_param_list(
1211 params: &[Vec<Token>],
1212 align: bool,
1213 _spec: &FormatSpec,
1214) -> Vec<Doc> {
1215 if !align {
1216 return params.iter()
1217 .map(|toks| doc::concat(smart_spaced_tokens(toks)))
1218 .collect();
1219 }
1220
1221 // Split each param at `:` into name-tokens and type-tokens.
1222 let mut parsed: Vec<(Vec<Token>, Vec<Token>)> = Vec::new();
1223 for param_toks in params {
1224 let colon_pos = param_toks.iter().position(|t| matches!(t.kind, TokenKind::Punct(':')));
1225 match colon_pos {
1226 Some(cp) => {
1227 let name_toks = param_toks[..cp].to_vec();
1228 let type_toks = param_toks[cp + 1..].to_vec();
1229 parsed.push((name_toks, type_toks));
1230 }
1231 None => {
1232 // No colon (e.g. `self`) — no alignment for this param.
1233 parsed.push((param_toks.clone(), Vec::new()));
1234 }
1235 }
1236 }
1237
1238 // Widths must be what the renderer will actually emit; see
1239 // rendered_width.
1240 let name_widths: Vec<usize> = parsed.iter()
1241 .map(|(name_toks, _)| rendered_width(name_toks))
1242 .collect();
1243 let max_name = name_widths.iter().copied().max().unwrap_or(0);
1244
1245 // Build docs: flat uses normal spacing, broken uses padded names.
1246 parsed.iter().zip(name_widths.iter()).map(|((name_toks, type_toks), &name_w)| {
1247 if type_toks.is_empty() {
1248 // No colon (self, etc.) — just emit as-is.
1249 return doc::concat(smart_spaced_tokens(name_toks));
1250 }
1251
1252 let name_doc = doc::concat(smart_spaced_tokens(name_toks));
1253 let type_doc = doc::concat(smart_spaced_tokens(type_toks));
1254
1255 // Flat: `name: type`
1256 let flat = doc::concat(vec![
1257 name_doc.clone(),
1258 doc::text(": "),
1259 type_doc.clone(),
1260 ]);
1261
1262 // Broken: `name: type` (padded to align type column).
1263 let pad_len = max_name - name_w;
1264 let pad = " ".repeat(pad_len + 1); // +1 for the space after `:`
1265 let broken = doc::concat(vec![
1266 name_doc,
1267 doc::text(":"),
1268 doc::text(pad.as_str()),
1269 type_doc,
1270 ]);
1271
1272 doc::if_break(flat, broken)
1273 }).collect()
1274}
1275
1276/// Check whether the `{` at position `brace_pos` is preceded by
1277/// a `struct` or `enum` keyword (possibly with intervening ident,
1278/// generics, where clause, etc.).
1279/// Format match arms with `=>` column alignment.
1280///
1281/// Splits the arm tokens at commas (respecting brace depth), then
1282/// aligns the `=>` across all simple (single-line) arms.
1283fn format_match_arms(tokens: &[Token], spec: &FormatSpec) -> Doc {
1284 // Split into individual arms at top-level commas.
1285 let mut arms: Vec<Vec<Token>> = split_by_comma_depth(tokens)
1286 .into_iter().filter(|a| !a.is_empty()).collect();
1287
1288 // A trailing group of carried comments is the block's last comment,
1289 // not an arm. Left in, it would be given an arm's comma, which is a
1290 // bare `,` on a line of its own and does not compile.
1291 let mut tail: Vec<Token> = Vec::new();
1292 while arms.last().map(|a| is_comment_only(a)).unwrap_or(false) {
1293 if let Some(a) = arms.pop() {
1294 tail.splice(0..0, a);
1295 }
1296 }
1297 let arms = arms;
1298
1299 if arms.is_empty() {
1300 let mut docs = Vec::new();
1301 emit_carried_comments(&mut docs, &tail, false);
1302 return doc::concat(docs);
1303 }
1304
1305 // Parse each arm: split prefix, find `=>`, split pattern/body.
1306 struct ArmParts {
1307 prefix: Vec<Token>,
1308 pattern: Vec<Token>,
1309 body: Vec<Token>,
1310 has_arrow: bool,
1311 }
1312 let mut parsed: Vec<ArmParts> = Vec::new();
1313 for arm_toks in &arms {
1314 let (prefix, content) = split_field_prefix(arm_toks);
1315 // Find `=>` at top-level depth.
1316 let mut arrow_pos = None;
1317 let mut depth = 0usize;
1318 for (j, t) in content.iter().enumerate() {
1319 match t.kind {
1320 TokenKind::Punct('(') | TokenKind::Punct('{') | TokenKind::Punct('[') => depth += 1,
1321 TokenKind::Punct(')') | TokenKind::Punct('}') | TokenKind::Punct(']') => {
1322 depth = depth.saturating_sub(1);
1323 }
1324 _ => {}
1325 }
1326 if depth == 0 && matches!(t.kind, TokenKind::Operator(ref op) if op == "=>") {
1327 arrow_pos = Some(j);
1328 break;
1329 }
1330 }
1331 match arrow_pos {
1332 Some(ap) => {
1333 parsed.push(ArmParts {
1334 prefix,
1335 pattern: content[..ap].to_vec(),
1336 body: content[ap + 1..].to_vec(),
1337 has_arrow: true,
1338 });
1339 }
1340 None => {
1341 parsed.push(ArmParts {
1342 prefix,
1343 pattern: content,
1344 body: Vec::new(),
1345 has_arrow: false,
1346 });
1347 }
1348 }
1349 }
1350
1351 // Pattern widths for alignment (prefix excluded). Measured as the
1352 // renderer will emit them: `Some(v)` is seven columns, not ten.
1353 let pat_widths: Vec<usize> = parsed.iter()
1354 .map(|a| rendered_width(&a.pattern))
1355 .collect();
1356 let max_pat = pat_widths.iter().copied().max().unwrap_or(0);
1357 let min_pat = pat_widths.iter().copied().min().unwrap_or(0);
1358 let align = (max_pat - min_pat) <= spec.field_align_threshold
1359 && parsed.iter().all(|a| a.has_arrow);
1360
1361 // Build docs.
1362 let mut docs = Vec::new();
1363 for (i, (arm, &pat_w)) in parsed.iter().zip(pat_widths.iter()).enumerate() {
1364 if i > 0 {
1365 docs.push(doc::text(","));
1366 docs.push(doc::hardline());
1367 }
1368
1369 emit_prefix(&mut docs, &arm.prefix);
1370 let pat_doc = doc::concat(smart_spaced_tokens(&arm.pattern));
1371
1372 if !arm.has_arrow {
1373 docs.push(pat_doc);
1374 continue;
1375 }
1376
1377 let body_doc = doc::concat(smart_spaced_tokens(&arm.body));
1378
1379 if align {
1380 let pad = " ".repeat(max_pat - pat_w);
1381 docs.push(doc::concat(vec![
1382 pat_doc,
1383 doc::text(pad.as_str()),
1384 doc::text(" => "),
1385 body_doc,
1386 ]));
1387 } else {
1388 docs.push(doc::concat(vec![
1389 pat_doc,
1390 doc::text(" => "),
1391 body_doc,
1392 ]));
1393 }
1394 }
1395 // Trailing comma on last arm.
1396 if spec.match_trailing_comma && !parsed.is_empty() {
1397 docs.push(doc::text(","));
1398 }
1399 emit_carried_comments(&mut docs, &tail, true);
1400 doc::concat(docs)
1401}
1402
1403/// Split a token sequence at commas, respecting bracket depth.
1404/// Unlike `split_by_comma`, this tracks `{`, `(`, `[` depth so
1405/// commas inside braced blocks don't split arms.
1406/// The column width a token run occupies once rendered.
1407///
1408/// Counting `sum(len) + n - 1` assumes a space between every pair of
1409/// tokens, which the renderer does not emit: `Some(v)` is four tokens
1410/// but seven columns, not ten. Alignment computed from the wrong width
1411/// puts the separators in different columns, which reads worse than no
1412/// alignment at all.
1413fn rendered_width(tokens: &[Token]) -> usize {
1414 let mut w = 0usize;
1415 for (i, tok) in tokens.iter().enumerate() {
1416 if i > 0 {
1417 let prev = &tokens[i - 1];
1418 if needs_separator_space(prev, tok)
1419 || (!suppress_space_before(tok) && !suppress_space_after(prev))
1420 {
1421 w += 1;
1422 }
1423 }
1424 w += tok.text.chars().count();
1425 }
1426 w
1427}
1428
1429/// Whether a field group carries no code, only carried comments.
1430///
1431/// `carry_close_comments` puts a block's last comment on a token that
1432/// renders nothing. Split by comma, that becomes a group of its own,
1433/// and treating it as a field earns it a comma it must not have.
1434fn is_comment_only(field: &[Token]) -> bool {
1435 field.iter().all(|t| t.text.is_empty())
1436}
1437
1438/// Give each field back the comment written after its comma.
1439///
1440/// A comment at the end of a field's line comes after the comma, so
1441/// the lexer attaches it to the *next* field as leading trivia. Left
1442/// there, the formatter prints it above that next field, where it
1443/// silently comes to describe the wrong one. Comments before the first
1444/// newline belong to the field just closed; everything from the
1445/// newline on is genuinely the next field's.
1446fn lift_trailing_comments(fields: &mut [Vec<Token>]) -> Vec<Vec<String>> {
1447 let mut trailing: Vec<Vec<String>> = vec![Vec::new(); fields.len()];
1448 for i in 1..fields.len() {
1449 let mut lifted: Vec<String> = Vec::new();
1450 if let Some(first) = fields[i].first_mut() {
1451 let mut keep: Vec<Trivia> = Vec::new();
1452 let mut seen_newline = false;
1453 for t in std::mem::take(&mut first.leading_trivia) {
1454 if seen_newline {
1455 keep.push(t);
1456 continue;
1457 }
1458 match &t {
1459 Trivia::Newline => {
1460 seen_newline = true;
1461 keep.push(t);
1462 }
1463 Trivia::LineComment(c) | Trivia::BlockComment(c) => {
1464 lifted.push(c.clone());
1465 }
1466 Trivia::Whitespace(_) => keep.push(t),
1467 }
1468 }
1469 first.leading_trivia = keep;
1470 }
1471 trailing[i - 1] = lifted;
1472 }
1473 trailing
1474}
1475
1476fn split_by_comma_depth(tokens: &[Token]) -> Vec<Vec<Token>> {
1477 let mut groups: Vec<Vec<Token>> = Vec::new();
1478 let mut current: Vec<Token> = Vec::new();
1479 let mut depth = 0usize;
1480 for tok in tokens {
1481 match tok.kind {
1482 TokenKind::Punct('(') | TokenKind::Punct('{') | TokenKind::Punct('[') => {
1483 depth += 1;
1484 current.push(tok.clone());
1485 }
1486 TokenKind::Punct(')') | TokenKind::Punct('}') | TokenKind::Punct(']') => {
1487 depth = depth.saturating_sub(1);
1488 current.push(tok.clone());
1489 }
1490 TokenKind::Punct(',') if depth == 0 => {
1491 if !current.is_empty() {
1492 groups.push(current);
1493 current = Vec::new();
1494 }
1495 }
1496 _ => {
1497 current.push(tok.clone());
1498 }
1499 }
1500 }
1501 if !current.is_empty() {
1502 groups.push(current);
1503 }
1504 groups
1505}
1506
1507fn is_preceded_by_struct_or_enum(tokens: &[Token], brace_pos: usize) -> bool {
1508 // Scan backwards for the nearest keyword.
1509 for j in (0..brace_pos).rev() {
1510 match &tokens[j].kind {
1511 TokenKind::Keyword(k) if k == "struct" || k == "enum" => return true,
1512 TokenKind::Keyword(_) => return false,
1513 _ => continue,
1514 }
1515 }
1516 false
1517}
1518
1519/// Format struct/enum fields with type-column alignment.
1520///
1521/// Splits the token sequence at commas, parses each field as
1522/// `name: Type`, and aligns the type column across all fields.
1523/// Format struct/enum fields with alignment.
1524///
1525/// Auto-detects the separator to align:
1526/// - `:` for struct fields (`name: Type`).
1527/// - `=` for enum discriminants (`Name = value`).
1528/// Falls back to no alignment if no separator is found.
1529/// Split a field's tokens into prefix (doc comments, attributes)
1530/// and content. Prefix items are emitted on separate lines.
1531fn split_field_prefix(tokens: &[Token]) -> (Vec<Token>, Vec<Token>) {
1532 let mut prefix = Vec::new();
1533 let mut content = Vec::new();
1534 let mut past = false;
1535 for tok in tokens {
1536 if !past && matches!(tok.kind, TokenKind::DocComment(_) | TokenKind::Attribute) {
1537 prefix.push(tok.clone());
1538 } else {
1539 past = true;
1540 content.push(tok.clone());
1541 }
1542 }
1543 (prefix, content)
1544}
1545
1546/// Emit doc comments and attributes on their own lines.
1547fn emit_prefix(docs: &mut Vec<Doc>, prefix: &[Token]) {
1548 for tok in prefix {
1549 // A plain comment written above a doc comment or an attribute
1550 // is attached to that token as leading trivia. Emitting only
1551 // the token's own text drops it -- and a `// ---- section ----`
1552 // banner above a documented field is exactly that shape.
1553 for t in &tok.leading_trivia {
1554 if let Trivia::LineComment(c) | Trivia::BlockComment(c) = t {
1555 docs.push(doc::text(c.as_str()));
1556 docs.push(doc::hardline());
1557 }
1558 }
1559 docs.push(doc::text(tok.text.as_str()));
1560 docs.push(doc::hardline());
1561 }
1562}
1563
1564/// Find the first occurrence of `ch` as a `Punct` at bracket depth 0.
1565fn find_at_depth0(tokens: &[Token], ch: char) -> Option<usize> {
1566 let mut depth = 0usize;
1567 for (i, tok) in tokens.iter().enumerate() {
1568 match tok.kind {
1569 TokenKind::Punct('(') | TokenKind::Punct('{') | TokenKind::Punct('[') => depth += 1,
1570 TokenKind::Punct(')') | TokenKind::Punct('}') | TokenKind::Punct(']') => {
1571 depth = depth.saturating_sub(1);
1572 }
1573 TokenKind::Punct(c) if c == ch && depth == 0 => return Some(i),
1574 _ => {}
1575 }
1576 }
1577 None
1578}
1579
1580/// Find the first `{` at bracket depth 0.
1581///
1582/// Unlike `find_at_depth0`, this checks the depth *before* the
1583/// open-brace increments it, so it actually finds a `{` that is
1584/// not nested inside other brackets.
1585fn find_brace_at_depth0(tokens: &[Token]) -> Option<usize> {
1586 let mut depth = 0usize;
1587 for (i, tok) in tokens.iter().enumerate() {
1588 match tok.kind {
1589 TokenKind::Punct('{') => {
1590 if depth == 0 { return Some(i); }
1591 depth += 1;
1592 }
1593 TokenKind::Punct('(') | TokenKind::Punct('[') => depth += 1,
1594 TokenKind::Punct('}') | TokenKind::Punct(')') | TokenKind::Punct(']') => {
1595 depth = depth.saturating_sub(1);
1596 }
1597 _ => {}
1598 }
1599 }
1600 None
1601}
1602
1603/// Format field or variant content, handling struct-like bodies.
1604///
1605/// When the content contains a `{` at bracket depth 0 (a struct-like
1606/// enum variant body), the part before the brace is emitted as the
1607/// variant header and the braced body is recursively formatted with
1608/// field alignment. Without a brace, falls back to smart spacing.
1609fn format_field_content(tokens: &[Token], spec: &FormatSpec) -> Doc {
1610 let brace_pos = find_brace_at_depth0(tokens);
1611 match brace_pos {
1612 Some(bp) if bp > 0 => {
1613 let header = &tokens[..bp];
1614 let header_doc = doc::concat(smart_spaced_tokens(header));
1615
1616 // Find the matching `}`.
1617 let mut depth = 0usize;
1618 let mut close = tokens.len();
1619 for j in bp..tokens.len() {
1620 match tokens[j].kind {
1621 TokenKind::Punct('{') => depth += 1,
1622 TokenKind::Punct('}') => {
1623 depth -= 1;
1624 if depth == 0 {
1625 close = j;
1626 break;
1627 }
1628 }
1629 _ => {}
1630 }
1631 }
1632
1633 let inner = &tokens[bp + 1..close];
1634 let inner_doc = format_aligned_fields(inner, spec);
1635
1636 let mut docs = vec![
1637 header_doc,
1638 doc::text(" {"),
1639 doc::nest(spec.indent_width, doc::concat(vec![
1640 doc::hardline(),
1641 inner_doc,
1642 ])),
1643 doc::hardline(),
1644 doc::text("}"),
1645 ];
1646
1647 // Tokens after `}` (rare but handle gracefully).
1648 if close + 1 < tokens.len() {
1649 docs.push(doc::concat(smart_spaced_tokens(&tokens[close + 1..])));
1650 }
1651
1652 doc::concat(docs)
1653 }
1654 _ => doc::concat(smart_spaced_tokens(tokens)),
1655 }
1656}
1657
1658fn format_aligned_fields(tokens: &[Token], spec: &FormatSpec) -> Doc {
1659 let mut fields: Vec<Vec<Token>> = split_by_comma_depth(tokens)
1660 .into_iter().filter(|f| !f.is_empty()).collect();
1661
1662 // A trailing group of carried comments is the block's last comment,
1663 // not a field, so it is emitted after the fields and earns no comma.
1664 let mut tail: Vec<Token> = Vec::new();
1665 while fields.last().map(|f| is_comment_only(f)).unwrap_or(false) {
1666 if let Some(f) = fields.pop() {
1667 tail.splice(0..0, f);
1668 }
1669 }
1670
1671 let trailing = lift_trailing_comments(&mut fields);
1672
1673 // Split each field into prefix (doc comments, attributes) and content.
1674 let split: Vec<(Vec<Token>, Vec<Token>)> = fields.iter()
1675 .map(|f| split_field_prefix(f))
1676 .collect();
1677
1678 // Detect separator at depth 0 only (ignore `:` inside `{...}`).
1679 let has_colon = split.iter().any(|(_, c)| find_at_depth0(c, ':').is_some());
1680 let has_eq = split.iter().any(|(_, c)| find_at_depth0(c, '=').is_some());
1681 let sep_char = if has_colon {
1682 Some(':')
1683 } else if has_eq {
1684 Some('=')
1685 } else {
1686 None
1687 };
1688
1689 // Split each field's content at the depth-0 separator, where there
1690 // is one. Without a separator there is nothing to align on, and the
1691 // field is emitted whole.
1692 let parsed: Vec<(Vec<Token>, Vec<Token>, Vec<Token>)> = split.iter()
1693 .map(|(prefix, content)| match sep_char.and_then(|c| find_at_depth0(content, c)) {
1694 Some(sp) => (
1695 prefix.clone(),
1696 content[..sp].to_vec(),
1697 content[sp + 1..].to_vec(),
1698 ),
1699 None => (prefix.clone(), content.clone(), Vec::new()),
1700 })
1701 .collect();
1702
1703 // Alignment is only worth doing when the names are of comparable
1704 // length; one long name would otherwise push every other value out.
1705 let name_widths: Vec<usize> = parsed.iter()
1706 .map(|(_, name_toks, _)| rendered_width(name_toks))
1707 .collect();
1708 let max_name = name_widths.iter().copied().max().unwrap_or(0);
1709 let min_name = name_widths.iter().copied().min().unwrap_or(0);
1710 let align = sep_char.is_some()
1711 && parsed.len() > 1
1712 && max_name.saturating_sub(min_name) <= spec.field_align_threshold;
1713
1714 let mut docs = Vec::new();
1715 for (i, ((prefix, name_toks, val_toks), &name_w)) in
1716 parsed.iter().zip(name_widths.iter()).enumerate()
1717 {
1718 emit_prefix(&mut docs, prefix);
1719 if val_toks.is_empty() {
1720 docs.push(format_field_content(name_toks, spec));
1721 } else {
1722 let name_doc = doc::concat(smart_spaced_tokens(name_toks));
1723 let val_doc = doc::concat(smart_spaced_tokens(val_toks));
1724 let pad = if align { " ".repeat(max_name - name_w) } else { String::new() };
1725 match sep_char {
1726 Some('=') => docs.push(doc::concat(vec![
1727 name_doc,
1728 doc::text(pad.as_str()),
1729 doc::text(" = "),
1730 val_doc,
1731 ])),
1732 _ => docs.push(doc::concat(vec![
1733 name_doc,
1734 doc::text(": "),
1735 doc::text(pad.as_str()),
1736 val_doc,
1737 ])),
1738 }
1739 }
1740 // The comma belongs to the field, and the field's own trailing
1741 // comment follows it on the same line.
1742 docs.push(doc::text(","));
1743 for c in &trailing[i] {
1744 docs.push(doc::text(" "));
1745 docs.push(doc::text(c.as_str()));
1746 }
1747 if i + 1 < parsed.len() {
1748 docs.push(doc::hardline());
1749 }
1750 }
1751
1752 // The block's last comment, which sat above the closing brace.
1753 for tok in &tail {
1754 for t in &tok.leading_trivia {
1755 match t {
1756 Trivia::LineComment(c) | Trivia::BlockComment(c) => {
1757 docs.push(doc::hardline());
1758 docs.push(doc::text(c.as_str()));
1759 }
1760 Trivia::Newline | Trivia::Whitespace(_) => {}
1761 }
1762 }
1763 }
1764
1765 doc::concat(docs)
1766}
1767
1768/// Check whether a token slice ends with a comma.
1769fn ends_with_comma(tokens: &[Token]) -> bool {
1770 tokens.last().map_or(false, |t| matches!(t.kind, TokenKind::Punct(',')))
1771}
1772
1773fn is_short_body(tokens: &[Token]) -> bool {
1774 // A body carrying a comment cannot be collapsed onto one line:
1775 // the single-line path renders through smart_spaced_tokens, which
1776 // emits no trivia, so the comment would be dropped.
1777 let has_comment = tokens
1778 .iter()
1779 .any(|t| t.leading_trivia.iter().any(is_comment_trivia));
1780 if has_comment { return false; }
1781 // Must be a single expression — no semicolons, no let, no braces.
1782 let has_semi = tokens.iter().any(|t| matches!(t.kind, TokenKind::Punct(';')));
1783 if has_semi { return false; }
1784 let has_let = tokens.iter().any(|t| matches!(t.kind, TokenKind::Keyword(ref k) if k == "let"));
1785 if has_let { return false; }
1786 let has_brace = tokens.iter().any(|t| matches!(t.kind, TokenKind::Punct('{')));
1787 if has_brace { return false; }
1788 // Short enough (heuristic: total text width).
1789 let width: usize = tokens.iter().map(|t| t.text.len() + 1).sum();
1790 width < 40
1791}
1792
1793
1794#[cfg(test)]
1795mod tests {
1796 use super::*;
1797 use crate::fmt::spec::FormatSpec;
1798
1799 fn fmt(source: &str) -> String {
1800 let spec = FormatSpec::fe2o3();
1801 format_rust(source, &spec).expect("format failed")
1802 }
1803
1804 fn fmt_with(source: &str, spec: &FormatSpec) -> String {
1805 format_rust(source, spec).expect("format failed")
1806 }
1807
1808 #[test]
1809 fn test_format_empty_fn() {
1810 let out = fmt("fn main() {}");
1811 assert!(out.contains("fn"), "output: {}", out);
1812 assert!(out.contains("main"), "output: {}", out);
1813 }
1814
1815 #[test]
1816 fn test_format_fn_with_body() {
1817 let out = fmt("fn main() { let x = 42; }");
1818 assert!(out.contains("fn main"), "output: {}", out);
1819 }
1820
1821 #[test]
1822 fn test_format_struct() {
1823 let out = fmt("struct Foo { x: u32, y: u64, }");
1824 assert!(out.contains("struct Foo"), "output: {}", out);
1825 }
1826
1827 #[test]
1828 fn test_format_use() {
1829 let out = fmt("use std::collections::HashMap;");
1830 assert!(out.contains("use"), "output: {}", out);
1831 assert!(out.contains("HashMap"), "output: {}", out);
1832 }
1833
1834 #[test]
1835 fn test_format_impl() {
1836 let out = fmt("impl Foo { fn bar(&self) {} }");
1837 assert!(out.contains("impl Foo"), "output: {}", out);
1838 }
1839
1840 #[test]
1841 fn test_format_preserves_comments() {
1842 let out = fmt("// A comment\nfn foo() {}");
1843 assert!(out.contains("// A comment"), "output: {:?}", out);
1844 }
1845
1846 #[test]
1847 fn test_format_preserves_doc_comments() {
1848 let out = fmt("/// A doc comment.\nfn foo() {}");
1849 assert!(out.contains("/// A doc comment."), "output: {:?}", out);
1850 }
1851
1852 #[test]
1853 fn test_format_multiple_items() {
1854 let out = fmt("use std::fmt;\n\nfn main() {}\n\nstruct Foo;");
1855 assert!(out.contains("use"), "output: {:?}", out);
1856 assert!(out.contains("fn main"), "output: {:?}", out);
1857 assert!(out.contains("struct Foo"), "output: {:?}", out);
1858 }
1859
1860 #[test]
1861 fn test_format_fn_with_params() {
1862 let out = fmt("fn foo(x: u32, y: u64) -> bool { true }");
1863 assert!(out.contains("fn foo"), "output: {:?}", out);
1864 assert!(out.contains("x: u32"), "output: {:?}", out);
1865 }
1866
1867 #[test]
1868 fn test_roundtrip_simple() {
1869 // Format twice — should be idempotent.
1870 let source = "fn main() {\n let x = 42;\n}";
1871 let first = fmt(source);
1872 let second = fmt(&first);
1873 assert_eq!(first, second, "not idempotent:\nfirst: {:?}\nsecond: {:?}", first, second);
1874 }
1875}