oxedyne/fe2o3/fe2o3_austenite/src/lang/lower.rs
12.6 KiB, 178 runs
created by r1870400018:36169, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Lowering from the Ingot surface tree to the block layer. |
| 2 | //! |
| 3 | //! The seam between front end and engine: a surface [`Item`](super::ast::Item) becomes a |
| 4 | //! [`doc::Block`](crate::doc::Block) the two-pass driver already knows how to set. A heading lowers to |
| 5 | //! a heading; a paragraph of plain prose to a plain [`Block::Paragraph`], and a paragraph carrying any |
| 6 | //! emphasis to a [`Block::RichParagraph`] of [`Segment`]s -- keeping the fast single-role break path |
| 7 | //! for the common case. The source spans are dropped here, since the block layer does not yet carry |
| 8 | //! them. As the surface language grows, this is where a richer item collapses to what the engine sets. |
| 9 | |
| 10 | use crate::doc::{ |
| 11 | Block, |
| 12 | ListEntry, |
| 13 | Segment, |
| 14 | }; |
| 15 | use crate::ir::Sp; |
| 16 | use crate::table::{ |
| 17 | Align, |
| 18 | Cell, |
| 19 | Row, |
| 20 | Table, |
| 21 | }; |
| 22 | |
| 23 | use super::ast::{ |
| 24 | AlignSpec, |
| 25 | FigureBody, |
| 26 | Inline, |
| 27 | Item, |
| 28 | ListItem, |
| 29 | TableSpec, |
| 30 | }; |
| 31 | |
| 32 | /// Lowers a surface item list to the block list the driver authors from. |
| 33 | pub fn blocks(items: &[Item]) -> Vec<Block> { |
| 34 | let mut out = Vec::with_capacity(items.len()); |
| 35 | for item in items { |
| 36 | match item { |
| 37 | Item::Heading { level, runs, label, .. } => out.push( |
| 38 | Block::heading_rich(*level, lower_runs(runs), label.clone())), |
| 39 | Item::Paragraph { runs, label, .. } => out.push(lower_paragraph(runs, label.clone())), |
| 40 | Item::List { ordered, items, loose, .. } => out.push(lower_list(*ordered, items, *loose)), |
| 41 | Item::Code { lines, .. } => out.push(Block::code(lines.clone())), |
| 42 | Item::Table { spec, .. } => out.push(Block::table(build_table(spec))), |
| 43 | Item::Rule { width, thickness, grey, .. } => out.push(Block::rule(*width, *thickness, *grey)), |
| 44 | Item::PageBreak { weak, .. } => out.push(Block::page_break(*weak)), |
| 45 | Item::ColBreak { weak, .. } => out.push(Block::ColBreak { weak: *weak }), |
| 46 | Item::Place { items, floating, clearance, .. } => out.push(Block::Place { |
| 47 | blocks: blocks(items), |
| 48 | floating: *floating, |
| 49 | clearance: *clearance, |
| 50 | }), |
| 51 | Item::Space { height, .. } => out.push(Block::space(*height)), |
| 52 | Item::Figure { body, caption, supplement, label, placement, .. } => { |
| 53 | let caption = caption.as_ref().map(|runs| lower_runs(runs)); |
| 54 | out.push(match body { |
| 55 | FigureBody::Table(spec) => Block::table_figure( |
| 56 | build_table(spec), caption, supplement.clone(), label.clone(), *placement), |
| 57 | FigureBody::Image { path, width, height, scale } => Block::image_figure( |
| 58 | path.clone(), *width, *height, *scale, |
| 59 | caption, supplement.clone(), label.clone(), *placement), |
| 60 | FigureBody::Code(figure) => Block::code_figure( |
| 61 | figure.clone(), caption, supplement.clone(), label.clone(), *placement), |
| 62 | }); |
| 63 | }, |
| 64 | Item::Image { path, width, height, scale, .. } => out.push( |
| 65 | Block::image(path.clone(), *width, *height, *scale)), |
| 66 | Item::SectionBanner { path, .. } => out.push(Block::section_banner(path.clone())), |
| 67 | Item::PrintGlossary { .. } => out.push(Block::Glossary), |
| 68 | Item::ClaimIndex { .. } => out.push(Block::ClaimIndex), |
| 69 | Item::Box { items, patch, placement, .. } => out.push(match placement { |
| 70 | Some(p) => Block::box_callout_float(blocks(items), patch.clone(), *p), |
| 71 | None => Block::box_callout(blocks(items), patch.clone()), |
| 72 | }), |
| 73 | Item::Scoped { patch, items } => out.push(Block::Scoped { patch: patch.clone(), blocks: blocks(items) }), |
| 74 | } |
| 75 | } |
| 76 | out |
| 77 | } |
| 78 | |
| 79 | /// Lowers a surface list, nesting and all, to a [`Block::List`]: each entry carries its own lowered runs |
| 80 | /// and its sub-lists, themselves lowered to nested [`Block::List`]s by [`blocks`]. The parent list keeps |
| 81 | /// its ordering regardless of what a child carries. |
| 82 | fn lower_list(ordered: bool, items: &[ListItem], loose: bool) -> Block { |
| 83 | let entries = items.iter() |
| 84 | .map(|it| ListEntry { segments: lower_runs(&it.runs), children: blocks(&it.children) }) |
| 85 | .collect(); |
| 86 | Block::list(ordered, entries, loose) |
| 87 | } |
| 88 | |
| 89 | /// Builds a [`Table`] from the parsed spec: the flat cells are chunked into rows of `ncols`, each cell |
| 90 | /// carrying its inline runs lowered to segments and its alignment from the [`AlignSpec`]. A header row's |
| 91 | /// cells set centred under the fixed forms; a closure is evaluated per cell, so a `(col, row) => ...` |
| 92 | /// spec sets each cell exactly as its own row/column logic dictates. |
| 93 | fn build_table(spec: &TableSpec) -> Table { |
| 94 | let ncols = spec.ncols.max(1); |
| 95 | let mut rows: Vec<Row> = Vec::new(); |
| 96 | for (r, chunk) in spec.cells.chunks(ncols).enumerate() { |
| 97 | let mut cells = Vec::with_capacity(ncols); |
| 98 | for (c, runs) in chunk.iter().enumerate() { |
| 99 | let content = lower_runs(runs); |
| 100 | cells.push(Cell::rich(content, cell_align(&spec.align, spec.header, r, c))); |
| 101 | } |
| 102 | rows.push(Row::new(cells)); |
| 103 | } |
| 104 | // A `columns: (2fr, 5fr, ...)` track list sizes the columns fractionally, as Typst does; a bare |
| 105 | // `columns: N` carries no weights and the columns fall back to content sizing. A weight list shorter |
| 106 | // than the columns is padded with content-sized zeros so every column has an entry. |
| 107 | let mut table = if spec.weights.iter().any(|&w| w > 0.0) { |
| 108 | let mut weights = spec.weights.clone(); |
| 109 | weights.resize(ncols, 0.0); |
| 110 | Table::with_weights(spec.header, rows, weights) |
| 111 | } else { |
| 112 | Table::new(spec.header, rows) |
| 113 | }; |
| 114 | // A `text(size: Npt)` wrapper (the books set their claim tables at 7 pt) and an explicit `inset:` cell |
| 115 | // padding carry through, so the table sets at the oracle's reduced size rather than the body size. |
| 116 | table.text_size = spec.text_pt.map(Sp::from_pt); |
| 117 | table.inset = spec.inset_pt.map(Sp::from_pt); |
| 118 | table |
| 119 | } |
| 120 | |
| 121 | /// The alignment of one cell at row `r`, column `c`, given the table's declared [`AlignSpec`]. A closure |
| 122 | /// carries its own row/column logic and is evaluated for every cell, header row included; the fixed |
| 123 | /// forms have no row dependence, so a header row centres its labels as Typst's book style does. |
| 124 | fn cell_align(spec: &AlignSpec, header: bool, r: usize, c: usize) -> Align { |
| 125 | if let AlignSpec::Closure(cl) = spec { |
| 126 | return cl.align_at(c, r); |
| 127 | } |
| 128 | if header && r == 0 { |
| 129 | return Align::Centre; // a header row centres its labels |
| 130 | } |
| 131 | match spec { |
| 132 | AlignSpec::Uniform(a) => *a, |
| 133 | AlignSpec::PerColumn(cols) => cols.get(c).copied().unwrap_or(Align::Left), |
| 134 | AlignSpec::Closure(cl) => cl.align_at(c, r), |
| 135 | } |
| 136 | } |
| 137 | |
| 138 | /// Lowers a paragraph's inline runs. A paragraph of one plain text run keeps the plain-paragraph path |
| 139 | /// (a single-role Knuth-Plass break); the moment it carries an emphasis run it becomes a rich paragraph |
| 140 | /// of segments, which the driver breaks with a face per run. A `label` is the paragraph's trailing |
| 141 | /// `<name>`, carried onto a display equation so an `@`-reference can resolve to it. |
| 142 | fn lower_paragraph(runs: &[Inline], label: Option<String>) -> Block { |
| 143 | match runs { |
| 144 | [Inline::Text(text)] => Block::paragraph(text.clone()), |
| 145 | // A paragraph that is nothing but one maths span is a display equation on its own line. The |
| 146 | // template sets `math.equation(numbering: "(1)")`, so every display equation takes the next |
| 147 | // number; inline maths, a run among others, never does. |
| 148 | [Inline::Math(atom)] => Block::equation(atom.clone(), true, label), |
| 149 | _ => Block::rich(lower_runs(runs)), |
| 150 | } |
| 151 | } |
| 152 | |
| 153 | /// Lowers a run of inlines to segments, then groups adjacent citations, so a `#cite ... #cite` |
| 154 | /// sequence parted by nothing but whitespace sets as one parenthesis. |
| 155 | pub(crate) fn lower_runs(runs: &[Inline]) -> Vec<Segment> { |
| 156 | group_adjacent_cites(runs.iter().map(lower_inline).collect()) |
| 157 | } |
| 158 | |
| 159 | /// Groups adjacent citations into one, as Typst does: two or more `#cite` calls separated by nothing |
| 160 | /// but whitespace set as a single parenthesis with a semicolon separator (`(A Year; B Year)`), while a |
| 161 | /// citation parted from the next by any prose keeps its own parenthesis. The whitespace between grouped |
| 162 | /// citations is dropped, matching the oracle. |
| 163 | fn group_adjacent_cites(segments: Vec<Segment>) -> Vec<Segment> { |
| 164 | // Grouping can only change a run holding two or more citations. |
| 165 | if segments.iter().filter(|s| matches!(s, Segment::Cite(_))).count() < 2 { |
| 166 | return segments; |
| 167 | } |
| 168 | let n = segments.len(); |
| 169 | let mut out: Vec<Segment> = Vec::with_capacity(n); |
| 170 | let mut i = 0usize; |
| 171 | while i < n { |
| 172 | if let Segment::Cite(keys) = &segments[i] { |
| 173 | let mut group = keys.clone(); |
| 174 | let mut j = i + 1; |
| 175 | // Absorb each following citation reachable across whitespace-only text alone. |
| 176 | loop { |
| 177 | let mut k = j; |
| 178 | while k < n && is_whitespace_text(&segments[k]) { |
| 179 | k += 1; |
| 180 | } |
| 181 | match segments.get(k) { |
| 182 | Some(Segment::Cite(more)) => { |
| 183 | group.extend(more.iter().cloned()); |
| 184 | j = k + 1; |
| 185 | }, |
| 186 | _ => break, |
| 187 | } |
| 188 | } |
| 189 | out.push(Segment::Cite(group)); |
| 190 | i = j; |
| 191 | } else { |
| 192 | out.push(segments[i].clone()); |
| 193 | i += 1; |
| 194 | } |
| 195 | } |
| 196 | out |
| 197 | } |
| 198 | |
| 199 | /// Is the segment a text run of nothing but whitespace? |
| 200 | fn is_whitespace_text(seg: &Segment) -> bool { |
| 201 | matches!(seg, Segment::Text(t) if t.chars().all(char::is_whitespace)) |
| 202 | } |
| 203 | |
| 204 | fn lower_inline(run: &Inline) -> Segment { |
| 205 | match run { |
| 206 | Inline::Text(text) => Segment::text(text.clone()), |
| 207 | Inline::Strong(text) => Segment::strong(text.clone()), |
| 208 | Inline::Emph(text) => Segment::emph(text.clone()), |
| 209 | Inline::BoldItalic(text) => Segment::bold_italic(text.clone()), |
| 210 | Inline::Super(text) => Segment::superscript(text.clone()), |
| 211 | Inline::SmallCaps(text) => Segment::SmallCaps(text.clone()), |
| 212 | Inline::Sub(text) => Segment::subscript(text.clone()), |
| 213 | Inline::PageRef(label) => Segment::page_ref(label.clone()), |
| 214 | Inline::Code(text) => Segment::code(text.clone()), |
| 215 | Inline::Math(atom) => Segment::math(atom.clone()), |
| 216 | Inline::Glossary { term, display } |
| 217 | => Segment::glossary(term.clone(), display.clone()), |
| 218 | Inline::Index { term, sub, display, main } |
| 219 | => Segment::index(term.clone(), sub.clone(), *main, lower_runs(display)), |
| 220 | Inline::Footnote(note) => Segment::footnote(lower_runs(note)), |
| 221 | Inline::Cite(keys) => Segment::cite(keys.clone()), |
| 222 | Inline::MarginNote { display, codes } => Segment::margin_note(display.clone(), codes.clone()), |
| 223 | } |
| 224 | } |
| 225 | |
| 226 | #[cfg(test)] |
| 227 | mod tests { |
| 228 | use super::*; |
| 229 | |
| 230 | fn cite(k: &str) -> Inline { Inline::Cite(vec![k.to_string()]) } |
| 231 | |
| 232 | /// Adjacent citations parted by nothing but whitespace collapse to one citation segment carrying |
| 233 | /// every key in source order, the whitespace between them dropped. |
| 234 | #[test] |
| 235 | fn adjacent_cites_group() { |
| 236 | let runs = vec![ |
| 237 | Inline::Text("A ".to_string()), |
| 238 | cite("a"), |
| 239 | Inline::Text(" ".to_string()), |
| 240 | cite("b"), |
| 241 | Inline::Text(" end.".to_string()), |
| 242 | ]; |
| 243 | let segs = lower_runs(&runs); |
| 244 | assert_eq!(segs.len(), 3, "got: {:?}", segs); |
| 245 | assert!(matches!(&segs[0], Segment::Text(t) if t == "A ")); |
| 246 | match &segs[1] { |
| 247 | Segment::Cite(keys) => assert_eq!(keys, &vec!["a".to_string(), "b".to_string()]), |
| 248 | other => panic!("expected one grouped cite, got {:?}", other), |
| 249 | } |
| 250 | assert!(matches!(&segs[2], Segment::Text(t) if t == " end.")); |
| 251 | } |
| 252 | |
| 253 | /// Three adjacent citations all fall into one group. |
| 254 | #[test] |
| 255 | fn three_adjacent_cites_group() { |
| 256 | let runs = vec![ |
| 257 | cite("a"), |
| 258 | Inline::Text(" ".to_string()), |
| 259 | cite("b"), |
| 260 | Inline::Text(" ".to_string()), |
| 261 | cite("c"), |
| 262 | ]; |
| 263 | let segs = lower_runs(&runs); |
| 264 | assert_eq!(segs.len(), 1, "got: {:?}", segs); |
| 265 | match &segs[0] { |
| 266 | Segment::Cite(keys) => assert_eq!( |
| 267 | keys, &vec!["a".to_string(), "b".to_string(), "c".to_string()]), |
| 268 | other => panic!("expected one grouped cite, got {:?}", other), |
| 269 | } |
| 270 | } |
| 271 | |
| 272 | /// A citation, then prose, then a citation stays two separate citation segments -- no over-grouping. |
| 273 | #[test] |
| 274 | fn cite_prose_cite_stays_separate() { |
| 275 | let runs = vec![ |
| 276 | cite("a"), |
| 277 | Inline::Text(" and then ".to_string()), |
| 278 | cite("b"), |
| 279 | ]; |
| 280 | let segs = lower_runs(&runs); |
| 281 | let cites: Vec<&Segment> = segs.iter() |
| 282 | .filter(|s| matches!(s, Segment::Cite(_))) |
| 283 | .collect(); |
| 284 | assert_eq!(cites.len(), 2, "expected two separate cites, got: {:?}", segs); |
| 285 | assert!(matches!(&cites[0], Segment::Cite(k) if k == &vec!["a".to_string()])); |
| 286 | assert!(matches!(&cites[1], Segment::Cite(k) if k == &vec!["b".to_string()])); |
| 287 | // The intervening prose survives. |
| 288 | assert!(segs.iter().any(|s| matches!(s, Segment::Text(t) if t == " and then "))); |
| 289 | } |
| 290 | |
| 291 | /// A single citation is left untouched. |
| 292 | #[test] |
| 293 | fn lone_cite_untouched() { |
| 294 | let runs = vec![ |
| 295 | Inline::Text("D ".to_string()), |
| 296 | cite("a"), |
| 297 | Inline::Text(" end.".to_string()), |
| 298 | ]; |
| 299 | let segs = lower_runs(&runs); |
| 300 | assert_eq!(segs.len(), 3); |
| 301 | assert!(matches!(&segs[1], Segment::Cite(k) if k == &vec!["a".to_string()])); |
| 302 | // The trailing space after a lone cite is preserved. |
| 303 | assert!(matches!(&segs[2], Segment::Text(t) if t == " end.")); |
| 304 | } |
| 305 | |
| 306 | /// A `#cite(<a>, <b>)` multi-key call already lowers to one segment; grouping leaves it whole. |
| 307 | #[test] |
| 308 | fn multikey_call_unchanged() { |
| 309 | let runs = vec![Inline::Cite(vec!["a".to_string(), "b".to_string()])]; |
| 310 | let segs = lower_runs(&runs); |
| 311 | assert_eq!(segs.len(), 1); |
| 312 | assert!(matches!(&segs[0], Segment::Cite(k) if k == &vec!["a".to_string(), "b".to_string()])); |
| 313 | } |
| 314 | } |