oxedyne/fe2o3/fe2o3_austenite/src/lang/ast.rs
15.1 KiB, 140 runs
created by r1870400018:36145, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! The surface syntax tree of a Typst source file, before lowering to [`doc::Block`](crate::doc::Block). |
| 2 | //! |
| 3 | //! Increment 2 adds inline emphasis to the markup spine: a document is a sequence of headings and |
| 4 | //! paragraphs, and a paragraph is a sequence of [`Inline`] runs -- plain text, `*strong*`, `_emph_`. |
| 5 | //! The `#` code mode and the declared-query references of `sec_language` are later increments -- the |
| 6 | //! tree names only what the engine can already set. |
| 7 | |
| 8 | use crate::ir::FloatPlacement; |
| 9 | use crate::ir::Floating; |
| 10 | use crate::ir::Length; |
| 11 | use crate::ir::Span; |
| 12 | use crate::ir::Sp; |
| 13 | use crate::lang::codefig::CodeFigure; |
| 14 | use crate::math::Atom; |
| 15 | use crate::table::Align; |
| 16 | use crate::theme::ThemePatch; |
| 17 | |
| 18 | /// One block of Ingot markup. The byte span is carried for a future diagnostic caret; the driver's |
| 19 | /// `Span` model already reserves it, so the front end records it from the first increment. |
| 20 | #[derive(Clone, Debug)] |
| 21 | pub enum Item { |
| 22 | Heading { level: u8, runs: Vec<Inline>, label: Option<String>, span: Span }, // label: a trailing <name>, runs: the title's inline markup |
| 23 | Paragraph { runs: Vec<Inline>, label: Option<String>, span: Span }, // label: a trailing <name>, anchoring a display equation for cross-reference |
| 24 | List { ordered: bool, items: Vec<ListItem>, loose: bool, span: Span }, // `-` bullets or `+` numbered; items may nest sub-lists by indent; loose when a blank line parts its items |
| 25 | Code { lines: Vec<String>, span: Span }, // a ```-fenced block, set verbatim in the mono face |
| 26 | Table { spec: TableSpec, span: Span }, // a bare `#table(...)`, not wrapped in a figure |
| 27 | // caption: the caption's inline markup; placement: Some when floated (`figure(placement: auto|top|bottom)`), |
| 28 | // carrying its `scope:` -- a parent-scoped float spans every column of the page |
| 29 | Figure { body: FigureBody, caption: Option<Vec<Inline>>, supplement: String, label: Option<String>, placement: Option<Floating>, span: Span }, |
| 30 | Image { path: String, width: Option<Length>, height: Option<Length>, scale: Option<f64>, span: Span }, // a line-leading `#padded-image(...)`/`#image(...)`, set centred without a figure number |
| 31 | SectionBanner { path: String, span: Span }, // a line-leading `#section-banner("logo")`, a full-width grey bar carrying a right-aligned section logo |
| 32 | Rule { width: Length, thickness: f64, grey: u8, span: Span }, // a standalone `#line(length:.., stroke:..)` horizontal divider |
| 33 | PageBreak { weak: bool, span: Span }, |
| 34 | ColBreak { weak: bool, span: Span }, // a line-leading `#colbreak()`: the next column, or a page break on a page of one column |
| 35 | // A line-leading `#place(<side>, float: true, scope: .., clearance: ..)[ ... ]`: its body set as a float at |
| 36 | // the top or foot of its column, or -- `scope: "parent"` -- spanning every column of the page. |
| 37 | Place { items: Vec<Item>, floating: Floating, clearance: Option<Spacing>, span: Span }, // a line-leading `#pagebreak()` (strong) or `#pagebreak(weak: true)`; a strong break always ejects, a weak one only a page carrying content |
| 38 | Space { height: Sp, span: Span }, // a line-leading `#v(<abs len>)`, a fixed vertical space (absolute units only) |
| 39 | PrintGlossary { span: Span }, // a line-leading `#print-glossary()`, a placeholder the book layer fills with the Term/Definition table |
| 40 | // A line-leading `#context { ... collect-claim-refs() ... }`: the reverse claim-reference index the Logic |
| 41 | // appendix builds. Recognised by the signature `collect-claim-refs(` in the block, not by evaluating the |
| 42 | // `#context` (the reader is hard-stratified and runs no query); it lowers to a `Block::ClaimIndex` the |
| 43 | // author fills from the claim references gathered walking the body. |
| 44 | ClaimIndex { span: Span }, |
| 45 | // a `#styled-box[...]` callout: its body re-parsed into items, set in a filled padded box, plus the |
| 46 | // theme patch its body's own top-level `#set` declarations lower to, scoped to the box (H3). placement is |
| 47 | // Some when the furniture floats its body (an `#aside-box(float: true)` re-wrapped in a floating figure). |
| 48 | Box { items: Vec<Item>, patch: ThemePatch, placement: Option<FloatPlacement>, span: Span }, |
| 49 | // A theme scope wrapping the items it governs -- a `#columns[...]` body whose own `#set` declarations |
| 50 | // scope to it (H1's flat-splice sibling). Nesting the items rather than bracketing them with a separate |
| 51 | // open/close marker makes an unmatched or missing close structurally impossible. Lowered to |
| 52 | // `Block::Scoped`. |
| 53 | Scoped { patch: ThemePatch, items: Vec<Item> }, |
| 54 | } |
| 55 | |
| 56 | /// A length that may be relative to the font size: absolute points, or ems of the text size in force. |
| 57 | #[derive(Clone, Copy, Debug, PartialEq)] |
| 58 | pub enum Spacing { |
| 59 | Pt(f64), |
| 60 | Em(f64), |
| 61 | } |
| 62 | |
| 63 | /// One item of an [`Item::List`]: its own inline runs, and any lists nested beneath it by deeper marker |
| 64 | /// indentation. `children` holds nested `Item::List`s, so a `+` step carrying indented `-` sub-bullets |
| 65 | /// keeps them under the step rather than breaking the parent enumeration. |
| 66 | #[derive(Clone, Debug)] |
| 67 | pub struct ListItem { |
| 68 | pub runs: Vec<Inline>, |
| 69 | pub children: Vec<Item>, |
| 70 | } |
| 71 | |
| 72 | /// What a `#figure(...)` wraps: a `#table(...)` this reader sets in full, or an image call whose ink is |
| 73 | /// deferred to a later increment and stood in for by a sized placeholder box. |
| 74 | #[derive(Clone, Debug)] |
| 75 | pub enum FigureBody { |
| 76 | Table(TableSpec), |
| 77 | // The image path and any sizing the call declared: `width`/`height` from `image(...)`, `scale` from |
| 78 | // `padded-image(...)`. A hint the call omits is `None`, and the figure fills the measure. |
| 79 | Image { path: String, width: Option<Length>, height: Option<Length>, scale: Option<f64> }, |
| 80 | // A figure the document draws by code -- a CeTZ/Fletcher diagram, a bar chart or a line plot -- read |
| 81 | // from the `#figure` body's source into a ready builder that draws real vector ink. |
| 82 | Code(CodeFigure), |
| 83 | } |
| 84 | |
| 85 | /// A parsed Typst `#table(...)` call, before it is built into a [`table::Table`](crate::table::Table). |
| 86 | /// Each cell keeps its inline runs, row-major, so a bold header, an italic word, a superscript or an |
| 87 | /// in-cell maths span sets with its own face rather than flattening to upright text; `header` is set when |
| 88 | /// a `fill:` keys the first row; `align` records the column alignment the call declared. |
| 89 | #[derive(Clone, Debug)] |
| 90 | pub struct TableSpec { |
| 91 | pub ncols: usize, |
| 92 | pub header: bool, |
| 93 | pub align: AlignSpec, |
| 94 | pub weights: Vec<f64>, // the `Nfr` weight per column; 0.0 for an `auto`/fixed track, empty for a bare `columns: N` |
| 95 | pub text_pt: Option<f64>, // a `text(size: Npt)[...]` wrapper's size, so a small table sets small |
| 96 | pub inset_pt: Option<f64>, // the `inset:` cell padding in points, overriding the default |
| 97 | pub cells: Vec<Vec<Inline>>, // flat, row-major; each cell a run of inline markup |
| 98 | } |
| 99 | |
| 100 | /// How a table's cells align. `Uniform` sets every cell alike; `PerColumn` gives each column its own |
| 101 | /// alignment (a header row still sets centred); `Closure` carries the `(col, row) => ...` idiom these |
| 102 | /// books use, evaluated per cell so the source's own row/column logic is honoured rather than guessed. |
| 103 | #[derive(Clone, Debug)] |
| 104 | pub enum AlignSpec { |
| 105 | Uniform(Align), |
| 106 | PerColumn(Vec<Align>), |
| 107 | Closure(ClosureAlign), |
| 108 | } |
| 109 | |
| 110 | /// A table `align:` closure captured for evaluation at each cell. Typst passes the closure the cell's |
| 111 | /// 0-based `(col, row)` and expects an alignment back; the parameter names and body are kept verbatim so |
| 112 | /// a cell's alignment is computed from the expression the source declared. The supported body is the |
| 113 | /// idiom these books use: an `if`/`else if`/`else` chain over `col`/`row` comparisons (`==`, `!=`, `<`, |
| 114 | /// `<=`, `>`, `>=`, combined with `or`/`and`), and alignment expressions that are a word (`left`, |
| 115 | /// `center`, `right`, with any `+ horizon`/`+ top` ignored) or a `(a, b, ...).at(col)` tuple pick. |
| 116 | #[derive(Clone, Debug)] |
| 117 | pub struct ClosureAlign { |
| 118 | pub col_var: String, |
| 119 | pub row_var: String, |
| 120 | pub body: String, |
| 121 | } |
| 122 | |
| 123 | impl ClosureAlign { |
| 124 | /// The alignment the closure yields for the cell at 0-based `col`, `row`. An expression the grammar |
| 125 | /// does not cover falls back to left, the Typst default for an unrecognised alignment. |
| 126 | pub fn align_at(&self, col: usize, row: usize) -> Align { |
| 127 | self.eval_body(self.body.trim(), col, row) |
| 128 | } |
| 129 | |
| 130 | fn eval_body(&self, s: &str, col: usize, row: usize) -> Align { |
| 131 | let s = strip_braces(s.trim()).trim(); |
| 132 | if let Some(rest) = s.strip_prefix("if ").or_else(|| s.strip_prefix("if(")) { |
| 133 | // The condition runs to the top-level `{` that opens the then-block. |
| 134 | let chars: Vec<char> = rest.chars().collect(); |
| 135 | if let Some(open) = find_top_brace(&chars) { |
| 136 | let cond: String = chars[..open].iter().collect(); |
| 137 | if let Some((block, after)) = read_brace_group(&chars, open) { |
| 138 | if self.eval_cond(&cond, col, row) { |
| 139 | return self.eval_body(&block, col, row); |
| 140 | } |
| 141 | let tail: String = chars[after..].iter().collect(); |
| 142 | let tail = tail.trim(); |
| 143 | if let Some(e) = tail.strip_prefix("else") { |
| 144 | return self.eval_body(e.trim(), col, row); |
| 145 | } |
| 146 | return Align::Left; // an `if` with no `else` leaves the rest flush left |
| 147 | } |
| 148 | } |
| 149 | return Align::Left; |
| 150 | } |
| 151 | self.eval_expr(s, col, row) |
| 152 | } |
| 153 | |
| 154 | fn eval_expr(&self, s: &str, col: usize, row: usize) -> Align { |
| 155 | let s = s.trim(); |
| 156 | // A `(a, b, ...).at(index)` tuple pick: split the tuple, index it by the resolved argument. |
| 157 | if let Some(pos) = s.find(".at(") { |
| 158 | let tuple = s[..pos].trim(); |
| 159 | let idx_src: String = s[pos + 4..].chars().take_while(|&c| c != ')').collect(); |
| 160 | let items = split_top_commas(strip_parens(tuple)); |
| 161 | let idx = self.eval_index(&idx_src, col, row); |
| 162 | return items.get(idx).map_or(Align::Left, |it| word_to_align(it)); |
| 163 | } |
| 164 | word_to_align(s) |
| 165 | } |
| 166 | |
| 167 | fn eval_index(&self, s: &str, col: usize, row: usize) -> usize { |
| 168 | let s = s.trim(); |
| 169 | if s == self.col_var { return col; } |
| 170 | if s == self.row_var { return row; } |
| 171 | s.parse::<usize>().unwrap_or(0) |
| 172 | } |
| 173 | |
| 174 | fn eval_cond(&self, s: &str, col: usize, row: usize) -> bool { |
| 175 | // `or` is the loosest binding, then `and`, then a single comparison. |
| 176 | if s.contains(" or ") { |
| 177 | return s.split(" or ").any(|p| self.eval_cond(p, col, row)); |
| 178 | } |
| 179 | if s.contains(" and ") { |
| 180 | return s.split(" and ").all(|p| self.eval_cond(p, col, row)); |
| 181 | } |
| 182 | self.eval_atom(s.trim(), col, row) |
| 183 | } |
| 184 | |
| 185 | fn eval_atom(&self, s: &str, col: usize, row: usize) -> bool { |
| 186 | for op in ["==", "!=", "<=", ">=", "<", ">"] { |
| 187 | if let Some((lhs, rhs)) = s.split_once(op) { |
| 188 | let lv = self.eval_index(lhs.trim(), col, row); |
| 189 | let rv = match rhs.trim().parse::<usize>() { |
| 190 | Ok(n) => n, |
| 191 | Err(_) => return false, |
| 192 | }; |
| 193 | return match op { |
| 194 | "==" => lv == rv, |
| 195 | "!=" => lv != rv, |
| 196 | "<=" => lv <= rv, |
| 197 | ">=" => lv >= rv, |
| 198 | "<" => lv < rv, |
| 199 | ">" => lv > rv, |
| 200 | _ => false, |
| 201 | }; |
| 202 | } |
| 203 | } |
| 204 | false |
| 205 | } |
| 206 | } |
| 207 | |
| 208 | /// Maps a Typst alignment word to an [`Align`], ignoring a `+ horizon`/`+ top` vertical component and |
| 209 | /// treating `start`/`end` as left/right. An unknown word is left-aligned, the Typst default. |
| 210 | fn word_to_align(s: &str) -> Align { |
| 211 | let first = s.trim().split(|c: char| c.is_whitespace() || c == '+').next().unwrap_or("").trim(); |
| 212 | match first { |
| 213 | "center" | "centre" => Align::Centre, |
| 214 | "right" | "end" => Align::Right, |
| 215 | _ => Align::Left, |
| 216 | } |
| 217 | } |
| 218 | |
| 219 | /// Strips one matching outer `{ ... }` layer, leaving other text untouched. |
| 220 | fn strip_braces(s: &str) -> &str { |
| 221 | let t = s.trim(); |
| 222 | match (t.strip_prefix('{'), t.strip_suffix('}')) { |
| 223 | (Some(_), Some(_)) => &t[1..t.len() - 1], |
| 224 | _ => t, |
| 225 | } |
| 226 | } |
| 227 | |
| 228 | /// Strips one matching outer `( ... )` layer. |
| 229 | fn strip_parens(s: &str) -> &str { |
| 230 | let t = s.trim(); |
| 231 | match (t.strip_prefix('('), t.strip_suffix(')')) { |
| 232 | (Some(_), Some(_)) => &t[1..t.len() - 1], |
| 233 | _ => t, |
| 234 | } |
| 235 | } |
| 236 | |
| 237 | /// The index of the first `{` not nested inside parentheses, or `None`. |
| 238 | fn find_top_brace(chars: &[char]) -> Option<usize> { |
| 239 | let mut depth = 0i32; |
| 240 | for (i, &c) in chars.iter().enumerate() { |
| 241 | match c { |
| 242 | '(' | '[' => depth += 1, |
| 243 | ')' | ']' => depth -= 1, |
| 244 | '{' if depth == 0 => return Some(i), |
| 245 | _ => {}, |
| 246 | } |
| 247 | } |
| 248 | None |
| 249 | } |
| 250 | |
| 251 | /// Reads the `{ ... }` group whose opening brace sits at `open`, returning its inner text and the index |
| 252 | /// just past the closing brace. |
| 253 | fn read_brace_group(chars: &[char], open: usize) -> Option<(String, usize)> { |
| 254 | let mut depth = 0i32; |
| 255 | for i in open..chars.len() { |
| 256 | match chars[i] { |
| 257 | '{' => depth += 1, |
| 258 | '}' => { |
| 259 | depth -= 1; |
| 260 | if depth == 0 { |
| 261 | let inner: String = chars[open + 1..i].iter().collect(); |
| 262 | return Some((inner, i + 1)); |
| 263 | } |
| 264 | }, |
| 265 | _ => {}, |
| 266 | } |
| 267 | } |
| 268 | None |
| 269 | } |
| 270 | |
| 271 | /// Splits a tuple's inner text on top-level commas, ignoring commas nested in brackets. |
| 272 | fn split_top_commas(s: &str) -> Vec<String> { |
| 273 | let mut out = Vec::new(); |
| 274 | let mut depth = 0i32; |
| 275 | let mut cur = String::new(); |
| 276 | for c in s.chars() { |
| 277 | match c { |
| 278 | '(' | '[' | '{' => { depth += 1; cur.push(c); }, |
| 279 | ')' | ']' | '}' => { depth -= 1; cur.push(c); }, |
| 280 | ',' if depth == 0 => out.push(std::mem::take(&mut cur)), |
| 281 | _ => cur.push(c), |
| 282 | } |
| 283 | } |
| 284 | if !cur.trim().is_empty() { |
| 285 | out.push(cur); |
| 286 | } |
| 287 | out |
| 288 | } |
| 289 | |
| 290 | /// One inline run of a paragraph: ordinary prose, a run marked for emphasis, a cross-reference, or an |
| 291 | /// inline code span. A run's text is flat; the one nesting the vocabulary carries is strong-within-emph |
| 292 | /// or emph-within-strong (`*_x_*`, `_*x*_`), which collapses to a single [`Inline::BoldItalic`] run. |
| 293 | #[derive(Clone, Debug)] |
| 294 | pub enum Inline { |
| 295 | Text(String), |
| 296 | Strong(String), // *strong*, lowered to a bold segment |
| 297 | Emph(String), // _emph_ or #emph[...], lowered to an italic segment |
| 298 | BoldItalic(String), // *_x_* or _*x*_, the two faces nested, lowered to a bold-italic segment |
| 299 | Super(String), // #super[...], lowered to a raised, smaller segment |
| 300 | Sub(String), // #sub[...], lowered to a dropped, smaller segment |
| 301 | SmallCaps(String), // #smallcaps[...], lowered to a segment shaped with the font's `smcp` feature |
| 302 | PageRef(String), // @label, resolving to the labelled anchor's page number |
| 303 | Code(String), // `raw` or #raw("..."), set in the mono face |
| 304 | Math(Atom), // $...$, parsed to the engine's maths tree |
| 305 | Glossary { term: String, display: String }, // a glossary term: bold-italic on its first document use |
| 306 | // An index marker (`#index`, `#idx`, `#gsi`, `#idx-nested`, ...): records the term's occurrence for the |
| 307 | // back-matter index and sets nothing itself in the body. `term` is the sort key (markup flattened); `sub` |
| 308 | // carries a nested entry's child term; `display` is the styled runs the index page sets for this entry, so |
| 309 | // a `#idx-as[March, James][James March]` prints "James March" and a `#idx[_Case_]` sets italic there. A |
| 310 | // visible call sets that same display in the body too, as a separate run beside this marker. `main` marks a |
| 311 | // primary reference (`#idx-main`, `#index-main`, `#idx-main-as`): its folio sets bold on the index page, as |
| 312 | // in-dexter's `index-main = index.with(fmt: strong)` renders a main reference's page number. |
| 313 | Index { term: String, sub: Option<String>, display: Vec<Inline>, main: bool }, |
| 314 | Footnote(Vec<Inline>), // #footnote[...], its note markup set at the foot of the page its mark lands on |
| 315 | Cite(Vec<String>), // #cite(<key>) or #cite(<a>, <b>), resolved to (Author Year) against the bibliography |
| 316 | // #claim-label(<code>..) or #claim-refs(<code>..): a zero-width margin anchor. `display` is the compressed |
| 317 | // code a `#claim-label` draws in the outside margin (empty for a metadata-only `#claim-refs`); `codes` are |
| 318 | // the raw reference codes a `#claim-refs` registers for the reverse claim index (empty for a `#claim-label`). |
| 319 | MarginNote { display: String, codes: Vec<String> }, |
| 320 | } |