Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_austenite/src/lang/parse.rs

281 KiB, 1279 runs

created by r1870400018:36173, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! The Typst reader: line-oriented source into the surface tree of [`ast::Item`](super::ast::Item).
2//!
3//! The scan is deliberately simple, one pass over the lines. A line whose first non-blank character
4//! is `=` is a heading, its level the run of leading `=`; a line opening with `-` or `+` (and a space)
5//! is a list item, a run of them a bullet or numbered list; every other non-blank line accumulates into
6//! a paragraph. A blank line, a heading, or the start of the other kind of block closes whatever is
7//! open. Whitespace within a paragraph is made insignificant here, so the line breaker downstream owns
8//! the measure. Byte offsets are tracked across the raw lines so each [`Item`] carries a true [`Span`].
9//!
10//! A closed paragraph's text is then scanned for inline emphasis by [`parse_inlines`]: `*strong*` and
11//! `_emph_`. A delimiter pairs only when it flanks a word, so a stray asterisk, a date's slash, or
12//! `and/or` is left as ordinary text rather than opening an emphasis that never closes.
13//!
14//! A Typst code statement (`#import`/`#let`/`#set`/`#show`) or a line-leading standalone template call
15//! (`#name(...)` or `#name[...]`) is not set: it is skipped. When its delimiters do not balance on the
16//! opening line -- a `#figure(...)`, `#table(...)`, `#aside-box[...]`, or a `#let x = (...)` data array
17//! that spans many lines -- the reader consumes following lines, tracking nesting across `()`, `[]` and
18//! `{}` and respecting string literals, until the delimiters balance, so the whole span renders nothing.
19//! Every such skip is recorded by name into a [`Refusals`] the parse returns beside its items, so a
20//! caller reports the constructs it dropped rather than losing them silently. A `#columns(n)[ ... ]`
21//! wrapper is the exception the reader does not drop whole: its body is re-parsed and set single-column.
22//!
23//! Typst comments are stripped before a line is classified: a `//` runs to the line's end, and a
24//! `/* ... */` spans lines, both dropped -- except within a `"..."` string or a `` `code` `` span, and a
25//! `//` right after `:` is kept, so a bare URL survives. Inline glossary and index calls, defined in the
26//! book template (`#gs`, `#gscap`, `#gsi`, `#gscapi`, `#glossind`, `#glossindcap`, the term-dictionary
27//! family `#g`, `#gcap`, `#gi`, `#gcapi`, `#t`, `#tcap`, `#graw`, and `#idx`, `#idx-main`, `#idx-as`,
28//! `#idx-main-as`, `#index`, `#index-main`, `#idx-nested`), plus a `#link(dest)[text]` hyperlink, are read
29//! by [`parse_inlines`]: a glossary term sets its display text, bold-italic on its first document use; a
30//! visible term or index call sets its display text plain; a link sets its text and drops the destination;
31//! a pure index marker sets nothing. An inline `#func[...]` the reader does not know is consumed, recorded
32//! in the summary, and its bracketed body folded in, so its words survive but its raw markup never leaks.
33
34use crate::ir::FloatPlacement;
35use crate::ir::FloatScope;
36use crate::ir::Floating;
37use crate::ir::Length;
38use crate::ir::Span;
39use crate::table::Align;
40
41use super::ast::{AlignSpec, ClosureAlign, FigureBody, Inline, Item, ListItem, Spacing, TableSpec};
42use super::mathparse;
43
44use oxedyne_fe2o3_core::prelude::*;
45
46use std::collections::BTreeMap;
47use std::collections::HashMap;
48use std::sync::RwLock;
49
50// The book's `term-dict`, set once by the loader before parsing so the term-dictionary glossary family
51// (`t`, `tcap`, `graw`, `g`, `gi`, `gcap`, `gcapi`) resolves a key to its display value while the key
52// identity is still known. A process-global rather than a threaded argument, mirroring the image base:
53// the inline reader sets one run at a time and carries no book context of its own. `None` until the
54// loader installs one, in which case every key falls back to its own text.
55static TERM_DICT: RwLock<Option<HashMap<String, String>>> = RwLock::new(None);
56
57/// Records the book's `term-dict`, read from a sibling `terms.typ`, so the term-dictionary glossary
58/// family resolves each key to its value at parse time. Installing a fresh map replaces any prior one.
59pub fn set_term_dict(dict: HashMap<String, String>) -> Outcome<()> {
60 let mut guard = lock_write!(TERM_DICT, "While recording the term dictionary");
61 *guard = Some(dict);
62 Ok(())
63}
64
65/// The display value a `term-dict` key resolves to, or `None` when no map is installed or it holds no
66/// such key. A poisoned lock reads as absent rather than failing the parse: a missing value falls back
67/// to the key text, which is exactly the safe degradation here.
68pub(crate) fn term_value(key: &str) -> Option<String> {
69 match TERM_DICT.read() {
70 Ok(guard) => guard.as_ref().and_then(|m| m.get(key).cloned()),
71 Err(_) => None,
72 }
73}
74
75/// Why a construct was refused rather than set: the axis a per-site diagnostic reports alongside its
76/// name and location, so a reader can tell a categorical limit from a todo.
77#[derive(Clone, Copy, Debug, PartialEq, Eq)]
78pub enum RefusalClass {
79 // Typst's own general evaluation primitives -- `#import`, `#let`, `#set`, `#show` -- which run an
80 // arbitrary expression to a value. Austenite's driver converges a fixed, two-pass ledger to a point;
81 // it does not carry a Typst-style evaluator, so these are a categorical limit of the architecture
82 // (see the crate root's own note that Typst treats a document as a program and Austenite does not),
83 // not a specific feature waiting to be added.
84 FixedPoint,
85 // A construct that reads the page's own laid-out state back into the document -- `#context`,
86 // `#query`, `#locate`, a counter's or state's `.at`/`.get`, `#measure`, `#layout`. This is exactly
87 // Typst's self-observation model, which the ledger exists to answer in Austenite's own terms; a
88 // construct landing here is a real gap the ledger could plausibly close, not a limit of the design.
89 Introspective,
90 // Anything else skipped: a specific call or wrapper (`#columns`, an unknown standalone or inline
91 // `#func`, an unknown term-dictionary key) that names no fundamental barrier -- just not yet read.
92 Unsupported,
93}
94
95impl RefusalClass {
96 /// Classifies a refusal by the source name it was recorded under. Matched by substring rather than
97 /// an exact keyword, since the same construct is recorded under different shapes depending on how it
98 /// was written (`#context`, a bare `context` inside a longer call name) -- this is a diagnostic
99 /// classifier, not a parser, so a generous match that occasionally over-reaches is the right trade.
100 fn classify(name: &str) -> Self {
101 let lower = name.to_lowercase();
102 for kw in ["#import", "#let", "#set", "#show"] {
103 if lower.starts_with(kw) {
104 return RefusalClass::FixedPoint;
105 }
106 }
107 const INTROSPECTIVE: [&str; 8] =
108 ["context", "query", "locate", "counter.at", "counter.get", "state", "measure", "layout"];
109 if INTROSPECTIVE.iter().any(|kw| lower.contains(kw)) {
110 return RefusalClass::Introspective;
111 }
112 RefusalClass::Unsupported
113 }
114
115 /// The word `--explain` prints for this class.
116 pub fn label(&self) -> &'static str {
117 match self {
118 RefusalClass::FixedPoint => "fixed-point",
119 RefusalClass::Introspective => "introspective",
120 RefusalClass::Unsupported => "unsupported",
121 }
122 }
123}
124
125/// One site the reader passed over rather than set: the source name it was written with (carrying its
126/// leading `#`, so it reads back as source), the byte span it was found at, and why it was refused.
127/// The span is the whole containing line for a code statement or standalone call, or the whole
128/// containing item (a paragraph, a heading) for an inline call found within one -- Austenite's inline
129/// scanner does not keep the fine per-character offset once a paragraph's lines have been joined and its
130/// whitespace collapsed, so the enclosing item is the finest boundary available without a deeper rework
131/// of the reader than this diagnostic upgrade is for.
132#[derive(Clone, Debug)]
133pub struct Refusal {
134 pub name: String,
135 pub span: Span,
136 pub class: RefusalClass,
137 // The source file this site was read from, for `--explain`'s "file:line:col". Empty immediately
138 // after parsing, since a lone parse of a source string carries no filename of its own; the book
139 // assembler ([`crate::book::assemble`]) tags each chapter's (and the root's own) refusals with the
140 // real path once assembly is back in a context that has one -- see [`Refusals::tag_file`].
141 pub file: String,
142}
143
144/// Every site the reader refused across one parse (or, once [`Refusals::merge`] has folded chapters
145/// together, across a whole book). Kept as a flat list of [`Refusal`]s rather than the old name-keyed
146/// tally, so a caller can still print the terse one-line [`Refusals::report`] but can also walk every
147/// site for `--explain`'s per-site listing. Empty when the reader set everything it met.
148#[derive(Clone, Debug, Default)]
149pub struct Refusals {
150 sites: Vec<Refusal>,
151}
152
153impl Refusals {
154 /// Records one refused construct by the source name it was written with (with its leading `#`) and
155 /// the span it was found at, classifying it from the name.
156 pub(crate) fn record(&mut self, name: &str, span: Span) {
157 let class = RefusalClass::classify(name);
158 self.sites.push(Refusal { name: name.to_string(), span, class, file: String::new() });
159 }
160
161 /// Builds a table directly from a caller's own sites, for a test (or another future caller outside
162 /// the parser) that wants a known `Refusals` without driving a real parse to produce one.
163 pub fn from_sites(sites: Vec<Refusal>) -> Self {
164 Self { sites }
165 }
166
167 pub fn is_empty(&self) -> bool { self.sites.is_empty() }
168
169 /// Sets every site's `file` that is not already set, so a caller assembling several chapters can tag
170 /// each chapter's refusals with its own path right after parsing it, before folding them into the
171 /// book's running total with [`merge`](Self::merge) -- at which point every site already carries the
172 /// file it came from, and a second tagging (the root's own trailing markup, read after every
173 /// include) touches only the sites still unset.
174 pub fn tag_file(&mut self, file: &str) {
175 for r in &mut self.sites {
176 if r.file.is_empty() {
177 r.file = file.to_string();
178 }
179 }
180 }
181
182 /// Every refused site, in the order the reader met them.
183 pub fn sites(&self) -> &[Refusal] { &self.sites }
184
185 /// The number of distinct construct names refused.
186 pub fn kinds(&self) -> usize { self.entries().len() }
187
188 /// The total count of refused sites across every name.
189 pub fn total(&self) -> usize { self.sites.len() }
190
191 /// Each refused construct name with its count, ordered by descending count then name, so the report
192 /// leads with the construct that cost the most.
193 pub fn entries(&self) -> Vec<(String, usize)> {
194 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
195 for r in &self.sites {
196 *counts.entry(r.name.as_str()).or_insert(0) += 1;
197 }
198 let mut v: Vec<(String, usize)> = counts.into_iter().map(|(k, c)| (k.to_string(), c)).collect();
199 v.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
200 v
201 }
202
203 /// Folds another parse's refusals into this one, so a caller assembling several chapters reports one
204 /// total rather than a summary per file. `other` is consumed rather than borrowed: a caller merging a
205 /// just-parsed chapter's refusals into the book's running total has no further use for its own copy.
206 pub fn merge(&mut self, other: Refusals) {
207 self.sites.extend(other.sites);
208 }
209
210 /// A one-line report -- "skipped 3 unsupported constructs: #show (2), #columns (1)" -- or `None` when
211 /// nothing was skipped, so a caller prints the line only when it has something to say. Unchanged in
212 /// wording from before this unit: `--explain` is the new, detailed report, this terse one stays the
213 /// default.
214 pub fn report(&self) -> Option<String> {
215 if self.sites.is_empty() {
216 return None;
217 }
218 let parts: Vec<String> = self.entries().into_iter()
219 .map(|(n, c)| fmt!("{} ({})", n, c))
220 .collect();
221 let n = self.total();
222 Some(fmt!("skipped {} unsupported construct{}: {}",
223 n, if n == 1 { "" } else { "s" }, parts.join(", ")))
224 }
225}
226
227/// Parses a whole Ingot source string into its surface items. The only error is an empty heading --
228/// a `=` marker with no title -- which names the offending 1-based line.
229pub fn document(src: &str) -> Outcome<Vec<Item>> {
230 let (items, _) = res!(document_with_refusals(src));
231 Ok(items)
232}
233
234/// Parses a source string into its surface items and, alongside, the [`Refusals`] of every construct
235/// the reader passed over rather than set -- a `#let`/`#set`/`#show`/`#import` code line, an unknown
236/// line-leading `#func(...)` call, a `#columns` wrapper, and any unhandled inline `#func[...]`. The
237/// caller prints the summary so a dropped construct is reported rather than lost silently.
238pub fn document_with_refusals(src: &str) -> Outcome<(Vec<Item>, Refusals)> {
239 let tfns = crate::lang::rules::TemplateFns::new();
240 let cfns = crate::lang::rules::ContentFns::new();
241 document_with_templates(src, crate::lang::rules::Bindings::new(&tfns, &cfns))
242}
243
244/// Records a refusal for every claim reference (`#claim-refs`/`#claim-label`) that sits in a context the
245/// layout does not gather into the reverse claim index. A top-level body run -- a paragraph, a list entry, a
246/// callout body -- and a table cell both feed the index (see `doc::build_pieces`, reached for a cell through
247/// `doc::build_grid`); a claim code in a heading title, a figure or table caption, or a footnote body is
248/// dropped by the layout, so it would otherwise vanish from the index (and, for a `#claim-label`, from the
249/// margin) with no trace. Making that a refusal keeps the silent-loss class this project guards against out
250/// of the reverse index. Gathering from a heading or caption needs anchor support there and is a later
251/// increment; until then the code is reported, not dropped.
252fn flag_unindexed_claim_refs(items: &[Item], skips: &mut Refusals) {
253 for item in items {
254 match item {
255 // A body run and a list entry are gathered; only a claim reference nested inside a footnote of one
256 // escapes the index, so the top-level runs are scanned as indexed and their footnotes are not.
257 Item::Paragraph { runs, span, .. } => scan_claim_refs(runs, true, *span, "a paragraph", skips),
258 Item::List { items: entries, .. } => for e in entries { flag_list_item_claim_refs(e, skips); },
259 // A callout body is gathered like the main flow; recurse so a claim reference in it is indexed and
260 // only its non-body sub-contexts (a caption, a footnote) are flagged.
261 Item::Box { items: inner, .. } => flag_unindexed_claim_refs(inner, skips),
262 Item::Scoped { items: inner, .. } => flag_unindexed_claim_refs(inner, skips),
263 // A heading title and a caption are still not gathered, so a claim reference in either is refused.
264 // A table cell now runs through the body's own segment pipeline (`doc::build_grid` ->
265 // `doc::build_pieces`), which weaves the cell's `#claim-refs`/`#claim-label` anchor into the reverse
266 // claim index exactly as a body run does, so it is no longer refused.
267 Item::Heading { runs, span, .. } => scan_claim_refs(runs, false, *span, "a heading title", skips),
268 Item::Figure { caption, span, .. } => {
269 if let Some(cap) = caption {
270 scan_claim_refs(cap, false, *span, "a figure caption", skips);
271 }
272 },
273 _ => {},
274 }
275 }
276}
277
278/// [`flag_unindexed_claim_refs`] for one list entry: its own runs are gathered (indexed), and its nested
279/// child items are walked as their own contexts.
280fn flag_list_item_claim_refs(entry: &ListItem, skips: &mut Refusals) {
281 scan_claim_refs(&entry.runs, true, Span::new(0, 0), "a list entry", skips);
282 flag_unindexed_claim_refs(&entry.children, skips);
283}
284
285/// Scans an inline run for claim references and records a refusal for each that will not reach the reverse
286/// index. `indexed` is true for a top-level body run (a paragraph, list entry or callout body), where a
287/// claim reference IS gathered and so is left alone; a footnote body is never gathered, so its own runs are
288/// always scanned as unindexed regardless of where the footnote sits.
289fn scan_claim_refs(runs: &[Inline], indexed: bool, span: Span, context: &str, skips: &mut Refusals) {
290 for run in runs {
291 match run {
292 Inline::MarginNote { codes, .. } if !indexed && !codes.is_empty() =>
293 skips.record(&fmt!("claim reference in {} is not indexed", context), span),
294 Inline::Footnote(inner) => scan_claim_refs(inner, false, span, "a footnote body", skips),
295 _ => {},
296 }
297 }
298}
299
300/// As [`document_with_refusals`], with the `#let` bindings (`binds`) in scope: a call to a furniture
301/// function -- `#pr-note[ ... ]`, `#aside-box(title: [..])[ ... ]` -- expands into a padded box, and a
302/// reference to a content binding -- `#greet("world")`, a bare `#intro` -- expands into its re-read markup,
303/// rather than either being tallied as a skipped construct. A body re-parsed here carries the same `binds`,
304/// so a call nested inside another's body expands too. With empty maps this is exactly
305/// [`document_with_refusals`].
306///
307/// After the surface tree is built, [`flag_unindexed_claim_refs`] records a refusal for any claim reference
308/// that landed in a context the layout does not gather into the reverse claim index. This runs once, on the
309/// whole assembled tree -- the recursive re-parse of a `#columns`/`#styled-box` body reaches for
310/// [`parse_items`] directly, so a nested claim reference is flagged once here rather than again per level.
311pub fn document_with_templates(src: &str, binds: crate::lang::rules::Bindings<'_, '_>)
312 -> Outcome<(Vec<Item>, Refusals)>
313{
314 let (items, mut skips) = res!(parse_items(src, binds));
315 flag_unindexed_claim_refs(&items, &mut skips);
316 Ok((items, skips))
317}
318
319/// The surface-tree parse proper, without the [`flag_unindexed_claim_refs`] post-pass -- so a recursively
320/// re-parsed body (a `#columns`/`#styled-box` wrapper's content) is not validated twice, once here and again
321/// when its parent walks the spliced items. [`document_with_templates`] wraps this with that one validation.
322fn parse_items(src: &str, binds: crate::lang::rules::Bindings<'_, '_>)
323 -> Outcome<(Vec<Item>, Refusals)>
324{
325 let mut skips: Refusals = Refusals::default();
326 let mut items: Vec<Item> = Vec::new();
327 let mut lines: Vec<String> = Vec::new(); // the current paragraph's constituent lines
328 let mut para_start: u32 = 0; // byte offset of the paragraph's first line
329 let mut para_end: u32 = 0; // byte offset just past its last line's content
330 let mut offset: u32 = 0; // running byte offset of the current line's start
331 let mut line_no = 0usize; // 1-based, for a diagnostic
332
333 // The stack of open list levels, innermost last. Each level records the leading-space indent of its
334 // markers, so a deeper marker opens a sub-list under the current item and a shallower one closes back
335 // to the matching level; an empty stack means no list is open. A list is a run of marker lines that a
336 // blank line does not break (Typst continues an enum across a gap), but any other content flushes.
337 let mut stack: Vec<ListFrame> = Vec::new();
338
339 // A fenced code block, while one is open: the verbatim lines gathered so far and the byte offset it
340 // began at. A ```-fence opens it, the next ```-fence closes it; between them every line is kept as it
341 // stands, its indentation and markup untouched.
342 let mut code: Option<(Vec<String>, u32)> = None;
343
344 // A multi-line Typst code statement or standalone template call being skipped: the net bracket depth
345 // still open across the lines consumed so far, and whether a string literal is currently open. `None`
346 // when not skipping. While it is `Some`, every line is consumed and nothing is set until the delimiters
347 // balance.
348 let mut skip: Option<SkipState> = None;
349
350 // A multi-line construct whose whole text is gathered so it can be parsed rather than skipped: a
351 // `#figure(...)`, a bare `#table(...)`, or a `#let name = (...)` data array feeding a table. `None`
352 // when none is open. The accumulated text is dispatched by its kind when the delimiters balance.
353 let mut capture: Option<Capture> = None;
354
355 // Data arrays declared by `#let name = (...)` and referenced by a table's `..name.flatten()` spread:
356 // the name maps to the flat sequence of cells the array holds, each cell a run of inline markup.
357 // Populated as the arrays are read, so a later figure resolves its cells against them.
358 let mut arrays: HashMap<String, Vec<Vec<Inline>>> = HashMap::new();
359
360 // Whether a `/* ... */` block comment is open across the line break. A `//` line comment never
361 // straddles a line, so it needs no carried state.
362 let mut comment = CommentState { in_block: false };
363
364 // `split_inclusive` keeps the trailing newline on each piece, so the running offset stays a true
365 // byte position into the source rather than drifting by the count of stripped terminators.
366 for raw in src.split_inclusive('\n') {
367 line_no += 1;
368 let start = offset;
369 offset = offset.saturating_add(raw.len() as u32);
370
371 // Strip the line terminator without consuming a real character: the final line may carry
372 // neither a newline nor a carriage return.
373 let mut line = raw;
374 if let Some(s) = line.strip_suffix('\n') { line = s; }
375 if let Some(s) = line.strip_suffix('\r') { line = s; }
376 let end = start.saturating_add(line.len() as u32);
377
378 // Strip Typst comments before classifying the line, but not while a fenced code block or a
379 // multi-line call skip is open: inside a fence a `//` is verbatim, and a skipped span is dropped
380 // whole regardless. The span above is computed from the raw line, so a diagnostic caret still
381 // points into the source.
382 let stripped;
383 let line = if code.is_none() && skip.is_none() {
384 stripped = strip_comments(line, &mut comment);
385 stripped.as_str()
386 } else {
387 line
388 };
389
390 let trimmed = line.trim_start();
391
392 // A multi-line code statement or standalone call is being skipped: keep consuming lines, tracking
393 // bracket nesting across `()`, `[]` and `{}` and respecting string literals, until the delimiters
394 // balance. Nothing between the opener and its close is set. This takes precedence over every other
395 // rule, since the span is code, not markup.
396 if let Some(state) = skip.as_mut() {
397 scan_brackets(line, state);
398 if !state.has_open_bracket() {
399 skip = None;
400 }
401 continue;
402 }
403
404 // A multi-line construct is being gathered whole: keep appending its lines and tracking the bracket
405 // balance until the delimiters close, then dispatch the accumulated text by its kind. Like the skip
406 // above, this takes precedence over the markup rules, since the span is a code construct.
407 if let Some(cap) = capture.as_mut() {
408 cap.buf.push_str(line);
409 cap.buf.push('\n');
410 scan_brackets(line, &mut cap.state);
411 if !cap.state.has_open_bracket() {
412 let done = capture.take();
413 if let Some(cap) = done {
414 dispatch_capture(cap, &mut items, &mut arrays, &mut skips, binds);
415 }
416 }
417 continue;
418 }
419
420 // A fenced code block takes precedence over every other rule: inside it, only a closing fence is
421 // special and every other line is verbatim, so its own `=`, `-` or `*` carry no markup meaning.
422 if let Some((buf, cstart)) = code.as_mut() {
423 if is_fence(trimmed) {
424 items.push(Item::Code { lines: std::mem::take(buf), span: Span::new(*cstart, end) });
425 code = None;
426 } else {
427 buf.push(line.to_string());
428 }
429 continue;
430 }
431 if is_fence(trimmed) {
432 // An opening fence closes any paragraph or list, then begins a verbatim block. The fence line
433 // itself (and any language tag on it) is not kept.
434 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
435 flush_list(&mut items, &mut stack);
436 code = Some((Vec::new(), start));
437 continue;
438 }
439
440 // A display maths block ($ ... $) spans several source lines, but every branch below classifies a
441 // line on its own -- so, left unchecked, a `=`-lead alignment row reads as a heading, a `-`-lead row
442 // as a list marker, and a blank row between stacked lines flushes the paragraph early, each stealing
443 // the line before the paragraph ever reaches `mathparse` whole. While an odd number of unescaped `$`
444 // have accumulated in the running paragraph the block is still open, so the blank/heading/list checks
445 // below are skipped for as long as it is: a fence opening or a capture opener (a `#`-led construct)
446 // still takes precedence regardless, since neither shape occurs inside genuine display maths.
447 let math_block_open = math_open(&lines);
448
449 if trimmed.is_empty() && !math_block_open {
450 // A blank line closes the paragraph it follows, but not an open list: Typst continues an enum
451 // (or bullet list) across a blank line between items, restarting the numbering only when other
452 // content intervenes. The list is therefore held open here; the marker branch joins a following
453 // item of the same kind, while any other line -- a paragraph, heading, figure, fence or code
454 // line -- flushes it first, so two lists parted by real content still restart. A blank line held
455 // between two items of the open level marks it loose: Typst then sets its items with block
456 // spacing rather than the tight body pitch a blank-free list takes.
457 if let Some(top) = stack.last_mut() {
458 top.saw_blank = true;
459 }
460 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
461 } else if let Some(kind) = capture_opener(trimmed, binds)
462 .filter(|k| !(matches!(k, CaptureKind::ContentCall(_)) && !lines.is_empty()))
463 {
464 // A standalone content-binding reference mid-paragraph joins the paragraph inline rather than
465 // splicing a block, matching Typst's inline value flow: only a reference with no paragraph open
466 // splices its expanded blocks (the `filter` above lets an open-paragraph `#name` fall through to
467 // the paragraph arm). A markup builtin (`#lorem`, `#v`, `#pagebreak`) is block-position but not
468 // blank-line-gated: `capture_opener` already admits it only own-line (a balanced call with nothing
469 // but whitespace after its `)`, or a multi-line span), so a builtin directly beneath a prose line
470 // with no blank between (`prose\n#pagebreak()`) closes the paragraph and sets its own block, exactly
471 // as `= heading\n#pagebreak()` and `#section-banner` already do -- Typst turns the page there whether
472 // or not a blank line parts the two, so a paragraph before the break must not silently swallow it.
473 // A builtin with prose on the SAME line (`#lorem(5) more`) is not own-line, so it never reaches here;
474 // inline mid-prose support is a later unit. Every other capture kind -- a figure, a bare table, a
475 // data array -- flushes the paragraph and is gathered as before.
476 //
477 // A multi-line construct the reader sets rather than skips. It closes any open block, then its
478 // whole text is gathered by the check
479 // at the top of the loop until the delimiters balance, and parsed by [`dispatch_capture`].
480 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
481 flush_list(&mut items, &mut stack);
482 let mut state = SkipState::new();
483 scan_brackets(line, &mut state);
484 let mut buf = String::new();
485 buf.push_str(line);
486 buf.push('\n');
487 let cap = Capture { kind, buf, state, start };
488 if !cap.state.has_open_bracket() {
489 dispatch_capture(cap, &mut items, &mut arrays, &mut skips, binds); // the whole construct closed on one line
490 } else {
491 capture = Some(cap);
492 }
493 } else if trimmed.starts_with("#line(") && call_inner(trimmed, "line").is_some() {
494 // A standalone `#line(length:.., stroke:..)` horizontal divider (the appendix brackets a note
495 // with one above and below). It closes any open block and sets a stroked rule; a multi-line
496 // `#line(` that does not close on this line falls through to the skip path below.
497 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
498 flush_list(&mut items, &mut stack);
499 if let Some(rule) = parse_line_rule(trimmed) {
500 items.push(rule);
501 }
502 } else if trimmed.starts_with("#print-glossary(") {
503 // A line-leading `#print-glossary()`: the glossary section's Term/Definition table. It closes any
504 // open block and emits a placeholder the book layer fills once the whole document's glossary terms
505 // are known -- unlike the surrounding template calls it is set in place, not recorded as a skip.
506 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
507 flush_list(&mut items, &mut stack);
508 items.push(Item::PrintGlossary { span: Span::new(start, end) });
509 } else if let Some(decision) = code_skip(trimmed) {
510 // A Typst code statement (`#import`, `#let`, `#set`, `#show`) or a line-leading standalone call
511 // to a template function Austenite does not yet run: it closes any open block and is skipped.
512 // The styling and computation layer is a later increment; the prose around it still sets. When
513 // its delimiters do not balance on this line, the multi-line span is consumed by the check at the
514 // top of the loop until they do.
515 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
516 flush_list(&mut items, &mut stack);
517 // The recorded span is the opening line alone, even for a construct whose delimiters run on
518 // for several more: that is where a reader wants `--explain`'s caret to land, and the true
519 // closing offset is not known until the multi-line skip above closes, several iterations on.
520 skips.record(&construct_name(trimmed), Span::new(start, end));
521 if let CodeSkip::Multi(state) = decision {
522 skip = Some(state);
523 }
524 } else if lines.is_empty() && is_code_reference(trimmed) && !names_scalar_alone(trimmed, binds.sfns) {
525 // A line-leading code-mode reference the reader cannot run -- a bare `#name` bound to nothing, a
526 // field/method access `#name.foo`, an `#if`/`#for`/`#while` control keyword, or an anonymous
527 // `#{ ... }`/`#( ... )` block. A bound `#name` was expanded by `capture_opener` above; a
528 // `#name(`/`#name[` call and the `#let`/`#set`/`#show`/`#import` keywords were refused by
529 // `code_skip`. Each of these resolves to a value or runs code in Typst, so setting its source as
530 // literal prose would leak a `#` onto the page (the very thing an expanded content-binding body
531 // carrying `#if`/`#{` would do); it is refused with its span instead. The `lines.is_empty()` guard
532 // keeps a reference mid-paragraph joining the line inline, as Typst does, rather than refusing it.
533 // A bare `#name` naming a SCALAR binding is exempt (`names_scalar_alone`): it falls through to the
534 // paragraph arm below, where `flush_para`'s own `substitute_scalars` replaces it with the bound
535 // value, so a scalar standing alone on its line sets its value just as one mid-prose already does.
536 // An UNBOUND standalone name is not exempt, so it stays the visible refusal it has always been.
537 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
538 flush_list(&mut items, &mut stack);
539 skips.record(&construct_name(trimmed), Span::new(start, end));
540 } else if trimmed.starts_with('=') && !math_block_open {
541 // A heading closes any paragraph or list above it, then stands on its own line.
542 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
543 flush_list(&mut items, &mut stack);
544 let level = trimmed.chars().take_while(|&c| c == '=').count();
545 let raw = trimmed[level..].trim(); // '=' is ASCII, so a byte slice at the count is safe
546 if raw.is_empty() {
547 return Err(err!(
548 "Empty heading on line {}: a `=` marker must be followed by a title.", line_no;
549 Input, Invalid, Missing));
550 }
551 let (title, label) = split_label(raw);
552 if title.is_empty() {
553 return Err(err!(
554 "Heading on line {} has a label but no title.", line_no; Input, Invalid, Missing));
555 }
556 // The title carries inline markup like any run, so a glossary term, an index call, emphasis or a
557 // maths span in a heading sets its display text rather than leaking its raw source into the head
558 // and the table of contents. An inline reference to a content binding (`= Product #stamp`) splices
559 // its expanded body first, and a bare `#name` naming a scalar `#let` value binding (`= Product
560 // #version`) substitutes its display text next, the same as a paragraph's.
561 let head_span = Span::new(start, end);
562 let title = substitute_content_calls(&title, binds, &mut skips, head_span);
563 let title = substitute_scalars(&title, binds.sfns);
564 items.push(Item::Heading {
565 level: level as u8,
566 runs: parse_inlines_in(&title, head_span, &mut skips),
567 label,
568 span: head_span,
569 });
570 } else {
571 // A list marker joins the list stack, unless a display maths block is open, in which case a
572 // `-`/`+`-lead row is part of the equation, not a bullet: it falls through to the paragraph arm
573 // below like every other captured line.
574 match if math_block_open { None } else { marker(trimmed) } {
575 Some((ord, text)) => {
576 // A list item. It closes any open paragraph, then joins the list stack by its indentation:
577 // a deeper marker opens a sub-list under the current item, a shallower one closes back to
578 // the matching level, and a same-indent marker of the other kind ends the list and starts
579 // one of the new kind. The item's text carries inline emphasis like any run.
580 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
581 let indent = line.chars().take_while(|c| c.is_whitespace()).count();
582 // A list item is running prose like a paragraph, so an inline content-fn reference in it --
583 // `+ ... written as M#oxe.` -- splices its expanded body first, exactly as `flush_para` does
584 // for a paragraph. Without this the item bypassed the pre-pass and leaked a raw `#oxe`/`#name`
585 // while the same reference in a paragraph beside it expanded, an inconsistency a reader sees.
586 let item_span = Span::new(start, end);
587 let text = substitute_content_calls(&text, binds, &mut skips, item_span);
588 let runs = parse_inlines_in(&text, item_span, &mut skips);
589 list_marker(&mut items, &mut stack, indent, ord, runs, start, end);
590 },
591 None => {
592 // Any other non-blank line joins the running paragraph, closing a list first; its own line
593 // break and indentation carry no meaning, only its words. This is also where a blank,
594 // heading-lead or list-marker-lead line lands while a display maths block is open.
595 flush_list(&mut items, &mut stack);
596 if lines.is_empty() {
597 para_start = start;
598 }
599 lines.push(line.to_string());
600 para_end = end;
601 },
602 }
603 }
604 }
605
606 // A source that ends without a closing blank line still closes its last paragraph or list; an
607 // unterminated code fence still yields the block it had gathered.
608 flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds);
609 flush_list(&mut items, &mut stack);
610 if let Some((buf, cstart)) = code {
611 items.push(Item::Code { lines: buf, span: Span::new(cstart, offset) });
612 }
613 // A construct left open at end of source is dispatched with what it gathered, so a missing closer
614 // still yields its best-effort figure or table rather than swallowing the tail silently.
615 if let Some(cap) = capture {
616 dispatch_capture(cap, &mut items, &mut arrays, &mut skips, binds);
617 }
618 Ok((items, skips))
619}
620
621/// Is a display maths block still open across the paragraph lines gathered so far? A `$` toggles the
622/// state; a `\`-escaped one (`\$`) is skipped, mirroring the same escape in [`parse_inlines_in`], so it
623/// never toggles. An odd running count means the block opened on some earlier line and has not yet met
624/// its close, which is what lets [`document_with_refusals`] keep capturing lines the per-line classifier
625/// would otherwise steal as a heading, a list marker, or a paragraph-flushing blank.
626fn math_open(lines: &[String]) -> bool {
627 let mut open = false;
628 for line in lines {
629 let mut chars = line.chars();
630 while let Some(c) = chars.next() {
631 if c == '\\' {
632 chars.next(); // the escaped character, taken literally
633 } else if c == '$' {
634 open = !open;
635 }
636 }
637 }
638 open
639}
640
641/// Is this already-left-trimmed line a ```` ``` ```` code fence? An opening fence may carry a language
642/// tag (```` ```rust ````); a closing fence is bare. Either way it opens with three backticks.
643fn is_fence(trimmed: &str) -> bool {
644 trimmed.starts_with("```")
645}
646
647/// Reads a list marker at the start of an already-left-trimmed line: `-` opens a bullet item, `+` a
648/// numbered one. The marker must be the whole line or be followed by whitespace, so a dash inside a word
649/// or a `+1` is ordinary prose, not a marker. Returns the item's kind and its text with the marker and
650/// surrounding whitespace removed.
651fn marker(trimmed: &str) -> Option<(bool, String)> {
652 let first = trimmed.chars().next()?;
653 let ordered = match first {
654 '-' => false,
655 '+' => true,
656 _ => return None,
657 };
658 let rest = &trimmed[first.len_utf8()..];
659 if rest.is_empty() {
660 return Some((ordered, String::new()));
661 }
662 if rest.starts_with(|c: char| c.is_whitespace()) {
663 return Some((ordered, rest.trim().to_string()));
664 }
665 None
666}
667
668/// One open level of a possibly-nested list while the reader accumulates it. `indent` is the leading-space
669/// width of the level's markers, so a deeper marker opens a child level and a shallower one closes this
670/// level back into the item it hung under.
671struct ListFrame {
672 indent: usize,
673 ordered: bool,
674 items: Vec<ListItem>,
675 loose: bool, // a blank line parted two of this level's items, so Typst sets it loose (block-spaced)
676 saw_blank: bool, // a blank line is pending since the last item; it makes the level loose only if a sibling follows
677 start: u32,
678 end: u32,
679}
680
681/// Attaches a marker line to the open list stack by its `indent`, opening or closing nested levels as the
682/// indentation and kind require. A deeper marker opens a sub-list under the current item; a shallower one
683/// closes the deeper level(s) first; a same-indent marker continues the level when its kind matches and
684/// otherwise ends it and starts a fresh list of the new kind, as the flat reader did.
685fn list_marker(
686 items: &mut Vec<Item>,
687 stack: &mut Vec<ListFrame>,
688 indent: usize,
689 ord: bool,
690 runs: Vec<Inline>,
691 start: u32,
692 end: u32,
693)
694{
695 // Close every open level deeper than this marker: a dedent ends the nested list(s), each folding into
696 // the item it hung under.
697 while stack.last().map_or(false, |f| f.indent > indent) {
698 if let Some(frame) = stack.pop() {
699 fold(items, stack, frame);
700 }
701 }
702 match stack.last_mut() {
703 Some(top) if top.indent == indent && top.ordered == ord => {
704 // Same level, same kind: another item of the open list. A blank line pending since the previous
705 // item was a genuine inter-item gap, so the level sets loose.
706 if top.saw_blank {
707 top.loose = true;
708 }
709 top.saw_blank = false;
710 top.items.push(ListItem { runs, children: Vec::new() });
711 top.end = end;
712 },
713 Some(top) if top.indent == indent => {
714 // Same indent, the other kind: the open list ends and a fresh one of the new kind begins.
715 if let Some(frame) = stack.pop() {
716 fold(items, stack, frame);
717 }
718 stack.push(ListFrame {
719 indent, ordered: ord, items: vec![ListItem { runs, children: Vec::new() }], loose: false, saw_blank: false, start, end });
720 },
721 // Deeper than the current level (a sub-list), or the first marker of a list: open a new level. A
722 // deeper level becomes a child of the current item when it folds.
723 _ => stack.push(ListFrame {
724 indent, ordered: ord, items: vec![ListItem { runs, children: Vec::new() }], loose: false, saw_blank: false, start, end }),
725 }
726}
727
728/// Folds a closed list level into the tree: it becomes an [`Item::List`] hanging under the current item of
729/// the level below, or a top-level item when no level remains open.
730fn fold(items: &mut Vec<Item>, stack: &mut Vec<ListFrame>, frame: ListFrame) {
731 let list = Item::List {
732 ordered: frame.ordered,
733 items: frame.items,
734 loose: frame.loose,
735 span: Span::new(frame.start, frame.end),
736 };
737 match stack.last_mut() {
738 Some(parent) => match parent.items.last_mut() {
739 Some(item) => {
740 item.children.push(list);
741 parent.end = frame.end;
742 },
743 // A nested level always opens after its parent item exists, so this arm is unreachable in
744 // practice; a stray level is kept as a top-level item rather than dropped.
745 None => items.push(list),
746 },
747 None => items.push(list),
748 }
749}
750
751/// Closes every open list level into the item tree. The deepest level folds into its parent's current item
752/// first, so a nested list lands under the item it hung under; the outermost becomes a top-level
753/// [`Item::List`]. An empty stack flushes nothing, so a stray flush between two paragraphs costs nothing.
754fn flush_list(items: &mut Vec<Item>, stack: &mut Vec<ListFrame>) {
755 while let Some(frame) = stack.pop() {
756 fold(items, stack, frame);
757 }
758}
759
760/// Closes the paragraph being accumulated, if any: its lines are joined, their whitespace collapsed,
761/// and the result pushed as one [`Item::Paragraph`] spanning the source it came from. An empty
762/// accumulator flushes nothing, so a run of blank lines closes a paragraph only once.
763fn flush_para(
764 items: &mut Vec<Item>,
765 lines: &mut Vec<String>,
766 start: u32,
767 end: u32,
768 skips: &mut Refusals,
769 binds: crate::lang::rules::Bindings<'_, '_>,
770)
771{
772 if lines.is_empty() {
773 return;
774 }
775 let text = normalise_ws(&lines.join(" "));
776 // A trailing `<name>` labels the block -- in practice a display equation, `$ ... $ <eq_x>` -- and is
777 // stripped before the runs are read, so the maths span stands alone and lowers to a numbered equation
778 // rather than a rich paragraph. Ordinary prose ends in a full stop, so the conservative `split_label`
779 // (a single whitespace-free token in angle brackets at the very end) does not fire on it.
780 let (body, label) = split_label(&text);
781 let span = Span::new(start, end);
782 // An inline mid-prose reference to a bound content binding -- `M#oxe`, `see #stamp("v2") for details` --
783 // splices its argument-substituted body into the surrounding prose here, before the scalar pass and the
784 // inline scanner, exactly as the own-line reference splices its blocks: the words before and after the
785 // call are kept, and the body's own markup is then read by the one downstream scanner. Runs first so a
786 // scalar reference the body carries still substitutes below.
787 let body = substitute_content_calls(&body, binds, skips, span);
788 // A bare `#name` naming a scalar `#let` value binding substitutes its display text before the inline
789 // scanner runs, so "Version #version." reads the same as if the number or string had been typed in
790 // place. Runs before `parse_inlines_in`, never inside it, so it never touches a raw code span, inline
791 // maths, or any other captured construct (a table, a figure, a `#context` block) -- those are gathered
792 // and dispatched on a wholly separate path and never reach here.
793 let body = substitute_scalars(&body, binds.sfns);
794 let runs = parse_inlines_in(&body, span, skips);
795 items.push(Item::Paragraph { runs, label, span });
796 lines.clear();
797}
798
799/// Collapses every run of whitespace to a single space and trims the ends, so a paragraph's set width
800/// is left to the line breaker rather than to the source's own line breaks and indentation.
801fn normalise_ws(s: &str) -> String {
802 s.split_whitespace().collect::<Vec<_>>().join(" ")
803}
804
805/// Splits a whitespace-collapsed paragraph into inline runs in Typst's markup: `*strong*`, `_emph_`, a
806/// `@label` cross-reference, and `\`-escapes. An emphasis delimiter pairs only when it flanks a word --
807/// whitespace or an opening bracket before it and a non-space after to open, the reverse to close -- so
808/// `fe2o3_net`, `5 * 3` and a lone `_` are ordinary text. A backslash sets the next character literally,
809/// so `\$`, `\#`, `\_` and `\@` appear as themselves. An unpaired delimiter, or an `@` with no label
810/// after it, is ordinary text. Nesting is a later increment: the first valid closer ends a run.
811pub(crate) fn parse_inlines(text: &str) -> Vec<Inline> {
812 let mut skips = Refusals::default();
813 // A table cell, a caption or a flattened array cell has no item-level span of its own to attribute a
814 // refusal to (see `Refusal`'s own doc comment on why the item, not the character, is the finest
815 // boundary kept); this thin wrapper already threw the summary away before this unit, so a zero span
816 // changes nothing a caller could observe.
817 parse_inlines_in(text, Span::new(0, 0), &mut skips)
818}
819
820/// The inline scanner proper, recording every unhandled inline call into `skips`, at `span` (the whole
821/// containing item -- a paragraph, a heading, a list item -- rather than the call's own narrower
822/// position; see `Refusal`'s doc comment), so a `#func[...]` the reader cannot set is reported rather
823/// than leaked into the running text. [`parse_inlines`] is the thin wrapper for callers -- table cells,
824/// captions, flattening -- that do not surface the summary.
825fn parse_inlines_in(text: &str, span: Span, skips: &mut Refusals) -> Vec<Inline> {
826 let chars: Vec<char> = text.chars().collect();
827 let n = chars.len();
828 let mut runs: Vec<Inline> = Vec::new();
829 let mut plain = String::new(); // ordinary text gathered before the next run
830 let mut i = 0usize;
831 while i < n {
832 let c = chars[i];
833 // A backslash escapes the next character, which is then set as itself.
834 if c == '\\' && i + 1 < n {
835 plain.push(chars[i + 1]);
836 i += 2;
837 continue;
838 }
839 // An inline maths span between dollars. A `\$` was already turned into a literal above, so a `$`
840 // reaching here opens maths. If it parses, it is a maths run; if not, the literal `$...$` is kept.
841 if c == '$' {
842 if let Some(close) = (i + 1..n).find(|&j| chars[j] == '$') {
843 let inner: String = chars[i + 1..close].iter().collect();
844 if let Ok(atom) = mathparse::parse(&inner) {
845 if !plain.is_empty() {
846 runs.push(Inline::Text(std::mem::take(&mut plain)));
847 }
848 runs.push(Inline::Math(atom));
849 i = close + 1;
850 continue;
851 }
852 }
853 }
854 // An inline code span, `raw` between backticks: its content is verbatim, no markup within.
855 if c == '`' {
856 if let Some(close) = (i + 1..n).find(|&j| chars[j] == '`') {
857 if !plain.is_empty() {
858 runs.push(Inline::Text(std::mem::take(&mut plain)));
859 }
860 runs.push(Inline::Code(chars[i + 1..close].iter().collect()));
861 i = close + 1;
862 continue;
863 }
864 }
865 // An inline footnote. Its bracketed content is markup, carried as inline runs so the note sets its
866 // own emphasis at the foot of the page. The mark falls after the run before it.
867 if c == '#' {
868 if let Some((note, next)) = footnote_call(&chars, i, span, skips) {
869 if !plain.is_empty() {
870 runs.push(Inline::Text(std::mem::take(&mut plain)));
871 }
872 runs.push(Inline::Footnote(note));
873 i = next;
874 continue;
875 }
876 }
877 // The function-call form of emphasis, `#emph[...]`, set exactly as `_..._`: its bracketed content is
878 // markup, so it takes the same expansion, and a call nested within it renders its display text.
879 if c == '#' {
880 if let Some((inner, next)) = emph_call(&chars, i) {
881 if !plain.is_empty() {
882 runs.push(Inline::Text(std::mem::take(&mut plain)));
883 }
884 push_emphasis(&mut runs, false, &inner, span, skips);
885 i = next;
886 continue;
887 }
888 }
889 // The function-call form of strength, `#strong[...]` or `#strong("...")`, set exactly as `*...*`:
890 // the bracket form's content is markup and takes the same expansion as the emph call above; the
891 // paren form's quoted-string argument is plain text bold in full. A paren argument that is not a
892 // plain string -- a bare identifier or an expression this reader cannot evaluate -- is left for
893 // the generic call handler below, which records the refusal rather than guessing at its text.
894 if c == '#' {
895 if let Some((inner, next)) = strong_call(&chars, i) {
896 if !plain.is_empty() {
897 runs.push(Inline::Text(std::mem::take(&mut plain)));
898 }
899 push_emphasis(&mut runs, true, &inner, span, skips);
900 i = next;
901 continue;
902 }
903 }
904 // Typst's superscript, `#super[...]` or `#super("...")`. Its content is usually a short string or
905 // number, reduced to display text here and set raised and smaller by the block layer.
906 if c == '#' {
907 if let Some((text, next)) = super_call(&chars, i) {
908 if !plain.is_empty() {
909 runs.push(Inline::Text(std::mem::take(&mut plain)));
910 }
911 runs.push(Inline::Super(text));
912 i = next;
913 continue;
914 }
915 }
916 // Typst's subscript, `#sub[...]` or `#sub("...")`, e.g. `CO#sub[2]`. Its content reduces to
917 // display text here and is set dropped and smaller by the block layer.
918 if c == '#' {
919 if let Some((text, next)) = sub_call(&chars, i) {
920 if !plain.is_empty() {
921 runs.push(Inline::Text(std::mem::take(&mut plain)));
922 }
923 runs.push(Inline::Sub(text));
924 i = next;
925 continue;
926 }
927 }
928 // Typst's `#smallcaps[...]` or `#smallcaps("...")`: its content reduces to display text here and is
929 // shaped with the font's small-capitals feature by the block layer, as Typst asks the font for it.
930 if c == '#' {
931 if let Some((text, next)) = smallcaps_call(&chars, i) {
932 if !plain.is_empty() {
933 runs.push(Inline::Text(std::mem::take(&mut plain)));
934 }
935 runs.push(Inline::SmallCaps(text));
936 i = next;
937 continue;
938 }
939 }
940 // An inline glossary or index call defined in the book template. A glossary term is a run of its
941 // own, so [`doc::author`] can set it bold-italic on first use; a visible index call sets its
942 // display text, which may itself carry markup, so it is parsed and folded in; a pure index marker
943 // sets nothing.
944 if c == '#' {
945 if let Some((call, next)) = glossary_call(&chars, i, span, skips) {
946 // An index marker is a run of its own, woven in before the display so the block layer
947 // records the term's occurrence at this point; the display, where the call has one, follows.
948 let push_index = |runs: &mut Vec<Inline>, plain: &mut String, index: Option<IndexKey>, skips: &mut Refusals| {
949 if let Some(k) = index {
950 if !plain.is_empty() {
951 runs.push(Inline::Text(std::mem::take(plain)));
952 }
953 // The display markup becomes its own runs, so the index page sets an emphasised entry
954 // italic and a display/sort split shows the display -- parsed exactly as the body's is.
955 let display = parse_inlines_in(&k.display, span, skips);
956 runs.push(Inline::Index { term: k.term, sub: k.sub, display, main: k.main });
957 }
958 };
959 match call {
960 Call::Glossary { term, display, index } => {
961 push_index(&mut runs, &mut plain, index, skips);
962 if !plain.is_empty() {
963 runs.push(Inline::Text(std::mem::take(&mut plain)));
964 }
965 runs.push(Inline::Glossary { term, display });
966 },
967 Call::Visible { display, index } => {
968 push_index(&mut runs, &mut plain, index, skips);
969 let sub = parse_inlines_in(&display, span, skips);
970 // A plain display folds back into the running text, keeping the fast single-run
971 // path; a display carrying markup becomes its own runs.
972 if let [Inline::Text(t)] = sub.as_slice() {
973 plain.push_str(t);
974 } else {
975 if !plain.is_empty() {
976 runs.push(Inline::Text(std::mem::take(&mut plain)));
977 }
978 runs.extend(sub);
979 }
980 },
981 Call::Invisible { index } => {
982 push_index(&mut runs, &mut plain, index, skips);
983 },
984 }
985 i = next;
986 continue;
987 }
988 }
989 // An inline citation, `#cite(<key>)` or `#cite(<a>, <b>)`. Its keys become a cite run the block
990 // layer resolves to "(Author Year)" against the bibliography; a citation with no readable key is
991 // dropped rather than left as raw source.
992 if c == '#' {
993 if let Some((keys, next)) = cite_call(&chars, i) {
994 if !keys.is_empty() {
995 if !plain.is_empty() {
996 runs.push(Inline::Text(std::mem::take(&mut plain)));
997 }
998 runs.push(Inline::Cite(keys));
999 }
1000 i = next;
1001 continue;
1002 }
1003 }
1004 // An inline claim marker, `#claim-label(...)` or `#claim-refs(...)`, from the book's claims
1005 // machinery. A `claim-label` registers invisible metadata AND sets a compressed code in the outside
1006 // margin: it emits a `MarginNote` run, which sets nothing in the body column but records a zero-width
1007 // anchor the overlay pass draws the code against after convergence. A `claim-refs` registers metadata
1008 // only, with no visible output, so it is consumed and emits nothing. Either way the body prose closes
1009 // over the marker's place, matching Typst's own body flow.
1010 if c == '#' {
1011 if let Some((display, codes, next)) = claim_call(&chars, i) {
1012 // A call naming a margin display or any reference code emits a MarginNote; a call naming
1013 // neither is consumed and sets nothing.
1014 if !display.is_empty() || !codes.is_empty() {
1015 if !plain.is_empty() {
1016 runs.push(Inline::Text(std::mem::take(&mut plain)));
1017 }
1018 runs.push(Inline::MarginNote { display, codes });
1019 }
1020 i = next;
1021 continue;
1022 }
1023 }
1024 // The `#raw("...")` call form of inline code.
1025 if c == '#' {
1026 if let Some((text, next)) = raw_call(&chars, i) {
1027 if !plain.is_empty() {
1028 runs.push(Inline::Text(std::mem::take(&mut plain)));
1029 }
1030 runs.push(Inline::Code(text));
1031 i = next;
1032 continue;
1033 }
1034 }
1035 // A Typst hyperlink, `#link("url")[text]` or `#link(<label>)[text]`. The link text is what a print
1036 // reader sees, so its markup is parsed and folded into the running line; the destination has no place
1037 // on a page with no clickable annotation and is dropped. A `#link("url")` with no bracket sets the
1038 // URL itself as its text, as Typst does.
1039 if c == '#' {
1040 if let Some((body, next)) = link_call(&chars, i, span, skips) {
1041 if let [Inline::Text(t)] = body.as_slice() {
1042 plain.push_str(t);
1043 } else {
1044 if !plain.is_empty() {
1045 runs.push(Inline::Text(std::mem::take(&mut plain)));
1046 }
1047 runs.extend(body);
1048 }
1049 i = next;
1050 continue;
1051 }
1052 }
1053 // A Typst cross-reference: `@` then a label.
1054 if c == '@' {
1055 if let Some((label, next)) = at_label(&chars, i) {
1056 if !plain.is_empty() {
1057 runs.push(Inline::Text(std::mem::take(&mut plain)));
1058 }
1059 runs.push(Inline::PageRef(label));
1060 i = next;
1061 continue;
1062 }
1063 }
1064 if (c == '*' || c == '_') && is_opener(&chars, i) {
1065 if let Some(close) = find_closer(&chars, i + 1, c) {
1066 if !plain.is_empty() {
1067 runs.push(Inline::Text(std::mem::take(&mut plain)));
1068 }
1069 let inner: String = chars[i + 1..close].iter().collect();
1070 push_emphasis(&mut runs, c == '*', &inner, span, skips);
1071 i = close + 1;
1072 continue;
1073 }
1074 }
1075 // An inline `#func(...)` or `#func[...]` call none of the handlers above claimed: a template function
1076 // the reader cannot yet run. It is recorded by name and consumed whole, so its raw markup no longer
1077 // leaks into the set text. When it wraps a single `[...]` content group -- the common shape of a Typst
1078 // content function -- that body is parsed and folded in, keeping its words rather than dropping them;
1079 // a call with only paren arguments (`#v(1em)`, `#colbreak()`) sets nothing where it stood.
1080 if c == '#' {
1081 if let Some((body, next, name)) = unknown_call(&chars, i, span, skips) {
1082 skips.record(&name, span);
1083 if let Some(body) = body {
1084 if let [Inline::Text(t)] = body.as_slice() {
1085 plain.push_str(t);
1086 } else {
1087 if !plain.is_empty() {
1088 runs.push(Inline::Text(std::mem::take(&mut plain)));
1089 }
1090 runs.extend(body);
1091 }
1092 }
1093 i = next;
1094 continue;
1095 }
1096 }
1097 // Typst's smartypants: a run of hyphens in markup text becomes em/en dashes, longest match first
1098 // (`---` before `--`), and three-or-more dots become an ellipsis. Every other branch above has
1099 // already claimed `` ` `` and `$` before falling through here, so this never touches a raw code
1100 // span or a maths span -- only ordinary prose reaches this fallback. A `\-`/`\.` was already turned
1101 // into a literal above and never joins a run counted here, matching Typst's own escape.
1102 if c == '-' || c == '.' {
1103 let mut run = 0usize;
1104 while chars.get(i + run) == Some(&c) { run += 1; }
1105 let mut left = run;
1106 if c == '-' {
1107 while left > 0 {
1108 if left >= 3 { plain.push('\u{2014}'); left -= 3; } // --- em dash
1109 else if left == 2 { plain.push('\u{2013}'); left = 0; } // -- en dash
1110 else { plain.push('-'); left = 0; }
1111 }
1112 } else if left >= 3 {
1113 plain.push('\u{2026}'); // ... ellipsis
1114 left -= 3;
1115 for _ in 0..left { plain.push('.'); }
1116 } else {
1117 for _ in 0..left { plain.push('.'); }
1118 }
1119 i += run;
1120 continue;
1121 }
1122 plain.push(c);
1123 i += 1;
1124 }
1125 if !plain.is_empty() {
1126 runs.push(Inline::Text(plain));
1127 }
1128 if runs.is_empty() {
1129 runs.push(Inline::Text(String::new())); // a paragraph of pure delimiters keeps one empty run
1130 }
1131 runs
1132}
1133
1134/// Pushes an emphasised run (`*strong*` when `strong`, else `_emph_`) onto `runs`, reading its inner
1135/// markup. When the inner is plain text the run keeps the fast flat path -- one [`Inline::Strong`] or
1136/// [`Inline::Emph`]. When it carries a glossary term, an index call or a maths span -- as
1137/// `*The captation #gsi[attractor]*` does -- the emphasis is expanded: its plain stretches take the
1138/// emphasis face and the embedded calls become their own runs, so a call nested in emphasis renders its
1139/// display text rather than leaking its raw source. A glossary term keeps its own first-use bold-italic
1140/// (which subsumes the surrounding emphasis), so only the plain stretches carry the emphasis face.
1141fn push_emphasis(runs: &mut Vec<Inline>, strong: bool, inner: &str, span: Span, skips: &mut Refusals) {
1142 let sub = parse_inlines_in(inner, span, skips);
1143 if let [Inline::Text(t)] = sub.as_slice() {
1144 runs.push(if strong { Inline::Strong(t.clone()) } else { Inline::Emph(t.clone()) });
1145 return;
1146 }
1147 for run in sub {
1148 match run {
1149 // A plain stretch takes the emphasis face.
1150 Inline::Text(t) => runs.push(if strong { Inline::Strong(t) } else { Inline::Emph(t) }),
1151 // An inner run of the opposite face (`*_word_*`, `_*word*_`) multiplies the two faces to a
1152 // bold-italic run rather than losing the outer; a run of the same face, a glossary term or a
1153 // maths span keeps its own face, one level of nesting being all the flat vocabulary carries.
1154 Inline::Emph(t) if strong => runs.push(Inline::BoldItalic(t)),
1155 Inline::Strong(t) if !strong => runs.push(Inline::BoldItalic(t)),
1156 other => runs.push(other),
1157 }
1158 }
1159}
1160
1161/// Does the delimiter at `i` flank the left of a word? A non-space must follow it, and the start of the
1162/// paragraph, whitespace, or an opening bracket must precede it.
1163fn is_opener(chars: &[char], i: usize) -> bool {
1164 match chars.get(i + 1) {
1165 Some(c) if !c.is_whitespace() => {},
1166 _ => return false,
1167 }
1168 match i.checked_sub(1).and_then(|p| chars.get(p)) {
1169 None => true,
1170 Some(&p) => p.is_whitespace() || matches!(p, '(' | '[' | '{' | '"' | '\''),
1171 }
1172}
1173
1174/// Does the delimiter at `j` flank the right of a word? A non-space must precede it, and the end of the
1175/// paragraph, whitespace, or closing punctuation must follow.
1176fn is_closer(chars: &[char], j: usize) -> bool {
1177 match j.checked_sub(1).and_then(|p| chars.get(p)) {
1178 Some(p) if !p.is_whitespace() => {},
1179 _ => return false,
1180 }
1181 match chars.get(j + 1) {
1182 None => true,
1183 Some(&c) => c.is_whitespace()
1184 || matches!(c, ')' | ']' | '}' | '.' | ',' | ';' | ':' | '!' | '?' | '"' | '\''),
1185 }
1186}
1187
1188/// The index of the first valid closing `delim` at or after `start`, or `None` when the run never
1189/// closes -- in which case the opener is ordinary text.
1190fn find_closer(chars: &[char], start: usize, delim: char) -> Option<usize> {
1191 (start..chars.len()).find(|&j| chars[j] == delim && is_closer(chars, j))
1192}
1193
1194/// Reads a Typst cross-reference at `i` (an `@`): the label of letters, digits and `- _ :` that follows,
1195/// and the index just past it. `None` when no label char follows, so a bare or escaped `@` is ordinary
1196/// text. A trailing `.` is not a label character, so `@intro.` at the end of a sentence keeps its stop.
1197fn at_label(chars: &[char], i: usize) -> Option<(String, usize)> {
1198 let start = i + 1;
1199 let mut j = start;
1200 while j < chars.len() && is_label_char(chars[j]) {
1201 j += 1;
1202 }
1203 if j == start {
1204 return None;
1205 }
1206 Some((chars[start..j].iter().collect(), j))
1207}
1208
1209/// A character legal within a Typst label. Deliberately excludes `.`, so a label does not swallow the
1210/// full stop that ends a sentence.
1211fn is_label_char(c: char) -> bool {
1212 c.is_alphanumeric() || matches!(c, '-' | '_' | ':')
1213}
1214
1215/// Reads an inline `#raw("...")` at `i`, returning its literal content and the index past the closing
1216/// `")`. `None` when the shape does not match, so a `#raw` written any other way is left as ordinary
1217/// text. Escaped quotes inside the string are not handled -- a later refinement.
1218fn raw_call(chars: &[char], i: usize) -> Option<(String, usize)> {
1219 let open = at_lit(chars, i, "#raw(\"")?;
1220 let close = (open..chars.len()).find(|&j| chars[j] == '"')?;
1221 if chars.get(close + 1) != Some(&')') {
1222 return None;
1223 }
1224 Some((chars[open..close].iter().collect(), close + 2))
1225}
1226
1227/// Reads an inline `#link(dest)[text]` (or a bare `#link(dest)`) at `i` (a `#`), returning the link's
1228/// display runs and the index just past it. The destination -- a `"url"` string or a `<label>` -- is read
1229/// and discarded, the page carrying no clickable annotation; the bracketed text is what the reader sees,
1230/// so it is parsed for its own markup. A `#link(dest)` with no following `[...]` sets the destination
1231/// string itself as its text, as Typst does. Any unhandled inline call within the text is recorded into
1232/// `skips`. `None` when the shape is not a link call or its arguments do not close.
1233fn link_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) -> Option<(Vec<Inline>, usize)> {
1234 let Some(open) = at_lit(chars, i, "#link") else { return None; };
1235 if chars.get(open) != Some(&'(') {
1236 return None;
1237 }
1238 let Some((dest, after_dest)) = read_group(chars, open) else { return None; };
1239 // A following `[...]` group is the link text; without one, the destination stands as the text.
1240 if chars.get(after_dest) == Some(&'[') {
1241 let Some((body, next)) = read_group(chars, after_dest) else { return None; };
1242 return Some((parse_inlines_in(&body, span, skips), next));
1243 }
1244 let text = link_dest_text(&dest);
1245 Some((vec![Inline::Text(text)], after_dest))
1246}
1247
1248/// The display text of a bare `#link(dest)` with no bracketed body: a `"url"` string loses its quotes, a
1249/// `<label>` its angle brackets, and anything else stands as written.
1250fn link_dest_text(dest: &str) -> String {
1251 let t = dest.trim();
1252 if let Some(inner) = t.strip_prefix('<').and_then(|s| s.strip_suffix('>')) {
1253 return inner.to_string();
1254 }
1255 unwrap_arg(t)
1256}
1257
1258/// Reads an inline `#name(...)`/`#name[...]` call at `i` (a `#`) that no earlier handler claimed, so the
1259/// reader can consume it whole rather than leak its raw markup. Returns the bracketed body's runs (parsed,
1260/// so its own markup survives) when the call is a single `[...]` content group, `None` for the body when it
1261/// carries only paren arguments, together with the index just past the call and its `#name` for the skip
1262/// report. The final `None` is returned when `i` does not open a `#name(`/`#name[` call at all, so a bare
1263/// `#` or a `#variable` interpolation is left as ordinary text.
1264fn unknown_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals)
1265 -> Option<(Option<Vec<Inline>>, usize, String)>
1266{
1267 if chars.get(i) != Some(&'#') {
1268 return None;
1269 }
1270 let start = i + 1;
1271 let mut j = start;
1272 while j < chars.len() && (chars[j].is_ascii_alphanumeric() || chars[j] == '-' || chars[j] == '_' || chars[j] == '.') {
1273 j += 1;
1274 }
1275 if j == start {
1276 return None;
1277 }
1278 let name: String = chars[start..j].iter().collect();
1279 match chars.get(j) {
1280 // A `#name[body]`: the bracketed content is the call's displayable body.
1281 Some('[') => {
1282 let Some((body, next)) = read_group(chars, j) else { return None; };
1283 Some((Some(parse_inlines_in(&body, span, skips)), next, fmt!("#{}", name)))
1284 },
1285 // A `#name(args)` and any following `[body]`: read the arguments away, then fold a body if one trails.
1286 Some('(') => {
1287 let Some((_, after_args)) = read_group(chars, j) else { return None; };
1288 if chars.get(after_args) == Some(&'[') {
1289 let Some((body, next)) = read_group(chars, after_args) else { return None; };
1290 return Some((Some(parse_inlines_in(&body, span, skips)), next, fmt!("#{}", name)));
1291 }
1292 Some((None, after_args, fmt!("#{}", name)))
1293 },
1294 _ => None,
1295 }
1296}
1297
1298/// The `#name` of a skipped line-leading code statement or standalone call, for the skip report: the
1299/// keyword itself for a block statement (`#let`, `#set`, `#show`, `#import`), or `#` and the identifier of
1300/// a standalone call. Falls back to the first whitespace-delimited token when neither shape reads, so the
1301/// tally always names something rather than nothing.
1302fn construct_name(trimmed: &str) -> String {
1303 for kw in ["#import", "#let", "#set", "#show"] {
1304 if trimmed.starts_with(kw) {
1305 return kw.to_string();
1306 }
1307 }
1308 let mut cs = trimmed.chars();
1309 if cs.next() == Some('#') {
1310 let mut ident = String::new();
1311 for c in cs {
1312 if c.is_alphanumeric() || c == '-' || c == '_' || c == '.' {
1313 ident.push(c);
1314 } else {
1315 break;
1316 }
1317 }
1318 if !ident.is_empty() {
1319 return fmt!("#{}", ident);
1320 }
1321 }
1322 trimmed.split_whitespace().next().unwrap_or(trimmed).to_string()
1323}
1324
1325/// If the literal `s` sits at `i` in `chars`, the index just past it; otherwise `None`.
1326fn at_lit(chars: &[char], i: usize, s: &str) -> Option<usize> {
1327 let mut k = i;
1328 for ch in s.chars() {
1329 if chars.get(k) != Some(&ch) {
1330 return None;
1331 }
1332 k += 1;
1333 }
1334 Some(k)
1335}
1336
1337/// One open delimiter context on the scanner's stack. Which frame is on top decides whether the next
1338/// bracket is structural: a `(` inside a `[...]` content block is author prose, not nesting, and a `(`
1339/// or `$` inside a `"..."` string or a `$...$` maths span never counts at all. This is what lets a caption
1340/// whose prose carries an unbalanced `(` still close at its `]`, where a flat depth counter stuck open to
1341/// end of source and swallowed the figure and everything after it.
1342#[derive(Clone, Copy, PartialEq)]
1343enum Frame {
1344 Code, // a `(...)`/`{...}`/`#name(...)` group, or the top level: brackets nest, `,`/`:` part
1345 Content, // a `[...]` content block: only `[` `]` nest; author `(` `)` `{` `}` are literal prose
1346 Str, // a `"..."` string literal: every character is literal until the closing quote
1347 Math, // a `$...$` maths span: every character is literal until the closing `$`
1348 Comment, // a `/* ... */` block comment: every character, brackets included, is literal until `*/`
1349 Raw, // a `` `...` `` code span: every character, `//`/`/*` included, is literal until the closing backtick
1350}
1351
1352/// The running delimiter balance while a bracketed span is scanned. The stack of [`Frame`]s replaces the
1353/// old flat `depth`: the span is closed when the stack is empty (was `depth <= 0`), and a multi-line code
1354/// skip is still open while it is not. `escaped` records that the previous character was a `\` inside a
1355/// string, maths span or content block, so a `\"`, `\$` or `\]` is passed over rather than closing its
1356/// frame. Both persist across the lines of a span, since a frame may straddle the line break.
1357pub(crate) struct SkipState {
1358 frames: Vec<Frame>,
1359 escaped: bool,
1360 in_quote: bool, // an odd number of literal `"` seen since the start of the current line, in Content mode
1361}
1362
1363/// Does a `//` at `i` open a line comment, or is it a URL's double slash (`https://...`) and so literal?
1364/// Mirrors the `://` exception in [`strip_comments`]: a `/` immediately after a `:` never starts a
1365/// comment, in code or in content prose alike.
1366fn is_line_comment(chars: &[char], i: usize) -> bool {
1367 chars.get(i + 1) == Some(&'/') && !(i > 0 && chars[i - 1] == ':')
1368}
1369
1370/// How many characters a `//` line comment opened at `i` consumes. `chars` may be a whole multi-line
1371/// capture buffer -- [`read_group`], [`split_top_args`] and [`named_arg`] all run on one -- so the comment
1372/// is bounded to the next `'\n'`, not to the end of the slice; a `//` on one line must never eat the lines
1373/// that follow it.
1374fn line_comment_len(chars: &[char], i: usize) -> usize {
1375 match chars[i..].iter().position(|&c| c == '\n') {
1376 Some(off) => off,
1377 None => chars.len() - i,
1378 }
1379}
1380
1381impl SkipState {
1382 pub(crate) fn new() -> Self {
1383 SkipState { frames: Vec::new(), escaped: false, in_quote: false }
1384 }
1385
1386 /// Is any frame still open? The top-level test for [`read_group`], [`split_top_args`] and [`named_arg`],
1387 /// where a comma or colon parts only when nothing at all is open and a group closes when the stack empties.
1388 fn is_open(&self) -> bool {
1389 !self.frames.is_empty()
1390 }
1391
1392 /// Is a structural bracket -- a `(`/`{`/`[` group -- still unclosed? This is the multi-line skip and
1393 /// capture test, matching the old flat `depth > 0`: a dangling `"` or `$` left open at the end of a line
1394 /// does not keep a construct open, since in prose a stray quote (an author's `"no bound"` split across
1395 /// two lines after an inline `#raw("...")`) or a lone `$` is a character, not the start of a code span.
1396 pub(crate) fn has_open_bracket(&self) -> bool {
1397 self.frames.iter().any(|f| matches!(f, Frame::Code | Frame::Content))
1398 }
1399
1400 /// How many structural `(`/`{`/`[` frames are nested right now -- the depth [`has_open_bracket`] only
1401 /// asks a yes/no of. A guard tracking its own single opening bracket uses this to tell its own matching
1402 /// closer (depth falls to 1) from an inner content block's closer (depth still above 1) on the same
1403 /// `]` text.
1404 pub(crate) fn open_brackets(&self) -> usize {
1405 self.frames.iter().filter(|f| matches!(f, Frame::Code | Frame::Content)).count()
1406 }
1407
1408 /// Folds the character (or, in content mode, the `#ident` run) at `i` into the stack, returning how
1409 /// many characters were consumed from `chars` -- always at least one, more for a `#name(`/`#name[`/`#x`
1410 /// run whose opener decides the frame it enters. All four scanners share this one transition so a
1411 /// bracket is counted at exactly one place, whatever their outer loops do with the characters.
1412 fn step(&mut self, chars: &[char], i: usize) -> usize {
1413 let c = chars[i];
1414 match self.frames.last().copied() {
1415 Some(Frame::Str) => {
1416 if self.escaped { self.escaped = false; }
1417 else if c == '\\' { self.escaped = true; }
1418 else if c == '"' { self.frames.pop(); }
1419 1
1420 },
1421 Some(Frame::Math) => {
1422 if self.escaped { self.escaped = false; }
1423 else if c == '\\' { self.escaped = true; }
1424 else if c == '$' { self.frames.pop(); }
1425 1
1426 },
1427 // A `/* ... */` block comment: every character, including a stray `}`/`]`/`)` an author's note
1428 // mentions, is literal until the comment's own closer -- the twin of Str/Math above, so a
1429 // `#context` guard's brace balance is never corrupted by a comment inside its body.
1430 Some(Frame::Comment) => {
1431 if c == '*' && chars.get(i + 1) == Some(&'/') { self.frames.pop(); 2 }
1432 else { 1 }
1433 },
1434 // A `` `...` `` code span: literal until the closing backtick, the twin of Comment above, so a
1435 // `//`/`/*` a prose note quotes as a raw code token (`` the `//` operator ``) is never mistaken
1436 // for a comment opener. Mirrors [`strip_comments`]' `in_raw`, which does not persist an
1437 // unterminated span past its own line, so an unclosed backtick is dropped at the newline rather
1438 // than swallowing the lines that follow.
1439 Some(Frame::Raw) => {
1440 match c {
1441 '`' => { self.frames.pop(); 1 },
1442 '\n' => { self.frames.pop(); 1 },
1443 _ => 1,
1444 }
1445 },
1446 Some(Frame::Content) => {
1447 // A `\`-escaped `\$ \[ \] \#` is literal content, so the escaped character is passed over
1448 // before any of the structural cases below can act on it.
1449 if self.escaped {
1450 self.escaped = false;
1451 return 1;
1452 }
1453 match c {
1454 '\\' => { self.escaped = true; 1 },
1455 '[' => { self.frames.push(Frame::Content); 1 },
1456 ']' => { self.frames.pop(); 1 },
1457 '$' => { self.frames.push(Frame::Math); 1 },
1458 '#' => self.content_hash(chars, i),
1459 '`' => { self.frames.push(Frame::Raw); 1 },
1460 // A literal `"` in prose is not a string (content mode never opens `Frame::Str`), but
1461 // `strip_comments` still treats a quoted phrase as opaque to `//`/`/*`, so a bare count
1462 // mirrors that here without disturbing the bracket balance a real quote would otherwise
1463 // leave alone. Line-scoped, as `strip_comments` is called once per line.
1464 '"' => { self.in_quote = !self.in_quote; 1 },
1465 '\n' => { self.in_quote = false; 1 },
1466 // A line comment runs to the next `\n` in `chars` (which may hold a whole multi-line
1467 // capture buffer, not just this one line) -- never past it, and never at all inside a
1468 // quoted phrase or a raw span. The `://` exception mirrors `strip_comments`, so a bare
1469 // URL's slashes stay literal prose. A block comment opens a `Comment` frame that can
1470 // straddle the line break, same as Str/Math above.
1471 '/' if is_line_comment(chars, i) && !self.in_quote => line_comment_len(chars, i),
1472 '/' if chars.get(i + 1) == Some(&'*') && !self.in_quote => { self.frames.push(Frame::Comment); 2 },
1473 // A `(` `)` `{` `}` in content mode is author prose, never nesting: this is the whole
1474 // point of tracking the frame, so a caption's unbalanced paren does not stick.
1475 _ => 1,
1476 }
1477 },
1478 // A code frame, or the top level (an empty stack): brackets nest as the flat counter had them,
1479 // the closer kind is not checked, and a `[` opens a content child, a `$` a maths span. A `"`
1480 // opens a real `Str` frame here, which already keeps a `//`/`/*` inside it literal, so no
1481 // separate quote count is needed the way Content mode's prose-only quote does.
1482 _ => {
1483 match c {
1484 '"' => { self.frames.push(Frame::Str); 1 },
1485 '`' => { self.frames.push(Frame::Raw); 1 },
1486 '(' | '{' => { self.frames.push(Frame::Code); 1 },
1487 '[' => { self.frames.push(Frame::Content); 1 },
1488 '$' => { self.frames.push(Frame::Math); 1 },
1489 ')' | '}' => { self.frames.pop(); 1 },
1490 '/' if is_line_comment(chars, i) => line_comment_len(chars, i),
1491 '/' if chars.get(i + 1) == Some(&'*') => { self.frames.push(Frame::Comment); 2 },
1492 _ => 1,
1493 }
1494 },
1495 }
1496 }
1497
1498 /// Handles a `#` met in content mode: a `#name` identifier follows, and its first non-identifier
1499 /// character decides the frame -- `(` opens the call's code arguments, `[` a content block, anything
1500 /// else (or end of input) is a bare `#name` field access with no group. Returns the count consumed:
1501 /// the `#`, the identifier, and, for a call or content opener, that opener too.
1502 fn content_hash(&mut self, chars: &[char], i: usize) -> usize {
1503 let mut j = i + 1;
1504 while j < chars.len() && is_call_ident(chars[j]) {
1505 j += 1;
1506 }
1507 match chars.get(j) {
1508 Some('(') => { self.frames.push(Frame::Code); j + 1 - i },
1509 Some('[') => { self.frames.push(Frame::Content); j + 1 - i },
1510 _ => j - i, // a bare `#name` (or a lone `#`): open no frame
1511 }
1512 }
1513}
1514
1515/// What to do with a line-leading Typst code statement or standalone template call.
1516enum CodeSkip {
1517 Line, // the call closes on this line; skip the one line, as before
1518 Multi(SkipState), // the delimiters are still open; begin a multi-line skip carrying the depth
1519}
1520
1521/// If this already-left-trimmed line begins a Typst code statement Austenite skips for now, decides how
1522/// much to skip: `Line` for a statement or standalone call that closes on this line, `Multi` for one
1523/// whose delimiters are still open at the end of it. `None` when the line is not code the reader skips,
1524/// so the caller sets it as prose.
1525///
1526/// The four block statements (`#import`, `#let`, `#set`, `#show`) are always code; a line-leading call
1527/// (`#name(` or `#name[`) is skipped only as a whole -- either it closes on the line, or it opens a
1528/// multi-line span. A balanced `#name[...]` with prose trailing it (`#index-main[x]More prose...`) is
1529/// left to set, since its content is a marker within a real paragraph, not a standalone call.
1530fn code_skip(trimmed: &str) -> Option<CodeSkip> {
1531 let keyword = code_keyword(trimmed);
1532 if !keyword && !opens_standalone_call(trimmed) {
1533 return None;
1534 }
1535 let mut state = SkipState::new();
1536 scan_brackets(trimmed, &mut state);
1537 if state.has_open_bracket() {
1538 return Some(CodeSkip::Multi(state));
1539 }
1540 // The delimiters balance on this line. A block statement is skipped whatever trails it; a standalone
1541 // call is skipped only when it truly ends with its own closer, so a marker inside a paragraph sets.
1542 // The `}` closer is the code-block call form (`#context{ ... }`) that closes on its own line.
1543 if keyword || trimmed.ends_with(')') || trimmed.ends_with(']') || trimmed.ends_with('}') {
1544 return Some(CodeSkip::Line);
1545 }
1546 None
1547}
1548
1549/// Does this already-left-trimmed line open one of the Typst block statements the reader skips?
1550///
1551/// `#include` sits here too, guarding against the shape it would otherwise fall through to: it takes a
1552/// bare string argument, with neither a `(` nor a `[` for [`opens_standalone_call`] to catch, so with no
1553/// entry here it reads as an ordinary paragraph line and its raw `#include "path"` prints as literal body
1554/// text. The book assembler ([`crate::book::assemble`]) resolves and follows a real `#include` itself,
1555/// before a chapter's source ever reaches this parser -- this is the belt-and-braces net for any source
1556/// that bypasses that assembler (a bare `to_blocks`/`to_blocks_with_templates` call, a test fixture): the
1557/// line is skipped and recorded rather than ever standing a chance of being set as prose.
1558fn code_keyword(trimmed: &str) -> bool {
1559 for kw in ["#import ", "#import\"", "#let ", "#set ", "#show ", "#show:", "#include ", "#include\""] {
1560 if trimmed.starts_with(kw) {
1561 return true;
1562 }
1563 }
1564 false
1565}
1566
1567/// Does this already-left-trimmed line open with a standalone call -- `#`, an identifier, then `(`, `[`
1568/// or `{`? A crude test, enough to recognise the opener of a call to an unrecognised template function
1569/// without inspecting where or whether it closes; the balance decides single- versus multi-line.
1570///
1571/// The `{` opener catches the code-block call form -- `#context { ... }`, the brace twin of
1572/// `#context[ ... ]` -- which is always Typst code at a line's head (a `{` in content mode is literal
1573/// prose, never a call opener). Typst attaches such a block to its keyword with optional whitespace
1574/// (`#context {`, the shape a book's reverse-reference index is written with, Lucronics ch29.8), so a run
1575/// of spaces before the `{` is skipped; a space before `(`, by contrast, breaks a Typst call, so only an
1576/// immediate `(` counts. No inline-call name is written with a `{`, so the `is_inline_call` guard, which
1577/// still holds off the glossary/index family, never fires on the brace form.
1578fn opens_standalone_call(trimmed: &str) -> bool {
1579 let mut cs = trimmed.chars();
1580 if cs.next() != Some('#') {
1581 return false;
1582 }
1583 let mut ident = String::new();
1584 let mut after = None; // the first character past the identifier
1585 for c in cs {
1586 if c.is_alphanumeric() || c == '-' || c == '_' {
1587 ident.push(c);
1588 continue;
1589 }
1590 after = Some(c);
1591 break;
1592 }
1593 if ident.is_empty() || is_inline_call(&ident) {
1594 // A line-leading inline glossary or index call is content, not a skippable standalone call, even
1595 // when it closes on its own line, so [`parse_inlines`] sets its display text rather than dropping it.
1596 return false;
1597 }
1598 match after {
1599 Some('(') | Some('[') | Some('{') => true,
1600 // A `{` code block may follow the keyword across whitespace (`#context {`); a `(` may not, since a
1601 // space before it breaks a Typst call. The second whitespace-split word is the block opener.
1602 Some(c) if c.is_whitespace() =>
1603 trimmed.split_whitespace().nth(1).map(|w| w.starts_with('{')).unwrap_or(false),
1604 _ => false,
1605 }
1606}
1607
1608/// Is this already-left-trimmed line a line-leading code-mode reference the reader cannot run and must
1609/// refuse rather than leak as prose? Recognises an anonymous `#{ ... }`/`#( ... )` block, an `#if`/`#for`/
1610/// `#while` control keyword, a field or method access `#name.foo`, and a bare `#name` (with only whitespace
1611/// after -- a `// comment` is stripped upstream). A `#name(`/`#name[` standalone call and the `#let`/`#set`/
1612/// `#show`/`#import` keywords are refused earlier by [`code_skip`], and a bound `#name` is expanded by
1613/// [`capture_opener`], so this catches exactly what is left -- above all a `#if`/`#{` surfacing in a re-read
1614/// content-binding body, which must be refused, not set with its leading `#`.
1615fn is_code_reference(trimmed: &str) -> bool {
1616 let rest = match trimmed.strip_prefix('#') {
1617 Some(r) => r,
1618 None => return false,
1619 };
1620 let chars: Vec<char> = rest.chars().collect();
1621 // An anonymous code block or expression.
1622 if matches!(chars.first(), Some('{') | Some('(')) {
1623 return true;
1624 }
1625 // A control keyword: `#if`/`#for`/`#while` followed by whitespace, `(` or `{`.
1626 for kw in ["if", "for", "while"] {
1627 if let Some(after) = rest.strip_prefix(kw) {
1628 match after.chars().next() {
1629 Some(c) if c.is_whitespace() || c == '(' || c == '{' => return true,
1630 _ => {},
1631 }
1632 }
1633 }
1634 // A bare identifier, or a field/method access on one.
1635 let name_len = chars.iter().take_while(|&&c| c.is_alphanumeric() || c == '-' || c == '_').count();
1636 if name_len == 0 {
1637 return false;
1638 }
1639 match chars.get(name_len) {
1640 None => true, // a bare `#name`
1641 Some('.') => true, // a field or method access `#name.foo`
1642 // A bare `#name` with only whitespace trailing it (a comment was stripped upstream); a `#name` with
1643 // trailing prose is an inline reference the paragraph keeps, and a `#name(`/`#name[` is a call handled
1644 // elsewhere.
1645 Some(c) if c.is_whitespace() => chars[name_len..].iter().all(|c| c.is_whitespace()),
1646 _ => false,
1647 }
1648}
1649
1650/// Does this already-left-trimmed standalone line consist of a bare `#name` that names a scalar `#let`
1651/// binding in scope? Only a bare reference -- an identifier with nothing but whitespace after it -- with a
1652/// name `sfns` actually binds qualifies; a `#name.field` access, a `#name(`/`#name[` call and an unbound
1653/// name all return false. It exempts exactly the standalone scalar reference from the code-reference skip
1654/// so its value is substituted (through the paragraph's own [`substitute_scalars`]), while every other
1655/// standalone code reference, above all an unbound name, stays the visible refusal it is.
1656fn names_scalar_alone(trimmed: &str, sfns: &crate::lang::rules::ScalarFns) -> bool {
1657 let rest = match trimmed.strip_prefix('#') {
1658 Some(r) => r,
1659 None => return false,
1660 };
1661 let chars: Vec<char> = rest.chars().collect();
1662 let name_len = chars.iter().take_while(|&&c| c.is_alphanumeric() || c == '-' || c == '_').count();
1663 if name_len == 0 {
1664 return false;
1665 }
1666 // A bare `#name`: nothing but whitespace after the identifier (no `.field`, no `(args)`, no `[body]`).
1667 if !chars[name_len..].iter().all(|c| c.is_whitespace()) {
1668 return false;
1669 }
1670 let name: String = chars[..name_len].iter().collect();
1671 sfns.contains_key(&name)
1672}
1673
1674/// Is this identifier one of the book template's inline functions the reader sets in place -- a glossary
1675/// or index call, a term-dictionary lookup, a hyperlink, a citation or an emphasis call? These emit body
1676/// text (or an invisible marker) mid-paragraph, so a line that opens with one is prose the inline scanner
1677/// reads, never a standalone call the line scanner skips.
1678fn is_inline_call(name: &str) -> bool {
1679 matches!(name,
1680 // The simple string-keyed glossary/index family, keyed on their own display text.
1681 "gs" | "gscap" | "gsi" | "gscapi" | "glossind" | "glossindcap"
1682 // The term-dictionary family, keyed on a `term-dict` entry: `g`/`gcap`/`gi`/`gcapi` set the value
1683 // with first-use styling, `t`/`tcap` set it plain, `graw` sets it in the mono face.
1684 | "g" | "gcap" | "gi" | "gcapi" | "t" | "tcap" | "graw"
1685 | "idx" | "idx-main" | "idx-as" | "idx-main-as" | "idx-nested"
1686 | "index" | "index-main" | "cite" | "link"
1687 | "emph" | "strong" | "super" | "sub"
1688 | "claim-label" | "claim-refs")
1689}
1690
1691/// Folds one line's delimiters into the running [`SkipState`]. A bracket inside a `"..."` string, a `$...$`
1692/// maths span or a `[...]` content block is not counted as structural nesting; the frame stack decides.
1693/// The state carries into the next line, so a frame that straddles the break is tracked correctly.
1694pub(crate) fn scan_brackets(line: &str, state: &mut SkipState) {
1695 let chars: Vec<char> = line.chars().collect();
1696 let mut i = 0;
1697 while i < chars.len() {
1698 i += state.step(&chars, i);
1699 }
1700}
1701
1702/// Splits a trailing `<label>` off a heading title: a `<name>` with no inner whitespace at the very end
1703/// labels the heading and is removed from its text. A title that merely contains angle brackets, or a
1704/// `< >` with a space inside, keeps them as ordinary characters.
1705fn split_label(title: &str) -> (String, Option<String>) {
1706 let t = title.trim_end();
1707 if let Some(inner) = t.strip_suffix('>') {
1708 if let Some(p) = inner.rfind('<') {
1709 let label = &inner[p + 1..];
1710 if !label.is_empty() && !label.contains(char::is_whitespace) {
1711 return (inner[..p].trim_end().to_string(), Some(label.to_string()));
1712 }
1713 }
1714 }
1715 (t.to_string(), None)
1716}
1717
1718/// Reads an inline `#footnote[...]` at `i` (a `#`), returning the note's inline markup -- parsed so a
1719/// `*strong*` or `_emph_` in the note sets with its own face -- and the index just past the closing `]`.
1720/// `None` when the shape is not a footnote call or its bracket does not close, so anything else is left as
1721/// ordinary text.
1722fn footnote_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) -> Option<(Vec<Inline>, usize)> {
1723 let Some(open) = at_lit(chars, i, "#footnote") else { return None; };
1724 if chars.get(open) != Some(&'[') {
1725 return None;
1726 }
1727 let Some((inner, next)) = read_group(chars, open) else { return None; };
1728 Some((parse_inlines_in(&inner, span, skips), next))
1729}
1730
1731/// Reads an inline `#emph[...]` at `i` (a `#`), returning its inner markup unreduced -- it is the call
1732/// form of `_..._` and the caller expands it the same way -- and the index just past the closing `]`.
1733/// `None` when the shape is not an emph call or its bracket does not close, so anything else is left as
1734/// ordinary text.
1735fn emph_call(chars: &[char], i: usize) -> Option<(String, usize)> {
1736 let open = at_lit(chars, i, "#emph")?;
1737 if chars.get(open) != Some(&'[') {
1738 return None;
1739 }
1740 read_group(chars, open)
1741}
1742
1743/// Reads an inline `#strong[...]` or `#strong("...")` at `i` (a `#`), returning the text to set bold and
1744/// the index just past the closing bracket. The bracket form's content is markup, unreduced -- it is the
1745/// call form of `*...*` and the caller expands it the same way emph does. The paren form's argument is
1746/// only resolved when it is a plain `"..."` string; a bare identifier or any other expression is not a
1747/// text this reader can evaluate, so `None` is returned and the generic call handler records the refusal
1748/// instead of guessing. `None` also when the shape is not a strong call or its group does not close.
1749fn strong_call(chars: &[char], i: usize) -> Option<(String, usize)> {
1750 let open = at_lit(chars, i, "#strong")?;
1751 match chars.get(open) {
1752 Some('[') => read_group(chars, open),
1753 Some('(') => {
1754 let (inner, next) = read_group(chars, open)?;
1755 let t = inner.trim();
1756 if t.len() >= 2 && t.starts_with('"') && t.ends_with('"') {
1757 Some((t[1..t.len() - 1].to_string(), next))
1758 } else {
1759 None
1760 }
1761 },
1762 _ => None,
1763 }
1764}
1765
1766/// Reads an inline `#super[...]` or `#super("...")` at `i` (a `#`), returning its content reduced to
1767/// display text by [`flatten_markup`] -- usually a short string or number -- and the index just past the
1768/// closing bracket. `None` when the shape is not a super call or its argument does not close.
1769fn super_call(chars: &[char], i: usize) -> Option<(String, usize)> {
1770 let open = at_lit(chars, i, "#super")?;
1771 match chars.get(open) {
1772 Some('[') | Some('(') => {},
1773 _ => return None,
1774 }
1775 let (inner, next) = read_group(chars, open)?;
1776 Some((flatten_markup(&unwrap_arg(&inner)), next))
1777}
1778
1779/// Reads an inline `#sub[...]` or `#sub("...")` at `i` (a `#`), returning its content reduced to display
1780/// text by [`flatten_markup`] -- usually a short string or number, as in `CO#sub[2]` -- and the index just
1781/// past the closing bracket. `None` when the shape is not a sub call or its argument does not close.
1782fn sub_call(chars: &[char], i: usize) -> Option<(String, usize)> {
1783 let open = at_lit(chars, i, "#sub")?;
1784 match chars.get(open) {
1785 Some('[') | Some('(') => {},
1786 _ => return None,
1787 }
1788 let (inner, next) = read_group(chars, open)?;
1789 Some((flatten_markup(&unwrap_arg(&inner)), next))
1790}
1791
1792/// Reads an inline `#smallcaps[...]` or `#smallcaps("...")` at `i` (a `#`), returning its content reduced
1793/// to display text by [`flatten_markup`] and the index just past the closing bracket. `None` when the shape
1794/// is not a smallcaps call or its argument does not close.
1795fn smallcaps_call(chars: &[char], i: usize) -> Option<(String, usize)> {
1796 let Some(open) = at_lit(chars, i, "#smallcaps") else { return None; };
1797 match chars.get(open) {
1798 Some('[') | Some('(') => {},
1799 _ => return None,
1800 }
1801 read_group(chars, open).map(|(inner, next)| (flatten_markup(&unwrap_arg(&inner)), next))
1802}
1803
1804/// Reads an inline `#cite(...)` at `i` (a `#`), returning the citation keys and the index past the
1805/// closing `)`. Every `<label>` token inside the parentheses is a key; a named argument such as
1806/// `form: "prose"` carries no label and is ignored. `None` when the shape is not a cite call or its
1807/// parentheses do not close, so anything else is left as ordinary text.
1808fn cite_call(chars: &[char], i: usize) -> Option<(Vec<String>, usize)> {
1809 let open = at_lit(chars, i, "#cite")?;
1810 if chars.get(open) != Some(&'(') {
1811 return None;
1812 }
1813 let (inner, next) = read_group(chars, open)?;
1814 let keys = cite_keys(&inner);
1815 Some((keys, next))
1816}
1817
1818/// Extracts the `<label>` citation keys from the inside of a `#cite(...)` call, in order. A `<` opens a
1819/// key and the next `>` closes it; anything outside a `<...>` pair (a named argument, a separating comma)
1820/// is skipped.
1821fn cite_keys(inner: &str) -> Vec<String> {
1822 let chars: Vec<char> = inner.chars().collect();
1823 let mut keys = Vec::new();
1824 let mut i = 0usize;
1825 while i < chars.len() {
1826 if chars[i] == '<' {
1827 if let Some(close) = (i + 1..chars.len()).find(|&j| chars[j] == '>') {
1828 let key: String = chars[i + 1..close].iter().collect();
1829 let key = key.trim().to_string();
1830 if !key.is_empty() {
1831 keys.push(key);
1832 }
1833 i = close + 1;
1834 continue;
1835 }
1836 }
1837 i += 1;
1838 }
1839 keys
1840}
1841
1842/// Reads an inline `#claim-label(...)` or `#claim-refs(...)` at `i` (a `#`), returning the compressed
1843/// margin code it sets, the raw reference codes it registers for the reverse claim index, and the index
1844/// just past the closing `)`. `#claim-label(..codes)` places a compressed code string in the outside
1845/// margin, so its display is the codes joined with a run of three or more consecutive same-prefix codes
1846/// collapsed to a range (`B1 B2 B3 B4` -> `B1–4`), matching the book's `claims.typ` `_compress-codes`;
1847/// `#claim-refs(..codes)` sets no visible margin code (empty display). Both register each of their raw codes
1848/// for the reverse index, matching `claims.typ`, where a label and a bare reference both emit the
1849/// `<claim-ref>` metadata `collect-claim-refs()` queries -- so a code contributes to the page list whether it
1850/// was labelled or merely referenced. Neither sets anything in the body text column, so the surrounding prose
1851/// closes over the marker's place as Typst's own body flow does. The outer `None` is returned when the shape
1852/// is not a claim call or its parentheses do not close.
1853fn claim_call(chars: &[char], i: usize) -> Option<(String, Vec<String>, usize)> {
1854 let (open, is_label) = match at_lit(chars, i, "#claim-label") {
1855 Some(o) => (o, true),
1856 None => (at_lit(chars, i, "#claim-refs")?, false),
1857 };
1858 if chars.get(open) != Some(&'(') {
1859 return None;
1860 }
1861 let (inner, next) = read_group(chars, open)?;
1862 let codes = claim_codes(&inner);
1863 let display = if is_label { compress_codes(&codes) } else { String::new() };
1864 Some((display, codes, next))
1865}
1866
1867/// The claim codes named inside a `#claim-label`/`#claim-refs` argument list, in source order: each
1868/// `<name>` label's name and each `"string"` argument's text, a `<...>` or `"..."` wrapper stripped and
1869/// anything else taken as written -- `claims.typ`'s own `str(c)` fallback for a bare argument.
1870fn claim_codes(inner: &str) -> Vec<String> {
1871 let mut out = Vec::new();
1872 for arg in split_claim_args(inner) {
1873 let a = arg.trim();
1874 if a.is_empty() {
1875 continue;
1876 }
1877 let code = if let Some(rest) = a.strip_prefix('<') {
1878 rest.strip_suffix('>').unwrap_or(rest).trim().to_string()
1879 } else if a.starts_with('"') {
1880 unwrap_arg(a)
1881 } else {
1882 a.to_string()
1883 };
1884 if !code.is_empty() {
1885 out.push(code);
1886 }
1887 }
1888 out
1889}
1890
1891/// Splits a claim argument list on its top-level commas, holding a `<...>`, `(...)` or `[...]` nesting and
1892/// a `"..."` string together so a comma inside one does not split an argument.
1893fn split_claim_args(inner: &str) -> Vec<String> {
1894 let mut out = Vec::new();
1895 let mut cur = String::new();
1896 let mut depth = 0i32;
1897 let mut in_str = false;
1898 for c in inner.chars() {
1899 match c {
1900 '"' => { in_str = !in_str; cur.push(c); },
1901 '<' | '(' | '[' if !in_str => { depth += 1; cur.push(c); },
1902 '>' | ')' | ']' if !in_str => { depth -= 1; cur.push(c); },
1903 ',' if depth == 0 && !in_str => out.push(std::mem::take(&mut cur)),
1904 _ => cur.push(c),
1905 }
1906 }
1907 if !cur.trim().is_empty() {
1908 out.push(cur);
1909 }
1910 out
1911}
1912
1913/// The margin display for a set of claim codes, a port of `claims.typ`'s `_compress-codes`: two or fewer
1914/// codes join with a space unchanged; a run of three or more consecutive codes sharing a letter prefix and
1915/// ascending by one collapses to a `first–lastnum` range (an en dash, `B1 B2 B3 B4` -> `B1–4`), a run of
1916/// exactly two is left expanded, and a code that does not parse as letters-then-digits is passed through.
1917fn compress_codes(codes: &[String]) -> String {
1918 if codes.len() <= 2 {
1919 return codes.join(" ");
1920 }
1921 // (prefix, number) for each code that matches `^([A-Za-z]+)(\d+)$`, else `None` for one that does not.
1922 let parsed: Vec<Option<(String, u64)>> = codes.iter().map(|s| parse_code(s)).collect();
1923 let mut result: Vec<String> = Vec::new();
1924 let mut i = 0usize;
1925 while i < codes.len() {
1926 let prefix = match &parsed[i] {
1927 Some((p, _)) => p.clone(),
1928 None => { result.push(codes[i].clone()); i += 1; continue; },
1929 };
1930 // The longest run from `i` sharing the prefix and ascending by one, per the Typst helper.
1931 let mut run_end = i;
1932 while run_end + 1 < codes.len() {
1933 match (&parsed[run_end + 1], &parsed[run_end]) {
1934 (Some((np, nn)), Some((_, cn))) if *np == prefix && *nn == cn + 1 => run_end += 1,
1935 _ => break,
1936 }
1937 }
1938 if run_end > i + 1 {
1939 let last_num = match &parsed[run_end] { Some((_, n)) => *n, None => 0 };
1940 result.push(fmt!("{}\u{2013}{}", codes[i], last_num));
1941 } else if run_end > i {
1942 result.push(codes[i].clone());
1943 result.push(codes[run_end].clone());
1944 } else {
1945 result.push(codes[i].clone());
1946 }
1947 i = run_end + 1;
1948 }
1949 result.join(" ")
1950}
1951
1952/// A claim code split into its letter prefix and its number, as `claims.typ`'s `^([A-Za-z]+)(\d+)$` match
1953/// does: `None` when the code is not one or more ASCII letters followed by one or more ASCII digits and
1954/// nothing else.
1955fn parse_code(s: &str) -> Option<(String, u64)> {
1956 let s = s.trim();
1957 let split = s.find(|c: char| c.is_ascii_digit())?;
1958 let (pre, num) = s.split_at(split);
1959 if pre.is_empty() || !pre.chars().all(|c| c.is_ascii_alphabetic()) {
1960 return None;
1961 }
1962 if num.is_empty() || !num.chars().all(|c| c.is_ascii_digit()) {
1963 return None;
1964 }
1965 num.parse::<u64>().ok().map(|n| (pre.to_string(), n))
1966}
1967
1968/// Reduces a run of markup to plain display text: the words a reader sees, with the emphasis, code,
1969/// glossary and index delimiters removed. A glossary term contributes its display, a visible index call
1970/// its text, a pure index marker nothing; `*strong*` and `_emph_` contribute their inner words. Inline
1971/// maths and cross-references, which have no plain form here, contribute nothing. Used where the engine
1972/// takes a plain string -- a footnote's note, a table cell, a figure caption -- and cannot yet carry the
1973/// runs themselves.
1974pub fn flatten_markup(text: &str) -> String {
1975 let mut out = String::new();
1976 for run in parse_inlines(text) {
1977 match run {
1978 Inline::Text(t) => out.push_str(&t),
1979 // A `*strong*` or `_emph_` inner run may still carry markup -- `*_word_*` nests emphasis in
1980 // strong -- so it is flattened again to strip the inner delimiters. Parsing does not nest, so
1981 // each pass removes one layer and the recursion terminates on plain text.
1982 Inline::Strong(t) => out.push_str(&flatten_markup(&t)),
1983 Inline::Emph(t) => out.push_str(&flatten_markup(&t)),
1984 Inline::BoldItalic(t) => out.push_str(&t), // already the flat inner of a nested run
1985 Inline::Super(t) => out.push_str(&t), // a flattened string cannot raise; keep its text
1986 Inline::Sub(t) => out.push_str(&t), // a flattened string cannot drop; keep its text
1987 Inline::SmallCaps(t) => out.push_str(&t), // a flattened string has no small capitals; keep its text
1988 Inline::Code(t) => out.push_str(&t),
1989 Inline::Glossary { display, .. } => out.push_str(&display),
1990 Inline::PageRef(_) => {}, // a page number has no plain form before layout
1991 Inline::Math(_) => {}, // maths is dropped from a flattened string
1992 Inline::Footnote(_) => {}, // a nested footnote is not set within a flattened string
1993 Inline::Cite(_) => {}, // a citation has no plain form before the bibliography resolves it
1994 Inline::MarginNote { .. } => {}, // a margin code is not part of the flattened body text
1995 Inline::Index { .. } => {}, // an index marker sets no words in the body text
1996 }
1997 }
1998 out
1999}
2000
2001/// The index term a call records, with a nested entry's child term where one was given.
2002struct IndexKey {
2003 term: String, // the sort key (markup flattened), e.g. "March, James"
2004 sub: Option<String>,
2005 display: String, // the display markup the index page sets, e.g. "James March" or "_Browder v. Gayle_"
2006 main: bool, // a primary reference (`idx-main`/`index-main`/`idx-main-as`); its folio sets bold
2007}
2008
2009/// What an inline glossary or index call sets into the running text. `index` carries the term the call
2010/// adds to the back-matter index, `None` for a glossary-only call (`g`/`gs`/`gscap`/`gcap`) that indexes
2011/// nothing.
2012enum Call {
2013 Glossary { term: String, display: String, index: Option<IndexKey> }, // a glossary term, keyed by `term` for first-use styling
2014 Visible { display: String, index: Option<IndexKey> }, // display text set plain, its markup parsed by the caller
2015 Invisible { index: Option<IndexKey> }, // a pure index marker: nothing is set but the term is recorded
2016}
2017
2018/// Reads an inline glossary or index call at `i` (a `#`), returning what it sets and the index just past
2019/// it, or `None` when the `#name` is not one the reader knows or its argument brackets do not close.
2020///
2021/// The visible glossary functions set their bracket content as the term, capitalising the display for
2022/// the `-cap` variants; `idx`/`idx-main` set the content plain; `idx-as`/`idx-main-as` take a second
2023/// argument as the display and set that; `index`/`index-main`/`idx-nested` are pure markers and set
2024/// nothing. First use is keyed by the term as written, matching the template's own case-sensitive
2025/// `glossary-seen` set.
2026///
2027/// The term-dictionary family keys a `term-dict` entry rather than carrying its own display, and the
2028/// reader translates the key to that value at parse time (the loader installs the map from `terms.typ`
2029/// before parsing): `g`/`gi` set the value bold-italic on first use, `gcap`/`gcapi` capitalised, `t`/`tcap`
2030/// plain and `graw` plain (its mono face is not reproduced). A key with no `term-dict` entry falls back to
2031/// the key text and is recorded in `skips`, so an unknown key is visible on the terse skip line rather
2032/// than silently wrong -- the template panics on a miss, which the reader must not.
2033fn glossary_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) -> Option<(Call, usize)> {
2034 if chars.get(i) != Some(&'#') {
2035 return None;
2036 }
2037 let start = i + 1;
2038 let mut j = start;
2039 while j < chars.len() && (chars[j].is_ascii_alphanumeric() || chars[j] == '-' || chars[j] == '_') {
2040 j += 1;
2041 }
2042 if j == start {
2043 return None;
2044 }
2045 let name: String = chars[start..j].iter().collect();
2046 if !is_inline_call(&name) {
2047 return None;
2048 }
2049 match chars.get(j) {
2050 Some('(') | Some('[') => {},
2051 _ => return None,
2052 }
2053 let (a1, next1) = read_group(chars, j)?;
2054
2055 // The two-argument display functions take the first argument as the index term and the second as the
2056 // visible text (`#idx-as("publicani")[tax farmers]` sets "tax farmers", indexes "publicani").
2057 if name == "idx-as" || name == "idx-main-as" {
2058 let (a2, next2) = read_group(chars, next1)?;
2059 // The first argument is the sort key ("March, James"), the second the display shown in the body and
2060 // set in the index ("James March"); the sort key is flattened so any markup in it does not misfile it.
2061 let display = unwrap_arg(&a2);
2062 let index = Some(IndexKey {
2063 term: flatten_markup(&unwrap_arg(&a1)),
2064 sub: None,
2065 display: display.clone(),
2066 main: name == "idx-main-as", // `idx-main-as` marks a primary reference, its folio bold
2067 });
2068 return Some((Call::Visible { display, index }, next2));
2069 }
2070 // A nested index entry is a pure marker of `parent > child`; its second argument is the child term.
2071 if name == "idx-nested" {
2072 let (child, end) = match read_group(chars, next1) {
2073 Some((c, n2)) => (Some(unwrap_arg(&c)), n2),
2074 None => (None, next1),
2075 };
2076 let parent = unwrap_arg(&a1);
2077 let index = Some(IndexKey { term: flatten_markup(&parent), sub: child, display: parent, main: false });
2078 return Some((Call::Invisible { index }, end));
2079 }
2080
2081 let arg = unwrap_arg(&a1);
2082 // A glossary+index call indexes the term it displays; a glossary-only call indexes nothing. The `-i`
2083 // suffix families and the `glossind`/`glossindcap` and `idx`/`index` families are the indexing ones,
2084 // mirroring the template's `#index`/`#index-main` calls inside each. The display carries the value's own
2085 // markup (the index page sets it), and the sort key is that value flattened to plain text.
2086 let idx_of = |value: String| Some(IndexKey { term: flatten_markup(&value), sub: None, display: value, main: false });
2087 // A primary reference (`idx-main`/`index-main`): the same key, marked `main` so its folio sets bold.
2088 let idx_of_main = |value: String| Some(IndexKey { term: flatten_markup(&value), sub: None, display: value, main: true });
2089 let call = match name.as_str() {
2090 // The simple family keys its own display text (a `term-defs` entry), so no translation applies.
2091 "gs" => Call::Glossary { term: arg.clone(), display: arg, index: None },
2092 "gsi" => Call::Glossary { term: arg.clone(), display: arg.clone(), index: idx_of(arg) },
2093 "gscap" => Call::Glossary { term: arg.clone(), display: cap_first(&arg), index: None },
2094 "gscapi" => Call::Glossary { term: arg.clone(), display: cap_first(&arg), index: idx_of(arg) },
2095 // `glossind`/`glossindcap` auto-detect: a key that is in `term-dict` sets its value, otherwise the
2096 // key stands as its own display, matching the template's `if key in term-dict` branch. Both index the
2097 // displayed value (uncapitalised), as the template's `idx-term` default does.
2098 "glossind" => {
2099 let display = term_value(&arg).unwrap_or_else(|| arg.clone());
2100 Call::Glossary { term: arg.clone(), display: display.clone(), index: idx_of(display) }
2101 },
2102 "glossindcap" => {
2103 let display = term_value(&arg).unwrap_or_else(|| arg.clone());
2104 Call::Glossary { term: arg.clone(), display: cap_first(&display), index: idx_of(display) }
2105 },
2106 // The term-dictionary family translates the key to its value; first use is keyed by the key, as the
2107 // template keys `glossary-seen` by the key name rather than the value. The `-i` variants index the
2108 // translated value.
2109 "g" => Call::Glossary { term: arg.clone(), display: resolve_term(&arg, &name, span, skips), index: None },
2110 "gi" => {
2111 let display = resolve_term(&arg, &name, span, skips);
2112 Call::Glossary { term: arg.clone(), display: display.clone(), index: idx_of(display) }
2113 },
2114 "gcap" => Call::Glossary { term: arg.clone(), display: cap_first(&resolve_term(&arg, &name, span, skips)), index: None },
2115 "gcapi" => {
2116 let display = resolve_term(&arg, &name, span, skips);
2117 Call::Glossary { term: arg.clone(), display: cap_first(&display), index: idx_of(display) }
2118 },
2119 "t" | "graw" => Call::Visible { display: resolve_term(&arg, &name, span, skips), index: None },
2120 "tcap" => Call::Visible { display: cap_first(&resolve_term(&arg, &name, span, skips)), index: None },
2121 "idx" => Call::Visible { display: arg.clone(), index: idx_of(arg) },
2122 "idx-main" => Call::Visible { display: arg.clone(), index: idx_of_main(arg) },
2123 "index" => Call::Invisible { index: idx_of(arg) },
2124 "index-main" => Call::Invisible { index: idx_of_main(arg) },
2125 _ => return None,
2126 };
2127 Some((call, next1))
2128}
2129
2130/// Resolves a term-dictionary key to its display value, or -- when no map is installed or it holds no
2131/// such key -- falls back to the key text and records the miss in `skips` under the calling function's
2132/// name, so an unknown key shows on the terse skip line rather than rendering silently as the raw key.
2133fn resolve_term(key: &str, func: &str, span: Span, skips: &mut Refusals) -> String {
2134 match term_value(key) {
2135 Some(value) => value,
2136 None => {
2137 skips.record(&fmt!("#{} unknown term-dict key {:?}", func, key), span);
2138 key.to_string()
2139 },
2140 }
2141}
2142
2143/// Reads a bracket or paren group whose opener sits at `i`, returning its inner content and the index
2144/// just past the matching closer. The [`SkipState`] frame stack decides what nests: strings, maths spans
2145/// and content blocks are respected, so a bracket inside a quoted argument, a `$...$` span or the prose of
2146/// a `[...]` caption does not close the group early -- a `(` an author left unbalanced in caption prose is
2147/// literal, and the group still closes at its own delimiter. `None` when the group never closes, so a
2148/// malformed call is left as ordinary text.
2149pub(crate) fn read_group(chars: &[char], i: usize) -> Option<(String, usize)> {
2150 let open = *chars.get(i)?;
2151 if open != '[' && open != '(' {
2152 return None;
2153 }
2154 let mut state = SkipState::new();
2155 let mut inner = String::new();
2156 // The opener pushes its frame (`[` a content block, `(` a code group) but is not part of the inner
2157 // content, so it is stepped over here and never appended.
2158 let mut j = i + state.step(chars, i);
2159 while j < chars.len() {
2160 let consumed = state.step(chars, j);
2161 if !state.is_open() {
2162 // This character closed the outer group -- the matching closer -- so the group ends just past
2163 // it, and the closer is dropped from the inner as the outer opener was.
2164 return Some((inner, j + consumed));
2165 }
2166 // Every other character is inner content, verbatim: a nested opener or closer, a string with its
2167 // quotes, a maths span, or a `#name(` run in content mode.
2168 for k in j..j + consumed {
2169 inner.push(chars[k]);
2170 }
2171 j += consumed;
2172 }
2173 None
2174}
2175
2176/// Strips a `"..."` wrapper from a paren-string argument, so `#gs("surplus")` reads the same term as
2177/// `#gs[surplus]`. A bracket argument has no quotes to strip and is returned unchanged.
2178fn unwrap_arg(inner: &str) -> String {
2179 let t = inner.trim();
2180 if t.len() >= 2 && t.starts_with('"') && t.ends_with('"') {
2181 return t[1..t.len() - 1].to_string();
2182 }
2183 inner.to_string()
2184}
2185
2186/// Capitalises the first character of a term for the `-cap` glossary variants, leaving the rest as it
2187/// stands. The template capitalises the first grapheme cluster; the first `char` matches it for every
2188/// term in these books.
2189fn cap_first(s: &str) -> String {
2190 let mut cs = s.chars();
2191 match cs.next() {
2192 Some(c) => c.to_uppercase().collect::<String>() + cs.as_str(),
2193 None => String::new(),
2194 }
2195}
2196
2197/// Whether a `/* ... */` block comment is open across the line break.
2198struct CommentState {
2199 in_block: bool,
2200}
2201
2202/// Removes Typst comments from one line: a `//` to the line's end, and any `/* ... */` span, which may
2203/// have opened on an earlier line ([`CommentState::in_block`] carries that across). A `//` or `/*`
2204/// inside a `"..."` string or a `` `code` `` span is not a comment and is kept, and a `//` immediately
2205/// after `:` is kept so a bare URL survives. Quotes and backticks are treated as span delimiters here,
2206/// which is what the reader's markup needs; a real Typst code line with string literals is skipped whole
2207/// by the caller, so stripping it never reaches the output.
2208fn strip_comments(line: &str, st: &mut CommentState) -> String {
2209 let chars: Vec<char> = line.chars().collect();
2210 let mut out = String::new();
2211 let mut in_str = false;
2212 let mut in_raw = false;
2213 let mut prev = '\0';
2214 let mut i = 0usize;
2215 while i < chars.len() {
2216 let c = chars[i];
2217 if st.in_block {
2218 if c == '*' && chars.get(i + 1) == Some(&'/') {
2219 st.in_block = false;
2220 i += 2;
2221 prev = '\0';
2222 continue;
2223 }
2224 i += 1;
2225 continue;
2226 }
2227 if in_str {
2228 out.push(c);
2229 if c == '"' { in_str = false; }
2230 prev = c;
2231 i += 1;
2232 continue;
2233 }
2234 if in_raw {
2235 out.push(c);
2236 if c == '`' { in_raw = false; }
2237 prev = c;
2238 i += 1;
2239 continue;
2240 }
2241 if c == '"' {
2242 in_str = true;
2243 out.push(c);
2244 prev = c;
2245 i += 1;
2246 continue;
2247 }
2248 if c == '`' {
2249 in_raw = true;
2250 out.push(c);
2251 prev = c;
2252 i += 1;
2253 continue;
2254 }
2255 if c == '/' && chars.get(i + 1) == Some(&'/') {
2256 if prev == ':' {
2257 out.push(c); // a `://` is part of a URL, not a comment
2258 prev = c;
2259 i += 1;
2260 continue;
2261 }
2262 break; // a line comment: drop the rest of the line
2263 }
2264 if c == '/' && chars.get(i + 1) == Some(&'*') {
2265 st.in_block = true;
2266 i += 2;
2267 prev = '\0';
2268 continue;
2269 }
2270 out.push(c);
2271 prev = c;
2272 i += 1;
2273 }
2274 out
2275}
2276
2277// -- Multi-line figure, table and data-array capture ----------------------------------------------
2278
2279/// A multi-line construct gathered whole so it can be parsed. The buffer accumulates its lines; the
2280/// bracket state closes it when the delimiters balance; the kind decides how the buffer is dispatched.
2281struct Capture {
2282 kind: CaptureKind,
2283 buf: String,
2284 state: SkipState,
2285 start: u32, // byte offset of the construct's opening line, for a `#columns` refusal's span
2286}
2287
2288/// The backstop cap on content-binding expansion depth, for a pathological *non-cyclic* chain of distinct
2289/// bindings (a cycle is already caught precisely by the name stack, at its own length). Set well above any
2290/// honest nesting yet well below the level at which the reader's own frames overflow the wasm shadow stack --
2291/// a depth-64 cap alone was found to overflow, so it would trap rather than refuse, defeating its purpose.
2292const MAX_EXPANSION_DEPTH: usize = 32;
2293
2294/// Which multi-line construct is being gathered.
2295enum CaptureKind {
2296 Figure, // a `#figure(...)` call, possibly wrapping a table or an image
2297 Table, // a bare `#table(...)` call
2298 Image, // a line-leading `#padded-image(...)` or `#image(...)` set without a figure number
2299 SectionBanner, // a line-leading `#section-banner("logo")`, a full-width grey bar carrying a section logo
2300 Let(String), // a `#let name = (...)` data array bound to this name
2301 Columns, // a `#columns(n)[ ... ]` wrapper: its body is set single-column
2302 StyledBox, // a `#styled-box[ ... ]` callout: its body is set inside a filled, padded box
2303 DeclStyle, // a `#show: <t>.with(...)` application or a lowerable `#set <target>(...)`; lowered onto the theme, not refused
2304 TemplateCall(String), // a `#name(args)?[ ... ]` call to a bound `#let` furniture function, expanded into a box
2305 ContentCall(String), // a `#name`, `#name(args)` or `#name[ ... ]` reference to a bound content binding, expanded into re-read markup spliced in
2306 Context, // a line-leading `#context { ... }`/`#context[ ... ]`: gathered whole, then either the reverse claim index (its body calls `collect-claim-refs(`) or a refusal
2307 Place, // a line-leading `#place(...)[ ... ]`: a float, set spanning its column or the page, or a refused overlay
2308 Builtin(BuiltinKind), // a line-leading Typst markup builtin the reader now sets rather than skips (`#pagebreak`, `#lorem`, `#v`)
2309}
2310
2311/// A line-leading Typst markup builtin the reader recognises and sets on the block path, rather than
2312/// tallying as a skipped construct. Each is dispatched by [`dispatch_capture`], which reads the call's
2313/// arguments from the gathered buffer -- so a builtin whose parentheses run across several lines is still
2314/// read whole.
2315enum BuiltinKind {
2316 PageBreak, // `#pagebreak()` / `#pagebreak(weak: true)`: a forced page eject
2317 Lorem, // `#lorem(<n>)`: n words of the standard lorem-ipsum placeholder, set as a paragraph
2318 Vspace, // `#v(<abs len>)`: a fixed vertical space, absolute units only
2319 ColBreak, // `#colbreak()` / `#colbreak(weak: true)`: a forced column break
2320}
2321
2322/// Detects the opener of a multi-line construct the reader parses rather than skips: a `#figure(`, a
2323/// bare `#table(`, or a `#let name = (` data array. `None` for any other line, which the caller then
2324/// offers to [`code_skip`].
2325fn capture_opener(trimmed: &str, binds: crate::lang::rules::Bindings<'_, '_>) -> Option<CaptureKind> {
2326 // A call to a bound `#let` furniture function -- `#pr-note[ ... ]`, `#aside-box(title: [..])[ ... ]` --
2327 // is expanded rather than skipped. Recognised before the generic openers so a furniture name never
2328 // collides with one of them (none of the corpus names does), and only when the map holds it, so an
2329 // unbound `#name[...]` still falls through to be tallied as a skip exactly as before.
2330 if let Some(name) = template_call_name(trimmed, binds.tfns) {
2331 return Some(CaptureKind::TemplateCall(name));
2332 }
2333 // A reference to a bound content binding -- a bare `#intro`, a `#greet("world")`, a `#note[ ... ]` -- is
2334 // gathered whole and expanded into its re-read markup. Recognised only when the map holds the name and it
2335 // is not one of the inline-call family (a content binding named `idx`/`g` must not shadow the inline call
2336 // the scanner sets in place), so an unbound or inline reference still falls through unchanged.
2337 if let Some(name) = content_call_name(trimmed, binds.cfns) {
2338 return Some(CaptureKind::ContentCall(name));
2339 }
2340 // A line-leading `#context { ... }` (or the bracket twin `#context[ ... ]`): gathered whole so its body
2341 // can be inspected for the `collect-claim-refs(` signature that marks the reverse claim index, and
2342 // otherwise refused exactly as before. Recognised here, ahead of `code_skip`, so the block reaches
2343 // [`dispatch_capture`] rather than being skipped and lost -- the reader still evaluates no `#context`.
2344 if is_context_opener(trimmed) {
2345 return Some(CaptureKind::Context);
2346 }
2347 // A line-leading Typst markup builtin -- `#pagebreak()`, `#lorem(60)`, `#v(12pt)` -- the reader now sets
2348 // rather than skipping. Recognised here, ahead of `code_skip`, so the whole call reaches
2349 // [`dispatch_capture`] (which reads its arguments) rather than being tallied as an unsupported construct
2350 // and dropped. None of these names can be a bound furniture or content function -- they are reserved by
2351 // [`crate::lang::rules::is_reserved_construct`] -- so this never shadows a corpus binding.
2352 //
2353 // Own-line only, mirroring [`code_skip`]'s own rule: the call must either open a multi-line span (its
2354 // delimiters unbalanced on this line, gathered whole below) or close on this line with nothing but
2355 // whitespace after its `)`. A balanced call with trailing prose -- `#lorem(5) more`, `#v(12pt) text` --
2356 // is NOT own-line, so it falls through to the existing visible refusal, which keeps that trailing prose;
2357 // inline mid-prose support is a later unit.
2358 if let Some(kind) = builtin_opener(trimmed) {
2359 let mut state = SkipState::new();
2360 scan_brackets(trimmed, &mut state);
2361 if state.has_open_bracket() || trimmed.trim_end().ends_with(')') {
2362 return Some(CaptureKind::Builtin(kind));
2363 }
2364 }
2365 if trimmed.starts_with("#figure(") {
2366 return Some(CaptureKind::Figure);
2367 }
2368 if trimmed.starts_with("#table(") {
2369 return Some(CaptureKind::Table);
2370 }
2371 if trimmed.starts_with("#columns(") {
2372 return Some(CaptureKind::Columns);
2373 }
2374 // A `#place(...)[ ... ]` is gathered whole: a floating one is set, spanning its column or the page; an
2375 // absolutely placed overlay is refused visibly at dispatch.
2376 if trimmed.starts_with("#place(") {
2377 return Some(CaptureKind::Place);
2378 }
2379 // A `#styled-box[ ... ]` callout: a full-measure filled box wrapping running prose. Its body opens with
2380 // the `[` on this line and closes on a later one, so it is gathered whole and re-parsed rather than
2381 // skipped -- otherwise the bracket span reads as an unbalanced standalone call and its text is dropped.
2382 if trimmed.starts_with("#styled-box[") {
2383 return Some(CaptureKind::StyledBox);
2384 }
2385 // A documentation section opens with a line-leading `#section-banner("logo")` -- a full-width grey bar
2386 // carrying the section's logo -- captured here so the bar is drawn rather than the call dropped. Tried
2387 // before `#image(`, since the name contains none of the others as a prefix.
2388 if trimmed.starts_with("#section-banner(") {
2389 return Some(CaptureKind::SectionBanner);
2390 }
2391 // A section opener draws its logo with a line-leading `#padded-image(...)` (the Pearl section's pearlite
2392 // mark), and a bare `#image(...)` places a graphic likewise. Both are set centred without a figure
2393 // number, so they are captured here rather than skipped. The hyphen keeps `#image(` from matching the
2394 // tail of `#padded-image(`, which is tried first.
2395 if trimmed.starts_with("#padded-image(") || trimmed.starts_with("#image(") {
2396 return Some(CaptureKind::Image);
2397 }
2398 // A declarative styling construct the reader lowers onto the theme rather than refusing: a
2399 // `#show: <template>.with(...)` whole-document application, a top-level `#set` on an element the theme
2400 // carries a field for, or a per-element `#show <selector>: <transform>` rule the engine collects and
2401 // applies. Capturing the rule line here stops the reader tallying it as a skipped construct while the
2402 // rule engine separately reads and applies (or refuses) it -- otherwise an authored selector rule is
2403 // double-reported, skipped by the reader AND applied by the engine. A `#set rect(...)` names no theme
2404 // element and a `#show <selector>` whose selector no element answers to are not caught, so both fall
2405 // through to [`code_skip`], staying a visible refusal; an engine-refused rule (an introspective
2406 // `it => { ... }`) surfaces through the rule engine's own diagnostic, not the reader's skip tally.
2407 if is_show_doc_with(trimmed) || is_lowerable_set(trimmed) || crate::lang::rules::is_rule_line(trimmed) {
2408 return Some(CaptureKind::DeclStyle);
2409 }
2410 let_array_name(trimmed).map(CaptureKind::Let)
2411}
2412
2413/// Does this already-left-trimmed line open a line-leading `#context` code block -- `#context {`,
2414/// `#context{` or `#context[` -- the shape the Logic appendix's reverse claim index is written with, and
2415/// the shape whose brace form once leaked verbatim as prose? The identifier must be exactly `context`
2416/// (`#contextual` does not match), and the delimiter must be a `{` (with any run of spaces before it, the
2417/// way Typst attaches a code block to its keyword) or an immediate `[`. Recognising it here routes the
2418/// whole block through [`dispatch_capture`], which either builds the index (its body calls
2419/// `collect-claim-refs(`) or refuses it, rather than skipping it as an opaque code span.
2420fn is_context_opener(trimmed: &str) -> bool {
2421 let rest = match trimmed.strip_prefix("#context") {
2422 Some(r) => r,
2423 None => return false,
2424 };
2425 // `#context[` -- the bracket twin -- attaches with no space; `#context {` -- the code block -- attaches
2426 // across any run of spaces, the way Typst binds a block to its keyword.
2427 rest.starts_with('[') || rest.trim_start().starts_with('{')
2428}
2429
2430/// If this already-left-trimmed line opens a supported Typst markup builtin -- `#pagebreak(`, `#lorem(` or
2431/// `#v(` -- the builtin it names; else `None`. The name must be followed immediately by `(`, so `#voluptas(`
2432/// (a content-mode word that happens to start with `v`) does not match `#v`, and only a genuine call is
2433/// caught.
2434fn builtin_opener(trimmed: &str) -> Option<BuiltinKind> {
2435 let rest = trimmed.strip_prefix('#')?;
2436 for (name, kind) in [
2437 ("pagebreak", BuiltinKind::PageBreak),
2438 ("colbreak", BuiltinKind::ColBreak),
2439 ("lorem", BuiltinKind::Lorem),
2440 ("v", BuiltinKind::Vspace),
2441 ] {
2442 if let Some(after) = rest.strip_prefix(name) {
2443 if after.starts_with('(') {
2444 return Some(kind);
2445 }
2446 }
2447 }
2448 None
2449}
2450
2451/// If this line opens a call to a bound furniture function -- `#<name>(` or `#<name>[` where `<name>` is a
2452/// key of `tfns` -- that name; else `None`. The name must be followed immediately by `(` (a keyword-argument
2453/// call) or `[` (a bare content call), so `#pr-note[` matches but a word that merely starts with a bound
2454/// name does not.
2455fn template_call_name(trimmed: &str, tfns: &crate::lang::rules::TemplateFns) -> Option<String> {
2456 let rest = trimmed.strip_prefix('#')?;
2457 let name_len = rest.chars().take_while(|&c| c.is_alphanumeric() || c == '-' || c == '_').count();
2458 if name_len == 0 {
2459 return None;
2460 }
2461 let name: String = rest.chars().take(name_len).collect();
2462 let next = rest.chars().nth(name_len);
2463 if (next == Some('(') || next == Some('[')) && tfns.contains_key(&name) {
2464 Some(name)
2465 } else {
2466 None
2467 }
2468}
2469
2470/// If this line is a *standalone* reference to a bound content binding -- a bare `#name`, a `#name(args)`, a
2471/// `#name[body]` or a `#name(args)[body]` whose balanced group(s) are followed by nothing but whitespace --
2472/// where `<name>` is a key of `cfns` and not one of the inline-call family, that name; else `None`. The
2473/// only-whitespace-after rule is the guard against silent word loss: `#em[Note] the rest.` carries trailing
2474/// prose, so it is NOT a standalone call -- it falls through to the paragraph, where the trailing words
2475/// survive, rather than being expanded with the rest of the sentence discarded. The inline-call guard
2476/// mirrors [`opens_standalone_call`]'s: a binding named `idx`/`g` must never shadow the inline call.
2477fn content_call_name(trimmed: &str, cfns: &crate::lang::rules::ContentFns) -> Option<String> {
2478 let rest = trimmed.strip_prefix('#')?;
2479 let chars: Vec<char> = rest.chars().collect();
2480 let name_len = chars.iter().take_while(|&&c| c.is_alphanumeric() || c == '-' || c == '_').count();
2481 if name_len == 0 {
2482 return None;
2483 }
2484 let name: String = chars[..name_len].iter().collect();
2485 if is_inline_call(&name) || !cfns.contains_key(&name) {
2486 return None;
2487 }
2488 // Step past the call's balanced group(s): a `(args)` optionally followed by a `[body]`, or a lone
2489 // `[body]`. A bare `#name` opens neither.
2490 let mut j = name_len;
2491 if chars.get(j) == Some(&'(') {
2492 j = read_group(&chars, j)?.1;
2493 }
2494 if chars.get(j) == Some(&'[') {
2495 j = read_group(&chars, j)?.1;
2496 }
2497 // Only when nothing but whitespace trails the reference is it standalone; a trailing character makes it
2498 // an inline reference the paragraph keeps whole.
2499 if chars[j..].iter().all(|c| c.is_whitespace()) {
2500 Some(name)
2501 } else {
2502 None
2503 }
2504}
2505
2506/// The `[ ... ]` body of a captured furniture call, and its keyword arguments if any. `#name[ body ]` has
2507/// no arguments (`args` empty); `#name(title: [..])[ body ]` carries the argument group before the body.
2508/// `None` when no `[ ... ]` content group follows the name, so a malformed call contributes no body.
2509fn template_call_parts(buf: &str, name: &str) -> Option<(String, String)> {
2510 let chars: Vec<char> = buf.chars().collect();
2511 let at = find_lit(&chars, &fmt!("#{}", name))?;
2512 let mut j = at + name.chars().count() + 1; // past `#name`
2513 let mut args = String::new();
2514 // An optional keyword-argument group `( ... )` immediately after the name.
2515 if chars.get(j) == Some(&'(') {
2516 let (inner, after) = read_group(&chars, j)?;
2517 args = inner;
2518 j = after;
2519 }
2520 // Whitespace between the argument group and the content body.
2521 while j < chars.len() && chars[j].is_whitespace() {
2522 j += 1;
2523 }
2524 if chars.get(j) != Some(&'[') {
2525 return None;
2526 }
2527 let (body, _) = read_group(&chars, j)?;
2528 Some((args, body))
2529}
2530
2531/// Is this line a `#show: <ident>.with(` whole-document template application -- the form whose named
2532/// arguments lower onto the theme? Distinguished from an introspective `#show ...: it => { ... }`,
2533/// which carries no `.with(` and is left to be refused.
2534fn is_show_doc_with(trimmed: &str) -> bool {
2535 let rest = match trimmed.strip_prefix("#show:") {
2536 Some(r) => r.trim_start(),
2537 None => return false,
2538 };
2539 match rest.find(".with(") {
2540 Some(dot) => {
2541 let ident = &rest[..dot];
2542 !ident.is_empty() && ident.chars().all(|c| c.is_alphanumeric() || c == '-' || c == '_')
2543 },
2544 None => false,
2545 }
2546}
2547
2548/// Is this line a lowerable top-level `#set <target>(` -- one of the elements the theme carries a field
2549/// for? The target list is [`crate::lang::set::LOWERABLE_SET_TARGETS`], the single source of truth the
2550/// lowering itself matches on, so the reader and the lowering never drift apart. A `#set` on any other
2551/// target returns `false` and is left to [`code_skip`] to refuse, since the reader has no field for it.
2552fn is_lowerable_set(trimmed: &str) -> bool {
2553 let rest = match trimmed.strip_prefix("#set ") {
2554 Some(r) => r.trim_start(),
2555 None => return false,
2556 };
2557 crate::lang::set::LOWERABLE_SET_TARGETS.iter().any(|target| {
2558 // The target must be followed immediately by `(`, so `par` does not match a `#set part(...)`.
2559 rest.strip_prefix(target).map_or(false, |after| after.starts_with('('))
2560 })
2561}
2562
2563/// If the line is a `#let name = (` binding whose value opens a paren group, its name; else `None`. Only
2564/// an array or tuple value is captured -- a scalar or a function `#let` (whose name carries `(`) is left
2565/// to [`code_skip`].
2566fn let_array_name(trimmed: &str) -> Option<String> {
2567 let rest = trimmed.strip_prefix("#let ")?;
2568 let eq = rest.find('=')?;
2569 let name = rest[..eq].trim();
2570 if name.is_empty() || !name.chars().all(|c| c.is_alphanumeric() || c == '-' || c == '_') {
2571 return None;
2572 }
2573 let value = rest[eq + 1..].trim_start();
2574 if value.starts_with('(') {
2575 Some(name.to_string())
2576 } else {
2577 None
2578 }
2579}
2580
2581/// Dispatches a completed capture: a data array is evaluated and stored under its name; a table or a
2582/// figure is parsed into an [`Item`]. A construct that does not parse -- an unresolved spread, an empty
2583/// table -- yields no item rather than an error, so a stray call never fails the whole document.
2584fn dispatch_capture(
2585 cap: Capture,
2586 items: &mut Vec<Item>,
2587 arrays: &mut HashMap<String, Vec<Vec<Inline>>>,
2588 skips: &mut Refusals,
2589 binds: crate::lang::rules::Bindings<'_, '_>,
2590)
2591{
2592 match cap.kind {
2593 CaptureKind::Let(name) => {
2594 arrays.insert(name, parse_let_array(&cap.buf));
2595 },
2596 CaptureKind::Table => {
2597 if let Some(inner) = call_inner(&cap.buf, "table") {
2598 if let Some(spec) = parse_table_spec(&inner, arrays, outer_text_size(&cap.buf)) {
2599 items.push(Item::Table { spec, span: Span::new(0, 0) });
2600 }
2601 }
2602 },
2603 CaptureKind::Figure => {
2604 if let Some(item) = parse_figure(&cap.buf, arrays) {
2605 items.push(item);
2606 }
2607 },
2608 CaptureKind::Image => {
2609 // A line-leading image call: its path and sizing are read the same way a figure's image body is,
2610 // then set as a plain centred image with no figure number. A call naming no path draws nothing.
2611 let (path, width, height, scale) = image_call(&cap.buf);
2612 if !path.is_empty() {
2613 items.push(Item::Image { path, width, height, scale, span: Span::new(0, 0) });
2614 }
2615 },
2616 CaptureKind::SectionBanner => {
2617 // The first positional argument is the logo path; a call naming none draws nothing.
2618 if let Some(path) = call_inner(&cap.buf, "section-banner").as_deref().and_then(first_string) {
2619 items.push(Item::SectionBanner { path, span: Span::new(0, 0) });
2620 }
2621 },
2622 CaptureKind::Place => {
2623 // A floating `#place`: its body is a block sequence, read through the document parser again and
2624 // set as one float. A place that does not float -- an overlay at an absolute position -- or names a
2625 // side a float cannot take is refused visibly, its body dropped with it as Typst would draw it
2626 // elsewhere, not in the flow.
2627 let span = Span::new(cap.start, cap.start);
2628 match place_float_call(&cap.buf) {
2629 Some((floating, clearance, body)) => {
2630 if let Ok((mut inner, sub)) = parse_items(&body, binds) {
2631 skips.merge(sub);
2632 // A float is laid out as one unit, so a page or column break inside it cannot be
2633 // honoured; it is refused visibly rather than dropped.
2634 refuse_nested_page_breaks(&mut inner, skips);
2635 items.push(Item::Place { items: inner, floating, clearance, span });
2636 }
2637 },
2638 None => skips.record("#place", span),
2639 }
2640 },
2641 CaptureKind::Columns => {
2642 // The reader has no column model: the `#columns(n)[ ... ]` wrapper is recorded as skipped and
2643 // its body set single-column, so the words survive even though the multi-column layout does not.
2644 // The body is a block sequence, so it is read through the document parser again and its items
2645 // spliced in; a nested refusal (a `#colbreak()`, an unknown call) folds into the same table.
2646 // The span is the wrapper's own opening line; a refusal recorded inside the re-parsed body
2647 // carries a span relative to that body text alone, not the enclosing document -- a known,
2648 // accepted imprecision for a wrapper nested this way (see `Refusal`'s own doc comment).
2649 skips.record("#columns", Span::new(cap.start, cap.start));
2650 if let Some(body) = columns_body(&cap.buf) {
2651 if let Ok((mut inner, sub)) = parse_items(&body, binds) {
2652 skips.merge(sub);
2653 // The columns body's own top-level `#set` declarations scope to the spliced subtree, the
2654 // way an included chapter's do (H1): its items splice in flat, so a scope marker pair
2655 // brackets them. An empty patch -- a body that declares no styling -- adds no markers.
2656 let patch = crate::lang::set::lower_declarations(&body);
2657 if patch == crate::theme::ThemePatch::default() {
2658 items.append(&mut inner);
2659 } else {
2660 items.push(Item::Scoped { patch, items: inner });
2661 }
2662 }
2663 }
2664 },
2665 CaptureKind::StyledBox => {
2666 // A `#styled-box[ ... ]` callout. Its body is a block sequence, so it is read through the document
2667 // parser again and wrapped in a single [`Item::Box`] the lowering sets in a filled, padded box --
2668 // unlike `#columns`, whose body splices in flat. The construct is set, not skipped, so it is not
2669 // recorded itself; a refusal within the body (an unknown inline call) still folds in.
2670 if let Some(body) = styled_box_body(&cap.buf) {
2671 if let Ok((mut inner, sub)) = parse_items(&body, binds) {
2672 skips.merge(sub);
2673 // A `#pagebreak()` nested in a callout body cannot be honoured -- the box is laid out as one
2674 // keep unit -- so it is refused visibly rather than dropped silently at render (see
2675 // [`refuse_nested_page_breaks`]).
2676 refuse_nested_page_breaks(&mut inner, skips);
2677 // The box body's own top-level `#set`/`#show: doc.with(...)` declarations lower to a patch
2678 // scoped to the box, applied to the box's subtree at render (H3) rather than the document.
2679 let patch = crate::lang::set::lower_declarations(&body);
2680 items.push(Item::Box { items: inner, patch, placement: None, span: Span::new(0, 0) });
2681 }
2682 }
2683 },
2684 CaptureKind::DeclStyle => {
2685 // A declarative styling construct -- a `#show: <template>.with(...)` application or a lowerable
2686 // top-level `#set`. The reader gathers it whole so it is neither leaked into the prose nor
2687 // blindly refused; lowering its arguments onto the theme is the book assembler's job (see
2688 // [`crate::lang::set`] and [`crate::book`]), which reads the same source with the theme in hand,
2689 // so nothing is emitted into the item stream here. A `#set` that would lower to nothing -- one
2690 // applying no argument, or naming an unrecognised or unconvertible one -- is recorded as a
2691 // refusal (H2), so it is visible rather than a silent no-op; a `#set` that fully lowers, and a
2692 // `#show: doc.with(...)`, record nothing.
2693 if let Some(name) = crate::lang::set::declstyle_refusal(&cap.buf) {
2694 skips.record(&name, Span::new(cap.start, cap.start));
2695 }
2696 },
2697 CaptureKind::Context => {
2698 // A gathered `#context { ... }` block. The reader evaluates no `#context` -- it is hard-stratified
2699 // and runs no `query` -- so the decision is by signature alone: a block whose body calls
2700 // `collect-claim-refs(` is the Logic appendix's reverse claim index, lowered to a `Block::ClaimIndex`
2701 // the author fills from the references gathered walking the body (like recognising `#print-glossary(`,
2702 // not like running it). Any other `#context` block is refused exactly as before, recorded once by
2703 // name so the section renders absent-but-reported rather than leaking its source as prose.
2704 if cap.buf.contains("collect-claim-refs(") {
2705 items.push(Item::ClaimIndex { span: Span::new(cap.start, cap.start) });
2706 } else {
2707 skips.record("#context", Span::new(cap.start, cap.start));
2708 }
2709 },
2710 CaptureKind::TemplateCall(name) => {
2711 // A call to a bound `#let` furniture function. Its `[ ... ]` body is a block sequence, re-parsed
2712 // through the document parser (with the same furniture in scope, so a nested call expands too) and
2713 // wrapped in a single `Item::Box` under the definition's lowered patch -- the callout geometry and
2714 // the inner-set overlay. A `title:` argument becomes a leading bold paragraph in the box. The call
2715 // is set, not skipped, so it is not tallied; a refusal inside the body still folds in.
2716 let tf = match binds.tfns.get(&name) {
2717 Some(tf) => tf,
2718 None => return, // the opener only fires for a bound name, so this cannot happen
2719 };
2720 match template_call_parts(&cap.buf, &name) {
2721 Some((args, body)) => {
2722 if let Ok((mut inner, sub)) = parse_items(&body, binds) {
2723 skips.merge(sub);
2724 // A `#pagebreak()` nested in a furniture callout body cannot be honoured -- the box is one
2725 // keep unit -- so it is refused visibly rather than dropped silently at render.
2726 refuse_nested_page_breaks(&mut inner, skips);
2727 // A `title:` keyword argument, its content set as a leading bold paragraph. It is set at
2728 // the title size the definition named (`text(size: 0.85em)`) by nesting it in a scope, so a
2729 // title larger or smaller than the body reads at its own size.
2730 if tf.has_title {
2731 if let Some(title) = named_content_arg(&args, "title") {
2732 let runs = parse_inlines(&title);
2733 let para = Item::Paragraph {
2734 runs: vec![Inline::Strong(inline_plain(&runs))],
2735 label: None,
2736 span: Span::new(cap.start, cap.start),
2737 };
2738 let title_item = match tf.title_size {
2739 Some(sz) => {
2740 let mut patch = crate::theme::ThemePatch::default();
2741 patch.text.body_size = Some(sz);
2742 Item::Scoped { patch, items: vec![para] }
2743 },
2744 None => para,
2745 };
2746 inner.insert(0, title_item);
2747 }
2748 }
2749 items.push(Item::Box { items: inner, patch: tf.patch.clone(), placement: tf.float, span: Span::new(cap.start, cap.start) });
2750 }
2751 },
2752 // A bound call with no `[ ... ]` body -- e.g. `#pr-note([x])`, an argument-only call this reader
2753 // cannot place -- is tallied as a skipped construct rather than dropped silently, so the report
2754 // still names it.
2755 None => skips.record(&fmt!("#{}", name), Span::new(cap.start, cap.start)),
2756 }
2757 },
2758 CaptureKind::ContentCall(name) => {
2759 // A reference to a bound content binding. Its positional arguments (none for a bare reference) are
2760 // read and substituted for each `#param` in the body, then the expanded markup is read through the
2761 // document parser again -- with the same bindings in scope, so a nested call expands too -- and its
2762 // blocks are spliced in flat. Unlike a furniture call, the body is arbitrary block markup, not a
2763 // wrap, so a heading in the binding becomes a real heading rather than a boxed paragraph. The call
2764 // is set, not skipped, so it is not tallied; a refusal inside the expanded body still folds in.
2765 let cf = match binds.cfns.get(&name) {
2766 Some(cf) => cf,
2767 None => return, // the opener only fires for a bound name, so this cannot happen
2768 };
2769 // A self- or mutually-referential binding (`#let a = [#a]`, `#let a = [#b]`/`#let b = [#a]`, or a
2770 // function form `#let f(n) = [x #f(n)]`) would re-expand without bound. The name stack catches it
2771 // the instant a name recurs -- so a cycle unwinds at its own length, never deep enough to overflow
2772 // the native stack or the wasm shadow stack -- and the depth cap is the backstop for a pathological
2773 // non-cyclic chain of distinct bindings. Either way an unbounded re-read becomes a recorded refusal.
2774 if binds.expanding(&name) {
2775 skips.record(
2776 &fmt!("#{} (cycle: content binding refers back to itself)", name),
2777 Span::new(cap.start, cap.start));
2778 return;
2779 }
2780 if binds.depth() >= MAX_EXPANSION_DEPTH {
2781 skips.record(
2782 &fmt!("#{} (cycle: expansion depth exceeds {})", name, MAX_EXPANSION_DEPTH),
2783 Span::new(cap.start, cap.start));
2784 return;
2785 }
2786 let args = content_call_args(&cap.buf, &name);
2787 let expanded = expand_content_body(cf, &args);
2788 // A styled-box content binding (`#let stamp(s) = box(fill: ..)[*v: #s*]`): the inner text is set,
2789 // but the box's own styling this reader cannot draw is recorded as a visible skip here, so the
2790 // styling is never silently lost -- only the text is kept, and it is kept, never dropped.
2791 if let Some(w) = &cf.wrapper {
2792 skips.record(&fmt!("#{} (styling dropped; content set)", w), Span::new(cap.start, cap.start));
2793 }
2794 let mut nested: Vec<String> = binds.active.to_vec();
2795 nested.push(name.clone());
2796 if let Ok((mut inner, sub)) = parse_items(&expanded, binds.with_active(&nested)) {
2797 skips.merge(sub);
2798 items.append(&mut inner);
2799 }
2800 },
2801 CaptureKind::Builtin(kind) => {
2802 let span = Span::new(cap.start, cap.start);
2803 match kind {
2804 // `#pagebreak()` / `#pagebreak(weak: true)`: a forced eject. The default (`weak: false`) is a
2805 // STRONG break, which always opens a fresh page -- even a trailing one, and even when the current
2806 // page is already empty; `weak: true` ejects only a page that carries content, matching Typst
2807 // 0.15.1 (the strong/weak flag is honoured in the driver's compose). A `to:` argument
2808 // (`pagebreak(to: "odd")`) selects a parity target the reader does not model, so it is refused
2809 // visibly rather than set as a plain break that quietly ignores the argument.
2810 BuiltinKind::PageBreak => {
2811 let inner = call_inner(&cap.buf, "pagebreak").unwrap_or_default();
2812 if inner.contains("to:") {
2813 skips.record("#pagebreak", span);
2814 } else {
2815 items.push(Item::PageBreak { weak: pagebreak_is_weak(&inner), span });
2816 }
2817 },
2818 // `#lorem(<n>)`: n words of the standard placeholder, set as one plain paragraph. A malformed
2819 // count stays a visible refusal; a zero count sets nothing; a huge count is capped at the
2820 // embedded corpus by [`lorem_words`], so generation stays bounded.
2821 BuiltinKind::Lorem => {
2822 if let Some(n) = lorem_arg(&cap.buf) {
2823 let text = lorem_words(n);
2824 if !text.is_empty() {
2825 items.push(Item::Paragraph { runs: vec![Inline::Text(text)], label: None, span });
2826 }
2827 } else {
2828 skips.record("#lorem", span); // a non-numeric argument stays a visible refusal
2829 }
2830 },
2831 // `#v(<abs len>)`: a fixed vertical space. Only an absolute first argument (pt/mm/cm/in) is set;
2832 // an `em`, `%` or `fr` length has no running size here, and a `weak:` argument asks for a
2833 // collapsing space the reader does not model -- either is refused visibly rather than set as the
2834 // wrong space, exactly as the heading-template spacer does (see [`crate::lang::rules`]).
2835 // `#colbreak()` or `#colbreak(weak: true)`: a forced column break, which on a page of one column
2836 // breaks the page, as Typst makes it.
2837 BuiltinKind::ColBreak => {
2838 let inner = call_inner(&cap.buf, "colbreak").unwrap_or_default();
2839 items.push(Item::ColBreak { weak: pagebreak_is_weak(&inner), span });
2840 },
2841 BuiltinKind::Vspace => {
2842 let inner = call_inner(&cap.buf, "v").unwrap_or_default();
2843 match parse_length(first_arg(&inner).trim()) {
2844 Some(Length::Abs(pt)) if !inner.contains("weak:") =>
2845 items.push(Item::Space { height: crate::ir::Sp::from_pt(pt), span }),
2846 _ => skips.record("#v", span),
2847 }
2848 },
2849 }
2850 },
2851 }
2852}
2853
2854/// The standard lorem-ipsum passage Typst's `#lorem` draws from, the opening of Cicero's *De Finibus* as
2855/// the `lipsum` crate carries it, word for word as `typst 0.15.1` renders it. Held to the first 120 words:
2856/// every one is plain Latin with only commas and full stops, so a placeholder of any realistic length sets
2857/// byte-identically to the oracle, while the later passage's quotation marks, question marks and en dash --
2858/// which would need their own glyph handling to match -- are left out. `#lorem(n)` for n beyond this is
2859/// capped here, so a huge count generates a bounded, sane amount rather than looping.
2860const LOREM_CORPUS: &str = "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magnam aliquam quaerat voluptatem. Ut enim aeque doleamus animo, cum corpore dolemus, fieri tamen permagna accessio potest, si aliquod aeternum et infinitum impendere malum nobis opinemur. Quod idem licet transferre in voluptatem, ut postea variari voluptas distinguique possit, augeri amplificarique non possit. At etiam Athenis, ut e patre audiebam facete et urbane Stoicos irridente, statua est in quo a nobis philosophia defensa et collaudata est, cum id, quod maxime placeat, facere possimus, omnis voluptas assumenda est, omnis dolor repellendus. Temporibus autem quibusdam et aut officiis debitis aut rerum necessitatibus saepe eveniet, ut et voluptates repudiandae sint et molestiae non recusandae. Itaque earum rerum";
2861
2862/// The first `n` words of [`LOREM_CORPUS`], joined by single spaces, with the final word's trailing
2863/// punctuation replaced by a full stop -- the shape Typst's `#lorem(n)` produces. `n` is capped at the
2864/// corpus length, so a very large count returns the whole embedded passage rather than looping unboundedly.
2865/// An `n` of zero yields the empty string, so the caller sets no paragraph.
2866fn lorem_words(n: usize) -> String {
2867 let words: Vec<&str> = LOREM_CORPUS.split_whitespace().collect();
2868 let take = n.min(words.len());
2869 if take == 0 {
2870 return String::new();
2871 }
2872 let mut out = words[..take].join(" ");
2873 // Typst ends the blind text with a full stop, dropping any comma, colon or semicolon the last word
2874 // carried in the source.
2875 let trimmed = out.trim_end_matches([',', ';', ':', '.']);
2876 out.truncate(trimmed.len());
2877 out.push('.');
2878 out
2879}
2880
2881/// The word count of a `#lorem(<n>)` call read from the gathered buffer: the first positional argument
2882/// parsed as a non-negative integer. `None` when the argument is absent or not a number, so the caller
2883/// records a refusal rather than guessing.
2884fn lorem_arg(buf: &str) -> Option<usize> {
2885 let inner = call_inner(buf, "lorem")?;
2886 first_arg(&inner).trim().parse::<usize>().ok()
2887}
2888
2889/// Does a `#pagebreak(...)` argument list ask for a WEAK break? Only an explicit `weak: true` does; a
2890/// `weak: false` and an absent argument are both the STRONG default (Typst 0.15.1), which always ejects.
2891/// The value is read as the token immediately after `weak:`, so `weak: true`, `weak:true` and
2892/// `weak: false` all resolve correctly.
2893fn pagebreak_is_weak(inner: &str) -> bool {
2894 match inner.find("weak:") {
2895 Some(at) => inner[at + "weak:".len()..].trim_start().starts_with("true"),
2896 None => false,
2897 }
2898}
2899
2900/// The first positional argument of a call's inner argument text: the run up to the first top-level comma,
2901/// so `#v(12pt, weak: true)` yields `12pt` and `#lorem(60)` yields `60`.
2902fn first_arg(inner: &str) -> String {
2903 split_arg_commas(inner).into_iter().next().unwrap_or_default()
2904}
2905
2906/// Records a visible refusal for, and removes, every `#pagebreak()` nested in a callout box's body. A
2907/// forced page eject has no meaning inside a box the layout keeps whole -- the box-body renderer has no
2908/// page to turn -- so it is refused rather than silently dropped at render. Recurses through a nested scope
2909/// or a nested box, so a break buried in either is caught too. The document top level and a `#columns` body
2910/// (which splices into the main flow, not a box) are untouched: a break there is honoured.
2911fn refuse_nested_page_breaks(items: &mut Vec<Item>, skips: &mut Refusals) {
2912 let mut kept = Vec::with_capacity(items.len());
2913 for mut item in items.drain(..) {
2914 match &mut item {
2915 Item::PageBreak { span, .. } => { skips.record("#pagebreak", *span); continue; },
2916 Item::ColBreak { span, .. } => { skips.record("#colbreak", *span); continue; },
2917 Item::Box { items: inner, .. } => refuse_nested_page_breaks(inner, skips),
2918 Item::Place { items: inner, .. } => refuse_nested_page_breaks(inner, skips),
2919 Item::Scoped { items: inner, .. } => refuse_nested_page_breaks(inner, skips),
2920 _ => {},
2921 }
2922 kept.push(item);
2923 }
2924 *items = kept;
2925}
2926
2927/// The positional arguments of a captured content-binding reference, each evaluated to its substitution
2928/// text: a `"quoted string"` yields its contents, a `[bracketed content]` its inner markup, and any other
2929/// value (a number, an identifier) its trimmed source. A bare `#name` reference, or a `#name[ ... ]` whose
2930/// single argument is the bracket body, is handled too. An empty list when the reference takes none.
2931fn content_call_args(buf: &str, name: &str) -> Vec<String> {
2932 let chars: Vec<char> = buf.chars().collect();
2933 let at = match find_lit(&chars, &fmt!("#{}", name)) {
2934 Some(a) => a,
2935 None => return Vec::new(),
2936 };
2937 let j = at + name.chars().count() + 1; // past `#name`
2938 match chars.get(j) {
2939 Some('(') => {
2940 match read_group(&chars, j) {
2941 Some((inner, _)) => split_arg_commas(&inner).into_iter()
2942 .map(|a| content_arg_value(a.trim()))
2943 .collect(),
2944 None => Vec::new(),
2945 }
2946 },
2947 // A `#name[ ... ]` call: the bracket body is the single positional argument.
2948 Some('[') => match read_group(&chars, j) {
2949 Some((inner, _)) => vec![inner],
2950 None => Vec::new(),
2951 },
2952 _ => Vec::new(), // a bare `#name` reference
2953 }
2954}
2955
2956/// Evaluates one content-binding argument to its substitution text: a `"..."` string is unquoted, a
2957/// `[ ... ]` content block is unwrapped to its inner markup, and any other value is kept as its trimmed
2958/// source.
2959fn content_arg_value(arg: &str) -> String {
2960 let t = arg.trim();
2961 if t.starts_with('[') {
2962 let chars: Vec<char> = t.chars().collect();
2963 if let Some((inner, _)) = read_group(&chars, 0) {
2964 return inner;
2965 }
2966 }
2967 unwrap_arg(t)
2968}
2969
2970/// Splits an argument list on its top-level commas, honouring `(`/`[`/`{` nesting and `"..."` strings so a
2971/// comma inside a bracketed or quoted argument does not split it.
2972fn split_arg_commas(s: &str) -> Vec<String> {
2973 let mut out = Vec::new();
2974 let mut depth = 0i32;
2975 let mut in_str = false;
2976 let mut esc = false;
2977 let mut cur = String::new();
2978 for c in s.chars() {
2979 if in_str {
2980 cur.push(c);
2981 if esc { esc = false; }
2982 else if c == '\\' { esc = true; }
2983 else if c == '"' { in_str = false; }
2984 continue;
2985 }
2986 match c {
2987 '"' => { in_str = true; cur.push(c); },
2988 '(' | '[' | '{' => { depth += 1; cur.push(c); },
2989 ')' | ']' | '}' => { depth -= 1; cur.push(c); },
2990 ',' if depth == 0 => out.push(std::mem::take(&mut cur)),
2991 _ => cur.push(c),
2992 }
2993 }
2994 if !cur.trim().is_empty() {
2995 out.push(cur);
2996 }
2997 out
2998}
2999
3000/// Substitutes a content binding's positional arguments into its body: each line-leading or inline `#param`
3001/// token whose identifier names a parameter is replaced by the matching argument's text (a parameter with no
3002/// argument supplied substitutes empty). A `#` that opens no parameter name -- an inline call, an escaped
3003/// literal -- is left untouched, so the body's own markup survives the substitution. A binding with no
3004/// parameters returns its body verbatim.
3005///
3006/// A `#name(...)` call immediately following a non-parameter identifier -- `#t(w)`, `#g(w)` -- opens Typst's
3007/// own code mode inside its parentheses, where `w` is a bare variable reference rather than a markup-mode
3008/// `#w`. [`substitute_call_args`] re-quotes any such bare parameter reference found inside that argument
3009/// list, so the nested call -- most often a term-dictionary lookup -- resolves against the substituted value
3010/// rather than the parameter's own name. Substitution runs here, before the expanded body is re-parsed, so
3011/// term-dictionary (or any other) resolution downstream always sees the caller's argument, never the
3012/// parameter placeholder: the root fix is ordering, not a term-dictionary-specific patch.
3013fn expand_content_body(cf: &crate::lang::rules::ContentFn, args: &[String]) -> String {
3014 if cf.params.is_empty() {
3015 return cf.body.clone();
3016 }
3017 let chars: Vec<char> = cf.body.chars().collect();
3018 let mut out = String::new();
3019 let mut i = 0usize;
3020 while i < chars.len() {
3021 if chars[i] == '#' {
3022 let mut j = i + 1;
3023 while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') {
3024 j += 1;
3025 }
3026 let ident: String = chars[i + 1..j].iter().collect();
3027 if let Some(pos) = cf.params.iter().position(|p| *p == ident) {
3028 out.push_str(args.get(pos).map(|s| s.as_str()).unwrap_or(""));
3029 i = j;
3030 continue;
3031 }
3032 // Not a bare parameter reference; if it names a call (`#t(`, `#g(`, `#anything(`), its argument
3033 // list is code mode, where a parameter appears with no leading `#`. Substitute there too, then
3034 // resume scanning after the call's closing paren -- the call name and everything past it are
3035 // otherwise untouched.
3036 if !ident.is_empty() && chars.get(j) == Some(&'(') {
3037 if let Some((inner, after)) = read_group(&chars, j) {
3038 out.push('#');
3039 out.push_str(&ident);
3040 out.push('(');
3041 out.push_str(&substitute_call_args(&inner, &cf.params, args));
3042 out.push(')');
3043 i = after;
3044 continue;
3045 }
3046 }
3047 }
3048 out.push(chars[i]);
3049 i += 1;
3050 }
3051 out
3052}
3053
3054/// Substitutes a bare parameter identifier found inside a call's argument list -- Typst code mode, where a
3055/// parameter carries no leading `#` of its own -- with a quoted literal of the caller's argument text, so a
3056/// nested call such as `#t(w)` sees the substituted value rather than the parameter's own name. A quote or
3057/// backslash in the substituted text is escaped, so the result is always one valid string literal. Left alone
3058/// inside an existing `"..."` string, so a quoted argument that merely contains the parameter's name as a
3059/// word is not mistaken for a reference to it. An identifier that is not a parameter -- the call's own other
3060/// arguments, a keyword name, a literal -- passes through unchanged.
3061fn substitute_call_args(inner: &str, params: &[String], args: &[String]) -> String {
3062 let chars: Vec<char> = inner.chars().collect();
3063 let mut out = String::new();
3064 let mut i = 0usize;
3065 let mut in_str = false;
3066 while i < chars.len() {
3067 let c = chars[i];
3068 if in_str {
3069 out.push(c);
3070 if c == '\\' && i + 1 < chars.len() {
3071 out.push(chars[i + 1]);
3072 i += 2;
3073 continue;
3074 }
3075 if c == '"' {
3076 in_str = false;
3077 }
3078 i += 1;
3079 continue;
3080 }
3081 if c == '"' {
3082 in_str = true;
3083 out.push(c);
3084 i += 1;
3085 continue;
3086 }
3087 if c.is_alphabetic() || c == '_' {
3088 let start = i;
3089 let mut j = i + 1;
3090 while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') {
3091 j += 1;
3092 }
3093 let word: String = chars[start..j].iter().collect();
3094 match params.iter().position(|p| *p == word) {
3095 Some(pos) => {
3096 let value = args.get(pos).map(|s| s.as_str()).unwrap_or("");
3097 out.push('"');
3098 out.push_str(&value.replace('\\', "\\\\").replace('"', "\\\""));
3099 out.push('"');
3100 },
3101 None => out.push_str(&word),
3102 }
3103 i = j;
3104 continue;
3105 }
3106 out.push(c);
3107 i += 1;
3108 }
3109 out
3110}
3111
3112/// Expands every inline mid-prose reference to a bound content binding within a run of already-joined
3113/// paragraph (or heading) text, splicing each call's argument-substituted, re-expanded body into the
3114/// surrounding prose at the position it stood -- the inline twin of the own-line [`CaptureKind::ContentCall`]
3115/// path (which splices a binding referenced alone on its line as blocks). A `M#oxe`, a
3116/// `see #stamp("v2") for details` or a `#note[x]` mid-sentence therefore sets its expansion with the words
3117/// before AND after it kept, rather than leaking its raw `#name` onto the page or refusing the call and
3118/// losing it.
3119///
3120/// Runs as a textual pre-pass in [`flush_para`] (and on a heading title), BEFORE [`substitute_scalars`] and
3121/// [`parse_inlines_in`] -- the same shape the scalar pass takes, and for the same reason: a content binding's
3122/// value is markup that must be re-read as the surrounding prose's own, so expanding it here lets the one
3123/// inline scanner downstream read the spliced result. A glossary term, an emphasis or a maths span in the
3124/// body sets exactly as if it had been typed in place, and an unrenderable primitive in the body (a `#h`, a
3125/// `#box`) reaches that scanner's own visible-refusal path rather than a parallel one here -- so the root fix
3126/// reuses [`expand_content_body`] and the block path's cycle rule rather than building a second expander.
3127///
3128/// `active` (carried on `binds`, seeded from an enclosing block-level expansion) is the stack of binding
3129/// names currently expanding: a reference to a name already on it is refused as a cycle, exactly as the block
3130/// path refuses one, so a self- or mutually-referential body unwinds at its own length rather than looping;
3131/// [`MAX_EXPANSION_DEPTH`] is the backstop for a pathological non-cyclic chain. A `#name` that names no
3132/// content binding, that is one of the inline-call family (so a binding named `g`/`idx` never shadows the
3133/// inline call the scanner sets in place), that is a field/method access (`#name.foo`), or that sits in a raw
3134/// `` `...` `` span, a `$...$` maths span or behind a `\`-escape is left untouched, so the downstream scanner
3135/// still sets or refuses it exactly as before. A binding referenced own-line is handled earlier by
3136/// [`capture_opener`], so this only ever sees a genuinely mid-prose reference.
3137pub(crate) fn substitute_content_calls(
3138 text: &str,
3139 binds: crate::lang::rules::Bindings<'_, '_>,
3140 skips: &mut Refusals,
3141 span: Span,
3142)
3143 -> String
3144{
3145 // The common case, no content bindings in scope: costs nothing beyond the check, so a document that uses
3146 // none reads exactly as before.
3147 if binds.cfns.is_empty() {
3148 return text.to_string();
3149 }
3150 let chars: Vec<char> = text.chars().collect();
3151 let mut out = String::new();
3152 let mut i = 0usize;
3153 let mut in_raw = false;
3154 let mut in_math = false;
3155 while i < chars.len() {
3156 let c = chars[i];
3157 // An escaped `\#` (or any `\`-escape) is literal: the backslash and its character are passed straight
3158 // through, so the downstream scanner still turns `\#` into a literal `#`.
3159 if c == '\\' && i + 1 < chars.len() {
3160 out.push(c);
3161 out.push(chars[i + 1]);
3162 i += 2;
3163 continue;
3164 }
3165 if c == '`' {
3166 in_raw = !in_raw;
3167 out.push(c);
3168 i += 1;
3169 continue;
3170 }
3171 if c == '$' && !in_raw {
3172 in_math = !in_math;
3173 out.push(c);
3174 i += 1;
3175 continue;
3176 }
3177 if c == '#' && !in_raw && !in_math {
3178 let start = i + 1;
3179 let mut j = start;
3180 while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') {
3181 j += 1;
3182 }
3183 if j > start {
3184 let name: String = chars[start..j].iter().collect();
3185 // A field or method access on the name -- `#name.foo` -- is a code-mode expression, not a bare
3186 // content reference: leave it for the downstream refusal path rather than expanding the name and
3187 // stranding the `.foo`. Only a `.` FOLLOWED BY an identifier is an access, though: a `.` before
3188 // whitespace, end of input or punctuation is a sentence's full stop, which Typst sets literally
3189 // after expanding the reference (`M#oxe.` sets `MX.`), so it must not block the expansion.
3190 let field_access = chars.get(j) == Some(&'.')
3191 && chars.get(j + 1).map_or(false, |c| c.is_alphabetic() || *c == '_');
3192 if !field_access && !is_inline_call(&name) {
3193 if let Some(cf) = binds.cfns.get(&name) {
3194 // Step over the call's balanced group(s) -- a `(args)` optionally followed by a `[body]`,
3195 // or a lone `[body]` -- reading its positional arguments the same way [`content_call_args`]
3196 // reads a captured own-line call's, so both paths substitute identically.
3197 let mut k = j;
3198 let mut args: Vec<String> = Vec::new();
3199 if chars.get(k) == Some(&'(') {
3200 if let Some((inner, after)) = read_group(&chars, k) {
3201 args = split_arg_commas(&inner).into_iter()
3202 .map(|a| content_arg_value(a.trim()))
3203 .collect();
3204 k = after;
3205 }
3206 }
3207 if chars.get(k) == Some(&'[') {
3208 if let Some((inner, after)) = read_group(&chars, k) {
3209 // A `#name[ ... ]` call with no paren group: the bracket body is the single
3210 // positional argument, mirroring [`content_call_args`]'s own bracket arm.
3211 if args.is_empty() {
3212 args = vec![inner];
3213 }
3214 k = after;
3215 }
3216 }
3217 // A self- or mutually-referential binding is refused the instant its name recurs, so a
3218 // cycle unwinds at its own length; the depth cap is the backstop for a pathological chain
3219 // of distinct bindings. Either way the call is consumed (the surrounding prose is kept)
3220 // and a visible refusal recorded, exactly as the own-line path does.
3221 if binds.expanding(&name) {
3222 skips.record(
3223 &fmt!("#{} (cycle: content binding refers back to itself)", name), span);
3224 i = k;
3225 continue;
3226 }
3227 if binds.depth() >= MAX_EXPANSION_DEPTH {
3228 skips.record(
3229 &fmt!("#{} (cycle: expansion depth exceeds {})", name, MAX_EXPANSION_DEPTH), span);
3230 i = k;
3231 continue;
3232 }
3233 let expanded = expand_content_body(cf, &args);
3234 // A styled-box content binding used inline (`see #stamp("v2") for details`): the inner text
3235 // is spliced into the surrounding prose, and the box's own styling this reader cannot draw
3236 // is recorded as a visible skip -- the styling is never silently lost, the text never dropped.
3237 if let Some(w) = &cf.wrapper {
3238 skips.record(&fmt!("#{} (styling dropped; content set)", w), span);
3239 }
3240 let mut nested: Vec<String> = binds.active.to_vec();
3241 nested.push(name.clone());
3242 // The expanded body may itself reference another binding inline, so it is expanded in
3243 // turn with this name pushed onto the active stack -- the recursion the cycle guard bounds.
3244 out.push_str(&substitute_content_calls(&expanded, binds.with_active(&nested), skips, span));
3245 i = k;
3246 continue;
3247 }
3248 }
3249 }
3250 }
3251 out.push(c);
3252 i += 1;
3253 }
3254 out
3255}
3256
3257/// Substitutes a bare `#name` reference to a scalar `#let` value binding -- `#let title = "Foo"`, then a
3258/// later `#title` -- with its display text: a string's contents, or a number/length literal's own source
3259/// text (see [`crate::lang::rules::ScalarValue::display_text`]). Runs once, on a heading's title or a
3260/// flushed paragraph's joined text, BEFORE [`parse_inlines_in`] reads it, rather than as a case inside that
3261/// scanner: a scalar's value is plain text needing no further markup expansion, so a textual pass here is
3262/// the whole fix, and it leaves every other construct -- a table, a figure, a caption, a `#context` block,
3263/// a data array -- untouched, since none of those reach this function.
3264///
3265/// A `#name` inside a raw `` `...` `` code span or a `$...$` maths span is left alone, matching
3266/// [`parse_inlines_in`]'s own treatment of those as literal/foreign territory; an escaped `\#name` is left
3267/// alone too, so the backslash still reaches the inline scanner to produce a literal `#`. A name with no
3268/// scalar binding, or one immediately followed by `(` or `[` (a call, not a bare reference), is untouched --
3269/// it still reaches the inline scanner's own refusal path, so an unbound or non-scalar reference stays a
3270/// visible skip rather than becoming a silent drop here.
3271pub(crate) fn substitute_scalars(text: &str, sfns: &crate::lang::rules::ScalarFns) -> String {
3272 if sfns.is_empty() {
3273 return text.to_string(); // the common case, no scalar bindings in scope -- costs nothing beyond the check
3274 }
3275 let chars: Vec<char> = text.chars().collect();
3276 let mut out = String::new();
3277 let mut i = 0usize;
3278 let mut in_raw = false;
3279 let mut in_math = false;
3280 while i < chars.len() {
3281 let c = chars[i];
3282 if c == '\\' && i + 1 < chars.len() {
3283 out.push(c);
3284 out.push(chars[i + 1]);
3285 i += 2;
3286 continue;
3287 }
3288 if c == '`' {
3289 in_raw = !in_raw;
3290 out.push(c);
3291 i += 1;
3292 continue;
3293 }
3294 if c == '$' && !in_raw {
3295 in_math = !in_math;
3296 out.push(c);
3297 i += 1;
3298 continue;
3299 }
3300 if c == '#' && !in_raw && !in_math {
3301 let start = i + 1;
3302 let mut j = start;
3303 while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') {
3304 j += 1;
3305 }
3306 if j > start {
3307 let name: String = chars[start..j].iter().collect();
3308 if !matches!(chars.get(j), Some('(') | Some('[')) {
3309 if let Some(value) = sfns.get(&name) {
3310 out.push_str(value.display_text());
3311 i = j;
3312 continue;
3313 }
3314 }
3315 }
3316 }
3317 out.push(c);
3318 i += 1;
3319 }
3320 out
3321}
3322
3323/// The content of a `key: [ ... ]` keyword argument, without its brackets. Used to read a furniture call's
3324/// `title:` argument. `None` when the key is absent or its value is not a content block.
3325fn named_content_arg(args: &str, key: &str) -> Option<String> {
3326 let at = args.find(key)?;
3327 let after = &args[at + key.len()..];
3328 let after = after.trim_start();
3329 let after = after.strip_prefix(':')?.trim_start();
3330 if !after.starts_with('[') {
3331 return None;
3332 }
3333 let chars: Vec<char> = after.chars().collect();
3334 read_group(&chars, 0).map(|(inner, _)| inner)
3335}
3336
3337/// The plain text of a run of inline markup, dropping the markup and keeping the words -- a title is set as
3338/// one bold run, so its own emphasis is flattened rather than nested inside the bold.
3339fn inline_plain(runs: &[Inline]) -> String {
3340 let mut out = String::new();
3341 for run in runs {
3342 match run {
3343 Inline::Text(t) | Inline::Strong(t) | Inline::Emph(t) | Inline::BoldItalic(t)
3344 | Inline::Super(t) | Inline::Sub(t) | Inline::Code(t) => out.push_str(t),
3345 _ => {},
3346 }
3347 }
3348 out
3349}
3350
3351/// The `[ ... ]` body of a captured `#columns(n)[ ... ]` wrapper: the column count arguments are read and
3352/// dropped, and the bracketed block content returned for re-parsing. `None` when no `[...]` group follows
3353/// the arguments, so a malformed wrapper contributes no body.
3354fn columns_body(buf: &str) -> Option<String> {
3355 let chars: Vec<char> = buf.chars().collect();
3356 let Some(at) = find_lit(&chars, "#columns") else { return None; };
3357 let open = at + "#columns".chars().count();
3358 if chars.get(open) != Some(&'(') {
3359 return None;
3360 }
3361 let Some((_, after_args)) = read_group(&chars, open) else { return None; };
3362 let mut j = after_args;
3363 while j < chars.len() && chars[j].is_whitespace() {
3364 j += 1;
3365 }
3366 if chars.get(j) != Some(&'[') {
3367 return None;
3368 }
3369 read_group(&chars, j).map(|(body, _)| body)
3370}
3371
3372/// Reads a captured `#place(...)[ ... ]` as a float: its side from the alignment argument (`top`, `bottom`
3373/// or `auto`, any horizontal part aside), its `scope:` and `clearance:`, and its body for re-parsing. `None`
3374/// for a place that does not float (`float:` absent or false -- an overlay at an absolute position the
3375/// reader does not set), one naming no side a float can take, one carrying an offset (`dx:`, `dy:`), or
3376/// one with no body.
3377fn place_float_call(buf: &str) -> Option<(Floating, Option<Spacing>, String)> {
3378 let chars: Vec<char> = buf.chars().collect();
3379 let Some(at) = find_lit(&chars, "#place") else { return None; };
3380 let open = at + "#place".chars().count();
3381 if chars.get(open) != Some(&'(') {
3382 return None;
3383 }
3384 let Some((inner, after)) = read_group(&chars, open) else { return None; };
3385 let mut side = None;
3386 let mut float = false;
3387 let mut scope = FloatScope::Column;
3388 let mut clearance = None;
3389 let mut body = None;
3390 for arg in split_top_args(&inner) {
3391 let a = arg.trim();
3392 if a.is_empty() {
3393 continue;
3394 }
3395 if let Some((key, val)) = named_arg(a) {
3396 match key.as_str() {
3397 "float" => float = val.trim() == "true",
3398 "scope" => scope = parse_scope(&val),
3399 "clearance" => match parse_spacing(&val) {
3400 Some(c) => clearance = Some(c),
3401 None => return None,
3402 },
3403 _ => return None,
3404 }
3405 continue;
3406 }
3407 if a.starts_with('[') {
3408 let ac: Vec<char> = a.chars().collect();
3409 body = read_group(&ac, 0).map(|(b, _)| b);
3410 continue;
3411 }
3412 // The alignment: its vertical part decides the side; a horizontal part does not move a float.
3413 for part in a.split('+') {
3414 match part.trim() {
3415 "top" => side = Some(FloatPlacement::Top),
3416 "bottom" => side = Some(FloatPlacement::Bottom),
3417 "auto" => side = Some(FloatPlacement::Auto),
3418 "left" | "center" | "right" | "start" | "end" => {},
3419 _ => return None,
3420 }
3421 }
3422 }
3423 if !float {
3424 return None;
3425 }
3426 // The trailing content block, `#place(...)[ ... ]`, when the body was not passed as an argument.
3427 if body.is_none() {
3428 let mut j = after;
3429 while j < chars.len() && chars[j].is_whitespace() {
3430 j += 1;
3431 }
3432 if chars.get(j) == Some(&'[') {
3433 body = read_group(&chars, j).map(|(b, _)| b);
3434 }
3435 }
3436 // A float with no side named takes `auto`, the default a floating `place` resolves its alignment to.
3437 let side = side.unwrap_or(FloatPlacement::Auto);
3438 body.map(|b| (Floating { side, scope }, clearance, b))
3439}
3440
3441/// Reads a length that may be relative to the text size: `1.5em` in ems, anything [`parse_length`] reads
3442/// as points. `None` for a percentage or an unreadable value.
3443fn parse_spacing(val: &str) -> Option<Spacing> {
3444 let v = val.trim();
3445 if let Some(em) = v.strip_suffix("em") {
3446 return em.trim().parse::<f64>().ok().map(Spacing::Em);
3447 }
3448 match parse_length(v) {
3449 Some(Length::Abs(pt)) => Some(Spacing::Pt(pt)),
3450 _ => None,
3451 }
3452}
3453
3454/// The `[ ... ]` body of a captured `#styled-box[ ... ]` callout, returned for re-parsing. The call takes
3455/// no arguments, so the bracket group opens immediately after the name. `None` when no `[...]` follows, so
3456/// a malformed callout contributes no body.
3457fn styled_box_body(buf: &str) -> Option<String> {
3458 let chars: Vec<char> = buf.chars().collect();
3459 let Some(at) = find_lit(&chars, "#styled-box") else { return None; };
3460 let open = at + "#styled-box".chars().count();
3461 if chars.get(open) != Some(&'[') {
3462 return None;
3463 }
3464 read_group(&chars, open).map(|(body, _)| body)
3465}
3466
3467/// The index of the first occurrence of the literal `s` in `chars`, or `None`.
3468fn find_lit(chars: &[char], s: &str) -> Option<usize> {
3469 let pat: Vec<char> = s.chars().collect();
3470 if pat.is_empty() || chars.len() < pat.len() {
3471 return None;
3472 }
3473 (0..=chars.len() - pat.len()).find(|&start| chars[start..start + pat.len()] == pat[..])
3474}
3475
3476/// Evaluates a `#let name = (...)` value into the flat sequence of cells it holds, each cell its inline
3477/// runs. The value is the paren group after the `=`; every `[...]` group within it, at any depth, is one
3478/// cell -- which is what `array.flatten()` yields for an array of content tuples.
3479fn parse_let_array(buf: &str) -> Vec<Vec<Inline>> {
3480 let chars: Vec<char> = buf.chars().collect();
3481 let eq = match chars.iter().position(|&c| c == '=') {
3482 Some(e) => e,
3483 None => return Vec::new(),
3484 };
3485 let open = match (eq + 1..chars.len()).find(|&j| !chars[j].is_whitespace()) {
3486 Some(v) if chars[v] == '(' => v,
3487 _ => return Vec::new(),
3488 };
3489 match read_group(&chars, open) {
3490 Some((inner, _)) => collect_cells(&inner),
3491 None => Vec::new(),
3492 }
3493}
3494
3495/// Collects every `[...]` group in `inner`, in order, each parsed into its inline runs. A `[` inside a
3496/// string is not a cell. Once a group opens, its whole content is one cell and is not descended into.
3497///
3498/// A `table.cell(colspan: n)[...]` wrapper is expanded to `n` grid cells: the bracketed content, then
3499/// `n - 1` empty span placeholders. A spanning cell consumes several columns in Typst, so without the
3500/// placeholders every cell after it slides one column to the left and the last column overruns the
3501/// table; the placeholders keep the flat cell stream aligned to the column grid. A bare `table.cell(...)`
3502/// (no `colspan`) is one cell, like a plain `[...]`.
3503fn collect_cells(inner: &str) -> Vec<Vec<Inline>> {
3504 let chars: Vec<char> = inner.chars().collect();
3505 let mut cells = Vec::new();
3506 let mut in_str = false;
3507 let mut esc = false;
3508 let mut i = 0usize;
3509 while i < chars.len() {
3510 let c = chars[i];
3511 if in_str {
3512 if esc { esc = false; }
3513 else if c == '\\' { esc = true; }
3514 else if c == '"' { in_str = false; }
3515 i += 1;
3516 continue;
3517 }
3518 if c == '"' {
3519 in_str = true;
3520 i += 1;
3521 continue;
3522 }
3523 // A `table.cell(args)[content]` wrapper: read the args for a `colspan`, then take the following
3524 // `[...]` as the content and emit it across that many columns.
3525 if let Some(after) = at_lit(&chars, i, "table.cell") {
3526 if chars.get(after) == Some(&'(') {
3527 if let Some((args, past_args)) = read_group(&chars, after) {
3528 let span = cell_colspan(&args);
3529 let mut j = past_args;
3530 while j < chars.len() && chars[j].is_whitespace() {
3531 j += 1;
3532 }
3533 if chars.get(j) == Some(&'[') {
3534 if let Some((content, next)) = read_group(&chars, j) {
3535 cells.push(parse_inlines(&content));
3536 for _ in 1..span {
3537 cells.push(vec![Inline::Text(String::new())]);
3538 }
3539 i = next;
3540 continue;
3541 }
3542 }
3543 // No content group followed the wrapper; step past its args and carry on.
3544 i = past_args;
3545 continue;
3546 }
3547 }
3548 }
3549 if c == '[' {
3550 if let Some((content, next)) = read_group(&chars, i) {
3551 cells.push(parse_inlines(&content));
3552 i = next;
3553 continue;
3554 }
3555 }
3556 i += 1;
3557 }
3558 cells
3559}
3560
3561/// The `colspan:` of a `table.cell(...)` argument list, at least one. A missing or unreadable `colspan`
3562/// is a single column.
3563fn cell_colspan(args: &str) -> usize {
3564 for arg in split_top_args(args) {
3565 if let Some((key, val)) = named_arg(arg.trim()) {
3566 if key == "colspan" {
3567 if let Ok(n) = val.trim().parse::<usize>() {
3568 return n.max(1);
3569 }
3570 }
3571 }
3572 }
3573 1
3574}
3575
3576/// Parses the inner text of a `#table(...)` call into a [`TableSpec`]. `columns:` fixes the column
3577/// count, `align:` the alignment, a `fill:` keyed on `row == 0` marks a header row; cells come from
3578/// inline `[...]` groups and from a `..name.flatten()` spread (or the row-remap idiom [`resolve_spread`]
3579/// evaluates) resolved against the data arrays. `None` when no cells are found, so an empty or
3580/// unresolved table sets nothing.
3581fn parse_table_spec(
3582 inner: &str,
3583 arrays: &HashMap<String, Vec<Vec<Inline>>>,
3584 outer_text_pt: Option<f64>,
3585)
3586 -> Option<TableSpec>
3587{
3588 let mut ncols = 1usize;
3589 let mut align = AlignSpec::Uniform(Align::Left);
3590 let mut header = false;
3591 let mut inset_pt: Option<f64> = None;
3592 let mut weights: Vec<f64> = Vec::new();
3593 let mut cells: Vec<Vec<Inline>> = Vec::new();
3594 // A spread's row-remap idiom needs the column count to chunk its array back into rows, so `columns:`
3595 // is read ahead of the main pass -- it names the table's shape wherever it sits in the argument list.
3596 let pre_ncols: usize = split_top_args(inner).iter()
3597 .find_map(|arg| named_arg(arg.trim()).filter(|(k, _)| k.as_str() == "columns").map(|(_, v)| parse_columns(&v)))
3598 .unwrap_or(1);
3599 for arg in split_top_args(inner) {
3600 let a = arg.trim();
3601 if a.is_empty() {
3602 continue;
3603 }
3604 if let Some((key, val)) = named_arg(a) {
3605 match key.as_str() {
3606 "columns" => { ncols = parse_columns(&val); weights = parse_column_weights(&val); },
3607 "align" => align = parse_align(&val),
3608 "fill" => if fill_marks_header(&val) { header = true; },
3609 "inset" => inset_pt = parse_length(&val).map(length_pt),
3610 _ => {}, // stroke, gutter and the rest are not modelled
3611 }
3612 continue;
3613 }
3614 if spread_name(a).is_some() {
3615 if let Some(v) = resolve_spread(a, arrays, pre_ncols) {
3616 cells.extend(v);
3617 }
3618 continue;
3619 }
3620 // An inline positional cell, or a `table.header(...)`/`table.cell(...)` wrapper whose bracketed
3621 // content is the cell -- each `[...]` group in the argument is one cell, in order.
3622 if a.contains('[') {
3623 cells.extend(collect_cells(a));
3624 }
3625 }
3626 if cells.is_empty() {
3627 return None;
3628 }
3629 Some(TableSpec { ncols: ncols.max(1), header, align, weights, text_pt: outer_text_pt, inset_pt, cells })
3630}
3631
3632/// Resolves a `..name` spread argument to the cells it contributes: the array itself for a bare `..name`
3633/// or `..name.flatten()`, or -- for the row-remap idiom `..name.enumerate().map(((idx, row)) => { if COND
3634/// { row } else { table.cell(colspan: n)[...] } }).flatten()`, as Lucronics' E. coli comparison uses to
3635/// merge its section-heading rows into spanning cells -- each row kept or replaced exactly as that
3636/// closure would evaluate it (see [`remap_enumerated_rows`]). `None` for an unknown array name; an
3637/// unrecognised suffix on a known array falls back to its raw cells, so an idiom this reader cannot
3638/// evaluate still sets a table rather than an empty one.
3639fn resolve_spread(
3640 arg: &str,
3641 arrays: &HashMap<String, Vec<Vec<Inline>>>,
3642 ncols: usize,
3643)
3644 -> Option<Vec<Vec<Inline>>>
3645{
3646 let name = spread_name(arg)?;
3647 let base = arrays.get(&name)?;
3648 let rest = arg.trim().strip_prefix("..")?[name.len()..].trim();
3649 if rest.is_empty() || rest == ".flatten()" {
3650 return Some(base.clone());
3651 }
3652 Some(remap_enumerated_rows(rest, base, ncols).unwrap_or_else(|| base.clone()))
3653}
3654
3655/// Evaluates the `.enumerate().map(((idx, row)) => { if COND { A } else { B } }).flatten()` idiom
3656/// against `base` (chunked into `ncols`-wide row tuples), yielding the flat cell list the closure would
3657/// produce. `COND` is a bounded slice of Typst -- `idx == N`, `row.at(N) == [...]`/`!= [...]` literal
3658/// comparisons, joined by `and`/`or` and grouped by parens (see [`eval_bool`]); a branch that is the bare
3659/// loop variable keeps that row's own cells, one shaped `table.cell(colspan: n)[...]` replaces them with
3660/// one spanning cell -- its content may reference the loop variable's cells as `row.at(k)`, substituted
3661/// with that cell's plain text before the existing `table.cell(...)` reader ([`collect_cells`]) parses it,
3662/// so a `#strong(row.at(0))` inside renders through the ordinary strong-call path. `None` when the suffix
3663/// is not this shape, or any row's condition or branch does not evaluate, so the caller falls back to the
3664/// untransformed array rather than guess at a partial result.
3665fn remap_enumerated_rows(after: &str, base: &[Vec<Inline>], ncols: usize) -> Option<Vec<Vec<Inline>>> {
3666 if ncols == 0 || base.is_empty() || base.len() % ncols != 0 {
3667 return None;
3668 }
3669 let rest = after.strip_prefix(".enumerate()")?.trim_start().strip_prefix(".map")?;
3670 let rchars: Vec<char> = rest.chars().collect();
3671 if rchars.first() != Some(&'(') {
3672 return None;
3673 }
3674 let (map_args, _) = read_group(&rchars, 0)?;
3675 let arrow = map_args.find("=>")?;
3676 let params = map_args[..arrow].trim();
3677 let body = map_args[arrow + 2..].trim();
3678 // The closure's own `{ ... }` block wraps its one expression -- unwrap it so what remains starts at
3679 // the `if`, the shape [`parse_if_else`] reads.
3680 let bchars: Vec<char> = body.chars().collect();
3681 let body: String = if bchars.first() == Some(&'{') {
3682 read_brace(&bchars, 0).map(|(inner, _)| inner)?
3683 } else {
3684 body.to_string()
3685 };
3686
3687 // The parameter is `((idx, row))`: an extra pair of parens wraps the tuple destructure, since it is
3688 // `.map`'s single positional argument.
3689 let pchars: Vec<char> = params.chars().collect();
3690 let inner_params = if pchars.first() == Some(&'(') {
3691 read_group(&pchars, 0)?.0
3692 } else {
3693 params.to_string()
3694 };
3695 let inner_params = inner_params.trim();
3696 let inner_params = inner_params.strip_prefix('(').and_then(|s| s.strip_suffix(')')).unwrap_or(inner_params);
3697 let names: Vec<&str> = inner_params.split(',').map(|s| s.trim()).collect();
3698 if names.len() != 2 || names[0].is_empty() || names[1].is_empty() {
3699 return None;
3700 }
3701 let (idx_name, row_name) = (names[0], names[1]);
3702
3703 let (cond, a_branch, b_branch) = parse_if_else(&body)?;
3704
3705 let mut out = Vec::with_capacity(base.len());
3706 for (i, chunk) in base.chunks(ncols).enumerate() {
3707 let row_text: Vec<String> = chunk.iter().map(|cell| inline_plain(cell)).collect();
3708 let take_a = eval_bool(&cond, i, &row_text, idx_name, row_name)?;
3709 let branch = if take_a { &a_branch } else { &b_branch };
3710 if branch.trim() == row_name {
3711 out.extend(chunk.iter().cloned());
3712 continue;
3713 }
3714 let substituted = substitute_row_refs(branch, row_name, &row_text);
3715 let cells = collect_cells(&substituted);
3716 if cells.is_empty() {
3717 return None; // the branch is not a cell this reader can set; refuse the whole idiom
3718 }
3719 out.extend(cells);
3720 }
3721 Some(out)
3722}
3723
3724/// Splits an `if COND { A } else { B }` closure body into its condition and branch source texts. The
3725/// condition runs to the first `{` not nested inside `(...)`/`[...]`; each branch is then read as a
3726/// brace-balanced group. `None` when the body is not this shape.
3727fn parse_if_else(body: &str) -> Option<(String, String, String)> {
3728 let rest = body.trim().strip_prefix("if")?;
3729 let chars: Vec<char> = rest.chars().collect();
3730 let mut depth = 0i32;
3731 let mut brace_at = None;
3732 for (k, &c) in chars.iter().enumerate() {
3733 match c {
3734 '(' | '[' => depth += 1,
3735 ')' | ']' => depth -= 1,
3736 '{' if depth == 0 => { brace_at = Some(k); break; },
3737 _ => {},
3738 }
3739 }
3740 let brace_at = brace_at?;
3741 let cond: String = chars[..brace_at].iter().collect();
3742 let (a_branch, next) = read_brace(&chars, brace_at)?;
3743 let after: String = chars[next..].iter().collect();
3744 let after = after.trim().strip_prefix("else")?.trim();
3745 let bchars: Vec<char> = after.chars().collect();
3746 if bchars.first() != Some(&'{') {
3747 return None;
3748 }
3749 let (b_branch, _) = read_brace(&bchars, 0)?;
3750 Some((cond.trim().to_string(), a_branch.trim().to_string(), b_branch.trim().to_string()))
3751}
3752
3753/// Reads a `{ ... }` block beginning at `i`, matching only `{`/`}` depth -- the closure bodies this
3754/// idiom targets hold no literal braces of their own, so a flat counter is enough, unlike [`read_group`]'s
3755/// full frame tracking for `[`/`(`. `None` when the block never closes.
3756fn read_brace(chars: &[char], i: usize) -> Option<(String, usize)> {
3757 if chars.get(i) != Some(&'{') {
3758 return None;
3759 }
3760 let mut depth = 1i32;
3761 let start = i + 1;
3762 let mut j = start;
3763 while j < chars.len() {
3764 match chars[j] {
3765 '{' => depth += 1,
3766 '}' => {
3767 depth -= 1;
3768 if depth == 0 {
3769 return Some((chars[start..j].iter().collect(), j + 1));
3770 }
3771 },
3772 _ => {},
3773 }
3774 j += 1;
3775 }
3776 None
3777}
3778
3779/// Evaluates a bounded boolean expression -- comparisons of `idx`/`row.at(n)` against a literal or a
3780/// number, joined by `and`/`or` and grouped by parens -- against one row's values. `idx_name`/`row_name`
3781/// are the closure's own parameter names, so the expression is read against exactly the variables that
3782/// idiom bound. `None` when the expression falls outside this bounded grammar.
3783fn eval_bool(expr: &str, idx: usize, row: &[String], idx_name: &str, row_name: &str) -> Option<bool> {
3784 let expr = expr.trim();
3785 if let Some(parts) = split_top_level(expr, "or") {
3786 let mut acc = false;
3787 for p in &parts {
3788 acc = acc || eval_bool(p, idx, row, idx_name, row_name)?;
3789 }
3790 return Some(acc);
3791 }
3792 if let Some(parts) = split_top_level(expr, "and") {
3793 let mut acc = true;
3794 for p in &parts {
3795 acc = acc && eval_bool(p, idx, row, idx_name, row_name)?;
3796 }
3797 return Some(acc);
3798 }
3799 if expr.starts_with('(') && expr.ends_with(')') {
3800 // Confirm the outer parens actually wrap the whole expression, rather than two disjoint groups
3801 // that merely happen to open and close at the ends.
3802 let mut depth = 0i32;
3803 let mut wraps_whole = true;
3804 let last = expr.chars().count() - 1;
3805 for (k, c) in expr.chars().enumerate() {
3806 match c {
3807 '(' => depth += 1,
3808 ')' => {
3809 depth -= 1;
3810 if depth == 0 && k != last {
3811 wraps_whole = false;
3812 break;
3813 }
3814 },
3815 _ => {},
3816 }
3817 }
3818 if wraps_whole {
3819 return eval_bool(&expr[1..expr.len() - 1], idx, row, idx_name, row_name);
3820 }
3821 }
3822 eval_cmp(expr, idx, row, idx_name, row_name)
3823}
3824
3825/// Splits `expr` at every top-level (outside `(...)`/`[...]`) whole-word occurrence of `kw` (`"and"` or
3826/// `"or"`), or `None` when `kw` does not occur at the top level, so the caller falls through to the next
3827/// precedence rather than treating an absent operator as one empty operand.
3828fn split_top_level(expr: &str, kw: &str) -> Option<Vec<String>> {
3829 let chars: Vec<char> = expr.chars().collect();
3830 let kwc: Vec<char> = kw.chars().collect();
3831 let mut depth = 0i32;
3832 let mut parts = Vec::new();
3833 let mut start = 0usize;
3834 let mut i = 0usize;
3835 let mut found = false;
3836 while i < chars.len() {
3837 match chars[i] {
3838 '(' | '[' => { depth += 1; i += 1; continue; },
3839 ')' | ']' => { depth -= 1; i += 1; continue; },
3840 _ => {},
3841 }
3842 if depth == 0 && i + kwc.len() <= chars.len() && chars[i..i + kwc.len()] == kwc[..] {
3843 let before_ok = i == 0 || chars[i - 1].is_whitespace();
3844 let after_ok = chars.get(i + kwc.len()).map(|c| c.is_whitespace()).unwrap_or(true);
3845 if before_ok && after_ok {
3846 parts.push(chars[start..i].iter().collect::<String>());
3847 i = i + kwc.len();
3848 start = i;
3849 found = true;
3850 continue;
3851 }
3852 }
3853 i += 1;
3854 }
3855 if !found {
3856 return None;
3857 }
3858 parts.push(chars[start..].iter().collect::<String>());
3859 Some(parts.into_iter().map(|s| s.trim().to_string()).collect())
3860}
3861
3862/// Evaluates one `lhs (==|!=) rhs` comparison against `idx`/`row`, via [`eval_scalar`]. `None` when
3863/// neither `==` nor `!=` appears, or either side does not resolve.
3864fn eval_cmp(e: &str, idx: usize, row: &[String], idx_name: &str, row_name: &str) -> Option<bool> {
3865 let e = e.trim();
3866 let (eq, at) = if let Some(p) = e.find("!=") {
3867 (false, p)
3868 } else if let Some(p) = e.find("==") {
3869 (true, p)
3870 } else {
3871 return None;
3872 };
3873 let lhs = eval_scalar(e[..at].trim(), idx, row, idx_name, row_name)?;
3874 let rhs = eval_scalar(e[at + 2..].trim(), idx, row, idx_name, row_name)?;
3875 Some(if eq { lhs == rhs } else { lhs != rhs })
3876}
3877
3878/// Reads one side of a comparison: the loop index, a `row.at(n)` cell (its plain text), a `[...]` content
3879/// literal (its plain text) or a bare number. `None` for anything else, refusing rather than guessing.
3880fn eval_scalar(s: &str, idx: usize, row: &[String], idx_name: &str, row_name: &str) -> Option<String> {
3881 let s = s.trim();
3882 if s == idx_name {
3883 return Some(idx.to_string());
3884 }
3885 if let Some(rest) = s.strip_prefix(row_name) {
3886 let n = rest.strip_prefix(".at(")?.strip_suffix(')')?.trim().parse::<usize>().ok()?;
3887 return row.get(n).cloned();
3888 }
3889 if let Some(lit) = s.strip_prefix('[').and_then(|t| t.strip_suffix(']')) {
3890 return Some(lit.trim().to_string());
3891 }
3892 if s.parse::<usize>().is_ok() {
3893 return Some(s.to_string());
3894 }
3895 None
3896}
3897
3898/// Replaces every `<row_name>.at(k)` reference in `src` with a quoted string of that cell's plain text,
3899/// so a branch like `table.cell(colspan: 7)[#strong(row.at(0))]` becomes a call [`collect_cells`] (via
3900/// the ordinary `#strong("...")` reader) can already set, rather than teaching the cell reader to evaluate
3901/// arbitrary code.
3902fn substitute_row_refs(src: &str, row_name: &str, row: &[String]) -> String {
3903 let mut out = src.to_string();
3904 for (k, cell) in row.iter().enumerate() {
3905 let pat = fmt!("{}.at({})", row_name, k);
3906 if out.contains(&pat) {
3907 let escaped = cell.replace('\\', "\\\\").replace('"', "\\\"");
3908 out = out.replace(&pat, &fmt!("\"{}\"", escaped));
3909 }
3910 }
3911 out
3912}
3913
3914/// The absolute point value of a [`Length`], resolving a percentage against a nominal 100 pt so a
3915/// percentage inset still yields a sensible padding; a table's inset is in practice an absolute length.
3916pub(crate) fn length_pt(len: Length) -> f64 {
3917 match len {
3918 Length::Abs(pt) => pt,
3919 Length::Rel(f) => f * 100.0,
3920 }
3921}
3922
3923/// The size of a `text(size: Npt)[...]` (or `#text(...)`) wrapper at the start of `text`, in points, or
3924/// `None` when the body is not wrapped in a sized `text` call. This lets a figure's small-set table --
3925/// `text(size: 7pt)[#table(...)]` -- carry its reduced size, which Typst applies to the whole table.
3926fn outer_text_size(text: &str) -> Option<f64> {
3927 let inner = call_inner(text, "text")?;
3928 for arg in split_top_args(&inner) {
3929 let a = arg.trim();
3930 if let Some((key, val)) = named_arg(a) {
3931 if key == "size" {
3932 return parse_length(&val).map(length_pt);
3933 }
3934 } else if a.ends_with("pt") || a.ends_with("em") {
3935 // The first positional length is the size, as in `text(7pt)[...]`.
3936 return parse_length(a).map(length_pt);
3937 }
3938 }
3939 None
3940}
3941
3942/// Parses a `#figure(...)` call (its buffer, a trailing `<label>` and all) into an [`Item::Figure`]. The
3943/// positional argument is the body -- a wrapped `#table(...)` set in full, or an image call stood in for
3944/// by a placeholder; `caption:` sets the caption, `supplement:`/`kind:` the "Figure" or "Table" label.
3945fn parse_figure(buf: &str, arrays: &HashMap<String, Vec<Vec<Inline>>>) -> Option<Item> {
3946 let (body_src, label) = strip_trailing_label(buf);
3947 let inner = call_inner(&body_src, "figure")?;
3948
3949 let mut caption: Option<Vec<Inline>> = None;
3950 let mut supplement: Option<String> = None;
3951 let mut kind: Option<String> = None;
3952 let mut positional: Option<String> = None;
3953 let mut placement: Option<FloatPlacement> = None;
3954 let mut scope = FloatScope::Column;
3955 for arg in split_top_args(&inner) {
3956 let a = arg.trim();
3957 if a.is_empty() {
3958 continue;
3959 }
3960 if let Some((key, val)) = named_arg(a) {
3961 match key.as_str() {
3962 "caption" => caption = Some(caption_inlines(&val)),
3963 "supplement" => supplement = Some(unquote(&val)),
3964 "kind" => kind = Some(unquote(&val)),
3965 "placement" => placement = parse_placement(&val),
3966 "scope" => scope = parse_scope(&val),
3967 _ => {}, // the rest do not affect the set figure
3968 }
3969 continue;
3970 }
3971 if positional.is_none() {
3972 positional = Some(a.to_string()); // the first positional argument is the figure body
3973 }
3974 }
3975
3976 let body_text = positional.unwrap_or_default();
3977 let body = figure_body(&body_text, arrays);
3978 let supplement = supplement.unwrap_or_else(|| match kind.as_deref() {
3979 Some("table") => "Table".to_string(),
3980 _ => "Figure".to_string(),
3981 });
3982 // A scope matters only to a float: Typst accepts `scope: "parent"` on a floating figure alone.
3983 let placement = placement.map(|side| Floating { side, scope });
3984 Some(Item::Figure { body, caption, supplement, label, placement, span: Span::new(0, 0) })
3985}
3986
3987/// Reads a float's `scope:` value: `"parent"` spans every column of the page, anything else -- `"column"`,
3988/// Typst's default -- keeps it within its column.
3989fn parse_scope(val: &str) -> FloatScope {
3990 match unquote(val).as_str() {
3991 "parent" => FloatScope::Parent,
3992 _ => FloatScope::Column,
3993 }
3994}
3995
3996/// Reads a `#figure` `placement:` value into a float placement. `auto` and `top` float to the page top,
3997/// `bottom` to the foot; `none` (and anything unrecognised) leaves the figure in the flow. Typst's own
3998/// default for a figure is `none`, so a figure that names no placement is not a float.
3999fn parse_placement(val: &str) -> Option<FloatPlacement> {
4000 match val.trim() {
4001 "auto" => Some(FloatPlacement::Auto),
4002 "top" => Some(FloatPlacement::Top),
4003 "bottom" => Some(FloatPlacement::Bottom),
4004 _ => None,
4005 }
4006}
4007
4008/// Decides a figure's body from its positional text: a wrapped `#table(...)` if one is present and
4009/// parses, otherwise an image carrying the path and any declared sizing (empty path when none is found).
4010fn figure_body(text: &str, arrays: &HashMap<String, Vec<Vec<Inline>>>) -> FigureBody {
4011 if let Some(inner) = call_inner(text, "table") {
4012 if let Some(spec) = parse_table_spec(&inner, arrays, outer_text_size(text)) {
4013 return FigureBody::Table(spec);
4014 }
4015 }
4016 // A CeTZ/Fletcher diagram, bar chart or line plot drawn inline is read into a builder that draws it
4017 // for real; only when the body is none of these does it fall through to the image/placeholder path.
4018 if let Some(cf) = super::codefig::parse_code_figure(text) {
4019 return FigureBody::Code(cf);
4020 }
4021 let (path, width, height, scale) = image_call(text);
4022 FigureBody::Image { path, width, height, scale }
4023}
4024
4025/// The path and sizing of a `padded-image("...")` or `image("...")` call in `text`. The custom wrapper is
4026/// tried first, since `image` is a word boundary within it only after the hyphen. The first positional
4027/// argument is the path; `width`/`height` size an `image(...)`, `scale` a `padded-image(...)`. A path
4028/// that is not found gives an empty string, which the block layer stands in for with a placeholder.
4029fn image_call(text: &str) -> (String, Option<Length>, Option<Length>, Option<f64>) {
4030 for name in ["padded-image", "image"] {
4031 if let Some(inner) = call_inner(text, name) {
4032 let mut path: Option<String> = None;
4033 let mut width: Option<Length> = None;
4034 let mut height: Option<Length> = None;
4035 let mut scale: Option<f64> = None;
4036 for arg in split_top_args(&inner) {
4037 let a = arg.trim();
4038 if a.is_empty() {
4039 continue;
4040 }
4041 if let Some((key, val)) = named_arg(a) {
4042 match key.as_str() {
4043 "width" => width = parse_length(&val),
4044 "height" => height = parse_length(&val),
4045 "scale" => scale = parse_percent(&val),
4046 _ => {}, // padding and the rest do not size the set image
4047 }
4048 continue;
4049 }
4050 if path.is_none() {
4051 path = first_string(a);
4052 }
4053 }
4054 if let Some(p) = path {
4055 return (p, width, height, scale);
4056 }
4057 }
4058 }
4059 (String::new(), None, None, None)
4060}
4061
4062/// Reads a Typst length argument into a [`Length`]: a percentage as a fraction of the measure, a `pt`,
4063/// `mm`, `cm` or `in` length as absolute points, a bare number as points. `auto` and anything unreadable
4064/// give `None`, so the figure falls back to filling the measure.
4065pub(crate) fn parse_length(val: &str) -> Option<Length> {
4066 let v = val.trim();
4067 if let Some(pct) = v.strip_suffix('%') {
4068 return pct.trim().parse::<f64>().ok().map(|n| Length::Rel(n / 100.0));
4069 }
4070 for (unit, per_pt) in [("pt", 1.0), ("mm", 72.0 / 25.4), ("cm", 72.0 / 2.54), ("in", 72.0)] {
4071 if let Some(num) = v.strip_suffix(unit) {
4072 return num.trim().parse::<f64>().ok().map(|n| Length::Abs(n * per_pt));
4073 }
4074 }
4075 v.parse::<f64>().ok().map(Length::Abs)
4076}
4077
4078/// Parses a standalone `#line(length:.., stroke:..)` into an [`Item::Rule`]. The length is a fraction of
4079/// the measure (`100%`) or an absolute length; the stroke gives the rule's thickness (a `pt` length) and
4080/// its grey (a `luma(N)` component). A missing length fills the measure; a missing thickness is a hairline
4081/// half-point; a missing colour is black, Typst's default stroke.
4082fn parse_line_rule(trimmed: &str) -> Option<Item> {
4083 let inner = call_inner(trimmed, "line")?;
4084 let mut width = Length::Rel(1.0);
4085 let mut thickness = 0.5;
4086 let mut grey = 0u8;
4087 for arg in split_top_args(&inner) {
4088 let a = arg.trim();
4089 if let Some((key, val)) = named_arg(a) {
4090 match key.as_str() {
4091 "length" => if let Some(l) = parse_length(&val) { width = l; },
4092 "stroke" => {
4093 let (t, g) = parse_stroke(&val);
4094 if let Some(t) = t { thickness = t; }
4095 if let Some(g) = g { grey = g; }
4096 },
4097 _ => {}, // start, end, angle and the rest do not affect a horizontal divider
4098 }
4099 }
4100 }
4101 Some(Item::Rule { width, thickness, grey, span: Span::new(0, 0) })
4102}
4103
4104/// The thickness (a `pt` length) and grey (a `luma(N)` value, 0-255) of a `stroke:` value such as
4105/// `0.5pt + luma(180)`; either component may be absent. A `luma` is read as a grey level; a bare colour
4106/// name or an `rgb(...)` is not modelled and leaves the grey unset.
4107fn parse_stroke(val: &str) -> (Option<f64>, Option<u8>) {
4108 let mut thickness: Option<f64> = None;
4109 let mut grey: Option<u8> = None;
4110 for part in val.split('+') {
4111 let p = part.trim();
4112 if let Some(inner) = call_inner(p, "luma") {
4113 if let Ok(n) = inner.trim().trim_end_matches('%').trim().parse::<f64>() {
4114 grey = Some(n.clamp(0.0, 255.0) as u8);
4115 }
4116 } else if let Some(Length::Abs(pt)) = parse_length(p) {
4117 thickness = Some(pt);
4118 }
4119 }
4120 (thickness, grey)
4121}
4122
4123/// Reads a percentage argument (`100%`) into a fraction (`1.0`), or `None` when it is not a percentage.
4124fn parse_percent(val: &str) -> Option<f64> {
4125 val.trim().strip_suffix('%').and_then(|p| p.trim().parse::<f64>().ok()).map(|n| n / 100.0)
4126}
4127
4128/// The content of the first `name(...)` call in `text`, balanced across nesting and strings, or `None`.
4129/// `name` must sit at a word boundary, so a short name does not match inside a longer identifier.
4130pub(crate) fn call_inner(text: &str, name: &str) -> Option<String> {
4131 let chars: Vec<char> = text.chars().collect();
4132 let namev: Vec<char> = name.chars().collect();
4133 let paren = find_call(&chars, &namev, 0)?;
4134 read_group(&chars, paren).map(|(inner, _)| inner)
4135}
4136
4137/// The index of the `(` of the first `name(` at a word boundary at or after `from`, or `None`.
4138fn find_call(chars: &[char], name: &[char], from: usize) -> Option<usize> {
4139 if name.is_empty() {
4140 return None;
4141 }
4142 let mut i = from;
4143 while i + name.len() < chars.len() {
4144 if chars[i..].starts_with(name) && chars.get(i + name.len()) == Some(&'(') {
4145 let boundary = i == 0 || !is_call_ident(chars[i - 1]);
4146 if boundary {
4147 return Some(i + name.len());
4148 }
4149 }
4150 i += 1;
4151 }
4152 None
4153}
4154
4155/// A character that continues a Typst identifier, for the word-boundary test in [`find_call`].
4156fn is_call_ident(c: char) -> bool {
4157 c.is_alphanumeric() || c == '-' || c == '_'
4158}
4159
4160/// The first `"..."` string literal's content in `text`, or `None`.
4161pub(crate) fn first_string(text: &str) -> Option<String> {
4162 let chars: Vec<char> = text.chars().collect();
4163 let start = chars.iter().position(|&c| c == '"')?;
4164 let end = (start + 1..chars.len()).find(|&j| chars[j] == '"')?;
4165 Some(chars[start + 1..end].iter().collect())
4166}
4167
4168/// Splits the inner text of a call by its top-level commas, respecting `()[]{}` nesting and `"..."`
4169/// strings, so a comma inside a nested group or a string does not part an argument.
4170pub(crate) fn split_top_args(inner: &str) -> Vec<String> {
4171 let chars: Vec<char> = inner.chars().collect();
4172 let mut args: Vec<String> = Vec::new();
4173 let mut cur = String::new();
4174 let mut state = SkipState::new();
4175 let mut i = 0;
4176 while i < chars.len() {
4177 // A comma parts the arguments only at the top level; inside any frame -- a nested group, a string,
4178 // a maths span or a content block -- it is literal and joins the current argument.
4179 if !state.is_open() && chars[i] == ',' {
4180 args.push(std::mem::take(&mut cur));
4181 i += 1;
4182 continue;
4183 }
4184 let consumed = state.step(&chars, i);
4185 for k in i..i + consumed {
4186 cur.push(chars[k]);
4187 }
4188 i += consumed;
4189 }
4190 if !cur.trim().is_empty() {
4191 args.push(cur);
4192 }
4193 args
4194}
4195
4196/// Splits a `key: value` argument at its top-level colon, returning the key and the trimmed value, or
4197/// `None` when there is no top-level colon or the key is not a bare identifier -- so a positional cell
4198/// or a spread is not mistaken for a named argument.
4199pub(crate) fn named_arg(arg: &str) -> Option<(String, String)> {
4200 let chars: Vec<char> = arg.chars().collect();
4201 let mut state = SkipState::new();
4202 let mut i = 0;
4203 while i < chars.len() {
4204 // A colon names the argument only at the top level; inside any frame it is part of the value (an
4205 // alignment `align: (col, row) => ...`, a ratio in a caption, a dictionary key in code).
4206 if !state.is_open() && chars[i] == ':' {
4207 let key: String = chars[..i].iter().collect();
4208 let key = key.trim().to_string();
4209 if !key.is_empty() && key.chars().all(is_call_ident) {
4210 let val: String = chars[i + 1..].iter().collect();
4211 return Some((key, val.trim().to_string()));
4212 }
4213 return None;
4214 }
4215 i += state.step(&chars, i);
4216 }
4217 None
4218}
4219
4220/// The array name of a `..name` or `..name.flatten()` spread argument, or `None`.
4221fn spread_name(arg: &str) -> Option<String> {
4222 let rest = arg.trim().strip_prefix("..")?;
4223 let name: String = rest.chars().take_while(|&c| is_call_ident(c)).collect();
4224 if name.is_empty() {
4225 None
4226 } else {
4227 Some(name)
4228 }
4229}
4230
4231/// Parses a `columns:` value into a column count: an integer as itself, a track list `(a, b, c)` as its
4232/// entry count, anything else as one column.
4233fn parse_columns(val: &str) -> usize {
4234 let v = val.trim();
4235 if let Ok(n) = v.parse::<usize>() {
4236 return n.max(1);
4237 }
4238 if v.starts_with('(') {
4239 let ch: Vec<char> = v.chars().collect();
4240 if let Some((inner, _)) = read_group(&ch, 0) {
4241 let cnt = split_top_args(&inner).iter().filter(|p| !p.trim().is_empty()).count();
4242 return cnt.max(1);
4243 }
4244 }
4245 1
4246}
4247
4248/// The per-column fractional weights of a `columns:` track list: a track `Nfr` (or a bare `fr`, weight 1)
4249/// contributes its weight, an `auto` or a fixed length contributes `0.0` so the column is sized to its
4250/// content. A bare `columns: N` gives no weights (an empty vector), leaving every column content-sized.
4251/// The weights let [`table::lower`](crate::table) reproduce Typst's fractional column sizing rather than
4252/// sizing every column from its widest cell.
4253fn parse_column_weights(val: &str) -> Vec<f64> {
4254 let v = val.trim();
4255 if !v.starts_with('(') {
4256 return Vec::new();
4257 }
4258 let ch: Vec<char> = v.chars().collect();
4259 let inner = match read_group(&ch, 0) {
4260 Some((inner, _)) => inner,
4261 None => return Vec::new(),
4262 };
4263 let mut out = Vec::new();
4264 for track in split_top_args(&inner) {
4265 let t = track.trim();
4266 if t.is_empty() {
4267 continue;
4268 }
4269 out.push(track_weight(t));
4270 }
4271 out
4272}
4273
4274/// The fractional weight of one `columns:` track: `Nfr` reads as `N`, a bare `fr` as `1`, and any other
4275/// track -- `auto`, `3cm`, `40pt`, `20%` -- as `0.0`, which marks the column content-sized.
4276fn track_weight(track: &str) -> f64 {
4277 match track.strip_suffix("fr") {
4278 Some(num) => {
4279 let n = num.trim();
4280 if n.is_empty() { 1.0 } else { n.parse::<f64>().unwrap_or(0.0) }
4281 },
4282 None => 0.0,
4283 }
4284}
4285
4286/// Parses an `align:` value: a `(col, row) => ...` closure as [`AlignSpec::Closure`] (its parameter
4287/// names and body captured for per-cell evaluation), a tuple of column alignments as
4288/// [`AlignSpec::PerColumn`], a single alignment word as [`AlignSpec::Uniform`].
4289fn parse_align(val: &str) -> AlignSpec {
4290 let v = val.trim();
4291 if let Some(arrow) = v.find("=>") {
4292 let params = v[..arrow].trim();
4293 let body = v[arrow + 2..].trim().to_string();
4294 // The parameter list `(col, row)`; a bare single parameter has no parentheses. The first names the
4295 // column, the second the row, matching Typst's `(col, row)` order.
4296 let names: Vec<String> = {
4297 let pch: Vec<char> = params.chars().collect();
4298 match read_group(&pch, 0) {
4299 Some((inner, _)) => split_top_args(&inner).iter().map(|s| s.trim().to_string()).collect(),
4300 None => vec![params.trim().to_string()],
4301 }
4302 };
4303 let col_var = names.first().cloned().filter(|s| !s.is_empty()).unwrap_or_else(|| "col".to_string());
4304 let row_var = names.get(1).cloned().filter(|s| !s.is_empty()).unwrap_or_else(|| "row".to_string());
4305 return AlignSpec::Closure(ClosureAlign { col_var, row_var, body });
4306 }
4307 if v.starts_with('(') {
4308 let ch: Vec<char> = v.chars().collect();
4309 if let Some((inner, _)) = read_group(&ch, 0) {
4310 let cols: Vec<Align> = split_top_args(&inner).iter().map(|p| word_align(p)).collect();
4311 if !cols.is_empty() {
4312 return AlignSpec::PerColumn(cols);
4313 }
4314 }
4315 }
4316 AlignSpec::Uniform(word_align(v))
4317}
4318
4319/// Maps a Typst alignment word to an [`Align`], ignoring a `+ horizon`/`+ top` vertical component and
4320/// treating `start`/`end` as left/right. An unknown word is left-aligned.
4321fn word_align(s: &str) -> Align {
4322 let first = s.trim().split(|c: char| c.is_whitespace() || c == '+').next().unwrap_or("").trim();
4323 match first {
4324 "center" | "centre" => Align::Centre,
4325 "right" | "end" => Align::Right,
4326 _ => Align::Left,
4327 }
4328}
4329
4330/// Does a `fill:` value key on the first row, marking a header? A `fill: (col, row) => ...` closure whose
4331/// body tests the row index against zero (`row == 0` or the common `y == 0`) fills the first row, which is
4332/// the books' header idiom; a `y < n` band likewise begins at the first row. Written with or without
4333/// spaces, and matching either name the closure gives its second (row) parameter.
4334fn fill_marks_header(val: &str) -> bool {
4335 let compact: String = val.chars().filter(|c| !c.is_whitespace()).collect();
4336 compact.contains("row==0")
4337 || compact.contains("y==0")
4338 || compact.contains("row<")
4339 || compact.contains("y<")
4340}
4341
4342/// Parses a `caption: [...]` value into its inline runs: the bracket content scanned for markup, or the
4343/// whole value scanned when it is not a bracket group, so a caption's emphasis, superscript or in-caption
4344/// maths sets with its own face rather than flattening to upright text.
4345fn caption_inlines(val: &str) -> Vec<Inline> {
4346 let v = val.trim();
4347 let ch: Vec<char> = v.chars().collect();
4348 if ch.first() == Some(&'[') {
4349 if let Some((content, _)) = read_group(&ch, 0) {
4350 return parse_inlines(&content);
4351 }
4352 }
4353 parse_inlines(v)
4354}
4355
4356/// Strips a trailing `<label>` from a captured call, returning the call text without it and the label.
4357/// A `<name>` with no inner whitespace at the very end labels the figure; anything else keeps the text.
4358fn strip_trailing_label(buf: &str) -> (String, Option<String>) {
4359 let t = buf.trim_end();
4360 if let Some(inner) = t.strip_suffix('>') {
4361 if let Some(p) = inner.rfind('<') {
4362 let label = &inner[p + 1..];
4363 if !label.is_empty() && !label.contains(char::is_whitespace) {
4364 return (inner[..p].to_string(), Some(label.to_string()));
4365 }
4366 }
4367 }
4368 (buf.to_string(), None)
4369}
4370
4371/// Strips a surrounding `"..."` from a string-literal argument value, leaving anything else unchanged.
4372fn unquote(val: &str) -> String {
4373 let t = val.trim();
4374 if t.len() >= 2 && t.starts_with('"') && t.ends_with('"') {
4375 return t[1..t.len() - 1].to_string();
4376 }
4377 t.to_string()
4378}
4379
4380#[cfg(test)]
4381mod tests {
4382 use super::*;
4383 use crate::math::{Atom, MatKind};
4384
4385 /// A `#claim-label` emits a `MarginNote` carrying its compressed code as the margin display and its raw
4386 /// code for the reverse index, setting nothing in the body column; a `#claim-refs` emits a `MarginNote`
4387 /// with an empty display (nothing drawn) carrying its raw reference codes. The body prose on either side
4388 /// closes over the gap when flattened, so no raw markup leaks.
4389 #[test]
4390 fn claim_label_emits_a_margin_note_and_sets_nothing_inline() {
4391 let runs = parse_inlines("clinical authority#claim-label(<LS8>) bites hardest.");
4392 assert_eq!(runs.len(), 3, "text, margin note, text: got {:?}", runs);
4393 assert!(matches!(&runs[0], Inline::Text(t) if t == "clinical authority"));
4394 assert!(matches!(&runs[1], Inline::MarginNote { display, codes } if display == "LS8" && codes == &vec!["LS8".to_string()]),
4395 "the claim code rides in a margin note and registers for the reverse index: {:?}", runs[1]);
4396 assert!(matches!(&runs[2], Inline::Text(t) if t == " bites hardest."));
4397 // A `#claim-refs` emits a margin note with an empty display carrying its reference codes.
4398 let refs = parse_inlines("formalised#claim-refs(<A1>, <A2>).");
4399 assert!(refs.iter().any(|r| matches!(r, Inline::MarginNote { display, codes }
4400 if display.is_empty() && codes == &vec!["A1".to_string(), "A2".to_string()])),
4401 "a claim-refs registers its raw codes with no margin ink: {:?}", refs);
4402 // Neither the margin code nor a reference is part of the flattened body text.
4403 assert_eq!(
4404 flatten_markup("margins#claim-label(<CD14>, <CD15>, <CD4>) formalised#claim-refs(<A1>)."),
4405 "margins formalised.");
4406 }
4407
4408 /// A bare `#name` naming a scalar `#let` value binding substitutes its display text: a string's contents
4409 /// unquoted, and a number/length literal's own source text, wherever it stands in running text -- at the
4410 /// start, mid-sentence, or immediately before punctuation.
4411 #[test]
4412 fn substitute_scalars_replaces_a_bound_scalar_in_prose() {
4413 let mut sfns = crate::lang::rules::ScalarFns::new();
4414 sfns.insert("version".to_string(), crate::lang::rules::ScalarValue::Str("1.2".to_string()));
4415 sfns.insert("edition".to_string(), crate::lang::rules::ScalarValue::Number("3".to_string()));
4416 assert_eq!(
4417 substitute_scalars("This is version #version, edition #edition, of the guide.", &sfns),
4418 "This is version 1.2, edition 3, of the guide.");
4419 // A reference at the very start of the text (the shape a heading title reads).
4420 assert_eq!(substitute_scalars("#version Notes", &sfns), "1.2 Notes");
4421 }
4422
4423 /// A `#name` inside a raw `` `...` `` code span or a `$...$` maths span is left untouched -- neither is
4424 /// this reader's word, matching how [`parse_inlines_in`] treats them as literal/foreign territory -- and
4425 /// an escaped `\#name` is left alone too, so the backslash still reaches the inline scanner to produce a
4426 /// literal `#`. An unbound name, or a bound name immediately followed by `(`/`[` (a call, not a bare
4427 /// reference), is also untouched.
4428 #[test]
4429 fn substitute_scalars_leaves_raw_math_and_escaped_references_alone() {
4430 let mut sfns = crate::lang::rules::ScalarFns::new();
4431 sfns.insert("version".to_string(), crate::lang::rules::ScalarValue::Str("1.2".to_string()));
4432 assert_eq!(substitute_scalars("see `#version` in code", &sfns), "see `#version` in code");
4433 assert_eq!(substitute_scalars("the constant $#version$ here", &sfns), "the constant $#version$ here");
4434 assert_eq!(substitute_scalars("literal \\#version stays", &sfns), "literal \\#version stays");
4435 assert_eq!(substitute_scalars("#unknown-name here", &sfns), "#unknown-name here");
4436 assert_eq!(substitute_scalars("#version(1) call-shaped", &sfns), "#version(1) call-shaped");
4437 }
4438
4439 /// A scalar `#let` value binding substitutes at a bare `#name` reference in both a heading title and
4440 /// running prose, through the whole [`document_with_templates`] pipeline (a [`Bindings::with_scalars`]
4441 /// scope, as [`crate::book::Scope::bindings`] builds for a real compile) -- not just in the
4442 /// [`substitute_scalars`] unit above.
4443 #[test]
4444 fn scalar_substitution_applies_in_a_heading_and_in_prose() {
4445 let tfns = crate::lang::rules::TemplateFns::new();
4446 let cfns = crate::lang::rules::ContentFns::new();
4447 let mut sfns = crate::lang::rules::ScalarFns::new();
4448 sfns.insert("title".to_string(), crate::lang::rules::ScalarValue::Str("Guide".to_string()));
4449 let binds = crate::lang::rules::Bindings::with_scalars(&tfns, &cfns, &sfns);
4450 let src = "= #title\n\nThe #title is version one.\n";
4451 let (items, _skips) = document_with_templates(src, binds).expect("parses");
4452 match &items[0] {
4453 Item::Heading { runs, .. } => assert!(
4454 matches!(&runs[0], Inline::Text(t) if t.contains("Guide")),
4455 "the heading substitutes the bound scalar: {:?}", runs),
4456 other => panic!("expected a heading first, got {:?}", other),
4457 }
4458 match &items[1] {
4459 Item::Paragraph { runs, .. } => {
4460 let text: String = runs.iter().map(|r| match r {
4461 Inline::Text(t) => t.clone(),
4462 _ => String::new(),
4463 }).collect();
4464 assert!(text.contains("Guide"), "the paragraph substitutes the bound scalar: {:?}", text);
4465 },
4466 other => panic!("expected a paragraph second, got {:?}", other),
4467 }
4468 }
4469
4470 /// A bare `#name` standing alone on its own line -- bound to no scalar, content or template binding --
4471 /// is still a visible refusal, exactly as before scalar bindings existed: [`document_with_templates`]
4472 /// records it in the returned [`Refusals`] rather than leaking its raw source as a paragraph, whether or
4473 /// not the document also carries a scalar scope. Scalar substitution only ever fires for a NAME the
4474 /// scope actually binds ([`substitute_scalars`] is a no-op otherwise), so adding it never turns this
4475 /// existing refusal into a silent drop.
4476 #[test]
4477 fn unbound_standalone_reference_stays_a_visible_refusal() {
4478 let mut sfns = crate::lang::rules::ScalarFns::new();
4479 sfns.insert("title".to_string(), crate::lang::rules::ScalarValue::Str("Guide".to_string()));
4480 let tfns = crate::lang::rules::TemplateFns::new();
4481 let cfns = crate::lang::rules::ContentFns::new();
4482 let binds = crate::lang::rules::Bindings::with_scalars(&tfns, &cfns, &sfns);
4483 let src = "Some prose above.\n\n#nosuchname\n\nSome prose below.\n";
4484 let (items, skips) = document_with_templates(src, binds).expect("parses");
4485 assert!(items.iter().all(|it| !matches!(it, Item::Paragraph { runs, .. }
4486 if runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("#nosuchname"))))),
4487 "an unbound standalone reference must not leak as paragraph text: {:?}", items);
4488 assert!(skips.report().is_some_and(|r| r.contains("nosuchname")),
4489 "an unbound standalone reference is a visible refusal, not a silent drop: {:?}", skips.report());
4490 }
4491
4492 /// A scalar `#let` binding referenced by a bare `#name` standing ALONE on its own line substitutes its
4493 /// value, through the whole [`document_with_templates`] pipeline -- not only mid-prose (which item 2 already
4494 /// handled) but where the reference is the line's only content, the case the standalone code-reference skip
4495 /// path swallowed before. The gate is self-non-vacuous: with the `names_scalar_alone` exemption reverted the
4496 /// standalone `#edition` is recorded as a `#pagebreak`-style skip and never becomes a paragraph, so the
4497 /// substituted value is absent and a skip is reported -- both assertions below then red. The companion
4498 /// [`unbound_standalone_reference_stays_a_visible_refusal`] proves an UNBOUND standalone name still refuses.
4499 #[test]
4500 fn standalone_line_scalar_reference_substitutes() {
4501 let tfns = crate::lang::rules::TemplateFns::new();
4502 let cfns = crate::lang::rules::ContentFns::new();
4503 let mut sfns = crate::lang::rules::ScalarFns::new();
4504 sfns.insert("edition".to_string(), crate::lang::rules::ScalarValue::Number("3".to_string()));
4505 let binds = crate::lang::rules::Bindings::with_scalars(&tfns, &cfns, &sfns);
4506 let src = "Some prose above.\n\n#edition\n\nSome prose below.\n";
4507 let (items, skips) = document_with_templates(src, binds).expect("parses");
4508 assert!(items.iter().any(|it| matches!(it, Item::Paragraph { runs, .. }
4509 if matches!(runs.as_slice(), [Inline::Text(t)] if t == "3"))),
4510 "a standalone-line scalar reference substitutes its value as its own paragraph: {:?}", items);
4511 assert!(!skips.report().is_some_and(|r| r.contains("edition")),
4512 "a bound standalone scalar reference is substituted, not recorded as a skip: {:?}", skips.report());
4513 }
4514
4515 /// `lorem_words` reproduces `typst 0.15.1`'s `#lorem(n)` verbatim: the classic opening for small counts,
4516 /// the last word's trailing punctuation replaced by a full stop, a zero count empty, and a huge count
4517 /// capped at the embedded corpus rather than looping. The pinned strings are the exact oracle output
4518 /// (checked against the installed typst), so a regression in the corpus or the join reds here.
4519 #[test]
4520 fn lorem_matches_the_typst_oracle() {
4521 assert_eq!(lorem_words(1), "Lorem.");
4522 assert_eq!(lorem_words(5), "Lorem ipsum dolor sit amet.");
4523 // Word 20 ("quaerat") carries no source punctuation; the full stop is appended.
4524 assert_eq!(lorem_words(20),
4525 "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magnam aliquam quaerat.");
4526 // Word 49 ("voluptatem,") carries a comma in the source, replaced by the terminal full stop.
4527 assert!(lorem_words(49).ends_with("transferre in voluptatem."),
4528 "the terminal comma must become a full stop: {:?}", lorem_words(49));
4529 assert_eq!(lorem_words(0), "");
4530 // A count past the corpus is capped, not looped: it returns the whole embedded passage, ending in a
4531 // full stop, with exactly the corpus's word count.
4532 let corpus_len = LOREM_CORPUS.split_whitespace().count();
4533 assert_eq!(lorem_words(100_000).split_whitespace().count(), corpus_len);
4534 assert!(lorem_words(100_000).ends_with('.'));
4535 }
4536
4537 /// An own-line `#pagebreak()`, `#lorem(n)` and `#v(<abs len>)` are set on the block path rather than
4538 /// tallied as skipped constructs: the page break lowers to an `Item::PageBreak`, the lorem call to a
4539 /// plain paragraph of the oracle text, and the absolute vertical space to an `Item::Space`. A relative
4540 /// `#v(2em)` has no running size here, so it stays a visible refusal.
4541 #[test]
4542 fn block_builtins_are_set_not_skipped() -> Outcome<()> {
4543 let src = "Opening prose.\n\n#lorem(5)\n\n#v(12pt)\n\n#pagebreak()\n\nAfter.\n";
4544 let (items, skips) = res!(document_with_refusals(src));
4545 assert!(skips.report().is_none(), "no builtin should be tallied as skipped: {:?}", skips.report());
4546 assert!(items.iter().any(|it| matches!(it, Item::PageBreak { .. })), "the page break is set: {:?}", items);
4547 assert!(items.iter().any(|it| matches!(it, Item::Space { .. })), "the absolute #v is set: {:?}", items);
4548 assert!(items.iter().any(|it| matches!(it,
4549 Item::Paragraph { runs, .. } if matches!(runs.as_slice(), [Inline::Text(t)] if t == "Lorem ipsum dolor sit amet."))),
4550 "the #lorem call sets its oracle text as a paragraph: {:?}", items);
4551 // A relative `#v(2em)` cannot be resolved here, so it stays a visible refusal rather than a wrong space.
4552 let (items2, skips2) = res!(document_with_refusals("#v(2em)\n"));
4553 assert!(!items2.iter().any(|it| matches!(it, Item::Space { .. })), "a relative #v must not be set: {:?}", items2);
4554 assert!(skips2.report().is_some(), "a relative #v is refused visibly");
4555 Ok(())
4556 }
4557
4558 /// Gathers the plain text of every top-level paragraph, for the trailing-prose and continuation checks.
4559 #[cfg(test)]
4560 fn paragraph_text(items: &[Item]) -> String {
4561 items.iter().filter_map(|it| match it {
4562 Item::Paragraph { runs, .. } => Some(runs.iter().filter_map(|r| match r {
4563 Inline::Text(t) => Some(t.clone()),
4564 _ => None,
4565 }).collect::<String>()),
4566 _ => None,
4567 }).collect::<Vec<_>>().join(" ")
4568 }
4569
4570 /// A balanced builtin call with prose trailing its `)` is NOT own-line: it falls through to the existing
4571 /// visible refusal, which keeps the trailing prose. This is the silent-loss regression the milestone audit
4572 /// flagged -- `#pagebreak() text`, `#lorem(5) text`, `#v(12pt) text` must not drop the trailing words nor
4573 /// set the builtin as a block.
4574 #[test]
4575 fn trailing_prose_after_a_builtin_is_kept() -> Outcome<()> {
4576 for src in [
4577 "#pagebreak() and then more prose.\n",
4578 "#lorem(5) and then more prose.\n",
4579 "#v(12pt) and then more prose.\n",
4580 ] {
4581 let (items, _skips) = res!(document_with_refusals(src));
4582 assert!(!items.iter().any(|it| matches!(it, Item::PageBreak { .. } | Item::Space { .. })),
4583 "a builtin with trailing prose must not be set as a block: {:?} -> {:?}", src, items);
4584 let body = paragraph_text(&items);
4585 assert!(body.contains("and then more prose"),
4586 "the trailing prose must survive for {:?}: body {:?}", src, body);
4587 assert!(!body.contains("Lorem ipsum dolor sit amet"),
4588 "a trailing-prose #lorem must not expand as a block for {:?}", src);
4589 }
4590 Ok(())
4591 }
4592
4593 /// An own-line builtin on the line directly after prose, with no blank line between, closes the paragraph
4594 /// and sets its own block -- exactly as `= heading\n#pagebreak()` and `#section-banner` already do, and as
4595 /// Typst 0.15.1 renders it. A blank line before the builtin is not required: the earlier deferral silently
4596 /// swallowed a `#pagebreak()`/`#v()` set directly beneath a paragraph, which is the render bug the live
4597 /// drive found (a paragraph before the break collapsed the page count to one; a bare heading before it did
4598 /// not, because a heading line opens no paragraph). The preceding prose survives as its own paragraph. A
4599 /// builtin with prose on the SAME line (`#lorem(5) more`) is still not own-line -- see
4600 /// [`trailing_prose_after_a_builtin_is_kept`] -- so inline mid-prose support remains a later unit.
4601 #[test]
4602 fn own_line_builtin_beneath_prose_flushes_and_sets() -> Outcome<()> {
4603 // `#lorem` beneath prose expands its oracle text as a fresh paragraph, and the prose above it is kept.
4604 let (items, skips) = res!(document_with_refusals("Some opening prose here.\n#lorem(5)\n"));
4605 assert!(items.iter().any(|it| matches!(it,
4606 Item::Paragraph { runs, .. } if matches!(runs.as_slice(), [Inline::Text(t)] if t == "Lorem ipsum dolor sit amet."))),
4607 "a #lorem directly beneath prose sets its oracle text as a block: {:?}", items);
4608 assert!(paragraph_text(&items).contains("Some opening prose here"),
4609 "the preceding prose is kept: {:?}", items);
4610 assert!(skips.report().is_none(), "an own-line builtin beneath prose is set, not refused: {:?}", skips.report());
4611
4612 // `#pagebreak` beneath prose sets the break; the prose is kept.
4613 let (items, skips) = res!(document_with_refusals("Some opening prose here.\n#pagebreak()\n"));
4614 assert!(items.iter().any(|it| matches!(it, Item::PageBreak { .. })),
4615 "a #pagebreak directly beneath prose is set: {:?}", items);
4616 assert!(paragraph_text(&items).contains("Some opening prose here"), "the preceding prose is kept: {:?}", items);
4617 assert!(skips.report().is_none(), "the break is set, not refused: {:?}", skips.report());
4618
4619 // `#v(<abs>)` beneath prose sets the space; the prose is kept.
4620 let (items, skips) = res!(document_with_refusals("Some opening prose here.\n#v(12pt)\n"));
4621 assert!(items.iter().any(|it| matches!(it, Item::Space { .. })),
4622 "an absolute #v directly beneath prose is set: {:?}", items);
4623 assert!(paragraph_text(&items).contains("Some opening prose here"), "the preceding prose is kept: {:?}", items);
4624 assert!(skips.report().is_none(), "the space is set, not refused: {:?}", skips.report());
4625 Ok(())
4626 }
4627
4628 /// A `#pagebreak()` nested in a `#styled-box[ ... ]` callout body cannot be honoured -- the box is one keep
4629 /// unit -- so it is a visible refusal, not a silent drop, and no `Item::PageBreak` survives inside the box.
4630 #[test]
4631 fn page_break_inside_a_box_is_refused_not_dropped() -> Outcome<()> {
4632 let src = "#styled-box[\nInside the callout.\n\n#pagebreak()\n\nStill inside.\n]\n";
4633 let (items, skips) = res!(document_with_refusals(src));
4634 assert!(skips.report().map_or(false, |r| r.contains("#pagebreak")),
4635 "the boxed page break is refused visibly: {:?}", skips.report());
4636 fn has_page_break(items: &[Item]) -> bool {
4637 items.iter().any(|it| match it {
4638 Item::PageBreak { .. } => true,
4639 Item::Box { items, .. } => has_page_break(items),
4640 Item::Scoped { items, .. } => has_page_break(items),
4641 _ => false,
4642 })
4643 }
4644 assert!(!has_page_break(&items), "no page break may survive inside the box: {:?}", items);
4645 Ok(())
4646 }
4647
4648 /// An ignored keyword argument is refused, not silently set as the wrong thing: `#pagebreak(to: "odd")`
4649 /// selects a parity target the reader does not model, and `#v(24pt, weak: true)` asks for a collapsing
4650 /// space it does not model -- each stays a visible refusal rather than a plain break or a fixed space.
4651 #[test]
4652 fn unsupported_keyword_args_are_refused() -> Outcome<()> {
4653 let (items, skips) = res!(document_with_refusals("#pagebreak(to: \"odd\")\n"));
4654 assert!(!items.iter().any(|it| matches!(it, Item::PageBreak { .. })), "a `to:` pagebreak is not set: {:?}", items);
4655 assert!(skips.report().map_or(false, |r| r.contains("#pagebreak")), "a `to:` pagebreak is refused: {:?}", skips.report());
4656 let (items2, skips2) = res!(document_with_refusals("#v(24pt, weak: true)\n"));
4657 assert!(!items2.iter().any(|it| matches!(it, Item::Space { .. })), "a weak #v is not set: {:?}", items2);
4658 assert!(skips2.report().map_or(false, |r| r.contains("#v")), "a weak #v is refused: {:?}", skips2.report());
4659 Ok(())
4660 }
4661
4662 /// `_compress-codes`: a run of three or more consecutive same-prefix codes collapses to an en-dash
4663 /// range, a pair stays expanded, two or fewer codes join unchanged, and an unparseable run passes through.
4664 #[test]
4665 fn claim_codes_compress_consecutive_runs() {
4666 let display = |s: &str| parse_inlines(s).into_iter()
4667 .find_map(|r| match r { Inline::MarginNote { display, .. } => Some(display), _ => None })
4668 .unwrap_or_default();
4669 assert_eq!(display("x#claim-label(<B1>, <B2>, <B3>, <B4>)"), "B1\u{2013}4"); // B1–4
4670 assert_eq!(display("x#claim-label(<A1>, <A2>)"), "A1 A2");
4671 assert_eq!(display("x#claim-label(<CD14>, <CD15>, <CD4>)"), "CD14 CD15 CD4");
4672 assert_eq!(display("x#claim-label(<LS8>)"), "LS8");
4673 }
4674
4675 /// A line that opens with a claim marker is prose, not a standalone call the line scanner skips, so
4676 /// the sentence that follows the marker is set rather than dropped with it.
4677 #[test]
4678 fn line_leading_claim_marker_is_prose() {
4679 assert!(is_inline_call("claim-label"));
4680 assert!(is_inline_call("claim-refs"));
4681 assert!(code_skip("#claim-label(<CD18>). Equilibrium appropriation follows.").is_none());
4682 }
4683
4684 /// A claim reference in a context the layout does not gather into the reverse claim index -- here a
4685 /// heading title -- is recorded as a refusal rather than dropped silently, while the same reference in a
4686 /// body paragraph (which IS gathered) draws no refusal. Guards the silent-loss path the audit flagged.
4687 #[test]
4688 fn claim_ref_in_a_non_body_context_is_refused_not_dropped() {
4689 let (_items, skips) = document_with_refusals("= Heading #claim-refs(<Z9>) here\n\nBody text follows.\n").expect("parse");
4690 assert!(skips.sites().iter().any(|s| s.name.contains("claim reference") && s.name.contains("heading")),
4691 "a claim reference in a heading title must be a refusal: {:?}", skips.sites());
4692 // A claim reference in a body paragraph is gathered into the index, so it is NOT refused.
4693 let (_i2, skips2) = document_with_refusals("Body carrying a reference#claim-refs(<Z9>) here.\n").expect("parse");
4694 assert!(!skips2.sites().iter().any(|s| s.name.contains("claim reference")),
4695 "a body claim reference is indexed, not refused: {:?}", skips2.sites());
4696 }
4697
4698 /// A line-leading `#padded-image(...)` (a section opener's logo) is set as an [`Item::Image`] carrying
4699 /// its path and scale, not skipped as a template call and not wrapped in a numbered figure.
4700 #[test]
4701 fn line_leading_padded_image_reads_as_image() -> Outcome<()> {
4702 let (items, _skips) = res!(document_with_refusals(
4703 "= Pearl\n\n#padded-image(\"assets/svg/pearlite_logo_text_right.svg\", scale: 45%)\n\nPearl is the format.\n"));
4704 let img = res!(items.iter().find_map(|it| match it {
4705 Item::Image { path, scale, .. } => Some((path.clone(), *scale)),
4706 _ => None,
4707 }).ok_or_else(|| err!("no Item::Image was produced for the standalone padded-image"; Test, Bug)));
4708 assert_eq!(img.0, "assets/svg/pearlite_logo_text_right.svg", "the image path is read");
4709 assert_eq!(img.1, Some(0.45), "the padded-image scale is read as a fraction");
4710 // The line must not have been swallowed as a skipped construct, nor turned into a figure.
4711 assert!(!items.iter().any(|it| matches!(it, Item::Figure { .. })), "a section logo is not a figure");
4712 Ok(())
4713 }
4714
4715 /// A line-leading `#section-banner("logo")` (a documentation section opener) is read as an
4716 /// [`Item::SectionBanner`] carrying its logo path, not skipped as a template call and not confused with a
4717 /// plain `#image`.
4718 #[test]
4719 fn line_leading_section_banner_reads_as_banner() -> Outcome<()> {
4720 let (items, skips) = res!(document_with_refusals(
4721 "#section-banner(\"assets/svg/fe2o3_logo_text_right.svg\")\n\n= Steel Server\n\nSteel is the server.\n"));
4722 let path = res!(items.iter().find_map(|it| match it {
4723 Item::SectionBanner { path, .. } => Some(path.clone()),
4724 _ => None,
4725 }).ok_or_else(|| err!("no Item::SectionBanner was produced for the standalone section-banner"; Test, Bug)));
4726 assert_eq!(path, "assets/svg/fe2o3_logo_text_right.svg", "the banner logo path is read");
4727 // The call must not have been swallowed as a skipped construct, nor read as a plain centred image.
4728 assert!(!skips.entries().iter().any(|(name, _)| name == "#section-banner"),
4729 "a section banner must not be reported as a skipped construct");
4730 assert!(!items.iter().any(|it| matches!(it, Item::Image { .. })), "a section banner is not a plain image");
4731 Ok(())
4732 }
4733
4734 /// A line-leading `#styled-box[...]` callout, its body opening on the marker line and closing on a later
4735 /// one, is gathered whole and read as an [`Item::Box`] holding the re-parsed body -- not skipped as an
4736 /// unbalanced standalone call, which would drop the callout's text. The construct is set, so it is not
4737 /// reported as a skipped construct.
4738 #[test]
4739 fn line_leading_styled_box_reads_as_box() -> Outcome<()> {
4740 let (items, skips) = res!(document_with_refusals(
4741 "Lead prose.\n\n#styled-box[\n*Principle.* Every participant is accountable.\n]\n\nTrailing prose.\n"));
4742 let inner = res!(items.iter().find_map(|it| match it {
4743 Item::Box { items, .. } => Some(items.clone()),
4744 _ => None,
4745 }).ok_or_else(|| err!("no Item::Box was produced for the standalone styled-box"; Test, Bug)));
4746 // The body re-parses to a paragraph, and its lead-in bold survives as a strong run.
4747 let has_para = inner.iter().any(|it| matches!(it, Item::Paragraph { .. }));
4748 assert!(has_para, "the styled-box body must re-parse to a paragraph, got: {:?}", inner);
4749 let has_strong = inner.iter().any(|it| matches!(it,
4750 Item::Paragraph { runs, .. } if runs.iter().any(|r| matches!(r, Inline::Strong(t) if t == "Principle."))));
4751 assert!(has_strong, "the body's bold lead-in must survive, got: {:?}", inner);
4752 // The callout must not have been swallowed as a skipped construct.
4753 assert!(!skips.entries().iter().any(|(name, _)| name == "#styled-box"),
4754 "a styled-box must not be reported as a skipped construct");
4755 // The prose either side of the callout still sets.
4756 assert!(items.iter().any(|it| matches!(it, Item::Paragraph { runs, .. }
4757 if runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Lead prose."))))),
4758 "prose before the callout is dropped");
4759 Ok(())
4760 }
4761
4762 /// A call to a bound `#let` furniture function -- `#pr-note[ ... ]` -- is gathered whole and expanded
4763 /// into an [`Item::Box`] carrying the definition's lowered patch (a transparent-wash, asymmetric-inset
4764 /// block), its `[ ... ]` body re-parsed into the box. The call is set, not skipped, so it is not tallied;
4765 /// an unbound `#name[...]` (no definition in scope) still falls through to be reported as a skip.
4766 #[test]
4767 fn pr_note_call_expands_to_a_box_and_is_not_skipped() -> Outcome<()> {
4768 let def = "#let pr-note(body) = block(inset: (left: 1.2em, right: 0.6em), above: 0.9em, below: 1.1em, \
4769{ set text(size: 0.88em); set par(spacing: 0.55em, first-line-indent: 0em); body })\n";
4770 let mut tfns = crate::lang::rules::TemplateFns::new();
4771 crate::lang::rules::collect_template_fns(def, crate::ir::Sp::from_pt(10.0),
4772 &crate::lang::rules::Palette::new(), &mut tfns);
4773 assert!(tfns.contains_key("pr-note"), "the definition is collected");
4774
4775 let src = "Lead prose.\n\n#pr-note[\n*Baseline:* one measure.\n\nA second paragraph.\n]\n\nTrailing prose.\n";
4776 let (items, skips) = res!(document_with_templates(src, crate::lang::rules::Bindings::new(&tfns, &crate::lang::rules::ContentFns::new())));
4777 let (inner, patch) = res!(items.iter().find_map(|it| match it {
4778 Item::Box { items, patch, .. } => Some((items.clone(), patch.clone())),
4779 _ => None,
4780 }).ok_or_else(|| err!("no Item::Box was produced for the pr-note call"; Test, Bug)));
4781 // The box carries the pr-note geometry: an asymmetric inset and a transparent (no-wash) fill.
4782 assert_eq!(patch.callout.inset_left, Some(crate::ir::Sp::from_pt(12.0)), "left inset resolved at 10pt body");
4783 assert_eq!(patch.callout.fill.map(|c| c.a), Some(0), "no fill -- a plain indented block");
4784 assert_eq!(patch.text.body_size, Some(crate::ir::Sp::from_pt(8.8)), "the body sets at 0.88em");
4785 // The body re-parses to paragraphs, its bold lead-in surviving as a strong run.
4786 assert!(inner.iter().any(|it| matches!(it,
4787 Item::Paragraph { runs, .. } if runs.iter().any(|r| matches!(r, Inline::Strong(t) if t == "Baseline:")))),
4788 "the body's bold lead-in must survive, got: {:?}", inner);
4789 // The call is not tallied as a skipped construct, and the surrounding prose still sets.
4790 assert!(!skips.entries().iter().any(|(name, _)| name == "#pr-note"),
4791 "a bound furniture call must not be reported as a skip");
4792 assert!(items.iter().any(|it| matches!(it, Item::Paragraph { runs, .. }
4793 if runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Lead prose."))))),
4794 "prose before the call is dropped");
4795
4796 // With no definition in scope, the same call is left to be tallied as a skip, unchanged.
4797 let (_it2, skips2) = res!(document_with_refusals(src));
4798 assert!(skips2.entries().iter().any(|(name, _)| name == "#pr-note"),
4799 "an unbound furniture call still reports as a skip");
4800 Ok(())
4801 }
4802
4803 /// A `#aside-box(title: [...])[ ... ]` call expands into a washed box carrying the definition's fill and
4804 /// left stroke, its `title:` argument set as a leading bold paragraph ahead of the body, and it is not
4805 /// tallied as a skip. A bound call with no `[ ... ]` body (an argument-only `#aside-box(...)`) is TALLIED
4806 /// as a skip rather than dropped silently.
4807 #[test]
4808 fn aside_box_call_expands_with_title_and_stroke() -> Outcome<()> {
4809 let def = "#let aside-box(title: none, body) = box(width: 100%, inset: 8pt, \
4810fill: colours.yellow.lighten(50%), radius: 4pt, stroke: (left: 2pt + colours.yellow.darken(20%)), \
4811[#text(weight: \"bold\", size: 0.85em)[#title] #text(size: 0.85em)[#body]])\n";
4812 let mut palette = crate::lang::rules::Palette::new();
4813 crate::lang::rules::collect_palette("#let colours = (yellow: rgb(\"#f0f600\"),)\n", &mut palette);
4814 let mut tfns = crate::lang::rules::TemplateFns::new();
4815 crate::lang::rules::collect_template_fns(def, crate::ir::Sp::from_pt(11.0), &palette, &mut tfns);
4816 let tf = res!(tfns.get("aside-box").ok_or_else(|| err!("aside-box collected"; Test, Bug)));
4817 assert!(tf.patch.callout.fill.map(|c| c.a) == Some(255), "the yellow fill resolved (opaque)");
4818 assert!(tf.patch.callout.stroke_left_w.is_some(), "the left stroke width is set");
4819
4820 let src = "Lead.\n\n#aside-box(title: [The welfare theorems])[\nMarket efficiency proved.\n]\n\nTail.\n";
4821 let (items, skips) = res!(document_with_templates(src, crate::lang::rules::Bindings::new(&tfns, &crate::lang::rules::ContentFns::new())));
4822 let inner = res!(items.iter().find_map(|it| match it {
4823 Item::Box { items, .. } => Some(items.clone()),
4824 _ => None,
4825 }).ok_or_else(|| err!("no Item::Box for the aside-box call"; Test, Bug)));
4826 // The first item is the bold title (in a size scope), then the body.
4827 let has_title = inner.iter().any(|it| match it {
4828 Item::Scoped { items, .. } => items.iter().any(|p| matches!(p,
4829 Item::Paragraph { runs, .. } if runs.iter().any(|r| matches!(r, Inline::Strong(t) if t.contains("welfare"))))),
4830 Item::Paragraph { runs, .. } => runs.iter().any(|r| matches!(r, Inline::Strong(t) if t.contains("welfare"))),
4831 _ => false,
4832 });
4833 assert!(has_title, "the title is set as a leading bold paragraph, got: {:?}", inner);
4834 assert!(!skips.entries().iter().any(|(n, _)| n == "#aside-box"), "the call is not a skip");
4835
4836 // A bound call with no `[body]` is tallied as a skip, not dropped.
4837 let (_it, skips2) = res!(document_with_templates("#aside-box(title: [X])\n", crate::lang::rules::Bindings::new(&tfns, &crate::lang::rules::ContentFns::new())));
4838 assert!(skips2.entries().iter().any(|(n, _)| n == "#aside-box"),
4839 "an argument-only bound call with no body is tallied as a skip");
4840 Ok(())
4841 }
4842
4843 /// A `#columns[...]` body's own top-level `#set` declarations scope to the spliced subtree (H1's
4844 /// flat-splice sibling): the reader nests the spliced items inside one `Item::Scoped` carrying the
4845 /// lowered patch. A columns body that declares nothing splices in flat, with no scope.
4846 #[test]
4847 fn columns_body_set_scopes_the_spliced_subtree() -> Outcome<()> {
4848 let (items, _skips) = res!(document_with_refusals(
4849 "#columns(2)[\n#set text(size: 20pt)\n\nScoped body.\n]\n"));
4850 let (patch, inner) = res!(items.iter().find_map(|it| match it {
4851 Item::Scoped { patch, items } => Some((patch.clone(), items)),
4852 _ => None,
4853 }).ok_or_else(|| err!("no Item::Scoped was produced for a columns body with a #set"; Test, Bug)));
4854 assert_eq!(patch.text.body_size, Some(crate::ir::Sp::from_pt(20.0)),
4855 "the columns body's #set text(size:) did not lower into the scope patch");
4856 assert!(!inner.is_empty(), "the scope must carry the body's items");
4857
4858 // A columns body that declares nothing splices in flat, with no scope.
4859 let (plain, _) = res!(document_with_refusals("#columns(2)[\nPlain body.\n]\n"));
4860 assert!(!plain.iter().any(|it| matches!(it, Item::Scoped { .. })),
4861 "a columns body with no #set must not be wrapped in a scope");
4862 Ok(())
4863 }
4864
4865 /// H2: a lowerable `#set` that lowers to nothing (an unrecognised argument) is recorded as a visible
4866 /// refusal named for the set, rather than silently dropped, while one that fully lowers records none.
4867 #[test]
4868 fn unconsumed_set_is_recorded_as_a_refusal() -> Outcome<()> {
4869 let (_items, skips) = res!(document_with_refusals("#set text(lang: \"de\")\n\nBody.\n"));
4870 assert!(skips.entries().iter().any(|(name, _)| name == "#set text"),
4871 "an unconsumed #set must be recorded as a refusal, got: {:?}", skips.entries());
4872
4873 let (_items2, skips2) = res!(document_with_refusals("#set text(size: 12pt)\n\nBody.\n"));
4874 assert!(!skips2.entries().iter().any(|(name, _)| name == "#set text"),
4875 "a fully-lowered #set must not be recorded as a refusal, got: {:?}", skips2.entries());
4876 Ok(())
4877 }
4878
4879 /// `#emph[...]` is the call form of `_..._`: it yields an [`Inline::Emph`] run with the same inner
4880 /// text, and nothing raw leaks.
4881 #[test]
4882 fn emph_call_reads_as_emphasis() {
4883 let runs = parse_inlines("The cube asks #emph[who] does the extracting.");
4884 assert!(runs.iter().any(|r| matches!(r, Inline::Emph(t) if t == "who")),
4885 "emph run missing: {:?}", runs);
4886 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#emph"))),
4887 "raw #emph leaked: {:?}", runs);
4888 // A call carrying its own markup expands the same way `_..._` does.
4889 let nested = parse_inlines("#emph[the #idx[Harvard Business Review] weekly]");
4890 assert!(nested.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#emph") || t.contains("#idx"))),
4891 "raw markup leaked from nested emph: {:?}", nested);
4892 }
4893
4894 /// An index marker carries a sort key distinct from its display: `#idx-as[Abbott, Andrew][Andrew Abbott]`
4895 /// files under the surname but shows "Andrew Abbott" (in the body and on the index page), and `#idx[_x_]`
4896 /// keeps its emphasis as an [`Inline::Emph`] display run while its sort key is the flattened plain text --
4897 /// so the index sets the display, not the sort key, and an emphasised entry italicises rather than printing
4898 /// literal underscores.
4899 #[test]
4900 fn index_marker_splits_sort_key_from_styled_display() {
4901 let runs = parse_inlines("The sociologist #idx-as[Abbott, Andrew][Andrew Abbott] wrote widely.");
4902 let mut found = false;
4903 for r in &runs {
4904 if let Inline::Index { term, sub, display, .. } = r {
4905 assert_eq!(term, "Abbott, Andrew", "sort key wrong: {:?}", runs);
4906 assert!(sub.is_none());
4907 assert!(matches!(display.as_slice(), [Inline::Text(t)] if t == "Andrew Abbott"),
4908 "display wrong: {:?}", display);
4909 found = true;
4910 }
4911 }
4912 assert!(found, "no index marker found: {:?}", runs);
4913 // The visible display "Andrew Abbott" is set in the body beside the marker.
4914 assert!(runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Andrew Abbott"))),
4915 "body display missing: {:?}", runs);
4916
4917 // An emphasised entry: the sort key is flattened, the display keeps the emphasis.
4918 let ital = parse_inlines("the ruling #idx[_Browder v. Gayle_] held.");
4919 let mut seen = false;
4920 for r in &ital {
4921 if let Inline::Index { term, display, .. } = r {
4922 assert_eq!(term, "Browder v. Gayle", "italic sort key not flattened: {:?}", ital);
4923 assert!(matches!(display.as_slice(), [Inline::Emph(t)] if t == "Browder v. Gayle"),
4924 "italic display lost its emphasis: {:?}", display);
4925 seen = true;
4926 }
4927 }
4928 assert!(seen, "no italic index marker found: {:?}", ital);
4929 }
4930
4931 /// `#strong[...]` and `#strong("...")` are the call forms of `*...*`: both yield an [`Inline::Strong`]
4932 /// run rather than a skip, and nothing raw leaks. A paren argument the reader cannot evaluate -- a
4933 /// bare identifier here -- keeps the current refusal rather than guessing at its text.
4934 #[test]
4935 fn strong_call_reads_as_bold() -> Outcome<()> {
4936 let bracket = parse_inlines("You want: #strong[the short version] first.");
4937 assert!(bracket.iter().any(|r| matches!(r, Inline::Strong(t) if t == "the short version")),
4938 "strong run missing from bracket form: {:?}", bracket);
4939 assert!(bracket.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#strong"))),
4940 "raw #strong leaked: {:?}", bracket);
4941
4942 let paren = parse_inlines("#strong(\"the short version\") first.");
4943 assert!(paren.iter().any(|r| matches!(r, Inline::Strong(t) if t == "the short version")),
4944 "strong run missing from paren form: {:?}", paren);
4945
4946 // A call carrying its own markup expands the same way `*...*` does.
4947 let nested = parse_inlines("#strong[the #idx[Harvard Business Review] weekly]");
4948 assert!(nested.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#strong") || t.contains("#idx"))),
4949 "raw markup leaked from nested strong: {:?}", nested);
4950
4951 // A paren argument that is not a plain string is left for the generic call handler, so it is
4952 // still tallied as a skip rather than silently guessed at.
4953 let (_it, skips) = res!(document_with_refusals("#strong(ident)\n\nBody.\n"));
4954 assert!(skips.entries().iter().any(|(n, _)| n == "#strong"),
4955 "an unresolvable #strong(...) argument must still be recorded as a refusal, got: {:?}",
4956 skips.entries());
4957 Ok(())
4958 }
4959
4960 /// Emphasis nested one level -- `*_x_*` or `_*x*_` -- collapses to a single [`Inline::BoldItalic`]
4961 /// run rather than dropping the outer face, so a bold-italic term sets in the bold-italic face in
4962 /// prose, a footnote or a table cell alike.
4963 #[test]
4964 fn nested_emphasis_reads_as_bold_italic() {
4965 for src in ["here *_both_* faces", "here _*both*_ faces"] {
4966 let runs = parse_inlines(src);
4967 assert!(runs.iter().any(|r| matches!(r, Inline::BoldItalic(t) if t == "both")),
4968 "expected a bold-italic run from {:?}, got {:?}", src, runs);
4969 assert!(runs.iter().all(|r| !matches!(r, Inline::Emph(t) if t == "both")),
4970 "outer face dropped to plain emph in {:?}: {:?}", src, runs);
4971 }
4972 // A lone `*strong*` or `_emph_` still yields its single face, unchanged.
4973 assert!(parse_inlines("just *strong* here").iter().any(|r| matches!(r, Inline::Strong(t) if t == "strong")));
4974 assert!(parse_inlines("just _emph_ here").iter().any(|r| matches!(r, Inline::Emph(t) if t == "emph")));
4975 }
4976
4977 /// A `(col, row) => ...` alignment closure is evaluated per cell: a row-keyed closure centres the
4978 /// header and flushes the body left; a per-column closure honours each column's own alignment,
4979 /// including an `or` over several columns.
4980 #[test]
4981 fn align_closure_evaluates_per_cell() {
4982 let row_keyed = parse_align("(col, row) => { if row == 0 { center } else { left } }");
4983 match row_keyed {
4984 AlignSpec::Closure(cl) => {
4985 assert_eq!(cl.align_at(0, 0), Align::Centre); // header, column 0
4986 assert_eq!(cl.align_at(0, 1), Align::Left); // body, column 0 -- was wrongly centred before
4987 assert_eq!(cl.align_at(2, 3), Align::Left);
4988 },
4989 other => panic!("expected a closure, got {:?}", other),
4990 }
4991 let per_col = parse_align("(col, row) => { if row == 0 { center } else if col == 0 or col == 4 { left } else { center } }");
4992 match per_col {
4993 AlignSpec::Closure(cl) => {
4994 assert_eq!(cl.align_at(0, 1), Align::Left);
4995 assert_eq!(cl.align_at(4, 1), Align::Left);
4996 assert_eq!(cl.align_at(1, 1), Align::Centre);
4997 assert_eq!(cl.align_at(3, 2), Align::Centre);
4998 },
4999 other => panic!("expected a closure, got {:?}", other),
5000 }
5001 // A tuple pick indexed by the column parameter.
5002 let tuple = parse_align("(x, y) => (left, center, right).at(x)");
5003 match tuple {
5004 AlignSpec::Closure(cl) => {
5005 assert_eq!(cl.align_at(0, 5), Align::Left);
5006 assert_eq!(cl.align_at(1, 5), Align::Centre);
5007 assert_eq!(cl.align_at(2, 5), Align::Right);
5008 },
5009 other => panic!("expected a closure, got {:?}", other),
5010 }
5011 }
5012
5013 /// A floating `#place` reads its side, scope, clearance and body; one that does not float, or carries an
5014 /// offset, is not a float and is refused at dispatch.
5015 #[test]
5016 fn place_reads_as_a_float_or_not_at_all() {
5017 match place_float_call("#place(top + center, scope: \"parent\", float: true, clearance: 2em)[Wide.]") {
5018 Some((f, c, body)) => {
5019 assert_eq!(f, Floating { side: FloatPlacement::Top, scope: FloatScope::Parent });
5020 assert_eq!(c, Some(Spacing::Em(2.0)));
5021 assert_eq!(body, "Wide.");
5022 },
5023 None => panic!("a floating place must read"),
5024 }
5025 assert!(matches!(place_float_call("#place(bottom, float: true)[Low.]"),
5026 Some((Floating { side: FloatPlacement::Bottom, scope: FloatScope::Column }, None, _))));
5027 assert!(place_float_call("#place(top)[Overlay.]").is_none(), "a non-floating place is an overlay");
5028 assert!(place_float_call("#place(top, float: true, dx: 2pt)[Moved.]").is_none(), "an offset is not set");
5029 }
5030
5031 /// `#smallcaps[...]` yields an [`Inline::SmallCaps`] run of its content, in both argument forms, and
5032 /// never leaves raw source behind.
5033 #[test]
5034 fn smallcaps_call_reads_as_small_caps() {
5035 let runs = parse_inlines("The #smallcaps[Nato] treaty, and #smallcaps(\"un\") too.");
5036 let sc: Vec<&String> = runs.iter().filter_map(|r| match r {
5037 Inline::SmallCaps(t) => Some(t),
5038 _ => None,
5039 }).collect();
5040 assert_eq!(sc, vec!["Nato", "un"], "unexpected small-caps runs: {:?}", runs);
5041 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#smallcaps"))),
5042 "raw #smallcaps leaked: {:?}", runs);
5043 }
5044
5045 /// `#super[...]` yields an [`Inline::Super`] run of its content, in both the bracket and the string
5046 /// argument forms, and never leaves raw source behind.
5047 #[test]
5048 fn super_call_reads_as_superscript() {
5049 let runs = parse_inlines("The area is 10#super[6] units, split#super[†] on the case.");
5050 let sups: Vec<&String> = runs.iter().filter_map(|r| match r {
5051 Inline::Super(t) => Some(t),
5052 _ => None,
5053 }).collect();
5054 assert_eq!(sups, vec!["6", "†"], "unexpected superscript runs: {:?}", runs);
5055 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#super"))),
5056 "raw #super leaked: {:?}", runs);
5057 // The string-argument form reads the same, its quotes stripped.
5058 let quoted = parse_inlines("x#super(\"2\")");
5059 assert!(quoted.iter().any(|r| matches!(r, Inline::Super(t) if t == "2")),
5060 "quoted super run missing: {:?}", quoted);
5061 }
5062
5063 /// `#sub[...]` yields an [`Inline::Sub`] run of its content, as `CO#sub[2]` sets it in Lucronics, and
5064 /// never leaves raw source behind.
5065 #[test]
5066 fn sub_call_reads_as_subscript() {
5067 let runs = parse_inlines("CO#sub[2] scrubber and H#sub[2]O#sub[2] both drop.");
5068 let subs: Vec<&String> = runs.iter().filter_map(|r| match r {
5069 Inline::Sub(t) => Some(t),
5070 _ => None,
5071 }).collect();
5072 assert_eq!(subs, vec!["2", "2", "2"], "unexpected subscript runs: {:?}", runs);
5073 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#sub"))),
5074 "raw #sub leaked: {:?}", runs);
5075 // The string-argument form reads the same, its quotes stripped.
5076 let quoted = parse_inlines("x#sub(\"2\")");
5077 assert!(quoted.iter().any(|r| matches!(r, Inline::Sub(t) if t == "2")),
5078 "quoted sub run missing: {:?}", quoted);
5079 }
5080
5081 /// The `.enumerate().map(((idx, row)) => { if COND { row } else { table.cell(colspan: n)[...] } })
5082 /// .flatten()` row-remap idiom -- Lucronics' E. coli comparison uses it to merge a section-heading row
5083 /// into one bold spanning cell while an ordinary row passes through -- resolves through the spread, not
5084 /// just the untransformed array. This is the reduced shape of the real table (3 columns, one heading
5085 /// row) rather than the full 7-column original.
5086 #[test]
5087 fn table_spread_map_merges_heading_rows_into_bold_spanning_cells() {
5088 let let_src = "#let data = (\n ([Head A], [Head B], [Head C]),\n ([Section], [], []),\n ([Item one], [x], [y]),\n)\n";
5089 let mut arrays = HashMap::new();
5090 arrays.insert("data".to_string(), parse_let_array(let_src));
5091
5092 let table_src = "\n columns: 3,\n ..data.enumerate().map(((idx, row)) => {\n if idx == 0 or row.at(0) != [Section] {\n row\n } else {\n table.cell(colspan: 3)[#strong(row.at(0))]\n }\n }).flatten()\n";
5093 let spec = parse_table_spec(table_src, &arrays, None).expect("the spread must resolve to a table");
5094 assert_eq!(spec.cells.len(), 9, "3 rows x 3 columns, the merged row padded to width");
5095 // Row 0 (the header) and row 2 (an ordinary item) pass through unchanged.
5096 assert!(matches!(spec.cells[0].as_slice(), [Inline::Text(t)] if t == "Head A"));
5097 assert!(matches!(spec.cells[6].as_slice(), [Inline::Text(t)] if t == "Item one"));
5098 // Row 1 (the section heading) is merged into one bold cell, padded to the column count.
5099 assert!(matches!(spec.cells[3].as_slice(), [Inline::Strong(t)] if t == "Section"),
5100 "expected the merged heading cell, got {:?}", spec.cells[3]);
5101 assert!(matches!(spec.cells[4].as_slice(), [Inline::Text(t)] if t.is_empty()), "padding cell after the span");
5102 assert!(matches!(spec.cells[5].as_slice(), [Inline::Text(t)] if t.is_empty()), "padding cell after the span");
5103 }
5104
5105 /// A citation nested in emphasis keeps its own [`Inline::Cite`] run rather than leaking its source,
5106 /// while the surrounding words take the emphasis face.
5107 #[test]
5108 fn cite_survives_inside_emphasis() {
5109 let runs = parse_inlines("the classic _early work #cite(<coase1937nature>) here_.");
5110 assert!(runs.iter().any(|r| matches!(r, Inline::Cite(k) if k == &vec!["coase1937nature".to_string()])),
5111 "cite run missing: {:?}", runs);
5112 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#cite"))),
5113 "raw #cite leaked: {:?}", runs);
5114 }
5115
5116 /// `#link("url")[text]` sets the link text in the running line and drops the URL, so no raw `#link`
5117 /// leaks; a bare `#link("url")` with no bracket sets the URL as its own text.
5118 #[test]
5119 fn link_call_renders_text_not_markup() {
5120 let runs = parse_inlines("See #link(\"https://aistatement.com\")[Centre for AI Safety, May 2023] on risk.");
5121 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#link") || t.contains("http"))),
5122 "raw link markup leaked: {:?}", runs);
5123 let joined = flatten_markup("See #link(\"https://aistatement.com\")[Centre for AI Safety, May 2023] on risk.");
5124 assert_eq!(joined, "See Centre for AI Safety, May 2023 on risk.");
5125 // The label-destination form and the bare no-body form both set text without leaking.
5126 assert_eq!(flatten_markup("read #link(<intro>)[the opening]"), "read the opening");
5127 assert_eq!(flatten_markup("at #link(\"elearnity.io\")"), "at elearnity.io");
5128 // A line-leading link is prose the inline scanner reads, not a standalone call the line scanner drops.
5129 assert!(is_inline_call("link"));
5130 assert!(code_skip("#link(\"https://x.io\")[click]").is_none());
5131 }
5132
5133 /// Typst's smartypants: `--` becomes an en dash and `---` an em dash in ordinary prose, matched
5134 /// longest-first, but a raw code span, a maths span and an escaped hyphen are all left untouched.
5135 #[test]
5136 fn dash_runs_become_en_and_em_dashes_in_prose() {
5137 assert_eq!(flatten_markup("a--b"), "a\u{2013}b");
5138 assert_eq!(flatten_markup("a---b"), "a\u{2014}b");
5139 // Four hyphens is an em dash plus a literal hyphen, greedy left to right, not two en dashes.
5140 assert_eq!(flatten_markup("a----b"), "a\u{2014}-b");
5141 assert_eq!(flatten_markup("Council of Trent (1545--1563)"), "Council of Trent (1545\u{2013}1563)");
5142 // A raw code span keeps its `--` literal: the inline scanner claims `` ` `` before this fallback
5143 // is ever reached, so the substitution never sees the span's characters.
5144 let runs = parse_inlines("see `a--b` here");
5145 assert!(runs.iter().any(|r| matches!(r, Inline::Code(t) if t == "a--b")),
5146 "raw span lost or converted: {:?}", runs);
5147 // A maths span keeps its `-` as subtraction, for the same reason.
5148 let runs = parse_inlines("$a-b$");
5149 assert!(matches!(runs.as_slice(), [Inline::Math(_)]), "maths span not read as maths: {:?}", runs);
5150 // A `\-` escape is consumed on its own, so it never joins the hyphen after it into a run: the
5151 // pair stays two literal ASCII hyphens rather than folding to the single en-dash glyph `a--b`
5152 // (unescaped) becomes.
5153 assert_eq!(flatten_markup("a\\--b"), "a--b");
5154 // `#link`'s destination is dropped entirely (austenite renders no clickable URL), so a `--` inside
5155 // one is never seen at all -- the strictest match to Typst leaving an auto-linked URL unconverted.
5156 assert_eq!(flatten_markup("see #link(\"https://x--y.com\")[the site]"), "see the site");
5157 }
5158
5159 /// The same fallback trivially covers Typst's other markup substitution on dots: three or more become
5160 /// an ellipsis, fewer stay literal, and the rule is still skipped inside code and maths.
5161 #[test]
5162 fn dot_runs_become_ellipsis_in_prose() {
5163 assert_eq!(flatten_markup("wait..."), "wait\u{2026}");
5164 assert_eq!(flatten_markup("a....b"), "a\u{2026}.b");
5165 assert_eq!(flatten_markup("a..b"), "a..b"); // two dots: no defined symbol, left alone
5166 assert_eq!(flatten_markup("v1.2.3"), "v1.2.3"); // scattered single dots: unaffected
5167 let runs = parse_inlines("`a...b`");
5168 assert!(matches!(runs.as_slice(), [Inline::Code(t)] if t == "a...b"), "raw span converted: {:?}", runs);
5169 }
5170
5171 /// The term-dictionary aliases set their argument text with the styling of their `gs` siblings: `g`/`gi`
5172 /// a first-use glossary term, `t`/`tcap` plain text, and none of them leak raw markup. A line opening
5173 /// with one is prose, not a skipped standalone call.
5174 /// Installs a fixed term dictionary so the term-dictionary family resolves deterministically. The map
5175 /// is a process-global shared across the parallel tests, so every term-dependent test installs the
5176 /// same one and their order cannot matter.
5177 fn install_test_terms() {
5178 let mut m = HashMap::new();
5179 m.insert("org".to_string(), "Elearnity Pty Ltd".to_string());
5180 m.insert("org_short".to_string(), "Elearnity".to_string());
5181 m.insert("website".to_string(), "elearnity.oxegen.io".to_string());
5182 m.insert("iniverse".to_string(), "iniverse".to_string());
5183 set_term_dict(m).expect("install test term dict");
5184 }
5185
5186 #[test]
5187 fn term_dict_aliases_render_without_leaking() {
5188 install_test_terms();
5189 // `iniverse` translates to itself, so the display equals the key here whether or not a map is set.
5190 let runs = parse_inlines("Call it the #g[iniverse], your inner universe.");
5191 assert!(runs.iter().any(|r| matches!(r, Inline::Glossary { term, display } if term == "iniverse" && display == "iniverse")),
5192 "glossary alias missing: {:?}", runs);
5193 // A key that differs from its value now sets the value, not the key.
5194 assert_eq!(flatten_markup("Visit #t[website] today"), "Visit elearnity.oxegen.io today");
5195 // A key absent from the dictionary falls back to its own text, capitalised for the `-cap` form.
5196 assert_eq!(flatten_markup("#tcap[donate] to help"), "Donate to help");
5197 assert!(is_inline_call("g") && is_inline_call("t") && is_inline_call("graw"));
5198 assert!(code_skip("#t[website]").is_none());
5199 }
5200
5201 /// The term-dictionary family sets a key's value: `t`/`graw` plain, `tcap` capitalised, `g` bold-italic
5202 /// on first use keyed by the key; an unknown key falls back to the key text and is recorded, never a
5203 /// panic (the template panics on a miss, the reader must not).
5204 #[test]
5205 fn term_dict_resolves_key_to_value() {
5206 install_test_terms();
5207 assert_eq!(flatten_markup("Visit #t[website]"), "Visit elearnity.oxegen.io");
5208 assert_eq!(flatten_markup("#graw[org]"), "Elearnity Pty Ltd");
5209 let runs = parse_inlines("The #g[org] view.");
5210 assert!(runs.iter().any(|r| matches!(r, Inline::Glossary { term, display }
5211 if term == "org" && display == "Elearnity Pty Ltd")),
5212 "g did not translate the key to its value: {:?}", runs);
5213 // An unknown key: the key text stands and the miss is recorded on the skip tally.
5214 let mut skips = Refusals::default();
5215 let runs = parse_inlines_in("A #t[nonesuch] term.", Span::new(0, 0), &mut skips);
5216 assert!(runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("nonesuch"))),
5217 "unknown term-dict key did not fall back to its text: {:?}", runs);
5218 assert_eq!(skips.total(), 1, "an unknown term-dict key was not recorded");
5219 }
5220
5221 /// A term-dictionary lookup inside a `#let` content-function body, keyed on the function's own parameter
5222 /// (`#let cite-term(w) = [Learn about #t(w).]`), resolves against the argument the caller passed --
5223 /// `#cite-term("website")` must look up "website", not the literal parameter name "w" -- so substitution
5224 /// must run before the nested `#t` call is read. Reverting the ordering fix (substituting only bare
5225 /// `#param` markup references, never a bare identifier inside a call's argument list) reproduces exactly
5226 /// the reported failure: the key "w" is looked up, is not in the dictionary, and the fallback plus a
5227 /// recorded skip fire instead of the resolved value.
5228 #[test]
5229 fn term_dict_resolves_inside_a_content_fn_body_after_param_substitution() {
5230 install_test_terms();
5231 let mut cfns = crate::lang::rules::ContentFns::new();
5232 cfns.insert("cite-term".to_string(), crate::lang::rules::ContentFn {
5233 params: vec!["w".to_string()],
5234 body: "Learn about #t(w).".to_string(),
5235 wrapper: None,
5236 });
5237 let tfns = crate::lang::rules::TemplateFns::new();
5238 let binds = crate::lang::rules::Bindings::new(&tfns, &cfns);
5239 let (items, skips) = document_with_templates("#cite-term(\"website\")\n", binds).expect("parse");
5240 let resolved = items.iter().any(|it| matches!(it,
5241 Item::Paragraph { runs, .. } if runs.iter().any(|r|
5242 matches!(r, Inline::Text(t) if t.contains("elearnity.oxegen.io")))));
5243 assert!(resolved,
5244 "the content-fn's own parameter did not resolve as the term-dict key: {:?}", items);
5245 assert_eq!(skips.total(), 0, "a resolved key must not be recorded as an unknown term-dict miss: {:?}", skips);
5246 }
5247
5248 /// The ordering fix is scoped to a call's argument list, not to ordinary prose: a parameter's name
5249 /// appearing as a genuine word in the body's running text (outside any call) is left exactly as written,
5250 /// since a bare word in markup mode is prose, never a code-mode variable reference.
5251 #[test]
5252 fn call_arg_substitution_does_not_touch_ordinary_prose() {
5253 let cf = crate::lang::rules::ContentFn {
5254 params: vec!["w".to_string()],
5255 body: "The word w on its own is prose, not #t(w).".to_string(),
5256 wrapper: None,
5257 };
5258 let expanded = expand_content_body(&cf, &["website".to_string()]);
5259 assert_eq!(expanded, "The word w on its own is prose, not #t(\"website\").",
5260 "prose text was wrongly substituted, or the call argument was not: {:?}", expanded);
5261 }
5262
5263 /// A blank line between numbered items does not restart the enum: Typst continues the numbering across
5264 /// the gap, so the items form one list. Real content between two lists still starts a fresh one.
5265 #[test]
5266 fn blank_line_between_enum_items_continues_one_list() {
5267 let src = "+ first item\n\n+ second item\n\n+ third item\n";
5268 let (items, _) = document_with_refusals(src).expect("parse");
5269 let lists: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::List { .. })).collect();
5270 assert_eq!(lists.len(), 1, "blank lines split the enum: {:?}", items);
5271 match lists[0] {
5272 Item::List { ordered, items, .. } => {
5273 assert!(*ordered, "the continued list lost its ordered kind");
5274 assert_eq!(items.len(), 3, "the enum dropped items across the blanks: {:?}", items);
5275 },
5276 _ => unreachable!(),
5277 }
5278 // A paragraph between two lists still restarts, so genuinely separate lists are not merged.
5279 let src2 = "+ a\n\n+ b\n\nA paragraph between.\n\n+ c\n";
5280 let (items2, _) = document_with_refusals(src2).expect("parse");
5281 let lists2 = items2.iter().filter(|it| matches!(it, Item::List { .. })).count();
5282 assert_eq!(lists2, 2, "prose between two lists did not restart them: {:?}", items2);
5283 }
5284
5285 /// An indented `-` sub-bullet between two `+` steps nests under the step it follows rather than closing
5286 /// the enum: the ordered list stays one list of three items, and the sub-bullet hangs under the first.
5287 #[test]
5288 fn indented_sub_bullet_nests_and_enum_continues() {
5289 let src = "+ step one\n - a sub point\n - another sub point\n+ step two\n+ step three\n";
5290 let (items, _) = document_with_refusals(src).expect("parse");
5291 let lists: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::List { .. })).collect();
5292 assert_eq!(lists.len(), 1, "the sub-bullet split the enum into several lists: {:?}", items);
5293 match lists[0] {
5294 Item::List { ordered, items, .. } => {
5295 assert!(*ordered, "the parent list lost its ordered kind");
5296 assert_eq!(items.len(), 3, "the enum did not keep three steps: {:?}", items);
5297 // The sub-bullets hang under the first step, as an unordered child list of two items.
5298 assert_eq!(items[0].children.len(), 1, "the first step lost its sub-list: {:?}", items[0]);
5299 match &items[0].children[0] {
5300 Item::List { ordered: cord, items: citems, .. } => {
5301 assert!(!*cord, "the sub-list should be unordered");
5302 assert_eq!(citems.len(), 2, "the sub-list dropped an item: {:?}", citems);
5303 },
5304 other => panic!("the child was not a nested list: {:?}", other),
5305 }
5306 assert!(items[1].children.is_empty(), "step two wrongly gained children");
5307 },
5308 _ => unreachable!(),
5309 }
5310 }
5311
5312 /// Two levels of indentation parse to two levels of nesting: a `-` under a `+`, and a deeper `-` under
5313 /// that `-`, so the tree is enum -> bullet -> bullet.
5314 #[test]
5315 fn two_level_nesting_parses_to_two_levels() {
5316 let src = "+ outer step\n - middle bullet\n - inner bullet\n+ next step\n";
5317 let (items, _) = document_with_refusals(src).expect("parse");
5318 let lists: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::List { .. })).collect();
5319 assert_eq!(lists.len(), 1, "the deep nesting split the list: {:?}", items);
5320 match lists[0] {
5321 Item::List { items, .. } => {
5322 assert_eq!(items.len(), 2, "the outer enum did not keep two steps: {:?}", items);
5323 let mid = &items[0].children;
5324 assert_eq!(mid.len(), 1, "the middle level is missing: {:?}", items[0]);
5325 match &mid[0] {
5326 Item::List { items: mid_items, .. } => {
5327 assert_eq!(mid_items.len(), 1, "the middle list should hold one bullet");
5328 let inner = &mid_items[0].children;
5329 assert_eq!(inner.len(), 1, "the inner level is missing: {:?}", mid_items[0]);
5330 match &inner[0] {
5331 Item::List { items: inner_items, .. } =>
5332 assert_eq!(inner_items.len(), 1, "the inner list should hold one bullet"),
5333 other => panic!("the inner child was not a list: {:?}", other),
5334 }
5335 },
5336 other => panic!("the middle child was not a list: {:?}", other),
5337 }
5338 },
5339 _ => unreachable!(),
5340 }
5341 }
5342
5343 /// An unhandled inline `#func[...]` is consumed and recorded rather than left as raw markup, its
5344 /// bracketed body folded in so its words survive; a paren-only call sets nothing where it stood.
5345 #[test]
5346 fn unknown_inline_call_is_recorded_not_leaked() {
5347 let mut skips = Refusals::default();
5348 let runs = parse_inlines_in("a #overline[Nato] treaty and a #v(2pt) gap", Span::new(0, 0), &mut skips);
5349 assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#overline") || t.contains("#v("))),
5350 "raw unknown call leaked: {:?}", runs);
5351 assert!(runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Nato"))),
5352 "overline body dropped: {:?}", runs);
5353 assert_eq!(skips.total(), 2);
5354 let names: Vec<String> = skips.entries().into_iter().map(|(n, _)| n).collect();
5355 assert!(names.contains(&"#overline".to_string()) && names.contains(&"#v".to_string()),
5356 "unexpected skip names: {:?}", names);
5357 }
5358
5359 /// The reader tallies the code lines and unknown calls it skips, and reports them one line, so a
5360 /// dropped construct is visible rather than silent. Handled inline calls do not appear in the tally.
5361 #[test]
5362 fn skip_summary_reports_skipped_constructs() {
5363 install_test_terms(); // so `#g[iniverse]` resolves and adds no term-dict miss to the tally
5364 // `#import` and `#set rect` (which names no theme element) stay reader refusals. A per-element
5365 // `#show <selector>: ...` line is a rule the engine collects and applies (or refuses in its own
5366 // diagnostic), so the reader captures it as a declarative-styling construct and no longer tallies it
5367 // -- see `show_selector_rule_is_not_a_reader_skip` and the L0 double-report fix.
5368 let src = "#import \"x.typ\": *\n#set rect(stroke: 1pt)\n\nBody with #g[iniverse] and a #footnote[note].\n\n#show heading: it => it\n";
5369 let (_, skips) = document_with_refusals(src).expect("parse");
5370 assert_eq!(skips.total(), 2, "the selector show rule is captured for the engine, not tallied: {:?}", skips.sites());
5371 let report = skips.report().expect("a report");
5372 assert!(report.starts_with("skipped 2 unsupported constructs:"), "report was {:?}", report);
5373 for name in ["#import", "#set"] {
5374 assert!(report.contains(name), "{} missing from {:?}", name, report);
5375 }
5376 assert!(!report.contains("heading"),
5377 "a selector show rule must not appear in the reader's skip tally (it is the engine's to apply/refuse): {:?}", report);
5378 // A source the reader sets whole has nothing to report.
5379 let (_, clean) = document_with_refusals("Just prose with #g[iniverse].\n").expect("parse");
5380 assert!(clean.is_empty() && clean.report().is_none());
5381 }
5382
5383 /// A `#show <selector>: <transform>` line is a per-element rule the engine collects and applies, so the
5384 /// reader captures it as a declarative-styling construct rather than tallying it as a skipped one -- the
5385 /// L0 double-report fix. Both the lowerable set-fields form and the refused introspective form are
5386 /// captured; the refused one surfaces through the rule engine's own diagnostic, not the reader's tally.
5387 #[test]
5388 fn show_selector_rule_is_not_a_reader_skip() {
5389 let (_, lowerable) = document_with_refusals("#show par: set text(size: 9pt)\n\nBody.\n").expect("parse");
5390 assert_eq!(lowerable.total(), 0, "a lowerable selector rule is not a reader skip: {:?}", lowerable.sites());
5391 let (_, introspective) = document_with_refusals("#show heading: it => it\n\nBody.\n").expect("parse");
5392 assert_eq!(introspective.total(), 0,
5393 "a refused selector rule surfaces via the engine, not the reader tally: {:?}", introspective.sites());
5394 }
5395
5396 /// A `#show: <template>.with(...)` application and a lowerable top-level `#set` are captured rather
5397 /// than refused -- their styling lowers onto the theme -- so neither adds to the refusal tally, while
5398 /// an introspective `#show` and a `#set` on an unsupported target still do.
5399 #[test]
5400 fn lowerable_set_and_doc_with_are_captured_not_refused() {
5401 // A multi-line `#show: doc.with(...)` and a lowerable `#set text(...)`: both captured, no refusal.
5402 let lowered = "#show: doc.with(\n title: [X],\n heading-font: \"Graystroke\",\n)\n\n#set text(size: 11pt)\n\n= Heading\n\nBody.\n";
5403 let (_, skips) = document_with_refusals(lowered).expect("parse");
5404 assert_eq!(skips.total(), 0, "lowerable declarations should not be refused: {:?}", skips.sites());
5405
5406 // The unsupported `#set` still refuses; a `#show <selector>:` rule is now the engine's to apply or
5407 // refuse, so it is captured here rather than tallied by the reader (see
5408 // `show_selector_rule_is_not_a_reader_skip`).
5409 let refused = "#set rect(stroke: 1pt)\n\n#show heading: it => it\n";
5410 let (_, skips) = document_with_refusals(refused).expect("parse");
5411 assert_eq!(skips.total(), 1, "the unsupported #set still refuses; the show rule is captured for the engine: {:?}", skips.sites());
5412 }
5413
5414 /// A line-leading `#context[...]` -- Typst's self-observation entry point -- is refused as exactly
5415 /// one site, classed `Introspective`. It is now gathered as a capture (so its body can be inspected for
5416 /// the reverse-claim-index signature) and refused when that signature is absent, so its span is the
5417 /// zero-width caret at the construct's opening offset, as the reader's other capture refusals record.
5418 #[test]
5419 fn context_call_is_one_introspective_refusal() {
5420 let src = "#context[whatever]\n";
5421 let (_, refusals) = document_with_refusals(src).expect("parse");
5422 assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites());
5423 let site = &refusals.sites()[0];
5424 assert_eq!(site.name, "#context");
5425 assert_eq!(site.class, RefusalClass::Introspective);
5426 assert_eq!(site.span, Span::new(0, 0), "a captured-construct refusal records the caret at its opening offset");
5427 }
5428
5429 /// The brace twin of the above -- a line-leading `#context{ ... }` code-block call, the shape a book's
5430 /// reverse-reference index is written with (Lucronics ch29.8) -- is recognised and refused the same
5431 /// way, not set as prose. Both a one-line block and a multi-line one are refused as one introspective
5432 /// site, and neither leaks its source into the body: the reader must produce no paragraph at all.
5433 #[test]
5434 fn context_brace_block_is_refused_not_set_as_prose() {
5435 // One line, brace immediately after the keyword.
5436 let one = "#context{ let x = 1 }\n";
5437 let (items, refusals) = document_with_refusals(one).expect("parse");
5438 assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites());
5439 assert_eq!(refusals.sites()[0].name, "#context");
5440 assert_eq!(refusals.sites()[0].class, RefusalClass::Introspective);
5441 assert!(!items.iter().any(|it| matches!(it, Item::Paragraph { .. })),
5442 "a #context{{}} block must not survive as body text: {:?}", items);
5443
5444 // Several lines, with a space before the brace -- a multi-line `#context {` block that does NOT call
5445 // `collect-claim-refs(` (so it is a plain introspective refusal, not the reverse claim index). The
5446 // whole block, including its own `[...]` and nested `{...}`, is consumed by the capture, not one line
5447 // of it set as prose.
5448 let many = "Before.\n\n#context {\n let by = (:)\n if by.len() == 0 [\n _None._\n ] else {\n let n = 1\n }\n}\n\nAfter.\n";
5449 let (items, refusals) = document_with_refusals(many).expect("parse");
5450 assert_eq!(refusals.total(), 1, "the multi-line brace block is one refusal: {:?}", refusals.sites());
5451 assert_eq!(refusals.sites()[0].name, "#context");
5452 let bodies: Vec<String> = items.iter().filter_map(|it| match it {
5453 Item::Paragraph { runs, .. } => Some(fmt!("{:?}", runs)),
5454 _ => None,
5455 }).collect();
5456 assert!(!bodies.iter().any(|b| b.contains("None") || b.contains("let by") || b.contains("by-code")),
5457 "no line of the #context{{}} block may leak into a paragraph: {:?}", bodies);
5458 // The two real paragraphs around it still set.
5459 assert_eq!(bodies.len(), 2, "the prose on either side of the block must still set: {:?}", bodies);
5460 }
5461
5462 /// A `//` or `/* ... */` comment inside a `#context { ... }` body must not fold its own `}`/`]` into the
5463 /// skip scanner's bracket balance -- the G3 fix. Before it, `let c = 1 // }` popped the outer brace early,
5464 /// so the tail of the block (the `if`/`else` and the closing `}`) leaked into the body as raw prose; this
5465 /// reds on a reverted `step` exactly the way the earlier form's leak did. The body calls a non-claim query
5466 /// (`counter(page).display()`, not `collect-claim-refs(`), so it stays a plain introspective refusal and
5467 /// does not trip the reverse-claim-index recognition covered by
5468 /// `context_collect_claim_refs_lowers_to_a_claim_index` below.
5469 #[test]
5470 fn context_brace_block_comment_does_not_close_early() {
5471 let many = "Before.\n\n#context {\n let refs = counter(page).display()\n let c = 1 // }\n /* a note about } */\n if refs.len() == 0 [\n _None._\n ] else {\n let by = (:)\n }\n}\n\nAfter.\n";
5472 let (items, refusals) = document_with_refusals(many).expect("parse");
5473 assert_eq!(refusals.total(), 1, "the whole commented block is still one refusal: {:?}", refusals.sites());
5474 assert_eq!(refusals.sites()[0].name, "#context");
5475 let bodies: Vec<String> = items.iter().filter_map(|it| match it {
5476 Item::Paragraph { runs, .. } => Some(fmt!("{:?}", runs)),
5477 _ => None,
5478 }).collect();
5479 assert!(!bodies.iter().any(|b| b.contains("counter") || b.contains("let refs") || b.contains("let by")),
5480 "a comment's `}}` must not close the guard early and leak its tail: {:?}", bodies);
5481 assert_eq!(bodies.len(), 2, "the prose on either side of the block must still set: {:?}", bodies);
5482 }
5483
5484 /// The one `#context { ... }` block the reader does not refuse: the Logic appendix's reverse claim index,
5485 /// recognised by the `collect-claim-refs(` signature in its body (never by evaluating the `#context`). It
5486 /// lowers to a single `Item::ClaimIndex`, records no refusal, and leaks no line of its source as prose.
5487 #[test]
5488 fn context_collect_claim_refs_lowers_to_a_claim_index() {
5489 let src = "Before.\n\n#context {\n let refs = collect-claim-refs()\n if refs.len() == 0 [\n _None._\n ] else {\n let by = (:)\n }\n}\n\nAfter.\n";
5490 let (items, refusals) = document_with_refusals(src).expect("parse");
5491 assert_eq!(refusals.total(), 0, "the reverse claim index is recognised, not refused: {:?}", refusals.sites());
5492 assert_eq!(items.iter().filter(|it| matches!(it, Item::ClaimIndex { .. })).count(), 1,
5493 "the collect-claim-refs block lowers to exactly one ClaimIndex: {:?}", items);
5494 let bodies: Vec<String> = items.iter().filter_map(|it| match it {
5495 Item::Paragraph { runs, .. } => Some(fmt!("{:?}", runs)),
5496 _ => None,
5497 }).collect();
5498 assert!(!bodies.iter().any(|b| b.contains("collect") || b.contains("let refs") || b.contains("None")),
5499 "no line of the block may leak into a paragraph: {:?}", bodies);
5500 assert_eq!(bodies.len(), 2, "the prose on either side of the block must still set: {:?}", bodies);
5501 }
5502
5503 /// A line-leading `#query(...)` -- reading the document's own resolved structure back -- is refused
5504 /// as exactly one site, classed `Introspective`.
5505 #[test]
5506 fn query_call_is_one_introspective_refusal() {
5507 let src = "#query(heading)\n";
5508 let (_, refusals) = document_with_refusals(src).expect("parse");
5509 assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites());
5510 let site = &refusals.sites()[0];
5511 assert_eq!(site.name, "#query");
5512 assert_eq!(site.class, RefusalClass::Introspective);
5513 assert_eq!(site.span, Span::new(0, src.len() as u32 - 1));
5514 }
5515
5516 /// A line-leading `#state(...)` call -- a state read/write that only resolves against Typst's own
5517 /// layout observation -- is refused as exactly one site, classed `Introspective`.
5518 #[test]
5519 fn state_call_is_one_introspective_refusal() {
5520 let src = "#state(\"count\", 0)\n";
5521 let (_, refusals) = document_with_refusals(src).expect("parse");
5522 assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites());
5523 let site = &refusals.sites()[0];
5524 assert_eq!(site.name, "#state");
5525 assert_eq!(site.class, RefusalClass::Introspective);
5526 assert_eq!(site.span, Span::new(0, src.len() as u32 - 1));
5527 }
5528
5529 /// The three classes land where the classifier's own doc comment says they should: Typst's general
5530 /// evaluation primitives are `FixedPoint`, an unrecognised call or wrapper is `Unsupported`.
5531 #[test]
5532 fn refusal_class_sorts_fixed_point_and_unsupported_correctly() {
5533 assert_eq!(RefusalClass::classify("#let"), RefusalClass::FixedPoint);
5534 assert_eq!(RefusalClass::classify("#show"), RefusalClass::FixedPoint);
5535 assert_eq!(RefusalClass::classify("#columns"), RefusalClass::Unsupported);
5536 assert_eq!(RefusalClass::classify("#overline"), RefusalClass::Unsupported);
5537 }
5538
5539 /// A `#columns(n)[ ... ]` wrapper is recorded as skipped and its body set single-column, so the words
5540 /// survive and no raw wrapper leaks into the block stream.
5541 #[test]
5542 fn columns_wrapper_flattens_to_single_column() {
5543 let src = "#columns(2)[\nFirst paragraph here.\n\nSecond paragraph here.\n]\n";
5544 let (items, skips) = document_with_refusals(src).expect("parse");
5545 let paras = items.iter().filter(|it| matches!(it, Item::Paragraph { .. })).count();
5546 assert_eq!(paras, 2, "column body not set as paragraphs: {:?}", items);
5547 assert_eq!(skips.entries(), vec![("#columns".to_string(), 1)]);
5548 }
5549
5550 /// Reads a single [`Item::Paragraph`]'s runs out of a parse, failing loudly with the whole item list
5551 /// when the source did not yield exactly one paragraph -- the shape every `math_open` regression test
5552 /// below expects, since a display block that leaked a line would instead split the source into a
5553 /// paragraph plus a stray heading or list.
5554 fn one_paragraph(items: &[Item]) -> Outcome<(Vec<Inline>, Option<String>)> {
5555 let paras: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::Paragraph { .. })).collect();
5556 match paras.as_slice() {
5557 [Item::Paragraph { runs, label, .. }] => Ok((runs.clone(), label.clone())),
5558 _ => Err(err!("expected exactly one paragraph, got: {:?}", items; Test, Bug)),
5559 }
5560 }
5561
5562 /// A source line inside an open `$...$` display block that begins `=` (an alignment row such as
5563 /// `=> 2N &= ...`, Oxegen TechSpec `app_maths.typ:533-534`) must stay in the equation, not be read as
5564 /// a heading -- the heading branch is gated off by `math_open` for exactly this shape.
5565 #[test]
5566 fn equals_lead_row_inside_display_math_stays_in_block() -> Outcome<()> {
5567 let src = "Total hashes.\n\n$\nN &= sum_(j=1)^J n_j \\\n=> 2N &= sum_(j=2)^(J+1) 2^(j-1) \\\n=> 2N - N &= 2^J - 1 \\\n$\n\nEnd of block.\n";
5568 let (items, _skips) = res!(document_with_refusals(src));
5569 assert!(!items.iter().any(|it| matches!(it, Item::Heading { .. })),
5570 "a `=`-lead row inside the block must not become a heading: {:?}", items);
5571 let has_align = items.iter().any(|it| matches!(it,
5572 Item::Paragraph { runs, .. } if matches!(runs.as_slice(),
5573 [Inline::Math(Atom::Matrix { kind: MatKind::Align, .. })])));
5574 assert!(has_align, "no single-run display alignment paragraph was produced: {:?}", items);
5575 Ok(())
5576 }
5577
5578 /// A source line inside an open `$...$` display block that begins `-` (a row such as
5579 /// `- tilde(N)_(1 0) u (...) = 0`, Oxegen TechSpec `ch04_nodes.typ:1657`) must stay in the equation,
5580 /// not be read as a bullet -- the list-marker branch is gated off by `math_open` for exactly this shape.
5581 #[test]
5582 fn dash_lead_row_inside_display_math_stays_in_block() -> Outcome<()> {
5583 let src = "$\na &= b \\\n- tilde(N)_(1 0) u (x) = 0 \\\nc &= d\n$\n";
5584 let (items, _skips) = res!(document_with_refusals(src));
5585 assert!(!items.iter().any(|it| matches!(it, Item::List { .. })),
5586 "a `-`-lead row inside the block must not become a list: {:?}", items);
5587 let (runs, _label) = res!(one_paragraph(&items));
5588 assert!(matches!(runs.as_slice(), [Inline::Math(Atom::Matrix { kind: MatKind::Align, .. })]),
5589 "expected one display alignment run, got: {:?}", runs);
5590 Ok(())
5591 }
5592
5593 /// A blank source line inside an open `$...$` display block (Oxegen TechSpec `ch04_nodes.typ:1661-1666`)
5594 /// must stay in the equation rather than flush the paragraph early, and a closing `$ <label>` still
5595 /// labels the resulting equation.
5596 #[test]
5597 fn blank_line_inside_display_math_stays_in_block_and_label_still_attaches() -> Outcome<()> {
5598 let src = "$\na &= b \\\n\nc &= d\n$ <eq_test>\n";
5599 let (items, _skips) = res!(document_with_refusals(src));
5600 let (runs, label) = res!(one_paragraph(&items));
5601 assert!(matches!(runs.as_slice(), [Inline::Math(Atom::Matrix { kind: MatKind::Align, .. })]),
5602 "expected one display alignment run, got: {:?}", runs);
5603 assert_eq!(label, Some("eq_test".to_string()), "the closing label must still attach");
5604 Ok(())
5605 }
5606
5607 /// The plain text of a run of inline markup, for asserting a caption or paragraph's words without
5608 /// caring how they were split into text, emphasis, glossary or maths runs.
5609 fn plain(runs: &[Inline]) -> String {
5610 let mut s = String::new();
5611 for r in runs {
5612 match r {
5613 Inline::Text(t) | Inline::Strong(t) | Inline::Emph(t)
5614 | Inline::BoldItalic(t) | Inline::Super(t) | Inline::Sub(t) | Inline::Code(t) => s.push_str(t),
5615 Inline::Glossary { display, .. } => s.push_str(display),
5616 _ => {},
5617 }
5618 }
5619 s
5620 }
5621
5622 /// The one [`Item::Figure`] in a parse, failing loudly with the item list when there is not exactly one.
5623 fn one_figure(items: &[Item]) -> Outcome<(Option<Vec<Inline>>, String, Option<String>)> {
5624 let figs: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::Figure { .. })).collect();
5625 match figs.as_slice() {
5626 [Item::Figure { caption, supplement, label, .. }] =>
5627 Ok((caption.clone(), supplement.clone(), label.clone())),
5628 _ => Err(err!("expected exactly one figure, got: {:?}", items; Test, Bug)),
5629 }
5630 }
5631
5632 /// A `#figure(...)` whose caption prose carries an author's unbalanced `(` (Oxegen TechSpec
5633 /// `app_maths.typ:550-566`: "...network messages (latency $100 unit(\"ms\")$ ... chunk sizes $d$.")
5634 /// must still close at its own `)`: content mode treats the stray paren as literal prose, where the old
5635 /// flat depth counter stuck open to end of source and swallowed both the figure and the tail after it.
5636 #[test]
5637 fn figure_caption_with_unbalanced_paren_closes_and_tail_survives() -> Outcome<()> {
5638 let src = "\
5639#figure(
5640 block(width: 70%)[
5641 #table(
5642 columns: (10fr, 10fr),
5643 [a], [b],
5644 )
5645 ],
5646 caption: [Representative processing times for hashing and network messages (latency $100 unit(\"ms\")$ per message for a Merkle tree of $1 unit(\"GiB\")$ of datastate with various chunk sizes $d$.],
5647 kind: \"table\",
5648 supplement: \"Table\",
5649) <merkle_tree_chunk_size>
5650
5651Trailing prose.
5652";
5653 let (items, _skips) = res!(document_with_refusals(src));
5654 let (caption, supplement, label) = res!(one_figure(&items));
5655 let caption = res!(caption.ok_or_else(|| err!("the figure lost its caption"; Test, Bug)));
5656 assert!(plain(&caption).contains("Representative processing times"),
5657 "the caption prose was lost: {:?}", caption);
5658 assert_eq!(supplement, "Table", "the supplement was not read from the figure");
5659 assert_eq!(label, Some("merkle_tree_chunk_size".to_string()), "the figure label was lost");
5660 // The tail after the figure must survive rather than be swallowed by a stuck bracket count.
5661 let tail = items.iter().any(|it| matches!(it,
5662 Item::Paragraph { runs, .. } if plain(runs).contains("Trailing prose")));
5663 assert!(tail, "the prose after the figure was swallowed: {:?}", items);
5664 Ok(())
5665 }
5666
5667 /// A `(`, a `)`, a `[` or a `]` inside a `$...$` maths span in a caption is literal maths, never a
5668 /// structural bracket, so an interval or a parenthesised function does not miscount and close the
5669 /// caption early. Without the maths frame, the `]` in `$[a, b]$` would pop the caption content block.
5670 #[test]
5671 fn caption_maths_parens_do_not_count() -> Outcome<()> {
5672 let src = "\
5673#figure(
5674 image(\"fig.png\"),
5675 caption: [see $f(x)$ over $[a, b]$ where it holds.],
5676) <fig_maths>
5677
5678After the figure.
5679";
5680 let (items, _skips) = res!(document_with_refusals(src));
5681 let (caption, _supplement, label) = res!(one_figure(&items));
5682 let caption = res!(caption.ok_or_else(|| err!("the maths caption was lost"; Test, Bug)));
5683 assert!(plain(&caption).contains("see") && plain(&caption).contains("where it holds"),
5684 "the caption prose around the maths was truncated: {:?}", caption);
5685 assert!(caption.iter().any(|r| matches!(r, Inline::Math(_))),
5686 "the caption maths span was not parsed as maths: {:?}", caption);
5687 assert_eq!(label, Some("fig_maths".to_string()), "the label after a maths caption was lost");
5688 assert!(items.iter().any(|it| matches!(it,
5689 Item::Paragraph { runs, .. } if plain(runs).contains("After the figure"))),
5690 "the tail after a maths caption was swallowed: {:?}", items);
5691 Ok(())
5692 }
5693
5694 /// A `(` or `)` inside a `"..."` string argument -- here an image path `image("a(b).png")` -- is literal
5695 /// string content, not a structural paren, so a single-line figure carrying such a path closes on its
5696 /// line rather than opening a run-away capture.
5697 #[test]
5698 fn code_string_paren_does_not_count() -> Outcome<()> {
5699 let src = "#figure(image(\"a(b).png\"), caption: [c])\n\nNext paragraph.\n";
5700 let (items, _skips) = res!(document_with_refusals(src));
5701 let (caption, _supplement, _label) = res!(one_figure(&items));
5702 let caption = res!(caption.ok_or_else(|| err!("the figure lost its caption"; Test, Bug)));
5703 assert_eq!(plain(&caption), "c", "the caption was misread past the string paren: {:?}", caption);
5704 let path = items.iter().find_map(|it| match it {
5705 Item::Figure { body: FigureBody::Image { path, .. }, .. } => Some(path.clone()),
5706 _ => None,
5707 });
5708 assert_eq!(path, Some("a(b).png".to_string()), "the image path with parens was misread: {:?}", path);
5709 assert!(items.iter().any(|it| matches!(it,
5710 Item::Paragraph { runs, .. } if plain(runs).contains("Next paragraph"))),
5711 "the paragraph after a single-line figure was swallowed: {:?}", items);
5712 Ok(())
5713 }
5714
5715 /// A `#name(...)` call met inside a caption's content re-enters code mode, so the string inside it is
5716 /// protected and a following unbalanced `(` in the surrounding prose is still literal: the caption
5717 /// closes at its own `]` and the tail survives.
5718 #[test]
5719 fn content_call_inside_caption_reenters_code() -> Outcome<()> {
5720 let src = "\
5721#figure(
5722 image(\"g.png\"),
5723 caption: [see #link(\"http://x\")[y] and note (z],
5724) <fig_call>
5725
5726Following text.
5727";
5728 let (items, _skips) = res!(document_with_refusals(src));
5729 let (caption, _supplement, label) = res!(one_figure(&items));
5730 let caption = res!(caption.ok_or_else(|| err!("the call caption was lost"; Test, Bug)));
5731 assert!(plain(&caption).contains("see") && plain(&caption).contains("note (z"),
5732 "the caption prose around the call was truncated: {:?}", caption);
5733 assert_eq!(label, Some("fig_call".to_string()), "the label after a call caption was lost");
5734 assert!(items.iter().any(|it| matches!(it,
5735 Item::Paragraph { runs, .. } if plain(runs).contains("Following text"))),
5736 "the tail after a call caption was swallowed: {:?}", items);
5737 Ok(())
5738 }
5739
5740 /// [`read_group`] and [`split_top_args`] on a caption argument list directly: an unbalanced `(` in the
5741 /// caption prose is literal, so the group closes at its own `]` and the top-level commas still part the
5742 /// arguments, with the maths span and its string protected throughout.
5743 #[test]
5744 fn read_group_and_split_handle_caption_prose_paren() -> Outcome<()> {
5745 // read_group on a caption bracket whose prose carries an unbalanced `(` and a `$...$` span with a
5746 // quoted `unit("ms")` inside: the group must end at the caption's own `]`, keeping its prose and
5747 // stopping before the trailing text.
5748 let s: Vec<char> = "[messages (latency $100 unit(\"ms\")$ per $d$.] more".chars().collect();
5749 let (inner, next) = res!(read_group(&s, 0)
5750 .ok_or_else(|| err!("the caption group did not close"; Test, Bug)));
5751 assert!(inner.contains("(latency"), "the caption prose was lost: {:?}", inner);
5752 assert!(!inner.contains("more"), "the caption group over-ran its closing bracket: {:?}", inner);
5753 assert_eq!(s[next..].iter().collect::<String>(), " more", "the index past the closer is wrong");
5754
5755 // split_top_args across the same shape: three arguments, the middle a caption whose unbalanced paren
5756 // and maths span do not part it, and named_arg reads the caption key back off it.
5757 let args = split_top_args("image(\"p.png\"), caption: [x (y $z(w)$.], kind: \"table\"");
5758 assert_eq!(args.len(), 3, "the unbalanced caption paren split the arg list wrong: {:?}", args);
5759 let named = res!(named_arg(args[1].trim())
5760 .ok_or_else(|| err!("the caption argument did not read as named: {:?}", args[1]; Test, Bug)));
5761 assert_eq!(named.0, "caption", "the caption key was misread: {:?}", named);
5762 assert!(named.1.contains("(y"), "the caption value lost its unbalanced paren: {:?}", named);
5763 Ok(())
5764 }
5765
5766 /// A prose line that opens with an inline `#raw("...")` and carries a stray `"` from an author's
5767 /// quotation split across the line break must be set as prose, not read as the start of a multi-line
5768 /// code skip: a dangling quote is a character, not an open code string. Only an unclosed structural
5769 /// bracket keeps a skip open (Hematite `sec_net.typ:679-681`, where `express "no\nbound".` split a
5770 /// quotation over two lines after an inline `#raw`).
5771 #[test]
5772 fn prose_line_with_inline_raw_and_dangling_quote_is_not_skipped() -> Outcome<()> {
5773 let src = "passes its own pair to #raw(\"read_message\"), or to\n\
5774#raw(\"WebSocket::with_limits\"). There is deliberately no way to express \"no\n\
5775bound\".\n";
5776 let (items, _skips) = res!(document_with_refusals(src));
5777 let text: String = items.iter()
5778 .filter_map(|it| match it {
5779 Item::Paragraph { runs, .. } => Some(plain(runs)),
5780 _ => None,
5781 })
5782 .collect::<Vec<_>>()
5783 .join(" ");
5784 assert!(text.contains("no way to express"),
5785 "the prose after an inline #raw was swallowed as a skip: {:?}", items);
5786 assert!(text.contains("bound"), "the continuation line was swallowed: {:?}", items);
5787 Ok(())
5788 }
5789
5790 /// A `` `//` `` and `` `/* */` `` inside a backtick code span, in a prose caption naming the operators
5791 /// themselves, must not be read as comment openers: [`Frame::Raw`] keeps the span literal, so the group
5792 /// still closes at its own `]` and the raw span survives verbatim in the inner text.
5793 #[test]
5794 fn read_group_protects_backtick_span_with_comment_markers() -> Outcome<()> {
5795 let s: Vec<char> = "[the `//` and `/* */` operators]".chars().collect();
5796 let (inner, next) = res!(read_group(&s, 0)
5797 .ok_or_else(|| err!("the backtick span made the group mis-scan as a comment"; Test, Bug)));
5798 assert_eq!(inner, "the `//` and `/* */` operators",
5799 "the raw span was not kept literal: {:?}", inner);
5800 assert_eq!(next, s.len(), "the group did not close at its own bracket");
5801 Ok(())
5802 }
5803
5804 /// A `//` on one line of a multi-line capture buffer must be bounded to that line's own end, not eat
5805 /// every line after it: the buffer's real closing bracket, on the following line, must still be found.
5806 #[test]
5807 fn read_group_line_comment_does_not_eat_the_next_line() -> Outcome<()> {
5808 let s: Vec<char> = "[first line // trailing\nsecond line]".chars().collect();
5809 let (inner, next) = res!(read_group(&s, 0)
5810 .ok_or_else(|| err!("the // comment ate past its own line and swallowed the closer"; Test, Bug)));
5811 assert_eq!(inner, "first line // trailing\nsecond line",
5812 "the buffer content around the line comment was misread: {:?}", inner);
5813 assert_eq!(next, s.len(), "the group did not close at its own bracket");
5814 Ok(())
5815 }
5816}