oxedyne/fe2o3/fe2o3_austenite/src/lang/parse.rs
281 KiB, 1279 runs
created by r1870400018:36173, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! The Typst reader: line-oriented source into the surface tree of [`ast::Item`](super::ast::Item). |
| 2 | //! |
| 3 | //! The scan is deliberately simple, one pass over the lines. A line whose first non-blank character |
| 4 | //! is `=` is a heading, its level the run of leading `=`; a line opening with `-` or `+` (and a space) |
| 5 | //! is a list item, a run of them a bullet or numbered list; every other non-blank line accumulates into |
| 6 | //! a paragraph. A blank line, a heading, or the start of the other kind of block closes whatever is |
| 7 | //! open. Whitespace within a paragraph is made insignificant here, so the line breaker downstream owns |
| 8 | //! the measure. Byte offsets are tracked across the raw lines so each [`Item`] carries a true [`Span`]. |
| 9 | //! |
| 10 | //! A closed paragraph's text is then scanned for inline emphasis by [`parse_inlines`]: `*strong*` and |
| 11 | //! `_emph_`. A delimiter pairs only when it flanks a word, so a stray asterisk, a date's slash, or |
| 12 | //! `and/or` is left as ordinary text rather than opening an emphasis that never closes. |
| 13 | //! |
| 14 | //! A Typst code statement (`#import`/`#let`/`#set`/`#show`) or a line-leading standalone template call |
| 15 | //! (`#name(...)` or `#name[...]`) is not set: it is skipped. When its delimiters do not balance on the |
| 16 | //! opening line -- a `#figure(...)`, `#table(...)`, `#aside-box[...]`, or a `#let x = (...)` data array |
| 17 | //! that spans many lines -- the reader consumes following lines, tracking nesting across `()`, `[]` and |
| 18 | //! `{}` and respecting string literals, until the delimiters balance, so the whole span renders nothing. |
| 19 | //! Every such skip is recorded by name into a [`Refusals`] the parse returns beside its items, so a |
| 20 | //! caller reports the constructs it dropped rather than losing them silently. A `#columns(n)[ ... ]` |
| 21 | //! wrapper is the exception the reader does not drop whole: its body is re-parsed and set single-column. |
| 22 | //! |
| 23 | //! Typst comments are stripped before a line is classified: a `//` runs to the line's end, and a |
| 24 | //! `/* ... */` spans lines, both dropped -- except within a `"..."` string or a `` `code` `` span, and a |
| 25 | //! `//` right after `:` is kept, so a bare URL survives. Inline glossary and index calls, defined in the |
| 26 | //! book template (`#gs`, `#gscap`, `#gsi`, `#gscapi`, `#glossind`, `#glossindcap`, the term-dictionary |
| 27 | //! family `#g`, `#gcap`, `#gi`, `#gcapi`, `#t`, `#tcap`, `#graw`, and `#idx`, `#idx-main`, `#idx-as`, |
| 28 | //! `#idx-main-as`, `#index`, `#index-main`, `#idx-nested`), plus a `#link(dest)[text]` hyperlink, are read |
| 29 | //! by [`parse_inlines`]: a glossary term sets its display text, bold-italic on its first document use; a |
| 30 | //! visible term or index call sets its display text plain; a link sets its text and drops the destination; |
| 31 | //! a pure index marker sets nothing. An inline `#func[...]` the reader does not know is consumed, recorded |
| 32 | //! in the summary, and its bracketed body folded in, so its words survive but its raw markup never leaks. |
| 33 | |
| 34 | use crate::ir::FloatPlacement; |
| 35 | use crate::ir::FloatScope; |
| 36 | use crate::ir::Floating; |
| 37 | use crate::ir::Length; |
| 38 | use crate::ir::Span; |
| 39 | use crate::table::Align; |
| 40 | |
| 41 | use super::ast::{AlignSpec, ClosureAlign, FigureBody, Inline, Item, ListItem, Spacing, TableSpec}; |
| 42 | use super::mathparse; |
| 43 | |
| 44 | use oxedyne_fe2o3_core::prelude::*; |
| 45 | |
| 46 | use std::collections::BTreeMap; |
| 47 | use std::collections::HashMap; |
| 48 | use std::sync::RwLock; |
| 49 | |
| 50 | // The book's `term-dict`, set once by the loader before parsing so the term-dictionary glossary family |
| 51 | // (`t`, `tcap`, `graw`, `g`, `gi`, `gcap`, `gcapi`) resolves a key to its display value while the key |
| 52 | // identity is still known. A process-global rather than a threaded argument, mirroring the image base: |
| 53 | // the inline reader sets one run at a time and carries no book context of its own. `None` until the |
| 54 | // loader installs one, in which case every key falls back to its own text. |
| 55 | static TERM_DICT: RwLock<Option<HashMap<String, String>>> = RwLock::new(None); |
| 56 | |
| 57 | /// Records the book's `term-dict`, read from a sibling `terms.typ`, so the term-dictionary glossary |
| 58 | /// family resolves each key to its value at parse time. Installing a fresh map replaces any prior one. |
| 59 | pub fn set_term_dict(dict: HashMap<String, String>) -> Outcome<()> { |
| 60 | let mut guard = lock_write!(TERM_DICT, "While recording the term dictionary"); |
| 61 | *guard = Some(dict); |
| 62 | Ok(()) |
| 63 | } |
| 64 | |
| 65 | /// The display value a `term-dict` key resolves to, or `None` when no map is installed or it holds no |
| 66 | /// such key. A poisoned lock reads as absent rather than failing the parse: a missing value falls back |
| 67 | /// to the key text, which is exactly the safe degradation here. |
| 68 | pub(crate) fn term_value(key: &str) -> Option<String> { |
| 69 | match TERM_DICT.read() { |
| 70 | Ok(guard) => guard.as_ref().and_then(|m| m.get(key).cloned()), |
| 71 | Err(_) => None, |
| 72 | } |
| 73 | } |
| 74 | |
| 75 | /// Why a construct was refused rather than set: the axis a per-site diagnostic reports alongside its |
| 76 | /// name and location, so a reader can tell a categorical limit from a todo. |
| 77 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] |
| 78 | pub enum RefusalClass { |
| 79 | // Typst's own general evaluation primitives -- `#import`, `#let`, `#set`, `#show` -- which run an |
| 80 | // arbitrary expression to a value. Austenite's driver converges a fixed, two-pass ledger to a point; |
| 81 | // it does not carry a Typst-style evaluator, so these are a categorical limit of the architecture |
| 82 | // (see the crate root's own note that Typst treats a document as a program and Austenite does not), |
| 83 | // not a specific feature waiting to be added. |
| 84 | FixedPoint, |
| 85 | // A construct that reads the page's own laid-out state back into the document -- `#context`, |
| 86 | // `#query`, `#locate`, a counter's or state's `.at`/`.get`, `#measure`, `#layout`. This is exactly |
| 87 | // Typst's self-observation model, which the ledger exists to answer in Austenite's own terms; a |
| 88 | // construct landing here is a real gap the ledger could plausibly close, not a limit of the design. |
| 89 | Introspective, |
| 90 | // Anything else skipped: a specific call or wrapper (`#columns`, an unknown standalone or inline |
| 91 | // `#func`, an unknown term-dictionary key) that names no fundamental barrier -- just not yet read. |
| 92 | Unsupported, |
| 93 | } |
| 94 | |
| 95 | impl RefusalClass { |
| 96 | /// Classifies a refusal by the source name it was recorded under. Matched by substring rather than |
| 97 | /// an exact keyword, since the same construct is recorded under different shapes depending on how it |
| 98 | /// was written (`#context`, a bare `context` inside a longer call name) -- this is a diagnostic |
| 99 | /// classifier, not a parser, so a generous match that occasionally over-reaches is the right trade. |
| 100 | fn classify(name: &str) -> Self { |
| 101 | let lower = name.to_lowercase(); |
| 102 | for kw in ["#import", "#let", "#set", "#show"] { |
| 103 | if lower.starts_with(kw) { |
| 104 | return RefusalClass::FixedPoint; |
| 105 | } |
| 106 | } |
| 107 | const INTROSPECTIVE: [&str; 8] = |
| 108 | ["context", "query", "locate", "counter.at", "counter.get", "state", "measure", "layout"]; |
| 109 | if INTROSPECTIVE.iter().any(|kw| lower.contains(kw)) { |
| 110 | return RefusalClass::Introspective; |
| 111 | } |
| 112 | RefusalClass::Unsupported |
| 113 | } |
| 114 | |
| 115 | /// The word `--explain` prints for this class. |
| 116 | pub fn label(&self) -> &'static str { |
| 117 | match self { |
| 118 | RefusalClass::FixedPoint => "fixed-point", |
| 119 | RefusalClass::Introspective => "introspective", |
| 120 | RefusalClass::Unsupported => "unsupported", |
| 121 | } |
| 122 | } |
| 123 | } |
| 124 | |
| 125 | /// One site the reader passed over rather than set: the source name it was written with (carrying its |
| 126 | /// leading `#`, so it reads back as source), the byte span it was found at, and why it was refused. |
| 127 | /// The span is the whole containing line for a code statement or standalone call, or the whole |
| 128 | /// containing item (a paragraph, a heading) for an inline call found within one -- Austenite's inline |
| 129 | /// scanner does not keep the fine per-character offset once a paragraph's lines have been joined and its |
| 130 | /// whitespace collapsed, so the enclosing item is the finest boundary available without a deeper rework |
| 131 | /// of the reader than this diagnostic upgrade is for. |
| 132 | #[derive(Clone, Debug)] |
| 133 | pub struct Refusal { |
| 134 | pub name: String, |
| 135 | pub span: Span, |
| 136 | pub class: RefusalClass, |
| 137 | // The source file this site was read from, for `--explain`'s "file:line:col". Empty immediately |
| 138 | // after parsing, since a lone parse of a source string carries no filename of its own; the book |
| 139 | // assembler ([`crate::book::assemble`]) tags each chapter's (and the root's own) refusals with the |
| 140 | // real path once assembly is back in a context that has one -- see [`Refusals::tag_file`]. |
| 141 | pub file: String, |
| 142 | } |
| 143 | |
| 144 | /// Every site the reader refused across one parse (or, once [`Refusals::merge`] has folded chapters |
| 145 | /// together, across a whole book). Kept as a flat list of [`Refusal`]s rather than the old name-keyed |
| 146 | /// tally, so a caller can still print the terse one-line [`Refusals::report`] but can also walk every |
| 147 | /// site for `--explain`'s per-site listing. Empty when the reader set everything it met. |
| 148 | #[derive(Clone, Debug, Default)] |
| 149 | pub struct Refusals { |
| 150 | sites: Vec<Refusal>, |
| 151 | } |
| 152 | |
| 153 | impl Refusals { |
| 154 | /// Records one refused construct by the source name it was written with (with its leading `#`) and |
| 155 | /// the span it was found at, classifying it from the name. |
| 156 | pub(crate) fn record(&mut self, name: &str, span: Span) { |
| 157 | let class = RefusalClass::classify(name); |
| 158 | self.sites.push(Refusal { name: name.to_string(), span, class, file: String::new() }); |
| 159 | } |
| 160 | |
| 161 | /// Builds a table directly from a caller's own sites, for a test (or another future caller outside |
| 162 | /// the parser) that wants a known `Refusals` without driving a real parse to produce one. |
| 163 | pub fn from_sites(sites: Vec<Refusal>) -> Self { |
| 164 | Self { sites } |
| 165 | } |
| 166 | |
| 167 | pub fn is_empty(&self) -> bool { self.sites.is_empty() } |
| 168 | |
| 169 | /// Sets every site's `file` that is not already set, so a caller assembling several chapters can tag |
| 170 | /// each chapter's refusals with its own path right after parsing it, before folding them into the |
| 171 | /// book's running total with [`merge`](Self::merge) -- at which point every site already carries the |
| 172 | /// file it came from, and a second tagging (the root's own trailing markup, read after every |
| 173 | /// include) touches only the sites still unset. |
| 174 | pub fn tag_file(&mut self, file: &str) { |
| 175 | for r in &mut self.sites { |
| 176 | if r.file.is_empty() { |
| 177 | r.file = file.to_string(); |
| 178 | } |
| 179 | } |
| 180 | } |
| 181 | |
| 182 | /// Every refused site, in the order the reader met them. |
| 183 | pub fn sites(&self) -> &[Refusal] { &self.sites } |
| 184 | |
| 185 | /// The number of distinct construct names refused. |
| 186 | pub fn kinds(&self) -> usize { self.entries().len() } |
| 187 | |
| 188 | /// The total count of refused sites across every name. |
| 189 | pub fn total(&self) -> usize { self.sites.len() } |
| 190 | |
| 191 | /// Each refused construct name with its count, ordered by descending count then name, so the report |
| 192 | /// leads with the construct that cost the most. |
| 193 | pub fn entries(&self) -> Vec<(String, usize)> { |
| 194 | let mut counts: BTreeMap<&str, usize> = BTreeMap::new(); |
| 195 | for r in &self.sites { |
| 196 | *counts.entry(r.name.as_str()).or_insert(0) += 1; |
| 197 | } |
| 198 | let mut v: Vec<(String, usize)> = counts.into_iter().map(|(k, c)| (k.to_string(), c)).collect(); |
| 199 | v.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0))); |
| 200 | v |
| 201 | } |
| 202 | |
| 203 | /// Folds another parse's refusals into this one, so a caller assembling several chapters reports one |
| 204 | /// total rather than a summary per file. `other` is consumed rather than borrowed: a caller merging a |
| 205 | /// just-parsed chapter's refusals into the book's running total has no further use for its own copy. |
| 206 | pub fn merge(&mut self, other: Refusals) { |
| 207 | self.sites.extend(other.sites); |
| 208 | } |
| 209 | |
| 210 | /// A one-line report -- "skipped 3 unsupported constructs: #show (2), #columns (1)" -- or `None` when |
| 211 | /// nothing was skipped, so a caller prints the line only when it has something to say. Unchanged in |
| 212 | /// wording from before this unit: `--explain` is the new, detailed report, this terse one stays the |
| 213 | /// default. |
| 214 | pub fn report(&self) -> Option<String> { |
| 215 | if self.sites.is_empty() { |
| 216 | return None; |
| 217 | } |
| 218 | let parts: Vec<String> = self.entries().into_iter() |
| 219 | .map(|(n, c)| fmt!("{} ({})", n, c)) |
| 220 | .collect(); |
| 221 | let n = self.total(); |
| 222 | Some(fmt!("skipped {} unsupported construct{}: {}", |
| 223 | n, if n == 1 { "" } else { "s" }, parts.join(", "))) |
| 224 | } |
| 225 | } |
| 226 | |
| 227 | /// Parses a whole Ingot source string into its surface items. The only error is an empty heading -- |
| 228 | /// a `=` marker with no title -- which names the offending 1-based line. |
| 229 | pub fn document(src: &str) -> Outcome<Vec<Item>> { |
| 230 | let (items, _) = res!(document_with_refusals(src)); |
| 231 | Ok(items) |
| 232 | } |
| 233 | |
| 234 | /// Parses a source string into its surface items and, alongside, the [`Refusals`] of every construct |
| 235 | /// the reader passed over rather than set -- a `#let`/`#set`/`#show`/`#import` code line, an unknown |
| 236 | /// line-leading `#func(...)` call, a `#columns` wrapper, and any unhandled inline `#func[...]`. The |
| 237 | /// caller prints the summary so a dropped construct is reported rather than lost silently. |
| 238 | pub fn document_with_refusals(src: &str) -> Outcome<(Vec<Item>, Refusals)> { |
| 239 | let tfns = crate::lang::rules::TemplateFns::new(); |
| 240 | let cfns = crate::lang::rules::ContentFns::new(); |
| 241 | document_with_templates(src, crate::lang::rules::Bindings::new(&tfns, &cfns)) |
| 242 | } |
| 243 | |
| 244 | /// Records a refusal for every claim reference (`#claim-refs`/`#claim-label`) that sits in a context the |
| 245 | /// layout does not gather into the reverse claim index. A top-level body run -- a paragraph, a list entry, a |
| 246 | /// callout body -- and a table cell both feed the index (see `doc::build_pieces`, reached for a cell through |
| 247 | /// `doc::build_grid`); a claim code in a heading title, a figure or table caption, or a footnote body is |
| 248 | /// dropped by the layout, so it would otherwise vanish from the index (and, for a `#claim-label`, from the |
| 249 | /// margin) with no trace. Making that a refusal keeps the silent-loss class this project guards against out |
| 250 | /// of the reverse index. Gathering from a heading or caption needs anchor support there and is a later |
| 251 | /// increment; until then the code is reported, not dropped. |
| 252 | fn flag_unindexed_claim_refs(items: &[Item], skips: &mut Refusals) { |
| 253 | for item in items { |
| 254 | match item { |
| 255 | // A body run and a list entry are gathered; only a claim reference nested inside a footnote of one |
| 256 | // escapes the index, so the top-level runs are scanned as indexed and their footnotes are not. |
| 257 | Item::Paragraph { runs, span, .. } => scan_claim_refs(runs, true, *span, "a paragraph", skips), |
| 258 | Item::List { items: entries, .. } => for e in entries { flag_list_item_claim_refs(e, skips); }, |
| 259 | // A callout body is gathered like the main flow; recurse so a claim reference in it is indexed and |
| 260 | // only its non-body sub-contexts (a caption, a footnote) are flagged. |
| 261 | Item::Box { items: inner, .. } => flag_unindexed_claim_refs(inner, skips), |
| 262 | Item::Scoped { items: inner, .. } => flag_unindexed_claim_refs(inner, skips), |
| 263 | // A heading title and a caption are still not gathered, so a claim reference in either is refused. |
| 264 | // A table cell now runs through the body's own segment pipeline (`doc::build_grid` -> |
| 265 | // `doc::build_pieces`), which weaves the cell's `#claim-refs`/`#claim-label` anchor into the reverse |
| 266 | // claim index exactly as a body run does, so it is no longer refused. |
| 267 | Item::Heading { runs, span, .. } => scan_claim_refs(runs, false, *span, "a heading title", skips), |
| 268 | Item::Figure { caption, span, .. } => { |
| 269 | if let Some(cap) = caption { |
| 270 | scan_claim_refs(cap, false, *span, "a figure caption", skips); |
| 271 | } |
| 272 | }, |
| 273 | _ => {}, |
| 274 | } |
| 275 | } |
| 276 | } |
| 277 | |
| 278 | /// [`flag_unindexed_claim_refs`] for one list entry: its own runs are gathered (indexed), and its nested |
| 279 | /// child items are walked as their own contexts. |
| 280 | fn flag_list_item_claim_refs(entry: &ListItem, skips: &mut Refusals) { |
| 281 | scan_claim_refs(&entry.runs, true, Span::new(0, 0), "a list entry", skips); |
| 282 | flag_unindexed_claim_refs(&entry.children, skips); |
| 283 | } |
| 284 | |
| 285 | /// Scans an inline run for claim references and records a refusal for each that will not reach the reverse |
| 286 | /// index. `indexed` is true for a top-level body run (a paragraph, list entry or callout body), where a |
| 287 | /// claim reference IS gathered and so is left alone; a footnote body is never gathered, so its own runs are |
| 288 | /// always scanned as unindexed regardless of where the footnote sits. |
| 289 | fn scan_claim_refs(runs: &[Inline], indexed: bool, span: Span, context: &str, skips: &mut Refusals) { |
| 290 | for run in runs { |
| 291 | match run { |
| 292 | Inline::MarginNote { codes, .. } if !indexed && !codes.is_empty() => |
| 293 | skips.record(&fmt!("claim reference in {} is not indexed", context), span), |
| 294 | Inline::Footnote(inner) => scan_claim_refs(inner, false, span, "a footnote body", skips), |
| 295 | _ => {}, |
| 296 | } |
| 297 | } |
| 298 | } |
| 299 | |
| 300 | /// As [`document_with_refusals`], with the `#let` bindings (`binds`) in scope: a call to a furniture |
| 301 | /// function -- `#pr-note[ ... ]`, `#aside-box(title: [..])[ ... ]` -- expands into a padded box, and a |
| 302 | /// reference to a content binding -- `#greet("world")`, a bare `#intro` -- expands into its re-read markup, |
| 303 | /// rather than either being tallied as a skipped construct. A body re-parsed here carries the same `binds`, |
| 304 | /// so a call nested inside another's body expands too. With empty maps this is exactly |
| 305 | /// [`document_with_refusals`]. |
| 306 | /// |
| 307 | /// After the surface tree is built, [`flag_unindexed_claim_refs`] records a refusal for any claim reference |
| 308 | /// that landed in a context the layout does not gather into the reverse claim index. This runs once, on the |
| 309 | /// whole assembled tree -- the recursive re-parse of a `#columns`/`#styled-box` body reaches for |
| 310 | /// [`parse_items`] directly, so a nested claim reference is flagged once here rather than again per level. |
| 311 | pub fn document_with_templates(src: &str, binds: crate::lang::rules::Bindings<'_, '_>) |
| 312 | -> Outcome<(Vec<Item>, Refusals)> |
| 313 | { |
| 314 | let (items, mut skips) = res!(parse_items(src, binds)); |
| 315 | flag_unindexed_claim_refs(&items, &mut skips); |
| 316 | Ok((items, skips)) |
| 317 | } |
| 318 | |
| 319 | /// The surface-tree parse proper, without the [`flag_unindexed_claim_refs`] post-pass -- so a recursively |
| 320 | /// re-parsed body (a `#columns`/`#styled-box` wrapper's content) is not validated twice, once here and again |
| 321 | /// when its parent walks the spliced items. [`document_with_templates`] wraps this with that one validation. |
| 322 | fn parse_items(src: &str, binds: crate::lang::rules::Bindings<'_, '_>) |
| 323 | -> Outcome<(Vec<Item>, Refusals)> |
| 324 | { |
| 325 | let mut skips: Refusals = Refusals::default(); |
| 326 | let mut items: Vec<Item> = Vec::new(); |
| 327 | let mut lines: Vec<String> = Vec::new(); // the current paragraph's constituent lines |
| 328 | let mut para_start: u32 = 0; // byte offset of the paragraph's first line |
| 329 | let mut para_end: u32 = 0; // byte offset just past its last line's content |
| 330 | let mut offset: u32 = 0; // running byte offset of the current line's start |
| 331 | let mut line_no = 0usize; // 1-based, for a diagnostic |
| 332 | |
| 333 | // The stack of open list levels, innermost last. Each level records the leading-space indent of its |
| 334 | // markers, so a deeper marker opens a sub-list under the current item and a shallower one closes back |
| 335 | // to the matching level; an empty stack means no list is open. A list is a run of marker lines that a |
| 336 | // blank line does not break (Typst continues an enum across a gap), but any other content flushes. |
| 337 | let mut stack: Vec<ListFrame> = Vec::new(); |
| 338 | |
| 339 | // A fenced code block, while one is open: the verbatim lines gathered so far and the byte offset it |
| 340 | // began at. A ```-fence opens it, the next ```-fence closes it; between them every line is kept as it |
| 341 | // stands, its indentation and markup untouched. |
| 342 | let mut code: Option<(Vec<String>, u32)> = None; |
| 343 | |
| 344 | // A multi-line Typst code statement or standalone template call being skipped: the net bracket depth |
| 345 | // still open across the lines consumed so far, and whether a string literal is currently open. `None` |
| 346 | // when not skipping. While it is `Some`, every line is consumed and nothing is set until the delimiters |
| 347 | // balance. |
| 348 | let mut skip: Option<SkipState> = None; |
| 349 | |
| 350 | // A multi-line construct whose whole text is gathered so it can be parsed rather than skipped: a |
| 351 | // `#figure(...)`, a bare `#table(...)`, or a `#let name = (...)` data array feeding a table. `None` |
| 352 | // when none is open. The accumulated text is dispatched by its kind when the delimiters balance. |
| 353 | let mut capture: Option<Capture> = None; |
| 354 | |
| 355 | // Data arrays declared by `#let name = (...)` and referenced by a table's `..name.flatten()` spread: |
| 356 | // the name maps to the flat sequence of cells the array holds, each cell a run of inline markup. |
| 357 | // Populated as the arrays are read, so a later figure resolves its cells against them. |
| 358 | let mut arrays: HashMap<String, Vec<Vec<Inline>>> = HashMap::new(); |
| 359 | |
| 360 | // Whether a `/* ... */` block comment is open across the line break. A `//` line comment never |
| 361 | // straddles a line, so it needs no carried state. |
| 362 | let mut comment = CommentState { in_block: false }; |
| 363 | |
| 364 | // `split_inclusive` keeps the trailing newline on each piece, so the running offset stays a true |
| 365 | // byte position into the source rather than drifting by the count of stripped terminators. |
| 366 | for raw in src.split_inclusive('\n') { |
| 367 | line_no += 1; |
| 368 | let start = offset; |
| 369 | offset = offset.saturating_add(raw.len() as u32); |
| 370 | |
| 371 | // Strip the line terminator without consuming a real character: the final line may carry |
| 372 | // neither a newline nor a carriage return. |
| 373 | let mut line = raw; |
| 374 | if let Some(s) = line.strip_suffix('\n') { line = s; } |
| 375 | if let Some(s) = line.strip_suffix('\r') { line = s; } |
| 376 | let end = start.saturating_add(line.len() as u32); |
| 377 | |
| 378 | // Strip Typst comments before classifying the line, but not while a fenced code block or a |
| 379 | // multi-line call skip is open: inside a fence a `//` is verbatim, and a skipped span is dropped |
| 380 | // whole regardless. The span above is computed from the raw line, so a diagnostic caret still |
| 381 | // points into the source. |
| 382 | let stripped; |
| 383 | let line = if code.is_none() && skip.is_none() { |
| 384 | stripped = strip_comments(line, &mut comment); |
| 385 | stripped.as_str() |
| 386 | } else { |
| 387 | line |
| 388 | }; |
| 389 | |
| 390 | let trimmed = line.trim_start(); |
| 391 | |
| 392 | // A multi-line code statement or standalone call is being skipped: keep consuming lines, tracking |
| 393 | // bracket nesting across `()`, `[]` and `{}` and respecting string literals, until the delimiters |
| 394 | // balance. Nothing between the opener and its close is set. This takes precedence over every other |
| 395 | // rule, since the span is code, not markup. |
| 396 | if let Some(state) = skip.as_mut() { |
| 397 | scan_brackets(line, state); |
| 398 | if !state.has_open_bracket() { |
| 399 | skip = None; |
| 400 | } |
| 401 | continue; |
| 402 | } |
| 403 | |
| 404 | // A multi-line construct is being gathered whole: keep appending its lines and tracking the bracket |
| 405 | // balance until the delimiters close, then dispatch the accumulated text by its kind. Like the skip |
| 406 | // above, this takes precedence over the markup rules, since the span is a code construct. |
| 407 | if let Some(cap) = capture.as_mut() { |
| 408 | cap.buf.push_str(line); |
| 409 | cap.buf.push('\n'); |
| 410 | scan_brackets(line, &mut cap.state); |
| 411 | if !cap.state.has_open_bracket() { |
| 412 | let done = capture.take(); |
| 413 | if let Some(cap) = done { |
| 414 | dispatch_capture(cap, &mut items, &mut arrays, &mut skips, binds); |
| 415 | } |
| 416 | } |
| 417 | continue; |
| 418 | } |
| 419 | |
| 420 | // A fenced code block takes precedence over every other rule: inside it, only a closing fence is |
| 421 | // special and every other line is verbatim, so its own `=`, `-` or `*` carry no markup meaning. |
| 422 | if let Some((buf, cstart)) = code.as_mut() { |
| 423 | if is_fence(trimmed) { |
| 424 | items.push(Item::Code { lines: std::mem::take(buf), span: Span::new(*cstart, end) }); |
| 425 | code = None; |
| 426 | } else { |
| 427 | buf.push(line.to_string()); |
| 428 | } |
| 429 | continue; |
| 430 | } |
| 431 | if is_fence(trimmed) { |
| 432 | // An opening fence closes any paragraph or list, then begins a verbatim block. The fence line |
| 433 | // itself (and any language tag on it) is not kept. |
| 434 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 435 | flush_list(&mut items, &mut stack); |
| 436 | code = Some((Vec::new(), start)); |
| 437 | continue; |
| 438 | } |
| 439 | |
| 440 | // A display maths block ($ ... $) spans several source lines, but every branch below classifies a |
| 441 | // line on its own -- so, left unchecked, a `=`-lead alignment row reads as a heading, a `-`-lead row |
| 442 | // as a list marker, and a blank row between stacked lines flushes the paragraph early, each stealing |
| 443 | // the line before the paragraph ever reaches `mathparse` whole. While an odd number of unescaped `$` |
| 444 | // have accumulated in the running paragraph the block is still open, so the blank/heading/list checks |
| 445 | // below are skipped for as long as it is: a fence opening or a capture opener (a `#`-led construct) |
| 446 | // still takes precedence regardless, since neither shape occurs inside genuine display maths. |
| 447 | let math_block_open = math_open(&lines); |
| 448 | |
| 449 | if trimmed.is_empty() && !math_block_open { |
| 450 | // A blank line closes the paragraph it follows, but not an open list: Typst continues an enum |
| 451 | // (or bullet list) across a blank line between items, restarting the numbering only when other |
| 452 | // content intervenes. The list is therefore held open here; the marker branch joins a following |
| 453 | // item of the same kind, while any other line -- a paragraph, heading, figure, fence or code |
| 454 | // line -- flushes it first, so two lists parted by real content still restart. A blank line held |
| 455 | // between two items of the open level marks it loose: Typst then sets its items with block |
| 456 | // spacing rather than the tight body pitch a blank-free list takes. |
| 457 | if let Some(top) = stack.last_mut() { |
| 458 | top.saw_blank = true; |
| 459 | } |
| 460 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 461 | } else if let Some(kind) = capture_opener(trimmed, binds) |
| 462 | .filter(|k| !(matches!(k, CaptureKind::ContentCall(_)) && !lines.is_empty())) |
| 463 | { |
| 464 | // A standalone content-binding reference mid-paragraph joins the paragraph inline rather than |
| 465 | // splicing a block, matching Typst's inline value flow: only a reference with no paragraph open |
| 466 | // splices its expanded blocks (the `filter` above lets an open-paragraph `#name` fall through to |
| 467 | // the paragraph arm). A markup builtin (`#lorem`, `#v`, `#pagebreak`) is block-position but not |
| 468 | // blank-line-gated: `capture_opener` already admits it only own-line (a balanced call with nothing |
| 469 | // but whitespace after its `)`, or a multi-line span), so a builtin directly beneath a prose line |
| 470 | // with no blank between (`prose\n#pagebreak()`) closes the paragraph and sets its own block, exactly |
| 471 | // as `= heading\n#pagebreak()` and `#section-banner` already do -- Typst turns the page there whether |
| 472 | // or not a blank line parts the two, so a paragraph before the break must not silently swallow it. |
| 473 | // A builtin with prose on the SAME line (`#lorem(5) more`) is not own-line, so it never reaches here; |
| 474 | // inline mid-prose support is a later unit. Every other capture kind -- a figure, a bare table, a |
| 475 | // data array -- flushes the paragraph and is gathered as before. |
| 476 | // |
| 477 | // A multi-line construct the reader sets rather than skips. It closes any open block, then its |
| 478 | // whole text is gathered by the check |
| 479 | // at the top of the loop until the delimiters balance, and parsed by [`dispatch_capture`]. |
| 480 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 481 | flush_list(&mut items, &mut stack); |
| 482 | let mut state = SkipState::new(); |
| 483 | scan_brackets(line, &mut state); |
| 484 | let mut buf = String::new(); |
| 485 | buf.push_str(line); |
| 486 | buf.push('\n'); |
| 487 | let cap = Capture { kind, buf, state, start }; |
| 488 | if !cap.state.has_open_bracket() { |
| 489 | dispatch_capture(cap, &mut items, &mut arrays, &mut skips, binds); // the whole construct closed on one line |
| 490 | } else { |
| 491 | capture = Some(cap); |
| 492 | } |
| 493 | } else if trimmed.starts_with("#line(") && call_inner(trimmed, "line").is_some() { |
| 494 | // A standalone `#line(length:.., stroke:..)` horizontal divider (the appendix brackets a note |
| 495 | // with one above and below). It closes any open block and sets a stroked rule; a multi-line |
| 496 | // `#line(` that does not close on this line falls through to the skip path below. |
| 497 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 498 | flush_list(&mut items, &mut stack); |
| 499 | if let Some(rule) = parse_line_rule(trimmed) { |
| 500 | items.push(rule); |
| 501 | } |
| 502 | } else if trimmed.starts_with("#print-glossary(") { |
| 503 | // A line-leading `#print-glossary()`: the glossary section's Term/Definition table. It closes any |
| 504 | // open block and emits a placeholder the book layer fills once the whole document's glossary terms |
| 505 | // are known -- unlike the surrounding template calls it is set in place, not recorded as a skip. |
| 506 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 507 | flush_list(&mut items, &mut stack); |
| 508 | items.push(Item::PrintGlossary { span: Span::new(start, end) }); |
| 509 | } else if let Some(decision) = code_skip(trimmed) { |
| 510 | // A Typst code statement (`#import`, `#let`, `#set`, `#show`) or a line-leading standalone call |
| 511 | // to a template function Austenite does not yet run: it closes any open block and is skipped. |
| 512 | // The styling and computation layer is a later increment; the prose around it still sets. When |
| 513 | // its delimiters do not balance on this line, the multi-line span is consumed by the check at the |
| 514 | // top of the loop until they do. |
| 515 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 516 | flush_list(&mut items, &mut stack); |
| 517 | // The recorded span is the opening line alone, even for a construct whose delimiters run on |
| 518 | // for several more: that is where a reader wants `--explain`'s caret to land, and the true |
| 519 | // closing offset is not known until the multi-line skip above closes, several iterations on. |
| 520 | skips.record(&construct_name(trimmed), Span::new(start, end)); |
| 521 | if let CodeSkip::Multi(state) = decision { |
| 522 | skip = Some(state); |
| 523 | } |
| 524 | } else if lines.is_empty() && is_code_reference(trimmed) && !names_scalar_alone(trimmed, binds.sfns) { |
| 525 | // A line-leading code-mode reference the reader cannot run -- a bare `#name` bound to nothing, a |
| 526 | // field/method access `#name.foo`, an `#if`/`#for`/`#while` control keyword, or an anonymous |
| 527 | // `#{ ... }`/`#( ... )` block. A bound `#name` was expanded by `capture_opener` above; a |
| 528 | // `#name(`/`#name[` call and the `#let`/`#set`/`#show`/`#import` keywords were refused by |
| 529 | // `code_skip`. Each of these resolves to a value or runs code in Typst, so setting its source as |
| 530 | // literal prose would leak a `#` onto the page (the very thing an expanded content-binding body |
| 531 | // carrying `#if`/`#{` would do); it is refused with its span instead. The `lines.is_empty()` guard |
| 532 | // keeps a reference mid-paragraph joining the line inline, as Typst does, rather than refusing it. |
| 533 | // A bare `#name` naming a SCALAR binding is exempt (`names_scalar_alone`): it falls through to the |
| 534 | // paragraph arm below, where `flush_para`'s own `substitute_scalars` replaces it with the bound |
| 535 | // value, so a scalar standing alone on its line sets its value just as one mid-prose already does. |
| 536 | // An UNBOUND standalone name is not exempt, so it stays the visible refusal it has always been. |
| 537 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 538 | flush_list(&mut items, &mut stack); |
| 539 | skips.record(&construct_name(trimmed), Span::new(start, end)); |
| 540 | } else if trimmed.starts_with('=') && !math_block_open { |
| 541 | // A heading closes any paragraph or list above it, then stands on its own line. |
| 542 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 543 | flush_list(&mut items, &mut stack); |
| 544 | let level = trimmed.chars().take_while(|&c| c == '=').count(); |
| 545 | let raw = trimmed[level..].trim(); // '=' is ASCII, so a byte slice at the count is safe |
| 546 | if raw.is_empty() { |
| 547 | return Err(err!( |
| 548 | "Empty heading on line {}: a `=` marker must be followed by a title.", line_no; |
| 549 | Input, Invalid, Missing)); |
| 550 | } |
| 551 | let (title, label) = split_label(raw); |
| 552 | if title.is_empty() { |
| 553 | return Err(err!( |
| 554 | "Heading on line {} has a label but no title.", line_no; Input, Invalid, Missing)); |
| 555 | } |
| 556 | // The title carries inline markup like any run, so a glossary term, an index call, emphasis or a |
| 557 | // maths span in a heading sets its display text rather than leaking its raw source into the head |
| 558 | // and the table of contents. An inline reference to a content binding (`= Product #stamp`) splices |
| 559 | // its expanded body first, and a bare `#name` naming a scalar `#let` value binding (`= Product |
| 560 | // #version`) substitutes its display text next, the same as a paragraph's. |
| 561 | let head_span = Span::new(start, end); |
| 562 | let title = substitute_content_calls(&title, binds, &mut skips, head_span); |
| 563 | let title = substitute_scalars(&title, binds.sfns); |
| 564 | items.push(Item::Heading { |
| 565 | level: level as u8, |
| 566 | runs: parse_inlines_in(&title, head_span, &mut skips), |
| 567 | label, |
| 568 | span: head_span, |
| 569 | }); |
| 570 | } else { |
| 571 | // A list marker joins the list stack, unless a display maths block is open, in which case a |
| 572 | // `-`/`+`-lead row is part of the equation, not a bullet: it falls through to the paragraph arm |
| 573 | // below like every other captured line. |
| 574 | match if math_block_open { None } else { marker(trimmed) } { |
| 575 | Some((ord, text)) => { |
| 576 | // A list item. It closes any open paragraph, then joins the list stack by its indentation: |
| 577 | // a deeper marker opens a sub-list under the current item, a shallower one closes back to |
| 578 | // the matching level, and a same-indent marker of the other kind ends the list and starts |
| 579 | // one of the new kind. The item's text carries inline emphasis like any run. |
| 580 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 581 | let indent = line.chars().take_while(|c| c.is_whitespace()).count(); |
| 582 | // A list item is running prose like a paragraph, so an inline content-fn reference in it -- |
| 583 | // `+ ... written as M#oxe.` -- splices its expanded body first, exactly as `flush_para` does |
| 584 | // for a paragraph. Without this the item bypassed the pre-pass and leaked a raw `#oxe`/`#name` |
| 585 | // while the same reference in a paragraph beside it expanded, an inconsistency a reader sees. |
| 586 | let item_span = Span::new(start, end); |
| 587 | let text = substitute_content_calls(&text, binds, &mut skips, item_span); |
| 588 | let runs = parse_inlines_in(&text, item_span, &mut skips); |
| 589 | list_marker(&mut items, &mut stack, indent, ord, runs, start, end); |
| 590 | }, |
| 591 | None => { |
| 592 | // Any other non-blank line joins the running paragraph, closing a list first; its own line |
| 593 | // break and indentation carry no meaning, only its words. This is also where a blank, |
| 594 | // heading-lead or list-marker-lead line lands while a display maths block is open. |
| 595 | flush_list(&mut items, &mut stack); |
| 596 | if lines.is_empty() { |
| 597 | para_start = start; |
| 598 | } |
| 599 | lines.push(line.to_string()); |
| 600 | para_end = end; |
| 601 | }, |
| 602 | } |
| 603 | } |
| 604 | } |
| 605 | |
| 606 | // A source that ends without a closing blank line still closes its last paragraph or list; an |
| 607 | // unterminated code fence still yields the block it had gathered. |
| 608 | flush_para(&mut items, &mut lines, para_start, para_end, &mut skips, binds); |
| 609 | flush_list(&mut items, &mut stack); |
| 610 | if let Some((buf, cstart)) = code { |
| 611 | items.push(Item::Code { lines: buf, span: Span::new(cstart, offset) }); |
| 612 | } |
| 613 | // A construct left open at end of source is dispatched with what it gathered, so a missing closer |
| 614 | // still yields its best-effort figure or table rather than swallowing the tail silently. |
| 615 | if let Some(cap) = capture { |
| 616 | dispatch_capture(cap, &mut items, &mut arrays, &mut skips, binds); |
| 617 | } |
| 618 | Ok((items, skips)) |
| 619 | } |
| 620 | |
| 621 | /// Is a display maths block still open across the paragraph lines gathered so far? A `$` toggles the |
| 622 | /// state; a `\`-escaped one (`\$`) is skipped, mirroring the same escape in [`parse_inlines_in`], so it |
| 623 | /// never toggles. An odd running count means the block opened on some earlier line and has not yet met |
| 624 | /// its close, which is what lets [`document_with_refusals`] keep capturing lines the per-line classifier |
| 625 | /// would otherwise steal as a heading, a list marker, or a paragraph-flushing blank. |
| 626 | fn math_open(lines: &[String]) -> bool { |
| 627 | let mut open = false; |
| 628 | for line in lines { |
| 629 | let mut chars = line.chars(); |
| 630 | while let Some(c) = chars.next() { |
| 631 | if c == '\\' { |
| 632 | chars.next(); // the escaped character, taken literally |
| 633 | } else if c == '$' { |
| 634 | open = !open; |
| 635 | } |
| 636 | } |
| 637 | } |
| 638 | open |
| 639 | } |
| 640 | |
| 641 | /// Is this already-left-trimmed line a ```` ``` ```` code fence? An opening fence may carry a language |
| 642 | /// tag (```` ```rust ````); a closing fence is bare. Either way it opens with three backticks. |
| 643 | fn is_fence(trimmed: &str) -> bool { |
| 644 | trimmed.starts_with("```") |
| 645 | } |
| 646 | |
| 647 | /// Reads a list marker at the start of an already-left-trimmed line: `-` opens a bullet item, `+` a |
| 648 | /// numbered one. The marker must be the whole line or be followed by whitespace, so a dash inside a word |
| 649 | /// or a `+1` is ordinary prose, not a marker. Returns the item's kind and its text with the marker and |
| 650 | /// surrounding whitespace removed. |
| 651 | fn marker(trimmed: &str) -> Option<(bool, String)> { |
| 652 | let first = trimmed.chars().next()?; |
| 653 | let ordered = match first { |
| 654 | '-' => false, |
| 655 | '+' => true, |
| 656 | _ => return None, |
| 657 | }; |
| 658 | let rest = &trimmed[first.len_utf8()..]; |
| 659 | if rest.is_empty() { |
| 660 | return Some((ordered, String::new())); |
| 661 | } |
| 662 | if rest.starts_with(|c: char| c.is_whitespace()) { |
| 663 | return Some((ordered, rest.trim().to_string())); |
| 664 | } |
| 665 | None |
| 666 | } |
| 667 | |
| 668 | /// One open level of a possibly-nested list while the reader accumulates it. `indent` is the leading-space |
| 669 | /// width of the level's markers, so a deeper marker opens a child level and a shallower one closes this |
| 670 | /// level back into the item it hung under. |
| 671 | struct ListFrame { |
| 672 | indent: usize, |
| 673 | ordered: bool, |
| 674 | items: Vec<ListItem>, |
| 675 | loose: bool, // a blank line parted two of this level's items, so Typst sets it loose (block-spaced) |
| 676 | saw_blank: bool, // a blank line is pending since the last item; it makes the level loose only if a sibling follows |
| 677 | start: u32, |
| 678 | end: u32, |
| 679 | } |
| 680 | |
| 681 | /// Attaches a marker line to the open list stack by its `indent`, opening or closing nested levels as the |
| 682 | /// indentation and kind require. A deeper marker opens a sub-list under the current item; a shallower one |
| 683 | /// closes the deeper level(s) first; a same-indent marker continues the level when its kind matches and |
| 684 | /// otherwise ends it and starts a fresh list of the new kind, as the flat reader did. |
| 685 | fn list_marker( |
| 686 | items: &mut Vec<Item>, |
| 687 | stack: &mut Vec<ListFrame>, |
| 688 | indent: usize, |
| 689 | ord: bool, |
| 690 | runs: Vec<Inline>, |
| 691 | start: u32, |
| 692 | end: u32, |
| 693 | ) |
| 694 | { |
| 695 | // Close every open level deeper than this marker: a dedent ends the nested list(s), each folding into |
| 696 | // the item it hung under. |
| 697 | while stack.last().map_or(false, |f| f.indent > indent) { |
| 698 | if let Some(frame) = stack.pop() { |
| 699 | fold(items, stack, frame); |
| 700 | } |
| 701 | } |
| 702 | match stack.last_mut() { |
| 703 | Some(top) if top.indent == indent && top.ordered == ord => { |
| 704 | // Same level, same kind: another item of the open list. A blank line pending since the previous |
| 705 | // item was a genuine inter-item gap, so the level sets loose. |
| 706 | if top.saw_blank { |
| 707 | top.loose = true; |
| 708 | } |
| 709 | top.saw_blank = false; |
| 710 | top.items.push(ListItem { runs, children: Vec::new() }); |
| 711 | top.end = end; |
| 712 | }, |
| 713 | Some(top) if top.indent == indent => { |
| 714 | // Same indent, the other kind: the open list ends and a fresh one of the new kind begins. |
| 715 | if let Some(frame) = stack.pop() { |
| 716 | fold(items, stack, frame); |
| 717 | } |
| 718 | stack.push(ListFrame { |
| 719 | indent, ordered: ord, items: vec![ListItem { runs, children: Vec::new() }], loose: false, saw_blank: false, start, end }); |
| 720 | }, |
| 721 | // Deeper than the current level (a sub-list), or the first marker of a list: open a new level. A |
| 722 | // deeper level becomes a child of the current item when it folds. |
| 723 | _ => stack.push(ListFrame { |
| 724 | indent, ordered: ord, items: vec![ListItem { runs, children: Vec::new() }], loose: false, saw_blank: false, start, end }), |
| 725 | } |
| 726 | } |
| 727 | |
| 728 | /// Folds a closed list level into the tree: it becomes an [`Item::List`] hanging under the current item of |
| 729 | /// the level below, or a top-level item when no level remains open. |
| 730 | fn fold(items: &mut Vec<Item>, stack: &mut Vec<ListFrame>, frame: ListFrame) { |
| 731 | let list = Item::List { |
| 732 | ordered: frame.ordered, |
| 733 | items: frame.items, |
| 734 | loose: frame.loose, |
| 735 | span: Span::new(frame.start, frame.end), |
| 736 | }; |
| 737 | match stack.last_mut() { |
| 738 | Some(parent) => match parent.items.last_mut() { |
| 739 | Some(item) => { |
| 740 | item.children.push(list); |
| 741 | parent.end = frame.end; |
| 742 | }, |
| 743 | // A nested level always opens after its parent item exists, so this arm is unreachable in |
| 744 | // practice; a stray level is kept as a top-level item rather than dropped. |
| 745 | None => items.push(list), |
| 746 | }, |
| 747 | None => items.push(list), |
| 748 | } |
| 749 | } |
| 750 | |
| 751 | /// Closes every open list level into the item tree. The deepest level folds into its parent's current item |
| 752 | /// first, so a nested list lands under the item it hung under; the outermost becomes a top-level |
| 753 | /// [`Item::List`]. An empty stack flushes nothing, so a stray flush between two paragraphs costs nothing. |
| 754 | fn flush_list(items: &mut Vec<Item>, stack: &mut Vec<ListFrame>) { |
| 755 | while let Some(frame) = stack.pop() { |
| 756 | fold(items, stack, frame); |
| 757 | } |
| 758 | } |
| 759 | |
| 760 | /// Closes the paragraph being accumulated, if any: its lines are joined, their whitespace collapsed, |
| 761 | /// and the result pushed as one [`Item::Paragraph`] spanning the source it came from. An empty |
| 762 | /// accumulator flushes nothing, so a run of blank lines closes a paragraph only once. |
| 763 | fn flush_para( |
| 764 | items: &mut Vec<Item>, |
| 765 | lines: &mut Vec<String>, |
| 766 | start: u32, |
| 767 | end: u32, |
| 768 | skips: &mut Refusals, |
| 769 | binds: crate::lang::rules::Bindings<'_, '_>, |
| 770 | ) |
| 771 | { |
| 772 | if lines.is_empty() { |
| 773 | return; |
| 774 | } |
| 775 | let text = normalise_ws(&lines.join(" ")); |
| 776 | // A trailing `<name>` labels the block -- in practice a display equation, `$ ... $ <eq_x>` -- and is |
| 777 | // stripped before the runs are read, so the maths span stands alone and lowers to a numbered equation |
| 778 | // rather than a rich paragraph. Ordinary prose ends in a full stop, so the conservative `split_label` |
| 779 | // (a single whitespace-free token in angle brackets at the very end) does not fire on it. |
| 780 | let (body, label) = split_label(&text); |
| 781 | let span = Span::new(start, end); |
| 782 | // An inline mid-prose reference to a bound content binding -- `M#oxe`, `see #stamp("v2") for details` -- |
| 783 | // splices its argument-substituted body into the surrounding prose here, before the scalar pass and the |
| 784 | // inline scanner, exactly as the own-line reference splices its blocks: the words before and after the |
| 785 | // call are kept, and the body's own markup is then read by the one downstream scanner. Runs first so a |
| 786 | // scalar reference the body carries still substitutes below. |
| 787 | let body = substitute_content_calls(&body, binds, skips, span); |
| 788 | // A bare `#name` naming a scalar `#let` value binding substitutes its display text before the inline |
| 789 | // scanner runs, so "Version #version." reads the same as if the number or string had been typed in |
| 790 | // place. Runs before `parse_inlines_in`, never inside it, so it never touches a raw code span, inline |
| 791 | // maths, or any other captured construct (a table, a figure, a `#context` block) -- those are gathered |
| 792 | // and dispatched on a wholly separate path and never reach here. |
| 793 | let body = substitute_scalars(&body, binds.sfns); |
| 794 | let runs = parse_inlines_in(&body, span, skips); |
| 795 | items.push(Item::Paragraph { runs, label, span }); |
| 796 | lines.clear(); |
| 797 | } |
| 798 | |
| 799 | /// Collapses every run of whitespace to a single space and trims the ends, so a paragraph's set width |
| 800 | /// is left to the line breaker rather than to the source's own line breaks and indentation. |
| 801 | fn normalise_ws(s: &str) -> String { |
| 802 | s.split_whitespace().collect::<Vec<_>>().join(" ") |
| 803 | } |
| 804 | |
| 805 | /// Splits a whitespace-collapsed paragraph into inline runs in Typst's markup: `*strong*`, `_emph_`, a |
| 806 | /// `@label` cross-reference, and `\`-escapes. An emphasis delimiter pairs only when it flanks a word -- |
| 807 | /// whitespace or an opening bracket before it and a non-space after to open, the reverse to close -- so |
| 808 | /// `fe2o3_net`, `5 * 3` and a lone `_` are ordinary text. A backslash sets the next character literally, |
| 809 | /// so `\$`, `\#`, `\_` and `\@` appear as themselves. An unpaired delimiter, or an `@` with no label |
| 810 | /// after it, is ordinary text. Nesting is a later increment: the first valid closer ends a run. |
| 811 | pub(crate) fn parse_inlines(text: &str) -> Vec<Inline> { |
| 812 | let mut skips = Refusals::default(); |
| 813 | // A table cell, a caption or a flattened array cell has no item-level span of its own to attribute a |
| 814 | // refusal to (see `Refusal`'s own doc comment on why the item, not the character, is the finest |
| 815 | // boundary kept); this thin wrapper already threw the summary away before this unit, so a zero span |
| 816 | // changes nothing a caller could observe. |
| 817 | parse_inlines_in(text, Span::new(0, 0), &mut skips) |
| 818 | } |
| 819 | |
| 820 | /// The inline scanner proper, recording every unhandled inline call into `skips`, at `span` (the whole |
| 821 | /// containing item -- a paragraph, a heading, a list item -- rather than the call's own narrower |
| 822 | /// position; see `Refusal`'s doc comment), so a `#func[...]` the reader cannot set is reported rather |
| 823 | /// than leaked into the running text. [`parse_inlines`] is the thin wrapper for callers -- table cells, |
| 824 | /// captions, flattening -- that do not surface the summary. |
| 825 | fn parse_inlines_in(text: &str, span: Span, skips: &mut Refusals) -> Vec<Inline> { |
| 826 | let chars: Vec<char> = text.chars().collect(); |
| 827 | let n = chars.len(); |
| 828 | let mut runs: Vec<Inline> = Vec::new(); |
| 829 | let mut plain = String::new(); // ordinary text gathered before the next run |
| 830 | let mut i = 0usize; |
| 831 | while i < n { |
| 832 | let c = chars[i]; |
| 833 | // A backslash escapes the next character, which is then set as itself. |
| 834 | if c == '\\' && i + 1 < n { |
| 835 | plain.push(chars[i + 1]); |
| 836 | i += 2; |
| 837 | continue; |
| 838 | } |
| 839 | // An inline maths span between dollars. A `\$` was already turned into a literal above, so a `$` |
| 840 | // reaching here opens maths. If it parses, it is a maths run; if not, the literal `$...$` is kept. |
| 841 | if c == '$' { |
| 842 | if let Some(close) = (i + 1..n).find(|&j| chars[j] == '$') { |
| 843 | let inner: String = chars[i + 1..close].iter().collect(); |
| 844 | if let Ok(atom) = mathparse::parse(&inner) { |
| 845 | if !plain.is_empty() { |
| 846 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 847 | } |
| 848 | runs.push(Inline::Math(atom)); |
| 849 | i = close + 1; |
| 850 | continue; |
| 851 | } |
| 852 | } |
| 853 | } |
| 854 | // An inline code span, `raw` between backticks: its content is verbatim, no markup within. |
| 855 | if c == '`' { |
| 856 | if let Some(close) = (i + 1..n).find(|&j| chars[j] == '`') { |
| 857 | if !plain.is_empty() { |
| 858 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 859 | } |
| 860 | runs.push(Inline::Code(chars[i + 1..close].iter().collect())); |
| 861 | i = close + 1; |
| 862 | continue; |
| 863 | } |
| 864 | } |
| 865 | // An inline footnote. Its bracketed content is markup, carried as inline runs so the note sets its |
| 866 | // own emphasis at the foot of the page. The mark falls after the run before it. |
| 867 | if c == '#' { |
| 868 | if let Some((note, next)) = footnote_call(&chars, i, span, skips) { |
| 869 | if !plain.is_empty() { |
| 870 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 871 | } |
| 872 | runs.push(Inline::Footnote(note)); |
| 873 | i = next; |
| 874 | continue; |
| 875 | } |
| 876 | } |
| 877 | // The function-call form of emphasis, `#emph[...]`, set exactly as `_..._`: its bracketed content is |
| 878 | // markup, so it takes the same expansion, and a call nested within it renders its display text. |
| 879 | if c == '#' { |
| 880 | if let Some((inner, next)) = emph_call(&chars, i) { |
| 881 | if !plain.is_empty() { |
| 882 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 883 | } |
| 884 | push_emphasis(&mut runs, false, &inner, span, skips); |
| 885 | i = next; |
| 886 | continue; |
| 887 | } |
| 888 | } |
| 889 | // The function-call form of strength, `#strong[...]` or `#strong("...")`, set exactly as `*...*`: |
| 890 | // the bracket form's content is markup and takes the same expansion as the emph call above; the |
| 891 | // paren form's quoted-string argument is plain text bold in full. A paren argument that is not a |
| 892 | // plain string -- a bare identifier or an expression this reader cannot evaluate -- is left for |
| 893 | // the generic call handler below, which records the refusal rather than guessing at its text. |
| 894 | if c == '#' { |
| 895 | if let Some((inner, next)) = strong_call(&chars, i) { |
| 896 | if !plain.is_empty() { |
| 897 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 898 | } |
| 899 | push_emphasis(&mut runs, true, &inner, span, skips); |
| 900 | i = next; |
| 901 | continue; |
| 902 | } |
| 903 | } |
| 904 | // Typst's superscript, `#super[...]` or `#super("...")`. Its content is usually a short string or |
| 905 | // number, reduced to display text here and set raised and smaller by the block layer. |
| 906 | if c == '#' { |
| 907 | if let Some((text, next)) = super_call(&chars, i) { |
| 908 | if !plain.is_empty() { |
| 909 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 910 | } |
| 911 | runs.push(Inline::Super(text)); |
| 912 | i = next; |
| 913 | continue; |
| 914 | } |
| 915 | } |
| 916 | // Typst's subscript, `#sub[...]` or `#sub("...")`, e.g. `CO#sub[2]`. Its content reduces to |
| 917 | // display text here and is set dropped and smaller by the block layer. |
| 918 | if c == '#' { |
| 919 | if let Some((text, next)) = sub_call(&chars, i) { |
| 920 | if !plain.is_empty() { |
| 921 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 922 | } |
| 923 | runs.push(Inline::Sub(text)); |
| 924 | i = next; |
| 925 | continue; |
| 926 | } |
| 927 | } |
| 928 | // Typst's `#smallcaps[...]` or `#smallcaps("...")`: its content reduces to display text here and is |
| 929 | // shaped with the font's small-capitals feature by the block layer, as Typst asks the font for it. |
| 930 | if c == '#' { |
| 931 | if let Some((text, next)) = smallcaps_call(&chars, i) { |
| 932 | if !plain.is_empty() { |
| 933 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 934 | } |
| 935 | runs.push(Inline::SmallCaps(text)); |
| 936 | i = next; |
| 937 | continue; |
| 938 | } |
| 939 | } |
| 940 | // An inline glossary or index call defined in the book template. A glossary term is a run of its |
| 941 | // own, so [`doc::author`] can set it bold-italic on first use; a visible index call sets its |
| 942 | // display text, which may itself carry markup, so it is parsed and folded in; a pure index marker |
| 943 | // sets nothing. |
| 944 | if c == '#' { |
| 945 | if let Some((call, next)) = glossary_call(&chars, i, span, skips) { |
| 946 | // An index marker is a run of its own, woven in before the display so the block layer |
| 947 | // records the term's occurrence at this point; the display, where the call has one, follows. |
| 948 | let push_index = |runs: &mut Vec<Inline>, plain: &mut String, index: Option<IndexKey>, skips: &mut Refusals| { |
| 949 | if let Some(k) = index { |
| 950 | if !plain.is_empty() { |
| 951 | runs.push(Inline::Text(std::mem::take(plain))); |
| 952 | } |
| 953 | // The display markup becomes its own runs, so the index page sets an emphasised entry |
| 954 | // italic and a display/sort split shows the display -- parsed exactly as the body's is. |
| 955 | let display = parse_inlines_in(&k.display, span, skips); |
| 956 | runs.push(Inline::Index { term: k.term, sub: k.sub, display, main: k.main }); |
| 957 | } |
| 958 | }; |
| 959 | match call { |
| 960 | Call::Glossary { term, display, index } => { |
| 961 | push_index(&mut runs, &mut plain, index, skips); |
| 962 | if !plain.is_empty() { |
| 963 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 964 | } |
| 965 | runs.push(Inline::Glossary { term, display }); |
| 966 | }, |
| 967 | Call::Visible { display, index } => { |
| 968 | push_index(&mut runs, &mut plain, index, skips); |
| 969 | let sub = parse_inlines_in(&display, span, skips); |
| 970 | // A plain display folds back into the running text, keeping the fast single-run |
| 971 | // path; a display carrying markup becomes its own runs. |
| 972 | if let [Inline::Text(t)] = sub.as_slice() { |
| 973 | plain.push_str(t); |
| 974 | } else { |
| 975 | if !plain.is_empty() { |
| 976 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 977 | } |
| 978 | runs.extend(sub); |
| 979 | } |
| 980 | }, |
| 981 | Call::Invisible { index } => { |
| 982 | push_index(&mut runs, &mut plain, index, skips); |
| 983 | }, |
| 984 | } |
| 985 | i = next; |
| 986 | continue; |
| 987 | } |
| 988 | } |
| 989 | // An inline citation, `#cite(<key>)` or `#cite(<a>, <b>)`. Its keys become a cite run the block |
| 990 | // layer resolves to "(Author Year)" against the bibliography; a citation with no readable key is |
| 991 | // dropped rather than left as raw source. |
| 992 | if c == '#' { |
| 993 | if let Some((keys, next)) = cite_call(&chars, i) { |
| 994 | if !keys.is_empty() { |
| 995 | if !plain.is_empty() { |
| 996 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 997 | } |
| 998 | runs.push(Inline::Cite(keys)); |
| 999 | } |
| 1000 | i = next; |
| 1001 | continue; |
| 1002 | } |
| 1003 | } |
| 1004 | // An inline claim marker, `#claim-label(...)` or `#claim-refs(...)`, from the book's claims |
| 1005 | // machinery. A `claim-label` registers invisible metadata AND sets a compressed code in the outside |
| 1006 | // margin: it emits a `MarginNote` run, which sets nothing in the body column but records a zero-width |
| 1007 | // anchor the overlay pass draws the code against after convergence. A `claim-refs` registers metadata |
| 1008 | // only, with no visible output, so it is consumed and emits nothing. Either way the body prose closes |
| 1009 | // over the marker's place, matching Typst's own body flow. |
| 1010 | if c == '#' { |
| 1011 | if let Some((display, codes, next)) = claim_call(&chars, i) { |
| 1012 | // A call naming a margin display or any reference code emits a MarginNote; a call naming |
| 1013 | // neither is consumed and sets nothing. |
| 1014 | if !display.is_empty() || !codes.is_empty() { |
| 1015 | if !plain.is_empty() { |
| 1016 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 1017 | } |
| 1018 | runs.push(Inline::MarginNote { display, codes }); |
| 1019 | } |
| 1020 | i = next; |
| 1021 | continue; |
| 1022 | } |
| 1023 | } |
| 1024 | // The `#raw("...")` call form of inline code. |
| 1025 | if c == '#' { |
| 1026 | if let Some((text, next)) = raw_call(&chars, i) { |
| 1027 | if !plain.is_empty() { |
| 1028 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 1029 | } |
| 1030 | runs.push(Inline::Code(text)); |
| 1031 | i = next; |
| 1032 | continue; |
| 1033 | } |
| 1034 | } |
| 1035 | // A Typst hyperlink, `#link("url")[text]` or `#link(<label>)[text]`. The link text is what a print |
| 1036 | // reader sees, so its markup is parsed and folded into the running line; the destination has no place |
| 1037 | // on a page with no clickable annotation and is dropped. A `#link("url")` with no bracket sets the |
| 1038 | // URL itself as its text, as Typst does. |
| 1039 | if c == '#' { |
| 1040 | if let Some((body, next)) = link_call(&chars, i, span, skips) { |
| 1041 | if let [Inline::Text(t)] = body.as_slice() { |
| 1042 | plain.push_str(t); |
| 1043 | } else { |
| 1044 | if !plain.is_empty() { |
| 1045 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 1046 | } |
| 1047 | runs.extend(body); |
| 1048 | } |
| 1049 | i = next; |
| 1050 | continue; |
| 1051 | } |
| 1052 | } |
| 1053 | // A Typst cross-reference: `@` then a label. |
| 1054 | if c == '@' { |
| 1055 | if let Some((label, next)) = at_label(&chars, i) { |
| 1056 | if !plain.is_empty() { |
| 1057 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 1058 | } |
| 1059 | runs.push(Inline::PageRef(label)); |
| 1060 | i = next; |
| 1061 | continue; |
| 1062 | } |
| 1063 | } |
| 1064 | if (c == '*' || c == '_') && is_opener(&chars, i) { |
| 1065 | if let Some(close) = find_closer(&chars, i + 1, c) { |
| 1066 | if !plain.is_empty() { |
| 1067 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 1068 | } |
| 1069 | let inner: String = chars[i + 1..close].iter().collect(); |
| 1070 | push_emphasis(&mut runs, c == '*', &inner, span, skips); |
| 1071 | i = close + 1; |
| 1072 | continue; |
| 1073 | } |
| 1074 | } |
| 1075 | // An inline `#func(...)` or `#func[...]` call none of the handlers above claimed: a template function |
| 1076 | // the reader cannot yet run. It is recorded by name and consumed whole, so its raw markup no longer |
| 1077 | // leaks into the set text. When it wraps a single `[...]` content group -- the common shape of a Typst |
| 1078 | // content function -- that body is parsed and folded in, keeping its words rather than dropping them; |
| 1079 | // a call with only paren arguments (`#v(1em)`, `#colbreak()`) sets nothing where it stood. |
| 1080 | if c == '#' { |
| 1081 | if let Some((body, next, name)) = unknown_call(&chars, i, span, skips) { |
| 1082 | skips.record(&name, span); |
| 1083 | if let Some(body) = body { |
| 1084 | if let [Inline::Text(t)] = body.as_slice() { |
| 1085 | plain.push_str(t); |
| 1086 | } else { |
| 1087 | if !plain.is_empty() { |
| 1088 | runs.push(Inline::Text(std::mem::take(&mut plain))); |
| 1089 | } |
| 1090 | runs.extend(body); |
| 1091 | } |
| 1092 | } |
| 1093 | i = next; |
| 1094 | continue; |
| 1095 | } |
| 1096 | } |
| 1097 | // Typst's smartypants: a run of hyphens in markup text becomes em/en dashes, longest match first |
| 1098 | // (`---` before `--`), and three-or-more dots become an ellipsis. Every other branch above has |
| 1099 | // already claimed `` ` `` and `$` before falling through here, so this never touches a raw code |
| 1100 | // span or a maths span -- only ordinary prose reaches this fallback. A `\-`/`\.` was already turned |
| 1101 | // into a literal above and never joins a run counted here, matching Typst's own escape. |
| 1102 | if c == '-' || c == '.' { |
| 1103 | let mut run = 0usize; |
| 1104 | while chars.get(i + run) == Some(&c) { run += 1; } |
| 1105 | let mut left = run; |
| 1106 | if c == '-' { |
| 1107 | while left > 0 { |
| 1108 | if left >= 3 { plain.push('\u{2014}'); left -= 3; } // --- em dash |
| 1109 | else if left == 2 { plain.push('\u{2013}'); left = 0; } // -- en dash |
| 1110 | else { plain.push('-'); left = 0; } |
| 1111 | } |
| 1112 | } else if left >= 3 { |
| 1113 | plain.push('\u{2026}'); // ... ellipsis |
| 1114 | left -= 3; |
| 1115 | for _ in 0..left { plain.push('.'); } |
| 1116 | } else { |
| 1117 | for _ in 0..left { plain.push('.'); } |
| 1118 | } |
| 1119 | i += run; |
| 1120 | continue; |
| 1121 | } |
| 1122 | plain.push(c); |
| 1123 | i += 1; |
| 1124 | } |
| 1125 | if !plain.is_empty() { |
| 1126 | runs.push(Inline::Text(plain)); |
| 1127 | } |
| 1128 | if runs.is_empty() { |
| 1129 | runs.push(Inline::Text(String::new())); // a paragraph of pure delimiters keeps one empty run |
| 1130 | } |
| 1131 | runs |
| 1132 | } |
| 1133 | |
| 1134 | /// Pushes an emphasised run (`*strong*` when `strong`, else `_emph_`) onto `runs`, reading its inner |
| 1135 | /// markup. When the inner is plain text the run keeps the fast flat path -- one [`Inline::Strong`] or |
| 1136 | /// [`Inline::Emph`]. When it carries a glossary term, an index call or a maths span -- as |
| 1137 | /// `*The captation #gsi[attractor]*` does -- the emphasis is expanded: its plain stretches take the |
| 1138 | /// emphasis face and the embedded calls become their own runs, so a call nested in emphasis renders its |
| 1139 | /// display text rather than leaking its raw source. A glossary term keeps its own first-use bold-italic |
| 1140 | /// (which subsumes the surrounding emphasis), so only the plain stretches carry the emphasis face. |
| 1141 | fn push_emphasis(runs: &mut Vec<Inline>, strong: bool, inner: &str, span: Span, skips: &mut Refusals) { |
| 1142 | let sub = parse_inlines_in(inner, span, skips); |
| 1143 | if let [Inline::Text(t)] = sub.as_slice() { |
| 1144 | runs.push(if strong { Inline::Strong(t.clone()) } else { Inline::Emph(t.clone()) }); |
| 1145 | return; |
| 1146 | } |
| 1147 | for run in sub { |
| 1148 | match run { |
| 1149 | // A plain stretch takes the emphasis face. |
| 1150 | Inline::Text(t) => runs.push(if strong { Inline::Strong(t) } else { Inline::Emph(t) }), |
| 1151 | // An inner run of the opposite face (`*_word_*`, `_*word*_`) multiplies the two faces to a |
| 1152 | // bold-italic run rather than losing the outer; a run of the same face, a glossary term or a |
| 1153 | // maths span keeps its own face, one level of nesting being all the flat vocabulary carries. |
| 1154 | Inline::Emph(t) if strong => runs.push(Inline::BoldItalic(t)), |
| 1155 | Inline::Strong(t) if !strong => runs.push(Inline::BoldItalic(t)), |
| 1156 | other => runs.push(other), |
| 1157 | } |
| 1158 | } |
| 1159 | } |
| 1160 | |
| 1161 | /// Does the delimiter at `i` flank the left of a word? A non-space must follow it, and the start of the |
| 1162 | /// paragraph, whitespace, or an opening bracket must precede it. |
| 1163 | fn is_opener(chars: &[char], i: usize) -> bool { |
| 1164 | match chars.get(i + 1) { |
| 1165 | Some(c) if !c.is_whitespace() => {}, |
| 1166 | _ => return false, |
| 1167 | } |
| 1168 | match i.checked_sub(1).and_then(|p| chars.get(p)) { |
| 1169 | None => true, |
| 1170 | Some(&p) => p.is_whitespace() || matches!(p, '(' | '[' | '{' | '"' | '\''), |
| 1171 | } |
| 1172 | } |
| 1173 | |
| 1174 | /// Does the delimiter at `j` flank the right of a word? A non-space must precede it, and the end of the |
| 1175 | /// paragraph, whitespace, or closing punctuation must follow. |
| 1176 | fn is_closer(chars: &[char], j: usize) -> bool { |
| 1177 | match j.checked_sub(1).and_then(|p| chars.get(p)) { |
| 1178 | Some(p) if !p.is_whitespace() => {}, |
| 1179 | _ => return false, |
| 1180 | } |
| 1181 | match chars.get(j + 1) { |
| 1182 | None => true, |
| 1183 | Some(&c) => c.is_whitespace() |
| 1184 | || matches!(c, ')' | ']' | '}' | '.' | ',' | ';' | ':' | '!' | '?' | '"' | '\''), |
| 1185 | } |
| 1186 | } |
| 1187 | |
| 1188 | /// The index of the first valid closing `delim` at or after `start`, or `None` when the run never |
| 1189 | /// closes -- in which case the opener is ordinary text. |
| 1190 | fn find_closer(chars: &[char], start: usize, delim: char) -> Option<usize> { |
| 1191 | (start..chars.len()).find(|&j| chars[j] == delim && is_closer(chars, j)) |
| 1192 | } |
| 1193 | |
| 1194 | /// Reads a Typst cross-reference at `i` (an `@`): the label of letters, digits and `- _ :` that follows, |
| 1195 | /// and the index just past it. `None` when no label char follows, so a bare or escaped `@` is ordinary |
| 1196 | /// text. A trailing `.` is not a label character, so `@intro.` at the end of a sentence keeps its stop. |
| 1197 | fn at_label(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1198 | let start = i + 1; |
| 1199 | let mut j = start; |
| 1200 | while j < chars.len() && is_label_char(chars[j]) { |
| 1201 | j += 1; |
| 1202 | } |
| 1203 | if j == start { |
| 1204 | return None; |
| 1205 | } |
| 1206 | Some((chars[start..j].iter().collect(), j)) |
| 1207 | } |
| 1208 | |
| 1209 | /// A character legal within a Typst label. Deliberately excludes `.`, so a label does not swallow the |
| 1210 | /// full stop that ends a sentence. |
| 1211 | fn is_label_char(c: char) -> bool { |
| 1212 | c.is_alphanumeric() || matches!(c, '-' | '_' | ':') |
| 1213 | } |
| 1214 | |
| 1215 | /// Reads an inline `#raw("...")` at `i`, returning its literal content and the index past the closing |
| 1216 | /// `")`. `None` when the shape does not match, so a `#raw` written any other way is left as ordinary |
| 1217 | /// text. Escaped quotes inside the string are not handled -- a later refinement. |
| 1218 | fn raw_call(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1219 | let open = at_lit(chars, i, "#raw(\"")?; |
| 1220 | let close = (open..chars.len()).find(|&j| chars[j] == '"')?; |
| 1221 | if chars.get(close + 1) != Some(&')') { |
| 1222 | return None; |
| 1223 | } |
| 1224 | Some((chars[open..close].iter().collect(), close + 2)) |
| 1225 | } |
| 1226 | |
| 1227 | /// Reads an inline `#link(dest)[text]` (or a bare `#link(dest)`) at `i` (a `#`), returning the link's |
| 1228 | /// display runs and the index just past it. The destination -- a `"url"` string or a `<label>` -- is read |
| 1229 | /// and discarded, the page carrying no clickable annotation; the bracketed text is what the reader sees, |
| 1230 | /// so it is parsed for its own markup. A `#link(dest)` with no following `[...]` sets the destination |
| 1231 | /// string itself as its text, as Typst does. Any unhandled inline call within the text is recorded into |
| 1232 | /// `skips`. `None` when the shape is not a link call or its arguments do not close. |
| 1233 | fn link_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) -> Option<(Vec<Inline>, usize)> { |
| 1234 | let Some(open) = at_lit(chars, i, "#link") else { return None; }; |
| 1235 | if chars.get(open) != Some(&'(') { |
| 1236 | return None; |
| 1237 | } |
| 1238 | let Some((dest, after_dest)) = read_group(chars, open) else { return None; }; |
| 1239 | // A following `[...]` group is the link text; without one, the destination stands as the text. |
| 1240 | if chars.get(after_dest) == Some(&'[') { |
| 1241 | let Some((body, next)) = read_group(chars, after_dest) else { return None; }; |
| 1242 | return Some((parse_inlines_in(&body, span, skips), next)); |
| 1243 | } |
| 1244 | let text = link_dest_text(&dest); |
| 1245 | Some((vec![Inline::Text(text)], after_dest)) |
| 1246 | } |
| 1247 | |
| 1248 | /// The display text of a bare `#link(dest)` with no bracketed body: a `"url"` string loses its quotes, a |
| 1249 | /// `<label>` its angle brackets, and anything else stands as written. |
| 1250 | fn link_dest_text(dest: &str) -> String { |
| 1251 | let t = dest.trim(); |
| 1252 | if let Some(inner) = t.strip_prefix('<').and_then(|s| s.strip_suffix('>')) { |
| 1253 | return inner.to_string(); |
| 1254 | } |
| 1255 | unwrap_arg(t) |
| 1256 | } |
| 1257 | |
| 1258 | /// Reads an inline `#name(...)`/`#name[...]` call at `i` (a `#`) that no earlier handler claimed, so the |
| 1259 | /// reader can consume it whole rather than leak its raw markup. Returns the bracketed body's runs (parsed, |
| 1260 | /// so its own markup survives) when the call is a single `[...]` content group, `None` for the body when it |
| 1261 | /// carries only paren arguments, together with the index just past the call and its `#name` for the skip |
| 1262 | /// report. The final `None` is returned when `i` does not open a `#name(`/`#name[` call at all, so a bare |
| 1263 | /// `#` or a `#variable` interpolation is left as ordinary text. |
| 1264 | fn unknown_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) |
| 1265 | -> Option<(Option<Vec<Inline>>, usize, String)> |
| 1266 | { |
| 1267 | if chars.get(i) != Some(&'#') { |
| 1268 | return None; |
| 1269 | } |
| 1270 | let start = i + 1; |
| 1271 | let mut j = start; |
| 1272 | while j < chars.len() && (chars[j].is_ascii_alphanumeric() || chars[j] == '-' || chars[j] == '_' || chars[j] == '.') { |
| 1273 | j += 1; |
| 1274 | } |
| 1275 | if j == start { |
| 1276 | return None; |
| 1277 | } |
| 1278 | let name: String = chars[start..j].iter().collect(); |
| 1279 | match chars.get(j) { |
| 1280 | // A `#name[body]`: the bracketed content is the call's displayable body. |
| 1281 | Some('[') => { |
| 1282 | let Some((body, next)) = read_group(chars, j) else { return None; }; |
| 1283 | Some((Some(parse_inlines_in(&body, span, skips)), next, fmt!("#{}", name))) |
| 1284 | }, |
| 1285 | // A `#name(args)` and any following `[body]`: read the arguments away, then fold a body if one trails. |
| 1286 | Some('(') => { |
| 1287 | let Some((_, after_args)) = read_group(chars, j) else { return None; }; |
| 1288 | if chars.get(after_args) == Some(&'[') { |
| 1289 | let Some((body, next)) = read_group(chars, after_args) else { return None; }; |
| 1290 | return Some((Some(parse_inlines_in(&body, span, skips)), next, fmt!("#{}", name))); |
| 1291 | } |
| 1292 | Some((None, after_args, fmt!("#{}", name))) |
| 1293 | }, |
| 1294 | _ => None, |
| 1295 | } |
| 1296 | } |
| 1297 | |
| 1298 | /// The `#name` of a skipped line-leading code statement or standalone call, for the skip report: the |
| 1299 | /// keyword itself for a block statement (`#let`, `#set`, `#show`, `#import`), or `#` and the identifier of |
| 1300 | /// a standalone call. Falls back to the first whitespace-delimited token when neither shape reads, so the |
| 1301 | /// tally always names something rather than nothing. |
| 1302 | fn construct_name(trimmed: &str) -> String { |
| 1303 | for kw in ["#import", "#let", "#set", "#show"] { |
| 1304 | if trimmed.starts_with(kw) { |
| 1305 | return kw.to_string(); |
| 1306 | } |
| 1307 | } |
| 1308 | let mut cs = trimmed.chars(); |
| 1309 | if cs.next() == Some('#') { |
| 1310 | let mut ident = String::new(); |
| 1311 | for c in cs { |
| 1312 | if c.is_alphanumeric() || c == '-' || c == '_' || c == '.' { |
| 1313 | ident.push(c); |
| 1314 | } else { |
| 1315 | break; |
| 1316 | } |
| 1317 | } |
| 1318 | if !ident.is_empty() { |
| 1319 | return fmt!("#{}", ident); |
| 1320 | } |
| 1321 | } |
| 1322 | trimmed.split_whitespace().next().unwrap_or(trimmed).to_string() |
| 1323 | } |
| 1324 | |
| 1325 | /// If the literal `s` sits at `i` in `chars`, the index just past it; otherwise `None`. |
| 1326 | fn at_lit(chars: &[char], i: usize, s: &str) -> Option<usize> { |
| 1327 | let mut k = i; |
| 1328 | for ch in s.chars() { |
| 1329 | if chars.get(k) != Some(&ch) { |
| 1330 | return None; |
| 1331 | } |
| 1332 | k += 1; |
| 1333 | } |
| 1334 | Some(k) |
| 1335 | } |
| 1336 | |
| 1337 | /// One open delimiter context on the scanner's stack. Which frame is on top decides whether the next |
| 1338 | /// bracket is structural: a `(` inside a `[...]` content block is author prose, not nesting, and a `(` |
| 1339 | /// or `$` inside a `"..."` string or a `$...$` maths span never counts at all. This is what lets a caption |
| 1340 | /// whose prose carries an unbalanced `(` still close at its `]`, where a flat depth counter stuck open to |
| 1341 | /// end of source and swallowed the figure and everything after it. |
| 1342 | #[derive(Clone, Copy, PartialEq)] |
| 1343 | enum Frame { |
| 1344 | Code, // a `(...)`/`{...}`/`#name(...)` group, or the top level: brackets nest, `,`/`:` part |
| 1345 | Content, // a `[...]` content block: only `[` `]` nest; author `(` `)` `{` `}` are literal prose |
| 1346 | Str, // a `"..."` string literal: every character is literal until the closing quote |
| 1347 | Math, // a `$...$` maths span: every character is literal until the closing `$` |
| 1348 | Comment, // a `/* ... */` block comment: every character, brackets included, is literal until `*/` |
| 1349 | Raw, // a `` `...` `` code span: every character, `//`/`/*` included, is literal until the closing backtick |
| 1350 | } |
| 1351 | |
| 1352 | /// The running delimiter balance while a bracketed span is scanned. The stack of [`Frame`]s replaces the |
| 1353 | /// old flat `depth`: the span is closed when the stack is empty (was `depth <= 0`), and a multi-line code |
| 1354 | /// skip is still open while it is not. `escaped` records that the previous character was a `\` inside a |
| 1355 | /// string, maths span or content block, so a `\"`, `\$` or `\]` is passed over rather than closing its |
| 1356 | /// frame. Both persist across the lines of a span, since a frame may straddle the line break. |
| 1357 | pub(crate) struct SkipState { |
| 1358 | frames: Vec<Frame>, |
| 1359 | escaped: bool, |
| 1360 | in_quote: bool, // an odd number of literal `"` seen since the start of the current line, in Content mode |
| 1361 | } |
| 1362 | |
| 1363 | /// Does a `//` at `i` open a line comment, or is it a URL's double slash (`https://...`) and so literal? |
| 1364 | /// Mirrors the `://` exception in [`strip_comments`]: a `/` immediately after a `:` never starts a |
| 1365 | /// comment, in code or in content prose alike. |
| 1366 | fn is_line_comment(chars: &[char], i: usize) -> bool { |
| 1367 | chars.get(i + 1) == Some(&'/') && !(i > 0 && chars[i - 1] == ':') |
| 1368 | } |
| 1369 | |
| 1370 | /// How many characters a `//` line comment opened at `i` consumes. `chars` may be a whole multi-line |
| 1371 | /// capture buffer -- [`read_group`], [`split_top_args`] and [`named_arg`] all run on one -- so the comment |
| 1372 | /// is bounded to the next `'\n'`, not to the end of the slice; a `//` on one line must never eat the lines |
| 1373 | /// that follow it. |
| 1374 | fn line_comment_len(chars: &[char], i: usize) -> usize { |
| 1375 | match chars[i..].iter().position(|&c| c == '\n') { |
| 1376 | Some(off) => off, |
| 1377 | None => chars.len() - i, |
| 1378 | } |
| 1379 | } |
| 1380 | |
| 1381 | impl SkipState { |
| 1382 | pub(crate) fn new() -> Self { |
| 1383 | SkipState { frames: Vec::new(), escaped: false, in_quote: false } |
| 1384 | } |
| 1385 | |
| 1386 | /// Is any frame still open? The top-level test for [`read_group`], [`split_top_args`] and [`named_arg`], |
| 1387 | /// where a comma or colon parts only when nothing at all is open and a group closes when the stack empties. |
| 1388 | fn is_open(&self) -> bool { |
| 1389 | !self.frames.is_empty() |
| 1390 | } |
| 1391 | |
| 1392 | /// Is a structural bracket -- a `(`/`{`/`[` group -- still unclosed? This is the multi-line skip and |
| 1393 | /// capture test, matching the old flat `depth > 0`: a dangling `"` or `$` left open at the end of a line |
| 1394 | /// does not keep a construct open, since in prose a stray quote (an author's `"no bound"` split across |
| 1395 | /// two lines after an inline `#raw("...")`) or a lone `$` is a character, not the start of a code span. |
| 1396 | pub(crate) fn has_open_bracket(&self) -> bool { |
| 1397 | self.frames.iter().any(|f| matches!(f, Frame::Code | Frame::Content)) |
| 1398 | } |
| 1399 | |
| 1400 | /// How many structural `(`/`{`/`[` frames are nested right now -- the depth [`has_open_bracket`] only |
| 1401 | /// asks a yes/no of. A guard tracking its own single opening bracket uses this to tell its own matching |
| 1402 | /// closer (depth falls to 1) from an inner content block's closer (depth still above 1) on the same |
| 1403 | /// `]` text. |
| 1404 | pub(crate) fn open_brackets(&self) -> usize { |
| 1405 | self.frames.iter().filter(|f| matches!(f, Frame::Code | Frame::Content)).count() |
| 1406 | } |
| 1407 | |
| 1408 | /// Folds the character (or, in content mode, the `#ident` run) at `i` into the stack, returning how |
| 1409 | /// many characters were consumed from `chars` -- always at least one, more for a `#name(`/`#name[`/`#x` |
| 1410 | /// run whose opener decides the frame it enters. All four scanners share this one transition so a |
| 1411 | /// bracket is counted at exactly one place, whatever their outer loops do with the characters. |
| 1412 | fn step(&mut self, chars: &[char], i: usize) -> usize { |
| 1413 | let c = chars[i]; |
| 1414 | match self.frames.last().copied() { |
| 1415 | Some(Frame::Str) => { |
| 1416 | if self.escaped { self.escaped = false; } |
| 1417 | else if c == '\\' { self.escaped = true; } |
| 1418 | else if c == '"' { self.frames.pop(); } |
| 1419 | 1 |
| 1420 | }, |
| 1421 | Some(Frame::Math) => { |
| 1422 | if self.escaped { self.escaped = false; } |
| 1423 | else if c == '\\' { self.escaped = true; } |
| 1424 | else if c == '$' { self.frames.pop(); } |
| 1425 | 1 |
| 1426 | }, |
| 1427 | // A `/* ... */` block comment: every character, including a stray `}`/`]`/`)` an author's note |
| 1428 | // mentions, is literal until the comment's own closer -- the twin of Str/Math above, so a |
| 1429 | // `#context` guard's brace balance is never corrupted by a comment inside its body. |
| 1430 | Some(Frame::Comment) => { |
| 1431 | if c == '*' && chars.get(i + 1) == Some(&'/') { self.frames.pop(); 2 } |
| 1432 | else { 1 } |
| 1433 | }, |
| 1434 | // A `` `...` `` code span: literal until the closing backtick, the twin of Comment above, so a |
| 1435 | // `//`/`/*` a prose note quotes as a raw code token (`` the `//` operator ``) is never mistaken |
| 1436 | // for a comment opener. Mirrors [`strip_comments`]' `in_raw`, which does not persist an |
| 1437 | // unterminated span past its own line, so an unclosed backtick is dropped at the newline rather |
| 1438 | // than swallowing the lines that follow. |
| 1439 | Some(Frame::Raw) => { |
| 1440 | match c { |
| 1441 | '`' => { self.frames.pop(); 1 }, |
| 1442 | '\n' => { self.frames.pop(); 1 }, |
| 1443 | _ => 1, |
| 1444 | } |
| 1445 | }, |
| 1446 | Some(Frame::Content) => { |
| 1447 | // A `\`-escaped `\$ \[ \] \#` is literal content, so the escaped character is passed over |
| 1448 | // before any of the structural cases below can act on it. |
| 1449 | if self.escaped { |
| 1450 | self.escaped = false; |
| 1451 | return 1; |
| 1452 | } |
| 1453 | match c { |
| 1454 | '\\' => { self.escaped = true; 1 }, |
| 1455 | '[' => { self.frames.push(Frame::Content); 1 }, |
| 1456 | ']' => { self.frames.pop(); 1 }, |
| 1457 | '$' => { self.frames.push(Frame::Math); 1 }, |
| 1458 | '#' => self.content_hash(chars, i), |
| 1459 | '`' => { self.frames.push(Frame::Raw); 1 }, |
| 1460 | // A literal `"` in prose is not a string (content mode never opens `Frame::Str`), but |
| 1461 | // `strip_comments` still treats a quoted phrase as opaque to `//`/`/*`, so a bare count |
| 1462 | // mirrors that here without disturbing the bracket balance a real quote would otherwise |
| 1463 | // leave alone. Line-scoped, as `strip_comments` is called once per line. |
| 1464 | '"' => { self.in_quote = !self.in_quote; 1 }, |
| 1465 | '\n' => { self.in_quote = false; 1 }, |
| 1466 | // A line comment runs to the next `\n` in `chars` (which may hold a whole multi-line |
| 1467 | // capture buffer, not just this one line) -- never past it, and never at all inside a |
| 1468 | // quoted phrase or a raw span. The `://` exception mirrors `strip_comments`, so a bare |
| 1469 | // URL's slashes stay literal prose. A block comment opens a `Comment` frame that can |
| 1470 | // straddle the line break, same as Str/Math above. |
| 1471 | '/' if is_line_comment(chars, i) && !self.in_quote => line_comment_len(chars, i), |
| 1472 | '/' if chars.get(i + 1) == Some(&'*') && !self.in_quote => { self.frames.push(Frame::Comment); 2 }, |
| 1473 | // A `(` `)` `{` `}` in content mode is author prose, never nesting: this is the whole |
| 1474 | // point of tracking the frame, so a caption's unbalanced paren does not stick. |
| 1475 | _ => 1, |
| 1476 | } |
| 1477 | }, |
| 1478 | // A code frame, or the top level (an empty stack): brackets nest as the flat counter had them, |
| 1479 | // the closer kind is not checked, and a `[` opens a content child, a `$` a maths span. A `"` |
| 1480 | // opens a real `Str` frame here, which already keeps a `//`/`/*` inside it literal, so no |
| 1481 | // separate quote count is needed the way Content mode's prose-only quote does. |
| 1482 | _ => { |
| 1483 | match c { |
| 1484 | '"' => { self.frames.push(Frame::Str); 1 }, |
| 1485 | '`' => { self.frames.push(Frame::Raw); 1 }, |
| 1486 | '(' | '{' => { self.frames.push(Frame::Code); 1 }, |
| 1487 | '[' => { self.frames.push(Frame::Content); 1 }, |
| 1488 | '$' => { self.frames.push(Frame::Math); 1 }, |
| 1489 | ')' | '}' => { self.frames.pop(); 1 }, |
| 1490 | '/' if is_line_comment(chars, i) => line_comment_len(chars, i), |
| 1491 | '/' if chars.get(i + 1) == Some(&'*') => { self.frames.push(Frame::Comment); 2 }, |
| 1492 | _ => 1, |
| 1493 | } |
| 1494 | }, |
| 1495 | } |
| 1496 | } |
| 1497 | |
| 1498 | /// Handles a `#` met in content mode: a `#name` identifier follows, and its first non-identifier |
| 1499 | /// character decides the frame -- `(` opens the call's code arguments, `[` a content block, anything |
| 1500 | /// else (or end of input) is a bare `#name` field access with no group. Returns the count consumed: |
| 1501 | /// the `#`, the identifier, and, for a call or content opener, that opener too. |
| 1502 | fn content_hash(&mut self, chars: &[char], i: usize) -> usize { |
| 1503 | let mut j = i + 1; |
| 1504 | while j < chars.len() && is_call_ident(chars[j]) { |
| 1505 | j += 1; |
| 1506 | } |
| 1507 | match chars.get(j) { |
| 1508 | Some('(') => { self.frames.push(Frame::Code); j + 1 - i }, |
| 1509 | Some('[') => { self.frames.push(Frame::Content); j + 1 - i }, |
| 1510 | _ => j - i, // a bare `#name` (or a lone `#`): open no frame |
| 1511 | } |
| 1512 | } |
| 1513 | } |
| 1514 | |
| 1515 | /// What to do with a line-leading Typst code statement or standalone template call. |
| 1516 | enum CodeSkip { |
| 1517 | Line, // the call closes on this line; skip the one line, as before |
| 1518 | Multi(SkipState), // the delimiters are still open; begin a multi-line skip carrying the depth |
| 1519 | } |
| 1520 | |
| 1521 | /// If this already-left-trimmed line begins a Typst code statement Austenite skips for now, decides how |
| 1522 | /// much to skip: `Line` for a statement or standalone call that closes on this line, `Multi` for one |
| 1523 | /// whose delimiters are still open at the end of it. `None` when the line is not code the reader skips, |
| 1524 | /// so the caller sets it as prose. |
| 1525 | /// |
| 1526 | /// The four block statements (`#import`, `#let`, `#set`, `#show`) are always code; a line-leading call |
| 1527 | /// (`#name(` or `#name[`) is skipped only as a whole -- either it closes on the line, or it opens a |
| 1528 | /// multi-line span. A balanced `#name[...]` with prose trailing it (`#index-main[x]More prose...`) is |
| 1529 | /// left to set, since its content is a marker within a real paragraph, not a standalone call. |
| 1530 | fn code_skip(trimmed: &str) -> Option<CodeSkip> { |
| 1531 | let keyword = code_keyword(trimmed); |
| 1532 | if !keyword && !opens_standalone_call(trimmed) { |
| 1533 | return None; |
| 1534 | } |
| 1535 | let mut state = SkipState::new(); |
| 1536 | scan_brackets(trimmed, &mut state); |
| 1537 | if state.has_open_bracket() { |
| 1538 | return Some(CodeSkip::Multi(state)); |
| 1539 | } |
| 1540 | // The delimiters balance on this line. A block statement is skipped whatever trails it; a standalone |
| 1541 | // call is skipped only when it truly ends with its own closer, so a marker inside a paragraph sets. |
| 1542 | // The `}` closer is the code-block call form (`#context{ ... }`) that closes on its own line. |
| 1543 | if keyword || trimmed.ends_with(')') || trimmed.ends_with(']') || trimmed.ends_with('}') { |
| 1544 | return Some(CodeSkip::Line); |
| 1545 | } |
| 1546 | None |
| 1547 | } |
| 1548 | |
| 1549 | /// Does this already-left-trimmed line open one of the Typst block statements the reader skips? |
| 1550 | /// |
| 1551 | /// `#include` sits here too, guarding against the shape it would otherwise fall through to: it takes a |
| 1552 | /// bare string argument, with neither a `(` nor a `[` for [`opens_standalone_call`] to catch, so with no |
| 1553 | /// entry here it reads as an ordinary paragraph line and its raw `#include "path"` prints as literal body |
| 1554 | /// text. The book assembler ([`crate::book::assemble`]) resolves and follows a real `#include` itself, |
| 1555 | /// before a chapter's source ever reaches this parser -- this is the belt-and-braces net for any source |
| 1556 | /// that bypasses that assembler (a bare `to_blocks`/`to_blocks_with_templates` call, a test fixture): the |
| 1557 | /// line is skipped and recorded rather than ever standing a chance of being set as prose. |
| 1558 | fn code_keyword(trimmed: &str) -> bool { |
| 1559 | for kw in ["#import ", "#import\"", "#let ", "#set ", "#show ", "#show:", "#include ", "#include\""] { |
| 1560 | if trimmed.starts_with(kw) { |
| 1561 | return true; |
| 1562 | } |
| 1563 | } |
| 1564 | false |
| 1565 | } |
| 1566 | |
| 1567 | /// Does this already-left-trimmed line open with a standalone call -- `#`, an identifier, then `(`, `[` |
| 1568 | /// or `{`? A crude test, enough to recognise the opener of a call to an unrecognised template function |
| 1569 | /// without inspecting where or whether it closes; the balance decides single- versus multi-line. |
| 1570 | /// |
| 1571 | /// The `{` opener catches the code-block call form -- `#context { ... }`, the brace twin of |
| 1572 | /// `#context[ ... ]` -- which is always Typst code at a line's head (a `{` in content mode is literal |
| 1573 | /// prose, never a call opener). Typst attaches such a block to its keyword with optional whitespace |
| 1574 | /// (`#context {`, the shape a book's reverse-reference index is written with, Lucronics ch29.8), so a run |
| 1575 | /// of spaces before the `{` is skipped; a space before `(`, by contrast, breaks a Typst call, so only an |
| 1576 | /// immediate `(` counts. No inline-call name is written with a `{`, so the `is_inline_call` guard, which |
| 1577 | /// still holds off the glossary/index family, never fires on the brace form. |
| 1578 | fn opens_standalone_call(trimmed: &str) -> bool { |
| 1579 | let mut cs = trimmed.chars(); |
| 1580 | if cs.next() != Some('#') { |
| 1581 | return false; |
| 1582 | } |
| 1583 | let mut ident = String::new(); |
| 1584 | let mut after = None; // the first character past the identifier |
| 1585 | for c in cs { |
| 1586 | if c.is_alphanumeric() || c == '-' || c == '_' { |
| 1587 | ident.push(c); |
| 1588 | continue; |
| 1589 | } |
| 1590 | after = Some(c); |
| 1591 | break; |
| 1592 | } |
| 1593 | if ident.is_empty() || is_inline_call(&ident) { |
| 1594 | // A line-leading inline glossary or index call is content, not a skippable standalone call, even |
| 1595 | // when it closes on its own line, so [`parse_inlines`] sets its display text rather than dropping it. |
| 1596 | return false; |
| 1597 | } |
| 1598 | match after { |
| 1599 | Some('(') | Some('[') | Some('{') => true, |
| 1600 | // A `{` code block may follow the keyword across whitespace (`#context {`); a `(` may not, since a |
| 1601 | // space before it breaks a Typst call. The second whitespace-split word is the block opener. |
| 1602 | Some(c) if c.is_whitespace() => |
| 1603 | trimmed.split_whitespace().nth(1).map(|w| w.starts_with('{')).unwrap_or(false), |
| 1604 | _ => false, |
| 1605 | } |
| 1606 | } |
| 1607 | |
| 1608 | /// Is this already-left-trimmed line a line-leading code-mode reference the reader cannot run and must |
| 1609 | /// refuse rather than leak as prose? Recognises an anonymous `#{ ... }`/`#( ... )` block, an `#if`/`#for`/ |
| 1610 | /// `#while` control keyword, a field or method access `#name.foo`, and a bare `#name` (with only whitespace |
| 1611 | /// after -- a `// comment` is stripped upstream). A `#name(`/`#name[` standalone call and the `#let`/`#set`/ |
| 1612 | /// `#show`/`#import` keywords are refused earlier by [`code_skip`], and a bound `#name` is expanded by |
| 1613 | /// [`capture_opener`], so this catches exactly what is left -- above all a `#if`/`#{` surfacing in a re-read |
| 1614 | /// content-binding body, which must be refused, not set with its leading `#`. |
| 1615 | fn is_code_reference(trimmed: &str) -> bool { |
| 1616 | let rest = match trimmed.strip_prefix('#') { |
| 1617 | Some(r) => r, |
| 1618 | None => return false, |
| 1619 | }; |
| 1620 | let chars: Vec<char> = rest.chars().collect(); |
| 1621 | // An anonymous code block or expression. |
| 1622 | if matches!(chars.first(), Some('{') | Some('(')) { |
| 1623 | return true; |
| 1624 | } |
| 1625 | // A control keyword: `#if`/`#for`/`#while` followed by whitespace, `(` or `{`. |
| 1626 | for kw in ["if", "for", "while"] { |
| 1627 | if let Some(after) = rest.strip_prefix(kw) { |
| 1628 | match after.chars().next() { |
| 1629 | Some(c) if c.is_whitespace() || c == '(' || c == '{' => return true, |
| 1630 | _ => {}, |
| 1631 | } |
| 1632 | } |
| 1633 | } |
| 1634 | // A bare identifier, or a field/method access on one. |
| 1635 | let name_len = chars.iter().take_while(|&&c| c.is_alphanumeric() || c == '-' || c == '_').count(); |
| 1636 | if name_len == 0 { |
| 1637 | return false; |
| 1638 | } |
| 1639 | match chars.get(name_len) { |
| 1640 | None => true, // a bare `#name` |
| 1641 | Some('.') => true, // a field or method access `#name.foo` |
| 1642 | // A bare `#name` with only whitespace trailing it (a comment was stripped upstream); a `#name` with |
| 1643 | // trailing prose is an inline reference the paragraph keeps, and a `#name(`/`#name[` is a call handled |
| 1644 | // elsewhere. |
| 1645 | Some(c) if c.is_whitespace() => chars[name_len..].iter().all(|c| c.is_whitespace()), |
| 1646 | _ => false, |
| 1647 | } |
| 1648 | } |
| 1649 | |
| 1650 | /// Does this already-left-trimmed standalone line consist of a bare `#name` that names a scalar `#let` |
| 1651 | /// binding in scope? Only a bare reference -- an identifier with nothing but whitespace after it -- with a |
| 1652 | /// name `sfns` actually binds qualifies; a `#name.field` access, a `#name(`/`#name[` call and an unbound |
| 1653 | /// name all return false. It exempts exactly the standalone scalar reference from the code-reference skip |
| 1654 | /// so its value is substituted (through the paragraph's own [`substitute_scalars`]), while every other |
| 1655 | /// standalone code reference, above all an unbound name, stays the visible refusal it is. |
| 1656 | fn names_scalar_alone(trimmed: &str, sfns: &crate::lang::rules::ScalarFns) -> bool { |
| 1657 | let rest = match trimmed.strip_prefix('#') { |
| 1658 | Some(r) => r, |
| 1659 | None => return false, |
| 1660 | }; |
| 1661 | let chars: Vec<char> = rest.chars().collect(); |
| 1662 | let name_len = chars.iter().take_while(|&&c| c.is_alphanumeric() || c == '-' || c == '_').count(); |
| 1663 | if name_len == 0 { |
| 1664 | return false; |
| 1665 | } |
| 1666 | // A bare `#name`: nothing but whitespace after the identifier (no `.field`, no `(args)`, no `[body]`). |
| 1667 | if !chars[name_len..].iter().all(|c| c.is_whitespace()) { |
| 1668 | return false; |
| 1669 | } |
| 1670 | let name: String = chars[..name_len].iter().collect(); |
| 1671 | sfns.contains_key(&name) |
| 1672 | } |
| 1673 | |
| 1674 | /// Is this identifier one of the book template's inline functions the reader sets in place -- a glossary |
| 1675 | /// or index call, a term-dictionary lookup, a hyperlink, a citation or an emphasis call? These emit body |
| 1676 | /// text (or an invisible marker) mid-paragraph, so a line that opens with one is prose the inline scanner |
| 1677 | /// reads, never a standalone call the line scanner skips. |
| 1678 | fn is_inline_call(name: &str) -> bool { |
| 1679 | matches!(name, |
| 1680 | // The simple string-keyed glossary/index family, keyed on their own display text. |
| 1681 | "gs" | "gscap" | "gsi" | "gscapi" | "glossind" | "glossindcap" |
| 1682 | // The term-dictionary family, keyed on a `term-dict` entry: `g`/`gcap`/`gi`/`gcapi` set the value |
| 1683 | // with first-use styling, `t`/`tcap` set it plain, `graw` sets it in the mono face. |
| 1684 | | "g" | "gcap" | "gi" | "gcapi" | "t" | "tcap" | "graw" |
| 1685 | | "idx" | "idx-main" | "idx-as" | "idx-main-as" | "idx-nested" |
| 1686 | | "index" | "index-main" | "cite" | "link" |
| 1687 | | "emph" | "strong" | "super" | "sub" |
| 1688 | | "claim-label" | "claim-refs") |
| 1689 | } |
| 1690 | |
| 1691 | /// Folds one line's delimiters into the running [`SkipState`]. A bracket inside a `"..."` string, a `$...$` |
| 1692 | /// maths span or a `[...]` content block is not counted as structural nesting; the frame stack decides. |
| 1693 | /// The state carries into the next line, so a frame that straddles the break is tracked correctly. |
| 1694 | pub(crate) fn scan_brackets(line: &str, state: &mut SkipState) { |
| 1695 | let chars: Vec<char> = line.chars().collect(); |
| 1696 | let mut i = 0; |
| 1697 | while i < chars.len() { |
| 1698 | i += state.step(&chars, i); |
| 1699 | } |
| 1700 | } |
| 1701 | |
| 1702 | /// Splits a trailing `<label>` off a heading title: a `<name>` with no inner whitespace at the very end |
| 1703 | /// labels the heading and is removed from its text. A title that merely contains angle brackets, or a |
| 1704 | /// `< >` with a space inside, keeps them as ordinary characters. |
| 1705 | fn split_label(title: &str) -> (String, Option<String>) { |
| 1706 | let t = title.trim_end(); |
| 1707 | if let Some(inner) = t.strip_suffix('>') { |
| 1708 | if let Some(p) = inner.rfind('<') { |
| 1709 | let label = &inner[p + 1..]; |
| 1710 | if !label.is_empty() && !label.contains(char::is_whitespace) { |
| 1711 | return (inner[..p].trim_end().to_string(), Some(label.to_string())); |
| 1712 | } |
| 1713 | } |
| 1714 | } |
| 1715 | (t.to_string(), None) |
| 1716 | } |
| 1717 | |
| 1718 | /// Reads an inline `#footnote[...]` at `i` (a `#`), returning the note's inline markup -- parsed so a |
| 1719 | /// `*strong*` or `_emph_` in the note sets with its own face -- and the index just past the closing `]`. |
| 1720 | /// `None` when the shape is not a footnote call or its bracket does not close, so anything else is left as |
| 1721 | /// ordinary text. |
| 1722 | fn footnote_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) -> Option<(Vec<Inline>, usize)> { |
| 1723 | let Some(open) = at_lit(chars, i, "#footnote") else { return None; }; |
| 1724 | if chars.get(open) != Some(&'[') { |
| 1725 | return None; |
| 1726 | } |
| 1727 | let Some((inner, next)) = read_group(chars, open) else { return None; }; |
| 1728 | Some((parse_inlines_in(&inner, span, skips), next)) |
| 1729 | } |
| 1730 | |
| 1731 | /// Reads an inline `#emph[...]` at `i` (a `#`), returning its inner markup unreduced -- it is the call |
| 1732 | /// form of `_..._` and the caller expands it the same way -- and the index just past the closing `]`. |
| 1733 | /// `None` when the shape is not an emph call or its bracket does not close, so anything else is left as |
| 1734 | /// ordinary text. |
| 1735 | fn emph_call(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1736 | let open = at_lit(chars, i, "#emph")?; |
| 1737 | if chars.get(open) != Some(&'[') { |
| 1738 | return None; |
| 1739 | } |
| 1740 | read_group(chars, open) |
| 1741 | } |
| 1742 | |
| 1743 | /// Reads an inline `#strong[...]` or `#strong("...")` at `i` (a `#`), returning the text to set bold and |
| 1744 | /// the index just past the closing bracket. The bracket form's content is markup, unreduced -- it is the |
| 1745 | /// call form of `*...*` and the caller expands it the same way emph does. The paren form's argument is |
| 1746 | /// only resolved when it is a plain `"..."` string; a bare identifier or any other expression is not a |
| 1747 | /// text this reader can evaluate, so `None` is returned and the generic call handler records the refusal |
| 1748 | /// instead of guessing. `None` also when the shape is not a strong call or its group does not close. |
| 1749 | fn strong_call(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1750 | let open = at_lit(chars, i, "#strong")?; |
| 1751 | match chars.get(open) { |
| 1752 | Some('[') => read_group(chars, open), |
| 1753 | Some('(') => { |
| 1754 | let (inner, next) = read_group(chars, open)?; |
| 1755 | let t = inner.trim(); |
| 1756 | if t.len() >= 2 && t.starts_with('"') && t.ends_with('"') { |
| 1757 | Some((t[1..t.len() - 1].to_string(), next)) |
| 1758 | } else { |
| 1759 | None |
| 1760 | } |
| 1761 | }, |
| 1762 | _ => None, |
| 1763 | } |
| 1764 | } |
| 1765 | |
| 1766 | /// Reads an inline `#super[...]` or `#super("...")` at `i` (a `#`), returning its content reduced to |
| 1767 | /// display text by [`flatten_markup`] -- usually a short string or number -- and the index just past the |
| 1768 | /// closing bracket. `None` when the shape is not a super call or its argument does not close. |
| 1769 | fn super_call(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1770 | let open = at_lit(chars, i, "#super")?; |
| 1771 | match chars.get(open) { |
| 1772 | Some('[') | Some('(') => {}, |
| 1773 | _ => return None, |
| 1774 | } |
| 1775 | let (inner, next) = read_group(chars, open)?; |
| 1776 | Some((flatten_markup(&unwrap_arg(&inner)), next)) |
| 1777 | } |
| 1778 | |
| 1779 | /// Reads an inline `#sub[...]` or `#sub("...")` at `i` (a `#`), returning its content reduced to display |
| 1780 | /// text by [`flatten_markup`] -- usually a short string or number, as in `CO#sub[2]` -- and the index just |
| 1781 | /// past the closing bracket. `None` when the shape is not a sub call or its argument does not close. |
| 1782 | fn sub_call(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1783 | let open = at_lit(chars, i, "#sub")?; |
| 1784 | match chars.get(open) { |
| 1785 | Some('[') | Some('(') => {}, |
| 1786 | _ => return None, |
| 1787 | } |
| 1788 | let (inner, next) = read_group(chars, open)?; |
| 1789 | Some((flatten_markup(&unwrap_arg(&inner)), next)) |
| 1790 | } |
| 1791 | |
| 1792 | /// Reads an inline `#smallcaps[...]` or `#smallcaps("...")` at `i` (a `#`), returning its content reduced |
| 1793 | /// to display text by [`flatten_markup`] and the index just past the closing bracket. `None` when the shape |
| 1794 | /// is not a smallcaps call or its argument does not close. |
| 1795 | fn smallcaps_call(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 1796 | let Some(open) = at_lit(chars, i, "#smallcaps") else { return None; }; |
| 1797 | match chars.get(open) { |
| 1798 | Some('[') | Some('(') => {}, |
| 1799 | _ => return None, |
| 1800 | } |
| 1801 | read_group(chars, open).map(|(inner, next)| (flatten_markup(&unwrap_arg(&inner)), next)) |
| 1802 | } |
| 1803 | |
| 1804 | /// Reads an inline `#cite(...)` at `i` (a `#`), returning the citation keys and the index past the |
| 1805 | /// closing `)`. Every `<label>` token inside the parentheses is a key; a named argument such as |
| 1806 | /// `form: "prose"` carries no label and is ignored. `None` when the shape is not a cite call or its |
| 1807 | /// parentheses do not close, so anything else is left as ordinary text. |
| 1808 | fn cite_call(chars: &[char], i: usize) -> Option<(Vec<String>, usize)> { |
| 1809 | let open = at_lit(chars, i, "#cite")?; |
| 1810 | if chars.get(open) != Some(&'(') { |
| 1811 | return None; |
| 1812 | } |
| 1813 | let (inner, next) = read_group(chars, open)?; |
| 1814 | let keys = cite_keys(&inner); |
| 1815 | Some((keys, next)) |
| 1816 | } |
| 1817 | |
| 1818 | /// Extracts the `<label>` citation keys from the inside of a `#cite(...)` call, in order. A `<` opens a |
| 1819 | /// key and the next `>` closes it; anything outside a `<...>` pair (a named argument, a separating comma) |
| 1820 | /// is skipped. |
| 1821 | fn cite_keys(inner: &str) -> Vec<String> { |
| 1822 | let chars: Vec<char> = inner.chars().collect(); |
| 1823 | let mut keys = Vec::new(); |
| 1824 | let mut i = 0usize; |
| 1825 | while i < chars.len() { |
| 1826 | if chars[i] == '<' { |
| 1827 | if let Some(close) = (i + 1..chars.len()).find(|&j| chars[j] == '>') { |
| 1828 | let key: String = chars[i + 1..close].iter().collect(); |
| 1829 | let key = key.trim().to_string(); |
| 1830 | if !key.is_empty() { |
| 1831 | keys.push(key); |
| 1832 | } |
| 1833 | i = close + 1; |
| 1834 | continue; |
| 1835 | } |
| 1836 | } |
| 1837 | i += 1; |
| 1838 | } |
| 1839 | keys |
| 1840 | } |
| 1841 | |
| 1842 | /// Reads an inline `#claim-label(...)` or `#claim-refs(...)` at `i` (a `#`), returning the compressed |
| 1843 | /// margin code it sets, the raw reference codes it registers for the reverse claim index, and the index |
| 1844 | /// just past the closing `)`. `#claim-label(..codes)` places a compressed code string in the outside |
| 1845 | /// margin, so its display is the codes joined with a run of three or more consecutive same-prefix codes |
| 1846 | /// collapsed to a range (`B1 B2 B3 B4` -> `B1–4`), matching the book's `claims.typ` `_compress-codes`; |
| 1847 | /// `#claim-refs(..codes)` sets no visible margin code (empty display). Both register each of their raw codes |
| 1848 | /// for the reverse index, matching `claims.typ`, where a label and a bare reference both emit the |
| 1849 | /// `<claim-ref>` metadata `collect-claim-refs()` queries -- so a code contributes to the page list whether it |
| 1850 | /// was labelled or merely referenced. Neither sets anything in the body text column, so the surrounding prose |
| 1851 | /// closes over the marker's place as Typst's own body flow does. The outer `None` is returned when the shape |
| 1852 | /// is not a claim call or its parentheses do not close. |
| 1853 | fn claim_call(chars: &[char], i: usize) -> Option<(String, Vec<String>, usize)> { |
| 1854 | let (open, is_label) = match at_lit(chars, i, "#claim-label") { |
| 1855 | Some(o) => (o, true), |
| 1856 | None => (at_lit(chars, i, "#claim-refs")?, false), |
| 1857 | }; |
| 1858 | if chars.get(open) != Some(&'(') { |
| 1859 | return None; |
| 1860 | } |
| 1861 | let (inner, next) = read_group(chars, open)?; |
| 1862 | let codes = claim_codes(&inner); |
| 1863 | let display = if is_label { compress_codes(&codes) } else { String::new() }; |
| 1864 | Some((display, codes, next)) |
| 1865 | } |
| 1866 | |
| 1867 | /// The claim codes named inside a `#claim-label`/`#claim-refs` argument list, in source order: each |
| 1868 | /// `<name>` label's name and each `"string"` argument's text, a `<...>` or `"..."` wrapper stripped and |
| 1869 | /// anything else taken as written -- `claims.typ`'s own `str(c)` fallback for a bare argument. |
| 1870 | fn claim_codes(inner: &str) -> Vec<String> { |
| 1871 | let mut out = Vec::new(); |
| 1872 | for arg in split_claim_args(inner) { |
| 1873 | let a = arg.trim(); |
| 1874 | if a.is_empty() { |
| 1875 | continue; |
| 1876 | } |
| 1877 | let code = if let Some(rest) = a.strip_prefix('<') { |
| 1878 | rest.strip_suffix('>').unwrap_or(rest).trim().to_string() |
| 1879 | } else if a.starts_with('"') { |
| 1880 | unwrap_arg(a) |
| 1881 | } else { |
| 1882 | a.to_string() |
| 1883 | }; |
| 1884 | if !code.is_empty() { |
| 1885 | out.push(code); |
| 1886 | } |
| 1887 | } |
| 1888 | out |
| 1889 | } |
| 1890 | |
| 1891 | /// Splits a claim argument list on its top-level commas, holding a `<...>`, `(...)` or `[...]` nesting and |
| 1892 | /// a `"..."` string together so a comma inside one does not split an argument. |
| 1893 | fn split_claim_args(inner: &str) -> Vec<String> { |
| 1894 | let mut out = Vec::new(); |
| 1895 | let mut cur = String::new(); |
| 1896 | let mut depth = 0i32; |
| 1897 | let mut in_str = false; |
| 1898 | for c in inner.chars() { |
| 1899 | match c { |
| 1900 | '"' => { in_str = !in_str; cur.push(c); }, |
| 1901 | '<' | '(' | '[' if !in_str => { depth += 1; cur.push(c); }, |
| 1902 | '>' | ')' | ']' if !in_str => { depth -= 1; cur.push(c); }, |
| 1903 | ',' if depth == 0 && !in_str => out.push(std::mem::take(&mut cur)), |
| 1904 | _ => cur.push(c), |
| 1905 | } |
| 1906 | } |
| 1907 | if !cur.trim().is_empty() { |
| 1908 | out.push(cur); |
| 1909 | } |
| 1910 | out |
| 1911 | } |
| 1912 | |
| 1913 | /// The margin display for a set of claim codes, a port of `claims.typ`'s `_compress-codes`: two or fewer |
| 1914 | /// codes join with a space unchanged; a run of three or more consecutive codes sharing a letter prefix and |
| 1915 | /// ascending by one collapses to a `first–lastnum` range (an en dash, `B1 B2 B3 B4` -> `B1–4`), a run of |
| 1916 | /// exactly two is left expanded, and a code that does not parse as letters-then-digits is passed through. |
| 1917 | fn compress_codes(codes: &[String]) -> String { |
| 1918 | if codes.len() <= 2 { |
| 1919 | return codes.join(" "); |
| 1920 | } |
| 1921 | // (prefix, number) for each code that matches `^([A-Za-z]+)(\d+)$`, else `None` for one that does not. |
| 1922 | let parsed: Vec<Option<(String, u64)>> = codes.iter().map(|s| parse_code(s)).collect(); |
| 1923 | let mut result: Vec<String> = Vec::new(); |
| 1924 | let mut i = 0usize; |
| 1925 | while i < codes.len() { |
| 1926 | let prefix = match &parsed[i] { |
| 1927 | Some((p, _)) => p.clone(), |
| 1928 | None => { result.push(codes[i].clone()); i += 1; continue; }, |
| 1929 | }; |
| 1930 | // The longest run from `i` sharing the prefix and ascending by one, per the Typst helper. |
| 1931 | let mut run_end = i; |
| 1932 | while run_end + 1 < codes.len() { |
| 1933 | match (&parsed[run_end + 1], &parsed[run_end]) { |
| 1934 | (Some((np, nn)), Some((_, cn))) if *np == prefix && *nn == cn + 1 => run_end += 1, |
| 1935 | _ => break, |
| 1936 | } |
| 1937 | } |
| 1938 | if run_end > i + 1 { |
| 1939 | let last_num = match &parsed[run_end] { Some((_, n)) => *n, None => 0 }; |
| 1940 | result.push(fmt!("{}\u{2013}{}", codes[i], last_num)); |
| 1941 | } else if run_end > i { |
| 1942 | result.push(codes[i].clone()); |
| 1943 | result.push(codes[run_end].clone()); |
| 1944 | } else { |
| 1945 | result.push(codes[i].clone()); |
| 1946 | } |
| 1947 | i = run_end + 1; |
| 1948 | } |
| 1949 | result.join(" ") |
| 1950 | } |
| 1951 | |
| 1952 | /// A claim code split into its letter prefix and its number, as `claims.typ`'s `^([A-Za-z]+)(\d+)$` match |
| 1953 | /// does: `None` when the code is not one or more ASCII letters followed by one or more ASCII digits and |
| 1954 | /// nothing else. |
| 1955 | fn parse_code(s: &str) -> Option<(String, u64)> { |
| 1956 | let s = s.trim(); |
| 1957 | let split = s.find(|c: char| c.is_ascii_digit())?; |
| 1958 | let (pre, num) = s.split_at(split); |
| 1959 | if pre.is_empty() || !pre.chars().all(|c| c.is_ascii_alphabetic()) { |
| 1960 | return None; |
| 1961 | } |
| 1962 | if num.is_empty() || !num.chars().all(|c| c.is_ascii_digit()) { |
| 1963 | return None; |
| 1964 | } |
| 1965 | num.parse::<u64>().ok().map(|n| (pre.to_string(), n)) |
| 1966 | } |
| 1967 | |
| 1968 | /// Reduces a run of markup to plain display text: the words a reader sees, with the emphasis, code, |
| 1969 | /// glossary and index delimiters removed. A glossary term contributes its display, a visible index call |
| 1970 | /// its text, a pure index marker nothing; `*strong*` and `_emph_` contribute their inner words. Inline |
| 1971 | /// maths and cross-references, which have no plain form here, contribute nothing. Used where the engine |
| 1972 | /// takes a plain string -- a footnote's note, a table cell, a figure caption -- and cannot yet carry the |
| 1973 | /// runs themselves. |
| 1974 | pub fn flatten_markup(text: &str) -> String { |
| 1975 | let mut out = String::new(); |
| 1976 | for run in parse_inlines(text) { |
| 1977 | match run { |
| 1978 | Inline::Text(t) => out.push_str(&t), |
| 1979 | // A `*strong*` or `_emph_` inner run may still carry markup -- `*_word_*` nests emphasis in |
| 1980 | // strong -- so it is flattened again to strip the inner delimiters. Parsing does not nest, so |
| 1981 | // each pass removes one layer and the recursion terminates on plain text. |
| 1982 | Inline::Strong(t) => out.push_str(&flatten_markup(&t)), |
| 1983 | Inline::Emph(t) => out.push_str(&flatten_markup(&t)), |
| 1984 | Inline::BoldItalic(t) => out.push_str(&t), // already the flat inner of a nested run |
| 1985 | Inline::Super(t) => out.push_str(&t), // a flattened string cannot raise; keep its text |
| 1986 | Inline::Sub(t) => out.push_str(&t), // a flattened string cannot drop; keep its text |
| 1987 | Inline::SmallCaps(t) => out.push_str(&t), // a flattened string has no small capitals; keep its text |
| 1988 | Inline::Code(t) => out.push_str(&t), |
| 1989 | Inline::Glossary { display, .. } => out.push_str(&display), |
| 1990 | Inline::PageRef(_) => {}, // a page number has no plain form before layout |
| 1991 | Inline::Math(_) => {}, // maths is dropped from a flattened string |
| 1992 | Inline::Footnote(_) => {}, // a nested footnote is not set within a flattened string |
| 1993 | Inline::Cite(_) => {}, // a citation has no plain form before the bibliography resolves it |
| 1994 | Inline::MarginNote { .. } => {}, // a margin code is not part of the flattened body text |
| 1995 | Inline::Index { .. } => {}, // an index marker sets no words in the body text |
| 1996 | } |
| 1997 | } |
| 1998 | out |
| 1999 | } |
| 2000 | |
| 2001 | /// The index term a call records, with a nested entry's child term where one was given. |
| 2002 | struct IndexKey { |
| 2003 | term: String, // the sort key (markup flattened), e.g. "March, James" |
| 2004 | sub: Option<String>, |
| 2005 | display: String, // the display markup the index page sets, e.g. "James March" or "_Browder v. Gayle_" |
| 2006 | main: bool, // a primary reference (`idx-main`/`index-main`/`idx-main-as`); its folio sets bold |
| 2007 | } |
| 2008 | |
| 2009 | /// What an inline glossary or index call sets into the running text. `index` carries the term the call |
| 2010 | /// adds to the back-matter index, `None` for a glossary-only call (`g`/`gs`/`gscap`/`gcap`) that indexes |
| 2011 | /// nothing. |
| 2012 | enum Call { |
| 2013 | Glossary { term: String, display: String, index: Option<IndexKey> }, // a glossary term, keyed by `term` for first-use styling |
| 2014 | Visible { display: String, index: Option<IndexKey> }, // display text set plain, its markup parsed by the caller |
| 2015 | Invisible { index: Option<IndexKey> }, // a pure index marker: nothing is set but the term is recorded |
| 2016 | } |
| 2017 | |
| 2018 | /// Reads an inline glossary or index call at `i` (a `#`), returning what it sets and the index just past |
| 2019 | /// it, or `None` when the `#name` is not one the reader knows or its argument brackets do not close. |
| 2020 | /// |
| 2021 | /// The visible glossary functions set their bracket content as the term, capitalising the display for |
| 2022 | /// the `-cap` variants; `idx`/`idx-main` set the content plain; `idx-as`/`idx-main-as` take a second |
| 2023 | /// argument as the display and set that; `index`/`index-main`/`idx-nested` are pure markers and set |
| 2024 | /// nothing. First use is keyed by the term as written, matching the template's own case-sensitive |
| 2025 | /// `glossary-seen` set. |
| 2026 | /// |
| 2027 | /// The term-dictionary family keys a `term-dict` entry rather than carrying its own display, and the |
| 2028 | /// reader translates the key to that value at parse time (the loader installs the map from `terms.typ` |
| 2029 | /// before parsing): `g`/`gi` set the value bold-italic on first use, `gcap`/`gcapi` capitalised, `t`/`tcap` |
| 2030 | /// plain and `graw` plain (its mono face is not reproduced). A key with no `term-dict` entry falls back to |
| 2031 | /// the key text and is recorded in `skips`, so an unknown key is visible on the terse skip line rather |
| 2032 | /// than silently wrong -- the template panics on a miss, which the reader must not. |
| 2033 | fn glossary_call(chars: &[char], i: usize, span: Span, skips: &mut Refusals) -> Option<(Call, usize)> { |
| 2034 | if chars.get(i) != Some(&'#') { |
| 2035 | return None; |
| 2036 | } |
| 2037 | let start = i + 1; |
| 2038 | let mut j = start; |
| 2039 | while j < chars.len() && (chars[j].is_ascii_alphanumeric() || chars[j] == '-' || chars[j] == '_') { |
| 2040 | j += 1; |
| 2041 | } |
| 2042 | if j == start { |
| 2043 | return None; |
| 2044 | } |
| 2045 | let name: String = chars[start..j].iter().collect(); |
| 2046 | if !is_inline_call(&name) { |
| 2047 | return None; |
| 2048 | } |
| 2049 | match chars.get(j) { |
| 2050 | Some('(') | Some('[') => {}, |
| 2051 | _ => return None, |
| 2052 | } |
| 2053 | let (a1, next1) = read_group(chars, j)?; |
| 2054 | |
| 2055 | // The two-argument display functions take the first argument as the index term and the second as the |
| 2056 | // visible text (`#idx-as("publicani")[tax farmers]` sets "tax farmers", indexes "publicani"). |
| 2057 | if name == "idx-as" || name == "idx-main-as" { |
| 2058 | let (a2, next2) = read_group(chars, next1)?; |
| 2059 | // The first argument is the sort key ("March, James"), the second the display shown in the body and |
| 2060 | // set in the index ("James March"); the sort key is flattened so any markup in it does not misfile it. |
| 2061 | let display = unwrap_arg(&a2); |
| 2062 | let index = Some(IndexKey { |
| 2063 | term: flatten_markup(&unwrap_arg(&a1)), |
| 2064 | sub: None, |
| 2065 | display: display.clone(), |
| 2066 | main: name == "idx-main-as", // `idx-main-as` marks a primary reference, its folio bold |
| 2067 | }); |
| 2068 | return Some((Call::Visible { display, index }, next2)); |
| 2069 | } |
| 2070 | // A nested index entry is a pure marker of `parent > child`; its second argument is the child term. |
| 2071 | if name == "idx-nested" { |
| 2072 | let (child, end) = match read_group(chars, next1) { |
| 2073 | Some((c, n2)) => (Some(unwrap_arg(&c)), n2), |
| 2074 | None => (None, next1), |
| 2075 | }; |
| 2076 | let parent = unwrap_arg(&a1); |
| 2077 | let index = Some(IndexKey { term: flatten_markup(&parent), sub: child, display: parent, main: false }); |
| 2078 | return Some((Call::Invisible { index }, end)); |
| 2079 | } |
| 2080 | |
| 2081 | let arg = unwrap_arg(&a1); |
| 2082 | // A glossary+index call indexes the term it displays; a glossary-only call indexes nothing. The `-i` |
| 2083 | // suffix families and the `glossind`/`glossindcap` and `idx`/`index` families are the indexing ones, |
| 2084 | // mirroring the template's `#index`/`#index-main` calls inside each. The display carries the value's own |
| 2085 | // markup (the index page sets it), and the sort key is that value flattened to plain text. |
| 2086 | let idx_of = |value: String| Some(IndexKey { term: flatten_markup(&value), sub: None, display: value, main: false }); |
| 2087 | // A primary reference (`idx-main`/`index-main`): the same key, marked `main` so its folio sets bold. |
| 2088 | let idx_of_main = |value: String| Some(IndexKey { term: flatten_markup(&value), sub: None, display: value, main: true }); |
| 2089 | let call = match name.as_str() { |
| 2090 | // The simple family keys its own display text (a `term-defs` entry), so no translation applies. |
| 2091 | "gs" => Call::Glossary { term: arg.clone(), display: arg, index: None }, |
| 2092 | "gsi" => Call::Glossary { term: arg.clone(), display: arg.clone(), index: idx_of(arg) }, |
| 2093 | "gscap" => Call::Glossary { term: arg.clone(), display: cap_first(&arg), index: None }, |
| 2094 | "gscapi" => Call::Glossary { term: arg.clone(), display: cap_first(&arg), index: idx_of(arg) }, |
| 2095 | // `glossind`/`glossindcap` auto-detect: a key that is in `term-dict` sets its value, otherwise the |
| 2096 | // key stands as its own display, matching the template's `if key in term-dict` branch. Both index the |
| 2097 | // displayed value (uncapitalised), as the template's `idx-term` default does. |
| 2098 | "glossind" => { |
| 2099 | let display = term_value(&arg).unwrap_or_else(|| arg.clone()); |
| 2100 | Call::Glossary { term: arg.clone(), display: display.clone(), index: idx_of(display) } |
| 2101 | }, |
| 2102 | "glossindcap" => { |
| 2103 | let display = term_value(&arg).unwrap_or_else(|| arg.clone()); |
| 2104 | Call::Glossary { term: arg.clone(), display: cap_first(&display), index: idx_of(display) } |
| 2105 | }, |
| 2106 | // The term-dictionary family translates the key to its value; first use is keyed by the key, as the |
| 2107 | // template keys `glossary-seen` by the key name rather than the value. The `-i` variants index the |
| 2108 | // translated value. |
| 2109 | "g" => Call::Glossary { term: arg.clone(), display: resolve_term(&arg, &name, span, skips), index: None }, |
| 2110 | "gi" => { |
| 2111 | let display = resolve_term(&arg, &name, span, skips); |
| 2112 | Call::Glossary { term: arg.clone(), display: display.clone(), index: idx_of(display) } |
| 2113 | }, |
| 2114 | "gcap" => Call::Glossary { term: arg.clone(), display: cap_first(&resolve_term(&arg, &name, span, skips)), index: None }, |
| 2115 | "gcapi" => { |
| 2116 | let display = resolve_term(&arg, &name, span, skips); |
| 2117 | Call::Glossary { term: arg.clone(), display: cap_first(&display), index: idx_of(display) } |
| 2118 | }, |
| 2119 | "t" | "graw" => Call::Visible { display: resolve_term(&arg, &name, span, skips), index: None }, |
| 2120 | "tcap" => Call::Visible { display: cap_first(&resolve_term(&arg, &name, span, skips)), index: None }, |
| 2121 | "idx" => Call::Visible { display: arg.clone(), index: idx_of(arg) }, |
| 2122 | "idx-main" => Call::Visible { display: arg.clone(), index: idx_of_main(arg) }, |
| 2123 | "index" => Call::Invisible { index: idx_of(arg) }, |
| 2124 | "index-main" => Call::Invisible { index: idx_of_main(arg) }, |
| 2125 | _ => return None, |
| 2126 | }; |
| 2127 | Some((call, next1)) |
| 2128 | } |
| 2129 | |
| 2130 | /// Resolves a term-dictionary key to its display value, or -- when no map is installed or it holds no |
| 2131 | /// such key -- falls back to the key text and records the miss in `skips` under the calling function's |
| 2132 | /// name, so an unknown key shows on the terse skip line rather than rendering silently as the raw key. |
| 2133 | fn resolve_term(key: &str, func: &str, span: Span, skips: &mut Refusals) -> String { |
| 2134 | match term_value(key) { |
| 2135 | Some(value) => value, |
| 2136 | None => { |
| 2137 | skips.record(&fmt!("#{} unknown term-dict key {:?}", func, key), span); |
| 2138 | key.to_string() |
| 2139 | }, |
| 2140 | } |
| 2141 | } |
| 2142 | |
| 2143 | /// Reads a bracket or paren group whose opener sits at `i`, returning its inner content and the index |
| 2144 | /// just past the matching closer. The [`SkipState`] frame stack decides what nests: strings, maths spans |
| 2145 | /// and content blocks are respected, so a bracket inside a quoted argument, a `$...$` span or the prose of |
| 2146 | /// a `[...]` caption does not close the group early -- a `(` an author left unbalanced in caption prose is |
| 2147 | /// literal, and the group still closes at its own delimiter. `None` when the group never closes, so a |
| 2148 | /// malformed call is left as ordinary text. |
| 2149 | pub(crate) fn read_group(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 2150 | let open = *chars.get(i)?; |
| 2151 | if open != '[' && open != '(' { |
| 2152 | return None; |
| 2153 | } |
| 2154 | let mut state = SkipState::new(); |
| 2155 | let mut inner = String::new(); |
| 2156 | // The opener pushes its frame (`[` a content block, `(` a code group) but is not part of the inner |
| 2157 | // content, so it is stepped over here and never appended. |
| 2158 | let mut j = i + state.step(chars, i); |
| 2159 | while j < chars.len() { |
| 2160 | let consumed = state.step(chars, j); |
| 2161 | if !state.is_open() { |
| 2162 | // This character closed the outer group -- the matching closer -- so the group ends just past |
| 2163 | // it, and the closer is dropped from the inner as the outer opener was. |
| 2164 | return Some((inner, j + consumed)); |
| 2165 | } |
| 2166 | // Every other character is inner content, verbatim: a nested opener or closer, a string with its |
| 2167 | // quotes, a maths span, or a `#name(` run in content mode. |
| 2168 | for k in j..j + consumed { |
| 2169 | inner.push(chars[k]); |
| 2170 | } |
| 2171 | j += consumed; |
| 2172 | } |
| 2173 | None |
| 2174 | } |
| 2175 | |
| 2176 | /// Strips a `"..."` wrapper from a paren-string argument, so `#gs("surplus")` reads the same term as |
| 2177 | /// `#gs[surplus]`. A bracket argument has no quotes to strip and is returned unchanged. |
| 2178 | fn unwrap_arg(inner: &str) -> String { |
| 2179 | let t = inner.trim(); |
| 2180 | if t.len() >= 2 && t.starts_with('"') && t.ends_with('"') { |
| 2181 | return t[1..t.len() - 1].to_string(); |
| 2182 | } |
| 2183 | inner.to_string() |
| 2184 | } |
| 2185 | |
| 2186 | /// Capitalises the first character of a term for the `-cap` glossary variants, leaving the rest as it |
| 2187 | /// stands. The template capitalises the first grapheme cluster; the first `char` matches it for every |
| 2188 | /// term in these books. |
| 2189 | fn cap_first(s: &str) -> String { |
| 2190 | let mut cs = s.chars(); |
| 2191 | match cs.next() { |
| 2192 | Some(c) => c.to_uppercase().collect::<String>() + cs.as_str(), |
| 2193 | None => String::new(), |
| 2194 | } |
| 2195 | } |
| 2196 | |
| 2197 | /// Whether a `/* ... */` block comment is open across the line break. |
| 2198 | struct CommentState { |
| 2199 | in_block: bool, |
| 2200 | } |
| 2201 | |
| 2202 | /// Removes Typst comments from one line: a `//` to the line's end, and any `/* ... */` span, which may |
| 2203 | /// have opened on an earlier line ([`CommentState::in_block`] carries that across). A `//` or `/*` |
| 2204 | /// inside a `"..."` string or a `` `code` `` span is not a comment and is kept, and a `//` immediately |
| 2205 | /// after `:` is kept so a bare URL survives. Quotes and backticks are treated as span delimiters here, |
| 2206 | /// which is what the reader's markup needs; a real Typst code line with string literals is skipped whole |
| 2207 | /// by the caller, so stripping it never reaches the output. |
| 2208 | fn strip_comments(line: &str, st: &mut CommentState) -> String { |
| 2209 | let chars: Vec<char> = line.chars().collect(); |
| 2210 | let mut out = String::new(); |
| 2211 | let mut in_str = false; |
| 2212 | let mut in_raw = false; |
| 2213 | let mut prev = '\0'; |
| 2214 | let mut i = 0usize; |
| 2215 | while i < chars.len() { |
| 2216 | let c = chars[i]; |
| 2217 | if st.in_block { |
| 2218 | if c == '*' && chars.get(i + 1) == Some(&'/') { |
| 2219 | st.in_block = false; |
| 2220 | i += 2; |
| 2221 | prev = '\0'; |
| 2222 | continue; |
| 2223 | } |
| 2224 | i += 1; |
| 2225 | continue; |
| 2226 | } |
| 2227 | if in_str { |
| 2228 | out.push(c); |
| 2229 | if c == '"' { in_str = false; } |
| 2230 | prev = c; |
| 2231 | i += 1; |
| 2232 | continue; |
| 2233 | } |
| 2234 | if in_raw { |
| 2235 | out.push(c); |
| 2236 | if c == '`' { in_raw = false; } |
| 2237 | prev = c; |
| 2238 | i += 1; |
| 2239 | continue; |
| 2240 | } |
| 2241 | if c == '"' { |
| 2242 | in_str = true; |
| 2243 | out.push(c); |
| 2244 | prev = c; |
| 2245 | i += 1; |
| 2246 | continue; |
| 2247 | } |
| 2248 | if c == '`' { |
| 2249 | in_raw = true; |
| 2250 | out.push(c); |
| 2251 | prev = c; |
| 2252 | i += 1; |
| 2253 | continue; |
| 2254 | } |
| 2255 | if c == '/' && chars.get(i + 1) == Some(&'/') { |
| 2256 | if prev == ':' { |
| 2257 | out.push(c); // a `://` is part of a URL, not a comment |
| 2258 | prev = c; |
| 2259 | i += 1; |
| 2260 | continue; |
| 2261 | } |
| 2262 | break; // a line comment: drop the rest of the line |
| 2263 | } |
| 2264 | if c == '/' && chars.get(i + 1) == Some(&'*') { |
| 2265 | st.in_block = true; |
| 2266 | i += 2; |
| 2267 | prev = '\0'; |
| 2268 | continue; |
| 2269 | } |
| 2270 | out.push(c); |
| 2271 | prev = c; |
| 2272 | i += 1; |
| 2273 | } |
| 2274 | out |
| 2275 | } |
| 2276 | |
| 2277 | // -- Multi-line figure, table and data-array capture ---------------------------------------------- |
| 2278 | |
| 2279 | /// A multi-line construct gathered whole so it can be parsed. The buffer accumulates its lines; the |
| 2280 | /// bracket state closes it when the delimiters balance; the kind decides how the buffer is dispatched. |
| 2281 | struct Capture { |
| 2282 | kind: CaptureKind, |
| 2283 | buf: String, |
| 2284 | state: SkipState, |
| 2285 | start: u32, // byte offset of the construct's opening line, for a `#columns` refusal's span |
| 2286 | } |
| 2287 | |
| 2288 | /// The backstop cap on content-binding expansion depth, for a pathological *non-cyclic* chain of distinct |
| 2289 | /// bindings (a cycle is already caught precisely by the name stack, at its own length). Set well above any |
| 2290 | /// honest nesting yet well below the level at which the reader's own frames overflow the wasm shadow stack -- |
| 2291 | /// a depth-64 cap alone was found to overflow, so it would trap rather than refuse, defeating its purpose. |
| 2292 | const MAX_EXPANSION_DEPTH: usize = 32; |
| 2293 | |
| 2294 | /// Which multi-line construct is being gathered. |
| 2295 | enum CaptureKind { |
| 2296 | Figure, // a `#figure(...)` call, possibly wrapping a table or an image |
| 2297 | Table, // a bare `#table(...)` call |
| 2298 | Image, // a line-leading `#padded-image(...)` or `#image(...)` set without a figure number |
| 2299 | SectionBanner, // a line-leading `#section-banner("logo")`, a full-width grey bar carrying a section logo |
| 2300 | Let(String), // a `#let name = (...)` data array bound to this name |
| 2301 | Columns, // a `#columns(n)[ ... ]` wrapper: its body is set single-column |
| 2302 | StyledBox, // a `#styled-box[ ... ]` callout: its body is set inside a filled, padded box |
| 2303 | DeclStyle, // a `#show: <t>.with(...)` application or a lowerable `#set <target>(...)`; lowered onto the theme, not refused |
| 2304 | TemplateCall(String), // a `#name(args)?[ ... ]` call to a bound `#let` furniture function, expanded into a box |
| 2305 | ContentCall(String), // a `#name`, `#name(args)` or `#name[ ... ]` reference to a bound content binding, expanded into re-read markup spliced in |
| 2306 | Context, // a line-leading `#context { ... }`/`#context[ ... ]`: gathered whole, then either the reverse claim index (its body calls `collect-claim-refs(`) or a refusal |
| 2307 | Place, // a line-leading `#place(...)[ ... ]`: a float, set spanning its column or the page, or a refused overlay |
| 2308 | Builtin(BuiltinKind), // a line-leading Typst markup builtin the reader now sets rather than skips (`#pagebreak`, `#lorem`, `#v`) |
| 2309 | } |
| 2310 | |
| 2311 | /// A line-leading Typst markup builtin the reader recognises and sets on the block path, rather than |
| 2312 | /// tallying as a skipped construct. Each is dispatched by [`dispatch_capture`], which reads the call's |
| 2313 | /// arguments from the gathered buffer -- so a builtin whose parentheses run across several lines is still |
| 2314 | /// read whole. |
| 2315 | enum BuiltinKind { |
| 2316 | PageBreak, // `#pagebreak()` / `#pagebreak(weak: true)`: a forced page eject |
| 2317 | Lorem, // `#lorem(<n>)`: n words of the standard lorem-ipsum placeholder, set as a paragraph |
| 2318 | Vspace, // `#v(<abs len>)`: a fixed vertical space, absolute units only |
| 2319 | ColBreak, // `#colbreak()` / `#colbreak(weak: true)`: a forced column break |
| 2320 | } |
| 2321 | |
| 2322 | /// Detects the opener of a multi-line construct the reader parses rather than skips: a `#figure(`, a |
| 2323 | /// bare `#table(`, or a `#let name = (` data array. `None` for any other line, which the caller then |
| 2324 | /// offers to [`code_skip`]. |
| 2325 | fn capture_opener(trimmed: &str, binds: crate::lang::rules::Bindings<'_, '_>) -> Option<CaptureKind> { |
| 2326 | // A call to a bound `#let` furniture function -- `#pr-note[ ... ]`, `#aside-box(title: [..])[ ... ]` -- |
| 2327 | // is expanded rather than skipped. Recognised before the generic openers so a furniture name never |
| 2328 | // collides with one of them (none of the corpus names does), and only when the map holds it, so an |
| 2329 | // unbound `#name[...]` still falls through to be tallied as a skip exactly as before. |
| 2330 | if let Some(name) = template_call_name(trimmed, binds.tfns) { |
| 2331 | return Some(CaptureKind::TemplateCall(name)); |
| 2332 | } |
| 2333 | // A reference to a bound content binding -- a bare `#intro`, a `#greet("world")`, a `#note[ ... ]` -- is |
| 2334 | // gathered whole and expanded into its re-read markup. Recognised only when the map holds the name and it |
| 2335 | // is not one of the inline-call family (a content binding named `idx`/`g` must not shadow the inline call |
| 2336 | // the scanner sets in place), so an unbound or inline reference still falls through unchanged. |
| 2337 | if let Some(name) = content_call_name(trimmed, binds.cfns) { |
| 2338 | return Some(CaptureKind::ContentCall(name)); |
| 2339 | } |
| 2340 | // A line-leading `#context { ... }` (or the bracket twin `#context[ ... ]`): gathered whole so its body |
| 2341 | // can be inspected for the `collect-claim-refs(` signature that marks the reverse claim index, and |
| 2342 | // otherwise refused exactly as before. Recognised here, ahead of `code_skip`, so the block reaches |
| 2343 | // [`dispatch_capture`] rather than being skipped and lost -- the reader still evaluates no `#context`. |
| 2344 | if is_context_opener(trimmed) { |
| 2345 | return Some(CaptureKind::Context); |
| 2346 | } |
| 2347 | // A line-leading Typst markup builtin -- `#pagebreak()`, `#lorem(60)`, `#v(12pt)` -- the reader now sets |
| 2348 | // rather than skipping. Recognised here, ahead of `code_skip`, so the whole call reaches |
| 2349 | // [`dispatch_capture`] (which reads its arguments) rather than being tallied as an unsupported construct |
| 2350 | // and dropped. None of these names can be a bound furniture or content function -- they are reserved by |
| 2351 | // [`crate::lang::rules::is_reserved_construct`] -- so this never shadows a corpus binding. |
| 2352 | // |
| 2353 | // Own-line only, mirroring [`code_skip`]'s own rule: the call must either open a multi-line span (its |
| 2354 | // delimiters unbalanced on this line, gathered whole below) or close on this line with nothing but |
| 2355 | // whitespace after its `)`. A balanced call with trailing prose -- `#lorem(5) more`, `#v(12pt) text` -- |
| 2356 | // is NOT own-line, so it falls through to the existing visible refusal, which keeps that trailing prose; |
| 2357 | // inline mid-prose support is a later unit. |
| 2358 | if let Some(kind) = builtin_opener(trimmed) { |
| 2359 | let mut state = SkipState::new(); |
| 2360 | scan_brackets(trimmed, &mut state); |
| 2361 | if state.has_open_bracket() || trimmed.trim_end().ends_with(')') { |
| 2362 | return Some(CaptureKind::Builtin(kind)); |
| 2363 | } |
| 2364 | } |
| 2365 | if trimmed.starts_with("#figure(") { |
| 2366 | return Some(CaptureKind::Figure); |
| 2367 | } |
| 2368 | if trimmed.starts_with("#table(") { |
| 2369 | return Some(CaptureKind::Table); |
| 2370 | } |
| 2371 | if trimmed.starts_with("#columns(") { |
| 2372 | return Some(CaptureKind::Columns); |
| 2373 | } |
| 2374 | // A `#place(...)[ ... ]` is gathered whole: a floating one is set, spanning its column or the page; an |
| 2375 | // absolutely placed overlay is refused visibly at dispatch. |
| 2376 | if trimmed.starts_with("#place(") { |
| 2377 | return Some(CaptureKind::Place); |
| 2378 | } |
| 2379 | // A `#styled-box[ ... ]` callout: a full-measure filled box wrapping running prose. Its body opens with |
| 2380 | // the `[` on this line and closes on a later one, so it is gathered whole and re-parsed rather than |
| 2381 | // skipped -- otherwise the bracket span reads as an unbalanced standalone call and its text is dropped. |
| 2382 | if trimmed.starts_with("#styled-box[") { |
| 2383 | return Some(CaptureKind::StyledBox); |
| 2384 | } |
| 2385 | // A documentation section opens with a line-leading `#section-banner("logo")` -- a full-width grey bar |
| 2386 | // carrying the section's logo -- captured here so the bar is drawn rather than the call dropped. Tried |
| 2387 | // before `#image(`, since the name contains none of the others as a prefix. |
| 2388 | if trimmed.starts_with("#section-banner(") { |
| 2389 | return Some(CaptureKind::SectionBanner); |
| 2390 | } |
| 2391 | // A section opener draws its logo with a line-leading `#padded-image(...)` (the Pearl section's pearlite |
| 2392 | // mark), and a bare `#image(...)` places a graphic likewise. Both are set centred without a figure |
| 2393 | // number, so they are captured here rather than skipped. The hyphen keeps `#image(` from matching the |
| 2394 | // tail of `#padded-image(`, which is tried first. |
| 2395 | if trimmed.starts_with("#padded-image(") || trimmed.starts_with("#image(") { |
| 2396 | return Some(CaptureKind::Image); |
| 2397 | } |
| 2398 | // A declarative styling construct the reader lowers onto the theme rather than refusing: a |
| 2399 | // `#show: <template>.with(...)` whole-document application, a top-level `#set` on an element the theme |
| 2400 | // carries a field for, or a per-element `#show <selector>: <transform>` rule the engine collects and |
| 2401 | // applies. Capturing the rule line here stops the reader tallying it as a skipped construct while the |
| 2402 | // rule engine separately reads and applies (or refuses) it -- otherwise an authored selector rule is |
| 2403 | // double-reported, skipped by the reader AND applied by the engine. A `#set rect(...)` names no theme |
| 2404 | // element and a `#show <selector>` whose selector no element answers to are not caught, so both fall |
| 2405 | // through to [`code_skip`], staying a visible refusal; an engine-refused rule (an introspective |
| 2406 | // `it => { ... }`) surfaces through the rule engine's own diagnostic, not the reader's skip tally. |
| 2407 | if is_show_doc_with(trimmed) || is_lowerable_set(trimmed) || crate::lang::rules::is_rule_line(trimmed) { |
| 2408 | return Some(CaptureKind::DeclStyle); |
| 2409 | } |
| 2410 | let_array_name(trimmed).map(CaptureKind::Let) |
| 2411 | } |
| 2412 | |
| 2413 | /// Does this already-left-trimmed line open a line-leading `#context` code block -- `#context {`, |
| 2414 | /// `#context{` or `#context[` -- the shape the Logic appendix's reverse claim index is written with, and |
| 2415 | /// the shape whose brace form once leaked verbatim as prose? The identifier must be exactly `context` |
| 2416 | /// (`#contextual` does not match), and the delimiter must be a `{` (with any run of spaces before it, the |
| 2417 | /// way Typst attaches a code block to its keyword) or an immediate `[`. Recognising it here routes the |
| 2418 | /// whole block through [`dispatch_capture`], which either builds the index (its body calls |
| 2419 | /// `collect-claim-refs(`) or refuses it, rather than skipping it as an opaque code span. |
| 2420 | fn is_context_opener(trimmed: &str) -> bool { |
| 2421 | let rest = match trimmed.strip_prefix("#context") { |
| 2422 | Some(r) => r, |
| 2423 | None => return false, |
| 2424 | }; |
| 2425 | // `#context[` -- the bracket twin -- attaches with no space; `#context {` -- the code block -- attaches |
| 2426 | // across any run of spaces, the way Typst binds a block to its keyword. |
| 2427 | rest.starts_with('[') || rest.trim_start().starts_with('{') |
| 2428 | } |
| 2429 | |
| 2430 | /// If this already-left-trimmed line opens a supported Typst markup builtin -- `#pagebreak(`, `#lorem(` or |
| 2431 | /// `#v(` -- the builtin it names; else `None`. The name must be followed immediately by `(`, so `#voluptas(` |
| 2432 | /// (a content-mode word that happens to start with `v`) does not match `#v`, and only a genuine call is |
| 2433 | /// caught. |
| 2434 | fn builtin_opener(trimmed: &str) -> Option<BuiltinKind> { |
| 2435 | let rest = trimmed.strip_prefix('#')?; |
| 2436 | for (name, kind) in [ |
| 2437 | ("pagebreak", BuiltinKind::PageBreak), |
| 2438 | ("colbreak", BuiltinKind::ColBreak), |
| 2439 | ("lorem", BuiltinKind::Lorem), |
| 2440 | ("v", BuiltinKind::Vspace), |
| 2441 | ] { |
| 2442 | if let Some(after) = rest.strip_prefix(name) { |
| 2443 | if after.starts_with('(') { |
| 2444 | return Some(kind); |
| 2445 | } |
| 2446 | } |
| 2447 | } |
| 2448 | None |
| 2449 | } |
| 2450 | |
| 2451 | /// If this line opens a call to a bound furniture function -- `#<name>(` or `#<name>[` where `<name>` is a |
| 2452 | /// key of `tfns` -- that name; else `None`. The name must be followed immediately by `(` (a keyword-argument |
| 2453 | /// call) or `[` (a bare content call), so `#pr-note[` matches but a word that merely starts with a bound |
| 2454 | /// name does not. |
| 2455 | fn template_call_name(trimmed: &str, tfns: &crate::lang::rules::TemplateFns) -> Option<String> { |
| 2456 | let rest = trimmed.strip_prefix('#')?; |
| 2457 | let name_len = rest.chars().take_while(|&c| c.is_alphanumeric() || c == '-' || c == '_').count(); |
| 2458 | if name_len == 0 { |
| 2459 | return None; |
| 2460 | } |
| 2461 | let name: String = rest.chars().take(name_len).collect(); |
| 2462 | let next = rest.chars().nth(name_len); |
| 2463 | if (next == Some('(') || next == Some('[')) && tfns.contains_key(&name) { |
| 2464 | Some(name) |
| 2465 | } else { |
| 2466 | None |
| 2467 | } |
| 2468 | } |
| 2469 | |
| 2470 | /// If this line is a *standalone* reference to a bound content binding -- a bare `#name`, a `#name(args)`, a |
| 2471 | /// `#name[body]` or a `#name(args)[body]` whose balanced group(s) are followed by nothing but whitespace -- |
| 2472 | /// where `<name>` is a key of `cfns` and not one of the inline-call family, that name; else `None`. The |
| 2473 | /// only-whitespace-after rule is the guard against silent word loss: `#em[Note] the rest.` carries trailing |
| 2474 | /// prose, so it is NOT a standalone call -- it falls through to the paragraph, where the trailing words |
| 2475 | /// survive, rather than being expanded with the rest of the sentence discarded. The inline-call guard |
| 2476 | /// mirrors [`opens_standalone_call`]'s: a binding named `idx`/`g` must never shadow the inline call. |
| 2477 | fn content_call_name(trimmed: &str, cfns: &crate::lang::rules::ContentFns) -> Option<String> { |
| 2478 | let rest = trimmed.strip_prefix('#')?; |
| 2479 | let chars: Vec<char> = rest.chars().collect(); |
| 2480 | let name_len = chars.iter().take_while(|&&c| c.is_alphanumeric() || c == '-' || c == '_').count(); |
| 2481 | if name_len == 0 { |
| 2482 | return None; |
| 2483 | } |
| 2484 | let name: String = chars[..name_len].iter().collect(); |
| 2485 | if is_inline_call(&name) || !cfns.contains_key(&name) { |
| 2486 | return None; |
| 2487 | } |
| 2488 | // Step past the call's balanced group(s): a `(args)` optionally followed by a `[body]`, or a lone |
| 2489 | // `[body]`. A bare `#name` opens neither. |
| 2490 | let mut j = name_len; |
| 2491 | if chars.get(j) == Some(&'(') { |
| 2492 | j = read_group(&chars, j)?.1; |
| 2493 | } |
| 2494 | if chars.get(j) == Some(&'[') { |
| 2495 | j = read_group(&chars, j)?.1; |
| 2496 | } |
| 2497 | // Only when nothing but whitespace trails the reference is it standalone; a trailing character makes it |
| 2498 | // an inline reference the paragraph keeps whole. |
| 2499 | if chars[j..].iter().all(|c| c.is_whitespace()) { |
| 2500 | Some(name) |
| 2501 | } else { |
| 2502 | None |
| 2503 | } |
| 2504 | } |
| 2505 | |
| 2506 | /// The `[ ... ]` body of a captured furniture call, and its keyword arguments if any. `#name[ body ]` has |
| 2507 | /// no arguments (`args` empty); `#name(title: [..])[ body ]` carries the argument group before the body. |
| 2508 | /// `None` when no `[ ... ]` content group follows the name, so a malformed call contributes no body. |
| 2509 | fn template_call_parts(buf: &str, name: &str) -> Option<(String, String)> { |
| 2510 | let chars: Vec<char> = buf.chars().collect(); |
| 2511 | let at = find_lit(&chars, &fmt!("#{}", name))?; |
| 2512 | let mut j = at + name.chars().count() + 1; // past `#name` |
| 2513 | let mut args = String::new(); |
| 2514 | // An optional keyword-argument group `( ... )` immediately after the name. |
| 2515 | if chars.get(j) == Some(&'(') { |
| 2516 | let (inner, after) = read_group(&chars, j)?; |
| 2517 | args = inner; |
| 2518 | j = after; |
| 2519 | } |
| 2520 | // Whitespace between the argument group and the content body. |
| 2521 | while j < chars.len() && chars[j].is_whitespace() { |
| 2522 | j += 1; |
| 2523 | } |
| 2524 | if chars.get(j) != Some(&'[') { |
| 2525 | return None; |
| 2526 | } |
| 2527 | let (body, _) = read_group(&chars, j)?; |
| 2528 | Some((args, body)) |
| 2529 | } |
| 2530 | |
| 2531 | /// Is this line a `#show: <ident>.with(` whole-document template application -- the form whose named |
| 2532 | /// arguments lower onto the theme? Distinguished from an introspective `#show ...: it => { ... }`, |
| 2533 | /// which carries no `.with(` and is left to be refused. |
| 2534 | fn is_show_doc_with(trimmed: &str) -> bool { |
| 2535 | let rest = match trimmed.strip_prefix("#show:") { |
| 2536 | Some(r) => r.trim_start(), |
| 2537 | None => return false, |
| 2538 | }; |
| 2539 | match rest.find(".with(") { |
| 2540 | Some(dot) => { |
| 2541 | let ident = &rest[..dot]; |
| 2542 | !ident.is_empty() && ident.chars().all(|c| c.is_alphanumeric() || c == '-' || c == '_') |
| 2543 | }, |
| 2544 | None => false, |
| 2545 | } |
| 2546 | } |
| 2547 | |
| 2548 | /// Is this line a lowerable top-level `#set <target>(` -- one of the elements the theme carries a field |
| 2549 | /// for? The target list is [`crate::lang::set::LOWERABLE_SET_TARGETS`], the single source of truth the |
| 2550 | /// lowering itself matches on, so the reader and the lowering never drift apart. A `#set` on any other |
| 2551 | /// target returns `false` and is left to [`code_skip`] to refuse, since the reader has no field for it. |
| 2552 | fn is_lowerable_set(trimmed: &str) -> bool { |
| 2553 | let rest = match trimmed.strip_prefix("#set ") { |
| 2554 | Some(r) => r.trim_start(), |
| 2555 | None => return false, |
| 2556 | }; |
| 2557 | crate::lang::set::LOWERABLE_SET_TARGETS.iter().any(|target| { |
| 2558 | // The target must be followed immediately by `(`, so `par` does not match a `#set part(...)`. |
| 2559 | rest.strip_prefix(target).map_or(false, |after| after.starts_with('(')) |
| 2560 | }) |
| 2561 | } |
| 2562 | |
| 2563 | /// If the line is a `#let name = (` binding whose value opens a paren group, its name; else `None`. Only |
| 2564 | /// an array or tuple value is captured -- a scalar or a function `#let` (whose name carries `(`) is left |
| 2565 | /// to [`code_skip`]. |
| 2566 | fn let_array_name(trimmed: &str) -> Option<String> { |
| 2567 | let rest = trimmed.strip_prefix("#let ")?; |
| 2568 | let eq = rest.find('=')?; |
| 2569 | let name = rest[..eq].trim(); |
| 2570 | if name.is_empty() || !name.chars().all(|c| c.is_alphanumeric() || c == '-' || c == '_') { |
| 2571 | return None; |
| 2572 | } |
| 2573 | let value = rest[eq + 1..].trim_start(); |
| 2574 | if value.starts_with('(') { |
| 2575 | Some(name.to_string()) |
| 2576 | } else { |
| 2577 | None |
| 2578 | } |
| 2579 | } |
| 2580 | |
| 2581 | /// Dispatches a completed capture: a data array is evaluated and stored under its name; a table or a |
| 2582 | /// figure is parsed into an [`Item`]. A construct that does not parse -- an unresolved spread, an empty |
| 2583 | /// table -- yields no item rather than an error, so a stray call never fails the whole document. |
| 2584 | fn dispatch_capture( |
| 2585 | cap: Capture, |
| 2586 | items: &mut Vec<Item>, |
| 2587 | arrays: &mut HashMap<String, Vec<Vec<Inline>>>, |
| 2588 | skips: &mut Refusals, |
| 2589 | binds: crate::lang::rules::Bindings<'_, '_>, |
| 2590 | ) |
| 2591 | { |
| 2592 | match cap.kind { |
| 2593 | CaptureKind::Let(name) => { |
| 2594 | arrays.insert(name, parse_let_array(&cap.buf)); |
| 2595 | }, |
| 2596 | CaptureKind::Table => { |
| 2597 | if let Some(inner) = call_inner(&cap.buf, "table") { |
| 2598 | if let Some(spec) = parse_table_spec(&inner, arrays, outer_text_size(&cap.buf)) { |
| 2599 | items.push(Item::Table { spec, span: Span::new(0, 0) }); |
| 2600 | } |
| 2601 | } |
| 2602 | }, |
| 2603 | CaptureKind::Figure => { |
| 2604 | if let Some(item) = parse_figure(&cap.buf, arrays) { |
| 2605 | items.push(item); |
| 2606 | } |
| 2607 | }, |
| 2608 | CaptureKind::Image => { |
| 2609 | // A line-leading image call: its path and sizing are read the same way a figure's image body is, |
| 2610 | // then set as a plain centred image with no figure number. A call naming no path draws nothing. |
| 2611 | let (path, width, height, scale) = image_call(&cap.buf); |
| 2612 | if !path.is_empty() { |
| 2613 | items.push(Item::Image { path, width, height, scale, span: Span::new(0, 0) }); |
| 2614 | } |
| 2615 | }, |
| 2616 | CaptureKind::SectionBanner => { |
| 2617 | // The first positional argument is the logo path; a call naming none draws nothing. |
| 2618 | if let Some(path) = call_inner(&cap.buf, "section-banner").as_deref().and_then(first_string) { |
| 2619 | items.push(Item::SectionBanner { path, span: Span::new(0, 0) }); |
| 2620 | } |
| 2621 | }, |
| 2622 | CaptureKind::Place => { |
| 2623 | // A floating `#place`: its body is a block sequence, read through the document parser again and |
| 2624 | // set as one float. A place that does not float -- an overlay at an absolute position -- or names a |
| 2625 | // side a float cannot take is refused visibly, its body dropped with it as Typst would draw it |
| 2626 | // elsewhere, not in the flow. |
| 2627 | let span = Span::new(cap.start, cap.start); |
| 2628 | match place_float_call(&cap.buf) { |
| 2629 | Some((floating, clearance, body)) => { |
| 2630 | if let Ok((mut inner, sub)) = parse_items(&body, binds) { |
| 2631 | skips.merge(sub); |
| 2632 | // A float is laid out as one unit, so a page or column break inside it cannot be |
| 2633 | // honoured; it is refused visibly rather than dropped. |
| 2634 | refuse_nested_page_breaks(&mut inner, skips); |
| 2635 | items.push(Item::Place { items: inner, floating, clearance, span }); |
| 2636 | } |
| 2637 | }, |
| 2638 | None => skips.record("#place", span), |
| 2639 | } |
| 2640 | }, |
| 2641 | CaptureKind::Columns => { |
| 2642 | // The reader has no column model: the `#columns(n)[ ... ]` wrapper is recorded as skipped and |
| 2643 | // its body set single-column, so the words survive even though the multi-column layout does not. |
| 2644 | // The body is a block sequence, so it is read through the document parser again and its items |
| 2645 | // spliced in; a nested refusal (a `#colbreak()`, an unknown call) folds into the same table. |
| 2646 | // The span is the wrapper's own opening line; a refusal recorded inside the re-parsed body |
| 2647 | // carries a span relative to that body text alone, not the enclosing document -- a known, |
| 2648 | // accepted imprecision for a wrapper nested this way (see `Refusal`'s own doc comment). |
| 2649 | skips.record("#columns", Span::new(cap.start, cap.start)); |
| 2650 | if let Some(body) = columns_body(&cap.buf) { |
| 2651 | if let Ok((mut inner, sub)) = parse_items(&body, binds) { |
| 2652 | skips.merge(sub); |
| 2653 | // The columns body's own top-level `#set` declarations scope to the spliced subtree, the |
| 2654 | // way an included chapter's do (H1): its items splice in flat, so a scope marker pair |
| 2655 | // brackets them. An empty patch -- a body that declares no styling -- adds no markers. |
| 2656 | let patch = crate::lang::set::lower_declarations(&body); |
| 2657 | if patch == crate::theme::ThemePatch::default() { |
| 2658 | items.append(&mut inner); |
| 2659 | } else { |
| 2660 | items.push(Item::Scoped { patch, items: inner }); |
| 2661 | } |
| 2662 | } |
| 2663 | } |
| 2664 | }, |
| 2665 | CaptureKind::StyledBox => { |
| 2666 | // A `#styled-box[ ... ]` callout. Its body is a block sequence, so it is read through the document |
| 2667 | // parser again and wrapped in a single [`Item::Box`] the lowering sets in a filled, padded box -- |
| 2668 | // unlike `#columns`, whose body splices in flat. The construct is set, not skipped, so it is not |
| 2669 | // recorded itself; a refusal within the body (an unknown inline call) still folds in. |
| 2670 | if let Some(body) = styled_box_body(&cap.buf) { |
| 2671 | if let Ok((mut inner, sub)) = parse_items(&body, binds) { |
| 2672 | skips.merge(sub); |
| 2673 | // A `#pagebreak()` nested in a callout body cannot be honoured -- the box is laid out as one |
| 2674 | // keep unit -- so it is refused visibly rather than dropped silently at render (see |
| 2675 | // [`refuse_nested_page_breaks`]). |
| 2676 | refuse_nested_page_breaks(&mut inner, skips); |
| 2677 | // The box body's own top-level `#set`/`#show: doc.with(...)` declarations lower to a patch |
| 2678 | // scoped to the box, applied to the box's subtree at render (H3) rather than the document. |
| 2679 | let patch = crate::lang::set::lower_declarations(&body); |
| 2680 | items.push(Item::Box { items: inner, patch, placement: None, span: Span::new(0, 0) }); |
| 2681 | } |
| 2682 | } |
| 2683 | }, |
| 2684 | CaptureKind::DeclStyle => { |
| 2685 | // A declarative styling construct -- a `#show: <template>.with(...)` application or a lowerable |
| 2686 | // top-level `#set`. The reader gathers it whole so it is neither leaked into the prose nor |
| 2687 | // blindly refused; lowering its arguments onto the theme is the book assembler's job (see |
| 2688 | // [`crate::lang::set`] and [`crate::book`]), which reads the same source with the theme in hand, |
| 2689 | // so nothing is emitted into the item stream here. A `#set` that would lower to nothing -- one |
| 2690 | // applying no argument, or naming an unrecognised or unconvertible one -- is recorded as a |
| 2691 | // refusal (H2), so it is visible rather than a silent no-op; a `#set` that fully lowers, and a |
| 2692 | // `#show: doc.with(...)`, record nothing. |
| 2693 | if let Some(name) = crate::lang::set::declstyle_refusal(&cap.buf) { |
| 2694 | skips.record(&name, Span::new(cap.start, cap.start)); |
| 2695 | } |
| 2696 | }, |
| 2697 | CaptureKind::Context => { |
| 2698 | // A gathered `#context { ... }` block. The reader evaluates no `#context` -- it is hard-stratified |
| 2699 | // and runs no `query` -- so the decision is by signature alone: a block whose body calls |
| 2700 | // `collect-claim-refs(` is the Logic appendix's reverse claim index, lowered to a `Block::ClaimIndex` |
| 2701 | // the author fills from the references gathered walking the body (like recognising `#print-glossary(`, |
| 2702 | // not like running it). Any other `#context` block is refused exactly as before, recorded once by |
| 2703 | // name so the section renders absent-but-reported rather than leaking its source as prose. |
| 2704 | if cap.buf.contains("collect-claim-refs(") { |
| 2705 | items.push(Item::ClaimIndex { span: Span::new(cap.start, cap.start) }); |
| 2706 | } else { |
| 2707 | skips.record("#context", Span::new(cap.start, cap.start)); |
| 2708 | } |
| 2709 | }, |
| 2710 | CaptureKind::TemplateCall(name) => { |
| 2711 | // A call to a bound `#let` furniture function. Its `[ ... ]` body is a block sequence, re-parsed |
| 2712 | // through the document parser (with the same furniture in scope, so a nested call expands too) and |
| 2713 | // wrapped in a single `Item::Box` under the definition's lowered patch -- the callout geometry and |
| 2714 | // the inner-set overlay. A `title:` argument becomes a leading bold paragraph in the box. The call |
| 2715 | // is set, not skipped, so it is not tallied; a refusal inside the body still folds in. |
| 2716 | let tf = match binds.tfns.get(&name) { |
| 2717 | Some(tf) => tf, |
| 2718 | None => return, // the opener only fires for a bound name, so this cannot happen |
| 2719 | }; |
| 2720 | match template_call_parts(&cap.buf, &name) { |
| 2721 | Some((args, body)) => { |
| 2722 | if let Ok((mut inner, sub)) = parse_items(&body, binds) { |
| 2723 | skips.merge(sub); |
| 2724 | // A `#pagebreak()` nested in a furniture callout body cannot be honoured -- the box is one |
| 2725 | // keep unit -- so it is refused visibly rather than dropped silently at render. |
| 2726 | refuse_nested_page_breaks(&mut inner, skips); |
| 2727 | // A `title:` keyword argument, its content set as a leading bold paragraph. It is set at |
| 2728 | // the title size the definition named (`text(size: 0.85em)`) by nesting it in a scope, so a |
| 2729 | // title larger or smaller than the body reads at its own size. |
| 2730 | if tf.has_title { |
| 2731 | if let Some(title) = named_content_arg(&args, "title") { |
| 2732 | let runs = parse_inlines(&title); |
| 2733 | let para = Item::Paragraph { |
| 2734 | runs: vec![Inline::Strong(inline_plain(&runs))], |
| 2735 | label: None, |
| 2736 | span: Span::new(cap.start, cap.start), |
| 2737 | }; |
| 2738 | let title_item = match tf.title_size { |
| 2739 | Some(sz) => { |
| 2740 | let mut patch = crate::theme::ThemePatch::default(); |
| 2741 | patch.text.body_size = Some(sz); |
| 2742 | Item::Scoped { patch, items: vec![para] } |
| 2743 | }, |
| 2744 | None => para, |
| 2745 | }; |
| 2746 | inner.insert(0, title_item); |
| 2747 | } |
| 2748 | } |
| 2749 | items.push(Item::Box { items: inner, patch: tf.patch.clone(), placement: tf.float, span: Span::new(cap.start, cap.start) }); |
| 2750 | } |
| 2751 | }, |
| 2752 | // A bound call with no `[ ... ]` body -- e.g. `#pr-note([x])`, an argument-only call this reader |
| 2753 | // cannot place -- is tallied as a skipped construct rather than dropped silently, so the report |
| 2754 | // still names it. |
| 2755 | None => skips.record(&fmt!("#{}", name), Span::new(cap.start, cap.start)), |
| 2756 | } |
| 2757 | }, |
| 2758 | CaptureKind::ContentCall(name) => { |
| 2759 | // A reference to a bound content binding. Its positional arguments (none for a bare reference) are |
| 2760 | // read and substituted for each `#param` in the body, then the expanded markup is read through the |
| 2761 | // document parser again -- with the same bindings in scope, so a nested call expands too -- and its |
| 2762 | // blocks are spliced in flat. Unlike a furniture call, the body is arbitrary block markup, not a |
| 2763 | // wrap, so a heading in the binding becomes a real heading rather than a boxed paragraph. The call |
| 2764 | // is set, not skipped, so it is not tallied; a refusal inside the expanded body still folds in. |
| 2765 | let cf = match binds.cfns.get(&name) { |
| 2766 | Some(cf) => cf, |
| 2767 | None => return, // the opener only fires for a bound name, so this cannot happen |
| 2768 | }; |
| 2769 | // A self- or mutually-referential binding (`#let a = [#a]`, `#let a = [#b]`/`#let b = [#a]`, or a |
| 2770 | // function form `#let f(n) = [x #f(n)]`) would re-expand without bound. The name stack catches it |
| 2771 | // the instant a name recurs -- so a cycle unwinds at its own length, never deep enough to overflow |
| 2772 | // the native stack or the wasm shadow stack -- and the depth cap is the backstop for a pathological |
| 2773 | // non-cyclic chain of distinct bindings. Either way an unbounded re-read becomes a recorded refusal. |
| 2774 | if binds.expanding(&name) { |
| 2775 | skips.record( |
| 2776 | &fmt!("#{} (cycle: content binding refers back to itself)", name), |
| 2777 | Span::new(cap.start, cap.start)); |
| 2778 | return; |
| 2779 | } |
| 2780 | if binds.depth() >= MAX_EXPANSION_DEPTH { |
| 2781 | skips.record( |
| 2782 | &fmt!("#{} (cycle: expansion depth exceeds {})", name, MAX_EXPANSION_DEPTH), |
| 2783 | Span::new(cap.start, cap.start)); |
| 2784 | return; |
| 2785 | } |
| 2786 | let args = content_call_args(&cap.buf, &name); |
| 2787 | let expanded = expand_content_body(cf, &args); |
| 2788 | // A styled-box content binding (`#let stamp(s) = box(fill: ..)[*v: #s*]`): the inner text is set, |
| 2789 | // but the box's own styling this reader cannot draw is recorded as a visible skip here, so the |
| 2790 | // styling is never silently lost -- only the text is kept, and it is kept, never dropped. |
| 2791 | if let Some(w) = &cf.wrapper { |
| 2792 | skips.record(&fmt!("#{} (styling dropped; content set)", w), Span::new(cap.start, cap.start)); |
| 2793 | } |
| 2794 | let mut nested: Vec<String> = binds.active.to_vec(); |
| 2795 | nested.push(name.clone()); |
| 2796 | if let Ok((mut inner, sub)) = parse_items(&expanded, binds.with_active(&nested)) { |
| 2797 | skips.merge(sub); |
| 2798 | items.append(&mut inner); |
| 2799 | } |
| 2800 | }, |
| 2801 | CaptureKind::Builtin(kind) => { |
| 2802 | let span = Span::new(cap.start, cap.start); |
| 2803 | match kind { |
| 2804 | // `#pagebreak()` / `#pagebreak(weak: true)`: a forced eject. The default (`weak: false`) is a |
| 2805 | // STRONG break, which always opens a fresh page -- even a trailing one, and even when the current |
| 2806 | // page is already empty; `weak: true` ejects only a page that carries content, matching Typst |
| 2807 | // 0.15.1 (the strong/weak flag is honoured in the driver's compose). A `to:` argument |
| 2808 | // (`pagebreak(to: "odd")`) selects a parity target the reader does not model, so it is refused |
| 2809 | // visibly rather than set as a plain break that quietly ignores the argument. |
| 2810 | BuiltinKind::PageBreak => { |
| 2811 | let inner = call_inner(&cap.buf, "pagebreak").unwrap_or_default(); |
| 2812 | if inner.contains("to:") { |
| 2813 | skips.record("#pagebreak", span); |
| 2814 | } else { |
| 2815 | items.push(Item::PageBreak { weak: pagebreak_is_weak(&inner), span }); |
| 2816 | } |
| 2817 | }, |
| 2818 | // `#lorem(<n>)`: n words of the standard placeholder, set as one plain paragraph. A malformed |
| 2819 | // count stays a visible refusal; a zero count sets nothing; a huge count is capped at the |
| 2820 | // embedded corpus by [`lorem_words`], so generation stays bounded. |
| 2821 | BuiltinKind::Lorem => { |
| 2822 | if let Some(n) = lorem_arg(&cap.buf) { |
| 2823 | let text = lorem_words(n); |
| 2824 | if !text.is_empty() { |
| 2825 | items.push(Item::Paragraph { runs: vec![Inline::Text(text)], label: None, span }); |
| 2826 | } |
| 2827 | } else { |
| 2828 | skips.record("#lorem", span); // a non-numeric argument stays a visible refusal |
| 2829 | } |
| 2830 | }, |
| 2831 | // `#v(<abs len>)`: a fixed vertical space. Only an absolute first argument (pt/mm/cm/in) is set; |
| 2832 | // an `em`, `%` or `fr` length has no running size here, and a `weak:` argument asks for a |
| 2833 | // collapsing space the reader does not model -- either is refused visibly rather than set as the |
| 2834 | // wrong space, exactly as the heading-template spacer does (see [`crate::lang::rules`]). |
| 2835 | // `#colbreak()` or `#colbreak(weak: true)`: a forced column break, which on a page of one column |
| 2836 | // breaks the page, as Typst makes it. |
| 2837 | BuiltinKind::ColBreak => { |
| 2838 | let inner = call_inner(&cap.buf, "colbreak").unwrap_or_default(); |
| 2839 | items.push(Item::ColBreak { weak: pagebreak_is_weak(&inner), span }); |
| 2840 | }, |
| 2841 | BuiltinKind::Vspace => { |
| 2842 | let inner = call_inner(&cap.buf, "v").unwrap_or_default(); |
| 2843 | match parse_length(first_arg(&inner).trim()) { |
| 2844 | Some(Length::Abs(pt)) if !inner.contains("weak:") => |
| 2845 | items.push(Item::Space { height: crate::ir::Sp::from_pt(pt), span }), |
| 2846 | _ => skips.record("#v", span), |
| 2847 | } |
| 2848 | }, |
| 2849 | } |
| 2850 | }, |
| 2851 | } |
| 2852 | } |
| 2853 | |
| 2854 | /// The standard lorem-ipsum passage Typst's `#lorem` draws from, the opening of Cicero's *De Finibus* as |
| 2855 | /// the `lipsum` crate carries it, word for word as `typst 0.15.1` renders it. Held to the first 120 words: |
| 2856 | /// every one is plain Latin with only commas and full stops, so a placeholder of any realistic length sets |
| 2857 | /// byte-identically to the oracle, while the later passage's quotation marks, question marks and en dash -- |
| 2858 | /// which would need their own glyph handling to match -- are left out. `#lorem(n)` for n beyond this is |
| 2859 | /// capped here, so a huge count generates a bounded, sane amount rather than looping. |
| 2860 | const LOREM_CORPUS: &str = "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magnam aliquam quaerat voluptatem. Ut enim aeque doleamus animo, cum corpore dolemus, fieri tamen permagna accessio potest, si aliquod aeternum et infinitum impendere malum nobis opinemur. Quod idem licet transferre in voluptatem, ut postea variari voluptas distinguique possit, augeri amplificarique non possit. At etiam Athenis, ut e patre audiebam facete et urbane Stoicos irridente, statua est in quo a nobis philosophia defensa et collaudata est, cum id, quod maxime placeat, facere possimus, omnis voluptas assumenda est, omnis dolor repellendus. Temporibus autem quibusdam et aut officiis debitis aut rerum necessitatibus saepe eveniet, ut et voluptates repudiandae sint et molestiae non recusandae. Itaque earum rerum"; |
| 2861 | |
| 2862 | /// The first `n` words of [`LOREM_CORPUS`], joined by single spaces, with the final word's trailing |
| 2863 | /// punctuation replaced by a full stop -- the shape Typst's `#lorem(n)` produces. `n` is capped at the |
| 2864 | /// corpus length, so a very large count returns the whole embedded passage rather than looping unboundedly. |
| 2865 | /// An `n` of zero yields the empty string, so the caller sets no paragraph. |
| 2866 | fn lorem_words(n: usize) -> String { |
| 2867 | let words: Vec<&str> = LOREM_CORPUS.split_whitespace().collect(); |
| 2868 | let take = n.min(words.len()); |
| 2869 | if take == 0 { |
| 2870 | return String::new(); |
| 2871 | } |
| 2872 | let mut out = words[..take].join(" "); |
| 2873 | // Typst ends the blind text with a full stop, dropping any comma, colon or semicolon the last word |
| 2874 | // carried in the source. |
| 2875 | let trimmed = out.trim_end_matches([',', ';', ':', '.']); |
| 2876 | out.truncate(trimmed.len()); |
| 2877 | out.push('.'); |
| 2878 | out |
| 2879 | } |
| 2880 | |
| 2881 | /// The word count of a `#lorem(<n>)` call read from the gathered buffer: the first positional argument |
| 2882 | /// parsed as a non-negative integer. `None` when the argument is absent or not a number, so the caller |
| 2883 | /// records a refusal rather than guessing. |
| 2884 | fn lorem_arg(buf: &str) -> Option<usize> { |
| 2885 | let inner = call_inner(buf, "lorem")?; |
| 2886 | first_arg(&inner).trim().parse::<usize>().ok() |
| 2887 | } |
| 2888 | |
| 2889 | /// Does a `#pagebreak(...)` argument list ask for a WEAK break? Only an explicit `weak: true` does; a |
| 2890 | /// `weak: false` and an absent argument are both the STRONG default (Typst 0.15.1), which always ejects. |
| 2891 | /// The value is read as the token immediately after `weak:`, so `weak: true`, `weak:true` and |
| 2892 | /// `weak: false` all resolve correctly. |
| 2893 | fn pagebreak_is_weak(inner: &str) -> bool { |
| 2894 | match inner.find("weak:") { |
| 2895 | Some(at) => inner[at + "weak:".len()..].trim_start().starts_with("true"), |
| 2896 | None => false, |
| 2897 | } |
| 2898 | } |
| 2899 | |
| 2900 | /// The first positional argument of a call's inner argument text: the run up to the first top-level comma, |
| 2901 | /// so `#v(12pt, weak: true)` yields `12pt` and `#lorem(60)` yields `60`. |
| 2902 | fn first_arg(inner: &str) -> String { |
| 2903 | split_arg_commas(inner).into_iter().next().unwrap_or_default() |
| 2904 | } |
| 2905 | |
| 2906 | /// Records a visible refusal for, and removes, every `#pagebreak()` nested in a callout box's body. A |
| 2907 | /// forced page eject has no meaning inside a box the layout keeps whole -- the box-body renderer has no |
| 2908 | /// page to turn -- so it is refused rather than silently dropped at render. Recurses through a nested scope |
| 2909 | /// or a nested box, so a break buried in either is caught too. The document top level and a `#columns` body |
| 2910 | /// (which splices into the main flow, not a box) are untouched: a break there is honoured. |
| 2911 | fn refuse_nested_page_breaks(items: &mut Vec<Item>, skips: &mut Refusals) { |
| 2912 | let mut kept = Vec::with_capacity(items.len()); |
| 2913 | for mut item in items.drain(..) { |
| 2914 | match &mut item { |
| 2915 | Item::PageBreak { span, .. } => { skips.record("#pagebreak", *span); continue; }, |
| 2916 | Item::ColBreak { span, .. } => { skips.record("#colbreak", *span); continue; }, |
| 2917 | Item::Box { items: inner, .. } => refuse_nested_page_breaks(inner, skips), |
| 2918 | Item::Place { items: inner, .. } => refuse_nested_page_breaks(inner, skips), |
| 2919 | Item::Scoped { items: inner, .. } => refuse_nested_page_breaks(inner, skips), |
| 2920 | _ => {}, |
| 2921 | } |
| 2922 | kept.push(item); |
| 2923 | } |
| 2924 | *items = kept; |
| 2925 | } |
| 2926 | |
| 2927 | /// The positional arguments of a captured content-binding reference, each evaluated to its substitution |
| 2928 | /// text: a `"quoted string"` yields its contents, a `[bracketed content]` its inner markup, and any other |
| 2929 | /// value (a number, an identifier) its trimmed source. A bare `#name` reference, or a `#name[ ... ]` whose |
| 2930 | /// single argument is the bracket body, is handled too. An empty list when the reference takes none. |
| 2931 | fn content_call_args(buf: &str, name: &str) -> Vec<String> { |
| 2932 | let chars: Vec<char> = buf.chars().collect(); |
| 2933 | let at = match find_lit(&chars, &fmt!("#{}", name)) { |
| 2934 | Some(a) => a, |
| 2935 | None => return Vec::new(), |
| 2936 | }; |
| 2937 | let j = at + name.chars().count() + 1; // past `#name` |
| 2938 | match chars.get(j) { |
| 2939 | Some('(') => { |
| 2940 | match read_group(&chars, j) { |
| 2941 | Some((inner, _)) => split_arg_commas(&inner).into_iter() |
| 2942 | .map(|a| content_arg_value(a.trim())) |
| 2943 | .collect(), |
| 2944 | None => Vec::new(), |
| 2945 | } |
| 2946 | }, |
| 2947 | // A `#name[ ... ]` call: the bracket body is the single positional argument. |
| 2948 | Some('[') => match read_group(&chars, j) { |
| 2949 | Some((inner, _)) => vec![inner], |
| 2950 | None => Vec::new(), |
| 2951 | }, |
| 2952 | _ => Vec::new(), // a bare `#name` reference |
| 2953 | } |
| 2954 | } |
| 2955 | |
| 2956 | /// Evaluates one content-binding argument to its substitution text: a `"..."` string is unquoted, a |
| 2957 | /// `[ ... ]` content block is unwrapped to its inner markup, and any other value is kept as its trimmed |
| 2958 | /// source. |
| 2959 | fn content_arg_value(arg: &str) -> String { |
| 2960 | let t = arg.trim(); |
| 2961 | if t.starts_with('[') { |
| 2962 | let chars: Vec<char> = t.chars().collect(); |
| 2963 | if let Some((inner, _)) = read_group(&chars, 0) { |
| 2964 | return inner; |
| 2965 | } |
| 2966 | } |
| 2967 | unwrap_arg(t) |
| 2968 | } |
| 2969 | |
| 2970 | /// Splits an argument list on its top-level commas, honouring `(`/`[`/`{` nesting and `"..."` strings so a |
| 2971 | /// comma inside a bracketed or quoted argument does not split it. |
| 2972 | fn split_arg_commas(s: &str) -> Vec<String> { |
| 2973 | let mut out = Vec::new(); |
| 2974 | let mut depth = 0i32; |
| 2975 | let mut in_str = false; |
| 2976 | let mut esc = false; |
| 2977 | let mut cur = String::new(); |
| 2978 | for c in s.chars() { |
| 2979 | if in_str { |
| 2980 | cur.push(c); |
| 2981 | if esc { esc = false; } |
| 2982 | else if c == '\\' { esc = true; } |
| 2983 | else if c == '"' { in_str = false; } |
| 2984 | continue; |
| 2985 | } |
| 2986 | match c { |
| 2987 | '"' => { in_str = true; cur.push(c); }, |
| 2988 | '(' | '[' | '{' => { depth += 1; cur.push(c); }, |
| 2989 | ')' | ']' | '}' => { depth -= 1; cur.push(c); }, |
| 2990 | ',' if depth == 0 => out.push(std::mem::take(&mut cur)), |
| 2991 | _ => cur.push(c), |
| 2992 | } |
| 2993 | } |
| 2994 | if !cur.trim().is_empty() { |
| 2995 | out.push(cur); |
| 2996 | } |
| 2997 | out |
| 2998 | } |
| 2999 | |
| 3000 | /// Substitutes a content binding's positional arguments into its body: each line-leading or inline `#param` |
| 3001 | /// token whose identifier names a parameter is replaced by the matching argument's text (a parameter with no |
| 3002 | /// argument supplied substitutes empty). A `#` that opens no parameter name -- an inline call, an escaped |
| 3003 | /// literal -- is left untouched, so the body's own markup survives the substitution. A binding with no |
| 3004 | /// parameters returns its body verbatim. |
| 3005 | /// |
| 3006 | /// A `#name(...)` call immediately following a non-parameter identifier -- `#t(w)`, `#g(w)` -- opens Typst's |
| 3007 | /// own code mode inside its parentheses, where `w` is a bare variable reference rather than a markup-mode |
| 3008 | /// `#w`. [`substitute_call_args`] re-quotes any such bare parameter reference found inside that argument |
| 3009 | /// list, so the nested call -- most often a term-dictionary lookup -- resolves against the substituted value |
| 3010 | /// rather than the parameter's own name. Substitution runs here, before the expanded body is re-parsed, so |
| 3011 | /// term-dictionary (or any other) resolution downstream always sees the caller's argument, never the |
| 3012 | /// parameter placeholder: the root fix is ordering, not a term-dictionary-specific patch. |
| 3013 | fn expand_content_body(cf: &crate::lang::rules::ContentFn, args: &[String]) -> String { |
| 3014 | if cf.params.is_empty() { |
| 3015 | return cf.body.clone(); |
| 3016 | } |
| 3017 | let chars: Vec<char> = cf.body.chars().collect(); |
| 3018 | let mut out = String::new(); |
| 3019 | let mut i = 0usize; |
| 3020 | while i < chars.len() { |
| 3021 | if chars[i] == '#' { |
| 3022 | let mut j = i + 1; |
| 3023 | while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') { |
| 3024 | j += 1; |
| 3025 | } |
| 3026 | let ident: String = chars[i + 1..j].iter().collect(); |
| 3027 | if let Some(pos) = cf.params.iter().position(|p| *p == ident) { |
| 3028 | out.push_str(args.get(pos).map(|s| s.as_str()).unwrap_or("")); |
| 3029 | i = j; |
| 3030 | continue; |
| 3031 | } |
| 3032 | // Not a bare parameter reference; if it names a call (`#t(`, `#g(`, `#anything(`), its argument |
| 3033 | // list is code mode, where a parameter appears with no leading `#`. Substitute there too, then |
| 3034 | // resume scanning after the call's closing paren -- the call name and everything past it are |
| 3035 | // otherwise untouched. |
| 3036 | if !ident.is_empty() && chars.get(j) == Some(&'(') { |
| 3037 | if let Some((inner, after)) = read_group(&chars, j) { |
| 3038 | out.push('#'); |
| 3039 | out.push_str(&ident); |
| 3040 | out.push('('); |
| 3041 | out.push_str(&substitute_call_args(&inner, &cf.params, args)); |
| 3042 | out.push(')'); |
| 3043 | i = after; |
| 3044 | continue; |
| 3045 | } |
| 3046 | } |
| 3047 | } |
| 3048 | out.push(chars[i]); |
| 3049 | i += 1; |
| 3050 | } |
| 3051 | out |
| 3052 | } |
| 3053 | |
| 3054 | /// Substitutes a bare parameter identifier found inside a call's argument list -- Typst code mode, where a |
| 3055 | /// parameter carries no leading `#` of its own -- with a quoted literal of the caller's argument text, so a |
| 3056 | /// nested call such as `#t(w)` sees the substituted value rather than the parameter's own name. A quote or |
| 3057 | /// backslash in the substituted text is escaped, so the result is always one valid string literal. Left alone |
| 3058 | /// inside an existing `"..."` string, so a quoted argument that merely contains the parameter's name as a |
| 3059 | /// word is not mistaken for a reference to it. An identifier that is not a parameter -- the call's own other |
| 3060 | /// arguments, a keyword name, a literal -- passes through unchanged. |
| 3061 | fn substitute_call_args(inner: &str, params: &[String], args: &[String]) -> String { |
| 3062 | let chars: Vec<char> = inner.chars().collect(); |
| 3063 | let mut out = String::new(); |
| 3064 | let mut i = 0usize; |
| 3065 | let mut in_str = false; |
| 3066 | while i < chars.len() { |
| 3067 | let c = chars[i]; |
| 3068 | if in_str { |
| 3069 | out.push(c); |
| 3070 | if c == '\\' && i + 1 < chars.len() { |
| 3071 | out.push(chars[i + 1]); |
| 3072 | i += 2; |
| 3073 | continue; |
| 3074 | } |
| 3075 | if c == '"' { |
| 3076 | in_str = false; |
| 3077 | } |
| 3078 | i += 1; |
| 3079 | continue; |
| 3080 | } |
| 3081 | if c == '"' { |
| 3082 | in_str = true; |
| 3083 | out.push(c); |
| 3084 | i += 1; |
| 3085 | continue; |
| 3086 | } |
| 3087 | if c.is_alphabetic() || c == '_' { |
| 3088 | let start = i; |
| 3089 | let mut j = i + 1; |
| 3090 | while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') { |
| 3091 | j += 1; |
| 3092 | } |
| 3093 | let word: String = chars[start..j].iter().collect(); |
| 3094 | match params.iter().position(|p| *p == word) { |
| 3095 | Some(pos) => { |
| 3096 | let value = args.get(pos).map(|s| s.as_str()).unwrap_or(""); |
| 3097 | out.push('"'); |
| 3098 | out.push_str(&value.replace('\\', "\\\\").replace('"', "\\\"")); |
| 3099 | out.push('"'); |
| 3100 | }, |
| 3101 | None => out.push_str(&word), |
| 3102 | } |
| 3103 | i = j; |
| 3104 | continue; |
| 3105 | } |
| 3106 | out.push(c); |
| 3107 | i += 1; |
| 3108 | } |
| 3109 | out |
| 3110 | } |
| 3111 | |
| 3112 | /// Expands every inline mid-prose reference to a bound content binding within a run of already-joined |
| 3113 | /// paragraph (or heading) text, splicing each call's argument-substituted, re-expanded body into the |
| 3114 | /// surrounding prose at the position it stood -- the inline twin of the own-line [`CaptureKind::ContentCall`] |
| 3115 | /// path (which splices a binding referenced alone on its line as blocks). A `M#oxe`, a |
| 3116 | /// `see #stamp("v2") for details` or a `#note[x]` mid-sentence therefore sets its expansion with the words |
| 3117 | /// before AND after it kept, rather than leaking its raw `#name` onto the page or refusing the call and |
| 3118 | /// losing it. |
| 3119 | /// |
| 3120 | /// Runs as a textual pre-pass in [`flush_para`] (and on a heading title), BEFORE [`substitute_scalars`] and |
| 3121 | /// [`parse_inlines_in`] -- the same shape the scalar pass takes, and for the same reason: a content binding's |
| 3122 | /// value is markup that must be re-read as the surrounding prose's own, so expanding it here lets the one |
| 3123 | /// inline scanner downstream read the spliced result. A glossary term, an emphasis or a maths span in the |
| 3124 | /// body sets exactly as if it had been typed in place, and an unrenderable primitive in the body (a `#h`, a |
| 3125 | /// `#box`) reaches that scanner's own visible-refusal path rather than a parallel one here -- so the root fix |
| 3126 | /// reuses [`expand_content_body`] and the block path's cycle rule rather than building a second expander. |
| 3127 | /// |
| 3128 | /// `active` (carried on `binds`, seeded from an enclosing block-level expansion) is the stack of binding |
| 3129 | /// names currently expanding: a reference to a name already on it is refused as a cycle, exactly as the block |
| 3130 | /// path refuses one, so a self- or mutually-referential body unwinds at its own length rather than looping; |
| 3131 | /// [`MAX_EXPANSION_DEPTH`] is the backstop for a pathological non-cyclic chain. A `#name` that names no |
| 3132 | /// content binding, that is one of the inline-call family (so a binding named `g`/`idx` never shadows the |
| 3133 | /// inline call the scanner sets in place), that is a field/method access (`#name.foo`), or that sits in a raw |
| 3134 | /// `` `...` `` span, a `$...$` maths span or behind a `\`-escape is left untouched, so the downstream scanner |
| 3135 | /// still sets or refuses it exactly as before. A binding referenced own-line is handled earlier by |
| 3136 | /// [`capture_opener`], so this only ever sees a genuinely mid-prose reference. |
| 3137 | pub(crate) fn substitute_content_calls( |
| 3138 | text: &str, |
| 3139 | binds: crate::lang::rules::Bindings<'_, '_>, |
| 3140 | skips: &mut Refusals, |
| 3141 | span: Span, |
| 3142 | ) |
| 3143 | -> String |
| 3144 | { |
| 3145 | // The common case, no content bindings in scope: costs nothing beyond the check, so a document that uses |
| 3146 | // none reads exactly as before. |
| 3147 | if binds.cfns.is_empty() { |
| 3148 | return text.to_string(); |
| 3149 | } |
| 3150 | let chars: Vec<char> = text.chars().collect(); |
| 3151 | let mut out = String::new(); |
| 3152 | let mut i = 0usize; |
| 3153 | let mut in_raw = false; |
| 3154 | let mut in_math = false; |
| 3155 | while i < chars.len() { |
| 3156 | let c = chars[i]; |
| 3157 | // An escaped `\#` (or any `\`-escape) is literal: the backslash and its character are passed straight |
| 3158 | // through, so the downstream scanner still turns `\#` into a literal `#`. |
| 3159 | if c == '\\' && i + 1 < chars.len() { |
| 3160 | out.push(c); |
| 3161 | out.push(chars[i + 1]); |
| 3162 | i += 2; |
| 3163 | continue; |
| 3164 | } |
| 3165 | if c == '`' { |
| 3166 | in_raw = !in_raw; |
| 3167 | out.push(c); |
| 3168 | i += 1; |
| 3169 | continue; |
| 3170 | } |
| 3171 | if c == '$' && !in_raw { |
| 3172 | in_math = !in_math; |
| 3173 | out.push(c); |
| 3174 | i += 1; |
| 3175 | continue; |
| 3176 | } |
| 3177 | if c == '#' && !in_raw && !in_math { |
| 3178 | let start = i + 1; |
| 3179 | let mut j = start; |
| 3180 | while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') { |
| 3181 | j += 1; |
| 3182 | } |
| 3183 | if j > start { |
| 3184 | let name: String = chars[start..j].iter().collect(); |
| 3185 | // A field or method access on the name -- `#name.foo` -- is a code-mode expression, not a bare |
| 3186 | // content reference: leave it for the downstream refusal path rather than expanding the name and |
| 3187 | // stranding the `.foo`. Only a `.` FOLLOWED BY an identifier is an access, though: a `.` before |
| 3188 | // whitespace, end of input or punctuation is a sentence's full stop, which Typst sets literally |
| 3189 | // after expanding the reference (`M#oxe.` sets `MX.`), so it must not block the expansion. |
| 3190 | let field_access = chars.get(j) == Some(&'.') |
| 3191 | && chars.get(j + 1).map_or(false, |c| c.is_alphabetic() || *c == '_'); |
| 3192 | if !field_access && !is_inline_call(&name) { |
| 3193 | if let Some(cf) = binds.cfns.get(&name) { |
| 3194 | // Step over the call's balanced group(s) -- a `(args)` optionally followed by a `[body]`, |
| 3195 | // or a lone `[body]` -- reading its positional arguments the same way [`content_call_args`] |
| 3196 | // reads a captured own-line call's, so both paths substitute identically. |
| 3197 | let mut k = j; |
| 3198 | let mut args: Vec<String> = Vec::new(); |
| 3199 | if chars.get(k) == Some(&'(') { |
| 3200 | if let Some((inner, after)) = read_group(&chars, k) { |
| 3201 | args = split_arg_commas(&inner).into_iter() |
| 3202 | .map(|a| content_arg_value(a.trim())) |
| 3203 | .collect(); |
| 3204 | k = after; |
| 3205 | } |
| 3206 | } |
| 3207 | if chars.get(k) == Some(&'[') { |
| 3208 | if let Some((inner, after)) = read_group(&chars, k) { |
| 3209 | // A `#name[ ... ]` call with no paren group: the bracket body is the single |
| 3210 | // positional argument, mirroring [`content_call_args`]'s own bracket arm. |
| 3211 | if args.is_empty() { |
| 3212 | args = vec![inner]; |
| 3213 | } |
| 3214 | k = after; |
| 3215 | } |
| 3216 | } |
| 3217 | // A self- or mutually-referential binding is refused the instant its name recurs, so a |
| 3218 | // cycle unwinds at its own length; the depth cap is the backstop for a pathological chain |
| 3219 | // of distinct bindings. Either way the call is consumed (the surrounding prose is kept) |
| 3220 | // and a visible refusal recorded, exactly as the own-line path does. |
| 3221 | if binds.expanding(&name) { |
| 3222 | skips.record( |
| 3223 | &fmt!("#{} (cycle: content binding refers back to itself)", name), span); |
| 3224 | i = k; |
| 3225 | continue; |
| 3226 | } |
| 3227 | if binds.depth() >= MAX_EXPANSION_DEPTH { |
| 3228 | skips.record( |
| 3229 | &fmt!("#{} (cycle: expansion depth exceeds {})", name, MAX_EXPANSION_DEPTH), span); |
| 3230 | i = k; |
| 3231 | continue; |
| 3232 | } |
| 3233 | let expanded = expand_content_body(cf, &args); |
| 3234 | // A styled-box content binding used inline (`see #stamp("v2") for details`): the inner text |
| 3235 | // is spliced into the surrounding prose, and the box's own styling this reader cannot draw |
| 3236 | // is recorded as a visible skip -- the styling is never silently lost, the text never dropped. |
| 3237 | if let Some(w) = &cf.wrapper { |
| 3238 | skips.record(&fmt!("#{} (styling dropped; content set)", w), span); |
| 3239 | } |
| 3240 | let mut nested: Vec<String> = binds.active.to_vec(); |
| 3241 | nested.push(name.clone()); |
| 3242 | // The expanded body may itself reference another binding inline, so it is expanded in |
| 3243 | // turn with this name pushed onto the active stack -- the recursion the cycle guard bounds. |
| 3244 | out.push_str(&substitute_content_calls(&expanded, binds.with_active(&nested), skips, span)); |
| 3245 | i = k; |
| 3246 | continue; |
| 3247 | } |
| 3248 | } |
| 3249 | } |
| 3250 | } |
| 3251 | out.push(c); |
| 3252 | i += 1; |
| 3253 | } |
| 3254 | out |
| 3255 | } |
| 3256 | |
| 3257 | /// Substitutes a bare `#name` reference to a scalar `#let` value binding -- `#let title = "Foo"`, then a |
| 3258 | /// later `#title` -- with its display text: a string's contents, or a number/length literal's own source |
| 3259 | /// text (see [`crate::lang::rules::ScalarValue::display_text`]). Runs once, on a heading's title or a |
| 3260 | /// flushed paragraph's joined text, BEFORE [`parse_inlines_in`] reads it, rather than as a case inside that |
| 3261 | /// scanner: a scalar's value is plain text needing no further markup expansion, so a textual pass here is |
| 3262 | /// the whole fix, and it leaves every other construct -- a table, a figure, a caption, a `#context` block, |
| 3263 | /// a data array -- untouched, since none of those reach this function. |
| 3264 | /// |
| 3265 | /// A `#name` inside a raw `` `...` `` code span or a `$...$` maths span is left alone, matching |
| 3266 | /// [`parse_inlines_in`]'s own treatment of those as literal/foreign territory; an escaped `\#name` is left |
| 3267 | /// alone too, so the backslash still reaches the inline scanner to produce a literal `#`. A name with no |
| 3268 | /// scalar binding, or one immediately followed by `(` or `[` (a call, not a bare reference), is untouched -- |
| 3269 | /// it still reaches the inline scanner's own refusal path, so an unbound or non-scalar reference stays a |
| 3270 | /// visible skip rather than becoming a silent drop here. |
| 3271 | pub(crate) fn substitute_scalars(text: &str, sfns: &crate::lang::rules::ScalarFns) -> String { |
| 3272 | if sfns.is_empty() { |
| 3273 | return text.to_string(); // the common case, no scalar bindings in scope -- costs nothing beyond the check |
| 3274 | } |
| 3275 | let chars: Vec<char> = text.chars().collect(); |
| 3276 | let mut out = String::new(); |
| 3277 | let mut i = 0usize; |
| 3278 | let mut in_raw = false; |
| 3279 | let mut in_math = false; |
| 3280 | while i < chars.len() { |
| 3281 | let c = chars[i]; |
| 3282 | if c == '\\' && i + 1 < chars.len() { |
| 3283 | out.push(c); |
| 3284 | out.push(chars[i + 1]); |
| 3285 | i += 2; |
| 3286 | continue; |
| 3287 | } |
| 3288 | if c == '`' { |
| 3289 | in_raw = !in_raw; |
| 3290 | out.push(c); |
| 3291 | i += 1; |
| 3292 | continue; |
| 3293 | } |
| 3294 | if c == '$' && !in_raw { |
| 3295 | in_math = !in_math; |
| 3296 | out.push(c); |
| 3297 | i += 1; |
| 3298 | continue; |
| 3299 | } |
| 3300 | if c == '#' && !in_raw && !in_math { |
| 3301 | let start = i + 1; |
| 3302 | let mut j = start; |
| 3303 | while j < chars.len() && (chars[j].is_alphanumeric() || chars[j] == '-' || chars[j] == '_') { |
| 3304 | j += 1; |
| 3305 | } |
| 3306 | if j > start { |
| 3307 | let name: String = chars[start..j].iter().collect(); |
| 3308 | if !matches!(chars.get(j), Some('(') | Some('[')) { |
| 3309 | if let Some(value) = sfns.get(&name) { |
| 3310 | out.push_str(value.display_text()); |
| 3311 | i = j; |
| 3312 | continue; |
| 3313 | } |
| 3314 | } |
| 3315 | } |
| 3316 | } |
| 3317 | out.push(c); |
| 3318 | i += 1; |
| 3319 | } |
| 3320 | out |
| 3321 | } |
| 3322 | |
| 3323 | /// The content of a `key: [ ... ]` keyword argument, without its brackets. Used to read a furniture call's |
| 3324 | /// `title:` argument. `None` when the key is absent or its value is not a content block. |
| 3325 | fn named_content_arg(args: &str, key: &str) -> Option<String> { |
| 3326 | let at = args.find(key)?; |
| 3327 | let after = &args[at + key.len()..]; |
| 3328 | let after = after.trim_start(); |
| 3329 | let after = after.strip_prefix(':')?.trim_start(); |
| 3330 | if !after.starts_with('[') { |
| 3331 | return None; |
| 3332 | } |
| 3333 | let chars: Vec<char> = after.chars().collect(); |
| 3334 | read_group(&chars, 0).map(|(inner, _)| inner) |
| 3335 | } |
| 3336 | |
| 3337 | /// The plain text of a run of inline markup, dropping the markup and keeping the words -- a title is set as |
| 3338 | /// one bold run, so its own emphasis is flattened rather than nested inside the bold. |
| 3339 | fn inline_plain(runs: &[Inline]) -> String { |
| 3340 | let mut out = String::new(); |
| 3341 | for run in runs { |
| 3342 | match run { |
| 3343 | Inline::Text(t) | Inline::Strong(t) | Inline::Emph(t) | Inline::BoldItalic(t) |
| 3344 | | Inline::Super(t) | Inline::Sub(t) | Inline::Code(t) => out.push_str(t), |
| 3345 | _ => {}, |
| 3346 | } |
| 3347 | } |
| 3348 | out |
| 3349 | } |
| 3350 | |
| 3351 | /// The `[ ... ]` body of a captured `#columns(n)[ ... ]` wrapper: the column count arguments are read and |
| 3352 | /// dropped, and the bracketed block content returned for re-parsing. `None` when no `[...]` group follows |
| 3353 | /// the arguments, so a malformed wrapper contributes no body. |
| 3354 | fn columns_body(buf: &str) -> Option<String> { |
| 3355 | let chars: Vec<char> = buf.chars().collect(); |
| 3356 | let Some(at) = find_lit(&chars, "#columns") else { return None; }; |
| 3357 | let open = at + "#columns".chars().count(); |
| 3358 | if chars.get(open) != Some(&'(') { |
| 3359 | return None; |
| 3360 | } |
| 3361 | let Some((_, after_args)) = read_group(&chars, open) else { return None; }; |
| 3362 | let mut j = after_args; |
| 3363 | while j < chars.len() && chars[j].is_whitespace() { |
| 3364 | j += 1; |
| 3365 | } |
| 3366 | if chars.get(j) != Some(&'[') { |
| 3367 | return None; |
| 3368 | } |
| 3369 | read_group(&chars, j).map(|(body, _)| body) |
| 3370 | } |
| 3371 | |
| 3372 | /// Reads a captured `#place(...)[ ... ]` as a float: its side from the alignment argument (`top`, `bottom` |
| 3373 | /// or `auto`, any horizontal part aside), its `scope:` and `clearance:`, and its body for re-parsing. `None` |
| 3374 | /// for a place that does not float (`float:` absent or false -- an overlay at an absolute position the |
| 3375 | /// reader does not set), one naming no side a float can take, one carrying an offset (`dx:`, `dy:`), or |
| 3376 | /// one with no body. |
| 3377 | fn place_float_call(buf: &str) -> Option<(Floating, Option<Spacing>, String)> { |
| 3378 | let chars: Vec<char> = buf.chars().collect(); |
| 3379 | let Some(at) = find_lit(&chars, "#place") else { return None; }; |
| 3380 | let open = at + "#place".chars().count(); |
| 3381 | if chars.get(open) != Some(&'(') { |
| 3382 | return None; |
| 3383 | } |
| 3384 | let Some((inner, after)) = read_group(&chars, open) else { return None; }; |
| 3385 | let mut side = None; |
| 3386 | let mut float = false; |
| 3387 | let mut scope = FloatScope::Column; |
| 3388 | let mut clearance = None; |
| 3389 | let mut body = None; |
| 3390 | for arg in split_top_args(&inner) { |
| 3391 | let a = arg.trim(); |
| 3392 | if a.is_empty() { |
| 3393 | continue; |
| 3394 | } |
| 3395 | if let Some((key, val)) = named_arg(a) { |
| 3396 | match key.as_str() { |
| 3397 | "float" => float = val.trim() == "true", |
| 3398 | "scope" => scope = parse_scope(&val), |
| 3399 | "clearance" => match parse_spacing(&val) { |
| 3400 | Some(c) => clearance = Some(c), |
| 3401 | None => return None, |
| 3402 | }, |
| 3403 | _ => return None, |
| 3404 | } |
| 3405 | continue; |
| 3406 | } |
| 3407 | if a.starts_with('[') { |
| 3408 | let ac: Vec<char> = a.chars().collect(); |
| 3409 | body = read_group(&ac, 0).map(|(b, _)| b); |
| 3410 | continue; |
| 3411 | } |
| 3412 | // The alignment: its vertical part decides the side; a horizontal part does not move a float. |
| 3413 | for part in a.split('+') { |
| 3414 | match part.trim() { |
| 3415 | "top" => side = Some(FloatPlacement::Top), |
| 3416 | "bottom" => side = Some(FloatPlacement::Bottom), |
| 3417 | "auto" => side = Some(FloatPlacement::Auto), |
| 3418 | "left" | "center" | "right" | "start" | "end" => {}, |
| 3419 | _ => return None, |
| 3420 | } |
| 3421 | } |
| 3422 | } |
| 3423 | if !float { |
| 3424 | return None; |
| 3425 | } |
| 3426 | // The trailing content block, `#place(...)[ ... ]`, when the body was not passed as an argument. |
| 3427 | if body.is_none() { |
| 3428 | let mut j = after; |
| 3429 | while j < chars.len() && chars[j].is_whitespace() { |
| 3430 | j += 1; |
| 3431 | } |
| 3432 | if chars.get(j) == Some(&'[') { |
| 3433 | body = read_group(&chars, j).map(|(b, _)| b); |
| 3434 | } |
| 3435 | } |
| 3436 | // A float with no side named takes `auto`, the default a floating `place` resolves its alignment to. |
| 3437 | let side = side.unwrap_or(FloatPlacement::Auto); |
| 3438 | body.map(|b| (Floating { side, scope }, clearance, b)) |
| 3439 | } |
| 3440 | |
| 3441 | /// Reads a length that may be relative to the text size: `1.5em` in ems, anything [`parse_length`] reads |
| 3442 | /// as points. `None` for a percentage or an unreadable value. |
| 3443 | fn parse_spacing(val: &str) -> Option<Spacing> { |
| 3444 | let v = val.trim(); |
| 3445 | if let Some(em) = v.strip_suffix("em") { |
| 3446 | return em.trim().parse::<f64>().ok().map(Spacing::Em); |
| 3447 | } |
| 3448 | match parse_length(v) { |
| 3449 | Some(Length::Abs(pt)) => Some(Spacing::Pt(pt)), |
| 3450 | _ => None, |
| 3451 | } |
| 3452 | } |
| 3453 | |
| 3454 | /// The `[ ... ]` body of a captured `#styled-box[ ... ]` callout, returned for re-parsing. The call takes |
| 3455 | /// no arguments, so the bracket group opens immediately after the name. `None` when no `[...]` follows, so |
| 3456 | /// a malformed callout contributes no body. |
| 3457 | fn styled_box_body(buf: &str) -> Option<String> { |
| 3458 | let chars: Vec<char> = buf.chars().collect(); |
| 3459 | let Some(at) = find_lit(&chars, "#styled-box") else { return None; }; |
| 3460 | let open = at + "#styled-box".chars().count(); |
| 3461 | if chars.get(open) != Some(&'[') { |
| 3462 | return None; |
| 3463 | } |
| 3464 | read_group(&chars, open).map(|(body, _)| body) |
| 3465 | } |
| 3466 | |
| 3467 | /// The index of the first occurrence of the literal `s` in `chars`, or `None`. |
| 3468 | fn find_lit(chars: &[char], s: &str) -> Option<usize> { |
| 3469 | let pat: Vec<char> = s.chars().collect(); |
| 3470 | if pat.is_empty() || chars.len() < pat.len() { |
| 3471 | return None; |
| 3472 | } |
| 3473 | (0..=chars.len() - pat.len()).find(|&start| chars[start..start + pat.len()] == pat[..]) |
| 3474 | } |
| 3475 | |
| 3476 | /// Evaluates a `#let name = (...)` value into the flat sequence of cells it holds, each cell its inline |
| 3477 | /// runs. The value is the paren group after the `=`; every `[...]` group within it, at any depth, is one |
| 3478 | /// cell -- which is what `array.flatten()` yields for an array of content tuples. |
| 3479 | fn parse_let_array(buf: &str) -> Vec<Vec<Inline>> { |
| 3480 | let chars: Vec<char> = buf.chars().collect(); |
| 3481 | let eq = match chars.iter().position(|&c| c == '=') { |
| 3482 | Some(e) => e, |
| 3483 | None => return Vec::new(), |
| 3484 | }; |
| 3485 | let open = match (eq + 1..chars.len()).find(|&j| !chars[j].is_whitespace()) { |
| 3486 | Some(v) if chars[v] == '(' => v, |
| 3487 | _ => return Vec::new(), |
| 3488 | }; |
| 3489 | match read_group(&chars, open) { |
| 3490 | Some((inner, _)) => collect_cells(&inner), |
| 3491 | None => Vec::new(), |
| 3492 | } |
| 3493 | } |
| 3494 | |
| 3495 | /// Collects every `[...]` group in `inner`, in order, each parsed into its inline runs. A `[` inside a |
| 3496 | /// string is not a cell. Once a group opens, its whole content is one cell and is not descended into. |
| 3497 | /// |
| 3498 | /// A `table.cell(colspan: n)[...]` wrapper is expanded to `n` grid cells: the bracketed content, then |
| 3499 | /// `n - 1` empty span placeholders. A spanning cell consumes several columns in Typst, so without the |
| 3500 | /// placeholders every cell after it slides one column to the left and the last column overruns the |
| 3501 | /// table; the placeholders keep the flat cell stream aligned to the column grid. A bare `table.cell(...)` |
| 3502 | /// (no `colspan`) is one cell, like a plain `[...]`. |
| 3503 | fn collect_cells(inner: &str) -> Vec<Vec<Inline>> { |
| 3504 | let chars: Vec<char> = inner.chars().collect(); |
| 3505 | let mut cells = Vec::new(); |
| 3506 | let mut in_str = false; |
| 3507 | let mut esc = false; |
| 3508 | let mut i = 0usize; |
| 3509 | while i < chars.len() { |
| 3510 | let c = chars[i]; |
| 3511 | if in_str { |
| 3512 | if esc { esc = false; } |
| 3513 | else if c == '\\' { esc = true; } |
| 3514 | else if c == '"' { in_str = false; } |
| 3515 | i += 1; |
| 3516 | continue; |
| 3517 | } |
| 3518 | if c == '"' { |
| 3519 | in_str = true; |
| 3520 | i += 1; |
| 3521 | continue; |
| 3522 | } |
| 3523 | // A `table.cell(args)[content]` wrapper: read the args for a `colspan`, then take the following |
| 3524 | // `[...]` as the content and emit it across that many columns. |
| 3525 | if let Some(after) = at_lit(&chars, i, "table.cell") { |
| 3526 | if chars.get(after) == Some(&'(') { |
| 3527 | if let Some((args, past_args)) = read_group(&chars, after) { |
| 3528 | let span = cell_colspan(&args); |
| 3529 | let mut j = past_args; |
| 3530 | while j < chars.len() && chars[j].is_whitespace() { |
| 3531 | j += 1; |
| 3532 | } |
| 3533 | if chars.get(j) == Some(&'[') { |
| 3534 | if let Some((content, next)) = read_group(&chars, j) { |
| 3535 | cells.push(parse_inlines(&content)); |
| 3536 | for _ in 1..span { |
| 3537 | cells.push(vec![Inline::Text(String::new())]); |
| 3538 | } |
| 3539 | i = next; |
| 3540 | continue; |
| 3541 | } |
| 3542 | } |
| 3543 | // No content group followed the wrapper; step past its args and carry on. |
| 3544 | i = past_args; |
| 3545 | continue; |
| 3546 | } |
| 3547 | } |
| 3548 | } |
| 3549 | if c == '[' { |
| 3550 | if let Some((content, next)) = read_group(&chars, i) { |
| 3551 | cells.push(parse_inlines(&content)); |
| 3552 | i = next; |
| 3553 | continue; |
| 3554 | } |
| 3555 | } |
| 3556 | i += 1; |
| 3557 | } |
| 3558 | cells |
| 3559 | } |
| 3560 | |
| 3561 | /// The `colspan:` of a `table.cell(...)` argument list, at least one. A missing or unreadable `colspan` |
| 3562 | /// is a single column. |
| 3563 | fn cell_colspan(args: &str) -> usize { |
| 3564 | for arg in split_top_args(args) { |
| 3565 | if let Some((key, val)) = named_arg(arg.trim()) { |
| 3566 | if key == "colspan" { |
| 3567 | if let Ok(n) = val.trim().parse::<usize>() { |
| 3568 | return n.max(1); |
| 3569 | } |
| 3570 | } |
| 3571 | } |
| 3572 | } |
| 3573 | 1 |
| 3574 | } |
| 3575 | |
| 3576 | /// Parses the inner text of a `#table(...)` call into a [`TableSpec`]. `columns:` fixes the column |
| 3577 | /// count, `align:` the alignment, a `fill:` keyed on `row == 0` marks a header row; cells come from |
| 3578 | /// inline `[...]` groups and from a `..name.flatten()` spread (or the row-remap idiom [`resolve_spread`] |
| 3579 | /// evaluates) resolved against the data arrays. `None` when no cells are found, so an empty or |
| 3580 | /// unresolved table sets nothing. |
| 3581 | fn parse_table_spec( |
| 3582 | inner: &str, |
| 3583 | arrays: &HashMap<String, Vec<Vec<Inline>>>, |
| 3584 | outer_text_pt: Option<f64>, |
| 3585 | ) |
| 3586 | -> Option<TableSpec> |
| 3587 | { |
| 3588 | let mut ncols = 1usize; |
| 3589 | let mut align = AlignSpec::Uniform(Align::Left); |
| 3590 | let mut header = false; |
| 3591 | let mut inset_pt: Option<f64> = None; |
| 3592 | let mut weights: Vec<f64> = Vec::new(); |
| 3593 | let mut cells: Vec<Vec<Inline>> = Vec::new(); |
| 3594 | // A spread's row-remap idiom needs the column count to chunk its array back into rows, so `columns:` |
| 3595 | // is read ahead of the main pass -- it names the table's shape wherever it sits in the argument list. |
| 3596 | let pre_ncols: usize = split_top_args(inner).iter() |
| 3597 | .find_map(|arg| named_arg(arg.trim()).filter(|(k, _)| k.as_str() == "columns").map(|(_, v)| parse_columns(&v))) |
| 3598 | .unwrap_or(1); |
| 3599 | for arg in split_top_args(inner) { |
| 3600 | let a = arg.trim(); |
| 3601 | if a.is_empty() { |
| 3602 | continue; |
| 3603 | } |
| 3604 | if let Some((key, val)) = named_arg(a) { |
| 3605 | match key.as_str() { |
| 3606 | "columns" => { ncols = parse_columns(&val); weights = parse_column_weights(&val); }, |
| 3607 | "align" => align = parse_align(&val), |
| 3608 | "fill" => if fill_marks_header(&val) { header = true; }, |
| 3609 | "inset" => inset_pt = parse_length(&val).map(length_pt), |
| 3610 | _ => {}, // stroke, gutter and the rest are not modelled |
| 3611 | } |
| 3612 | continue; |
| 3613 | } |
| 3614 | if spread_name(a).is_some() { |
| 3615 | if let Some(v) = resolve_spread(a, arrays, pre_ncols) { |
| 3616 | cells.extend(v); |
| 3617 | } |
| 3618 | continue; |
| 3619 | } |
| 3620 | // An inline positional cell, or a `table.header(...)`/`table.cell(...)` wrapper whose bracketed |
| 3621 | // content is the cell -- each `[...]` group in the argument is one cell, in order. |
| 3622 | if a.contains('[') { |
| 3623 | cells.extend(collect_cells(a)); |
| 3624 | } |
| 3625 | } |
| 3626 | if cells.is_empty() { |
| 3627 | return None; |
| 3628 | } |
| 3629 | Some(TableSpec { ncols: ncols.max(1), header, align, weights, text_pt: outer_text_pt, inset_pt, cells }) |
| 3630 | } |
| 3631 | |
| 3632 | /// Resolves a `..name` spread argument to the cells it contributes: the array itself for a bare `..name` |
| 3633 | /// or `..name.flatten()`, or -- for the row-remap idiom `..name.enumerate().map(((idx, row)) => { if COND |
| 3634 | /// { row } else { table.cell(colspan: n)[...] } }).flatten()`, as Lucronics' E. coli comparison uses to |
| 3635 | /// merge its section-heading rows into spanning cells -- each row kept or replaced exactly as that |
| 3636 | /// closure would evaluate it (see [`remap_enumerated_rows`]). `None` for an unknown array name; an |
| 3637 | /// unrecognised suffix on a known array falls back to its raw cells, so an idiom this reader cannot |
| 3638 | /// evaluate still sets a table rather than an empty one. |
| 3639 | fn resolve_spread( |
| 3640 | arg: &str, |
| 3641 | arrays: &HashMap<String, Vec<Vec<Inline>>>, |
| 3642 | ncols: usize, |
| 3643 | ) |
| 3644 | -> Option<Vec<Vec<Inline>>> |
| 3645 | { |
| 3646 | let name = spread_name(arg)?; |
| 3647 | let base = arrays.get(&name)?; |
| 3648 | let rest = arg.trim().strip_prefix("..")?[name.len()..].trim(); |
| 3649 | if rest.is_empty() || rest == ".flatten()" { |
| 3650 | return Some(base.clone()); |
| 3651 | } |
| 3652 | Some(remap_enumerated_rows(rest, base, ncols).unwrap_or_else(|| base.clone())) |
| 3653 | } |
| 3654 | |
| 3655 | /// Evaluates the `.enumerate().map(((idx, row)) => { if COND { A } else { B } }).flatten()` idiom |
| 3656 | /// against `base` (chunked into `ncols`-wide row tuples), yielding the flat cell list the closure would |
| 3657 | /// produce. `COND` is a bounded slice of Typst -- `idx == N`, `row.at(N) == [...]`/`!= [...]` literal |
| 3658 | /// comparisons, joined by `and`/`or` and grouped by parens (see [`eval_bool`]); a branch that is the bare |
| 3659 | /// loop variable keeps that row's own cells, one shaped `table.cell(colspan: n)[...]` replaces them with |
| 3660 | /// one spanning cell -- its content may reference the loop variable's cells as `row.at(k)`, substituted |
| 3661 | /// with that cell's plain text before the existing `table.cell(...)` reader ([`collect_cells`]) parses it, |
| 3662 | /// so a `#strong(row.at(0))` inside renders through the ordinary strong-call path. `None` when the suffix |
| 3663 | /// is not this shape, or any row's condition or branch does not evaluate, so the caller falls back to the |
| 3664 | /// untransformed array rather than guess at a partial result. |
| 3665 | fn remap_enumerated_rows(after: &str, base: &[Vec<Inline>], ncols: usize) -> Option<Vec<Vec<Inline>>> { |
| 3666 | if ncols == 0 || base.is_empty() || base.len() % ncols != 0 { |
| 3667 | return None; |
| 3668 | } |
| 3669 | let rest = after.strip_prefix(".enumerate()")?.trim_start().strip_prefix(".map")?; |
| 3670 | let rchars: Vec<char> = rest.chars().collect(); |
| 3671 | if rchars.first() != Some(&'(') { |
| 3672 | return None; |
| 3673 | } |
| 3674 | let (map_args, _) = read_group(&rchars, 0)?; |
| 3675 | let arrow = map_args.find("=>")?; |
| 3676 | let params = map_args[..arrow].trim(); |
| 3677 | let body = map_args[arrow + 2..].trim(); |
| 3678 | // The closure's own `{ ... }` block wraps its one expression -- unwrap it so what remains starts at |
| 3679 | // the `if`, the shape [`parse_if_else`] reads. |
| 3680 | let bchars: Vec<char> = body.chars().collect(); |
| 3681 | let body: String = if bchars.first() == Some(&'{') { |
| 3682 | read_brace(&bchars, 0).map(|(inner, _)| inner)? |
| 3683 | } else { |
| 3684 | body.to_string() |
| 3685 | }; |
| 3686 | |
| 3687 | // The parameter is `((idx, row))`: an extra pair of parens wraps the tuple destructure, since it is |
| 3688 | // `.map`'s single positional argument. |
| 3689 | let pchars: Vec<char> = params.chars().collect(); |
| 3690 | let inner_params = if pchars.first() == Some(&'(') { |
| 3691 | read_group(&pchars, 0)?.0 |
| 3692 | } else { |
| 3693 | params.to_string() |
| 3694 | }; |
| 3695 | let inner_params = inner_params.trim(); |
| 3696 | let inner_params = inner_params.strip_prefix('(').and_then(|s| s.strip_suffix(')')).unwrap_or(inner_params); |
| 3697 | let names: Vec<&str> = inner_params.split(',').map(|s| s.trim()).collect(); |
| 3698 | if names.len() != 2 || names[0].is_empty() || names[1].is_empty() { |
| 3699 | return None; |
| 3700 | } |
| 3701 | let (idx_name, row_name) = (names[0], names[1]); |
| 3702 | |
| 3703 | let (cond, a_branch, b_branch) = parse_if_else(&body)?; |
| 3704 | |
| 3705 | let mut out = Vec::with_capacity(base.len()); |
| 3706 | for (i, chunk) in base.chunks(ncols).enumerate() { |
| 3707 | let row_text: Vec<String> = chunk.iter().map(|cell| inline_plain(cell)).collect(); |
| 3708 | let take_a = eval_bool(&cond, i, &row_text, idx_name, row_name)?; |
| 3709 | let branch = if take_a { &a_branch } else { &b_branch }; |
| 3710 | if branch.trim() == row_name { |
| 3711 | out.extend(chunk.iter().cloned()); |
| 3712 | continue; |
| 3713 | } |
| 3714 | let substituted = substitute_row_refs(branch, row_name, &row_text); |
| 3715 | let cells = collect_cells(&substituted); |
| 3716 | if cells.is_empty() { |
| 3717 | return None; // the branch is not a cell this reader can set; refuse the whole idiom |
| 3718 | } |
| 3719 | out.extend(cells); |
| 3720 | } |
| 3721 | Some(out) |
| 3722 | } |
| 3723 | |
| 3724 | /// Splits an `if COND { A } else { B }` closure body into its condition and branch source texts. The |
| 3725 | /// condition runs to the first `{` not nested inside `(...)`/`[...]`; each branch is then read as a |
| 3726 | /// brace-balanced group. `None` when the body is not this shape. |
| 3727 | fn parse_if_else(body: &str) -> Option<(String, String, String)> { |
| 3728 | let rest = body.trim().strip_prefix("if")?; |
| 3729 | let chars: Vec<char> = rest.chars().collect(); |
| 3730 | let mut depth = 0i32; |
| 3731 | let mut brace_at = None; |
| 3732 | for (k, &c) in chars.iter().enumerate() { |
| 3733 | match c { |
| 3734 | '(' | '[' => depth += 1, |
| 3735 | ')' | ']' => depth -= 1, |
| 3736 | '{' if depth == 0 => { brace_at = Some(k); break; }, |
| 3737 | _ => {}, |
| 3738 | } |
| 3739 | } |
| 3740 | let brace_at = brace_at?; |
| 3741 | let cond: String = chars[..brace_at].iter().collect(); |
| 3742 | let (a_branch, next) = read_brace(&chars, brace_at)?; |
| 3743 | let after: String = chars[next..].iter().collect(); |
| 3744 | let after = after.trim().strip_prefix("else")?.trim(); |
| 3745 | let bchars: Vec<char> = after.chars().collect(); |
| 3746 | if bchars.first() != Some(&'{') { |
| 3747 | return None; |
| 3748 | } |
| 3749 | let (b_branch, _) = read_brace(&bchars, 0)?; |
| 3750 | Some((cond.trim().to_string(), a_branch.trim().to_string(), b_branch.trim().to_string())) |
| 3751 | } |
| 3752 | |
| 3753 | /// Reads a `{ ... }` block beginning at `i`, matching only `{`/`}` depth -- the closure bodies this |
| 3754 | /// idiom targets hold no literal braces of their own, so a flat counter is enough, unlike [`read_group`]'s |
| 3755 | /// full frame tracking for `[`/`(`. `None` when the block never closes. |
| 3756 | fn read_brace(chars: &[char], i: usize) -> Option<(String, usize)> { |
| 3757 | if chars.get(i) != Some(&'{') { |
| 3758 | return None; |
| 3759 | } |
| 3760 | let mut depth = 1i32; |
| 3761 | let start = i + 1; |
| 3762 | let mut j = start; |
| 3763 | while j < chars.len() { |
| 3764 | match chars[j] { |
| 3765 | '{' => depth += 1, |
| 3766 | '}' => { |
| 3767 | depth -= 1; |
| 3768 | if depth == 0 { |
| 3769 | return Some((chars[start..j].iter().collect(), j + 1)); |
| 3770 | } |
| 3771 | }, |
| 3772 | _ => {}, |
| 3773 | } |
| 3774 | j += 1; |
| 3775 | } |
| 3776 | None |
| 3777 | } |
| 3778 | |
| 3779 | /// Evaluates a bounded boolean expression -- comparisons of `idx`/`row.at(n)` against a literal or a |
| 3780 | /// number, joined by `and`/`or` and grouped by parens -- against one row's values. `idx_name`/`row_name` |
| 3781 | /// are the closure's own parameter names, so the expression is read against exactly the variables that |
| 3782 | /// idiom bound. `None` when the expression falls outside this bounded grammar. |
| 3783 | fn eval_bool(expr: &str, idx: usize, row: &[String], idx_name: &str, row_name: &str) -> Option<bool> { |
| 3784 | let expr = expr.trim(); |
| 3785 | if let Some(parts) = split_top_level(expr, "or") { |
| 3786 | let mut acc = false; |
| 3787 | for p in &parts { |
| 3788 | acc = acc || eval_bool(p, idx, row, idx_name, row_name)?; |
| 3789 | } |
| 3790 | return Some(acc); |
| 3791 | } |
| 3792 | if let Some(parts) = split_top_level(expr, "and") { |
| 3793 | let mut acc = true; |
| 3794 | for p in &parts { |
| 3795 | acc = acc && eval_bool(p, idx, row, idx_name, row_name)?; |
| 3796 | } |
| 3797 | return Some(acc); |
| 3798 | } |
| 3799 | if expr.starts_with('(') && expr.ends_with(')') { |
| 3800 | // Confirm the outer parens actually wrap the whole expression, rather than two disjoint groups |
| 3801 | // that merely happen to open and close at the ends. |
| 3802 | let mut depth = 0i32; |
| 3803 | let mut wraps_whole = true; |
| 3804 | let last = expr.chars().count() - 1; |
| 3805 | for (k, c) in expr.chars().enumerate() { |
| 3806 | match c { |
| 3807 | '(' => depth += 1, |
| 3808 | ')' => { |
| 3809 | depth -= 1; |
| 3810 | if depth == 0 && k != last { |
| 3811 | wraps_whole = false; |
| 3812 | break; |
| 3813 | } |
| 3814 | }, |
| 3815 | _ => {}, |
| 3816 | } |
| 3817 | } |
| 3818 | if wraps_whole { |
| 3819 | return eval_bool(&expr[1..expr.len() - 1], idx, row, idx_name, row_name); |
| 3820 | } |
| 3821 | } |
| 3822 | eval_cmp(expr, idx, row, idx_name, row_name) |
| 3823 | } |
| 3824 | |
| 3825 | /// Splits `expr` at every top-level (outside `(...)`/`[...]`) whole-word occurrence of `kw` (`"and"` or |
| 3826 | /// `"or"`), or `None` when `kw` does not occur at the top level, so the caller falls through to the next |
| 3827 | /// precedence rather than treating an absent operator as one empty operand. |
| 3828 | fn split_top_level(expr: &str, kw: &str) -> Option<Vec<String>> { |
| 3829 | let chars: Vec<char> = expr.chars().collect(); |
| 3830 | let kwc: Vec<char> = kw.chars().collect(); |
| 3831 | let mut depth = 0i32; |
| 3832 | let mut parts = Vec::new(); |
| 3833 | let mut start = 0usize; |
| 3834 | let mut i = 0usize; |
| 3835 | let mut found = false; |
| 3836 | while i < chars.len() { |
| 3837 | match chars[i] { |
| 3838 | '(' | '[' => { depth += 1; i += 1; continue; }, |
| 3839 | ')' | ']' => { depth -= 1; i += 1; continue; }, |
| 3840 | _ => {}, |
| 3841 | } |
| 3842 | if depth == 0 && i + kwc.len() <= chars.len() && chars[i..i + kwc.len()] == kwc[..] { |
| 3843 | let before_ok = i == 0 || chars[i - 1].is_whitespace(); |
| 3844 | let after_ok = chars.get(i + kwc.len()).map(|c| c.is_whitespace()).unwrap_or(true); |
| 3845 | if before_ok && after_ok { |
| 3846 | parts.push(chars[start..i].iter().collect::<String>()); |
| 3847 | i = i + kwc.len(); |
| 3848 | start = i; |
| 3849 | found = true; |
| 3850 | continue; |
| 3851 | } |
| 3852 | } |
| 3853 | i += 1; |
| 3854 | } |
| 3855 | if !found { |
| 3856 | return None; |
| 3857 | } |
| 3858 | parts.push(chars[start..].iter().collect::<String>()); |
| 3859 | Some(parts.into_iter().map(|s| s.trim().to_string()).collect()) |
| 3860 | } |
| 3861 | |
| 3862 | /// Evaluates one `lhs (==|!=) rhs` comparison against `idx`/`row`, via [`eval_scalar`]. `None` when |
| 3863 | /// neither `==` nor `!=` appears, or either side does not resolve. |
| 3864 | fn eval_cmp(e: &str, idx: usize, row: &[String], idx_name: &str, row_name: &str) -> Option<bool> { |
| 3865 | let e = e.trim(); |
| 3866 | let (eq, at) = if let Some(p) = e.find("!=") { |
| 3867 | (false, p) |
| 3868 | } else if let Some(p) = e.find("==") { |
| 3869 | (true, p) |
| 3870 | } else { |
| 3871 | return None; |
| 3872 | }; |
| 3873 | let lhs = eval_scalar(e[..at].trim(), idx, row, idx_name, row_name)?; |
| 3874 | let rhs = eval_scalar(e[at + 2..].trim(), idx, row, idx_name, row_name)?; |
| 3875 | Some(if eq { lhs == rhs } else { lhs != rhs }) |
| 3876 | } |
| 3877 | |
| 3878 | /// Reads one side of a comparison: the loop index, a `row.at(n)` cell (its plain text), a `[...]` content |
| 3879 | /// literal (its plain text) or a bare number. `None` for anything else, refusing rather than guessing. |
| 3880 | fn eval_scalar(s: &str, idx: usize, row: &[String], idx_name: &str, row_name: &str) -> Option<String> { |
| 3881 | let s = s.trim(); |
| 3882 | if s == idx_name { |
| 3883 | return Some(idx.to_string()); |
| 3884 | } |
| 3885 | if let Some(rest) = s.strip_prefix(row_name) { |
| 3886 | let n = rest.strip_prefix(".at(")?.strip_suffix(')')?.trim().parse::<usize>().ok()?; |
| 3887 | return row.get(n).cloned(); |
| 3888 | } |
| 3889 | if let Some(lit) = s.strip_prefix('[').and_then(|t| t.strip_suffix(']')) { |
| 3890 | return Some(lit.trim().to_string()); |
| 3891 | } |
| 3892 | if s.parse::<usize>().is_ok() { |
| 3893 | return Some(s.to_string()); |
| 3894 | } |
| 3895 | None |
| 3896 | } |
| 3897 | |
| 3898 | /// Replaces every `<row_name>.at(k)` reference in `src` with a quoted string of that cell's plain text, |
| 3899 | /// so a branch like `table.cell(colspan: 7)[#strong(row.at(0))]` becomes a call [`collect_cells`] (via |
| 3900 | /// the ordinary `#strong("...")` reader) can already set, rather than teaching the cell reader to evaluate |
| 3901 | /// arbitrary code. |
| 3902 | fn substitute_row_refs(src: &str, row_name: &str, row: &[String]) -> String { |
| 3903 | let mut out = src.to_string(); |
| 3904 | for (k, cell) in row.iter().enumerate() { |
| 3905 | let pat = fmt!("{}.at({})", row_name, k); |
| 3906 | if out.contains(&pat) { |
| 3907 | let escaped = cell.replace('\\', "\\\\").replace('"', "\\\""); |
| 3908 | out = out.replace(&pat, &fmt!("\"{}\"", escaped)); |
| 3909 | } |
| 3910 | } |
| 3911 | out |
| 3912 | } |
| 3913 | |
| 3914 | /// The absolute point value of a [`Length`], resolving a percentage against a nominal 100 pt so a |
| 3915 | /// percentage inset still yields a sensible padding; a table's inset is in practice an absolute length. |
| 3916 | pub(crate) fn length_pt(len: Length) -> f64 { |
| 3917 | match len { |
| 3918 | Length::Abs(pt) => pt, |
| 3919 | Length::Rel(f) => f * 100.0, |
| 3920 | } |
| 3921 | } |
| 3922 | |
| 3923 | /// The size of a `text(size: Npt)[...]` (or `#text(...)`) wrapper at the start of `text`, in points, or |
| 3924 | /// `None` when the body is not wrapped in a sized `text` call. This lets a figure's small-set table -- |
| 3925 | /// `text(size: 7pt)[#table(...)]` -- carry its reduced size, which Typst applies to the whole table. |
| 3926 | fn outer_text_size(text: &str) -> Option<f64> { |
| 3927 | let inner = call_inner(text, "text")?; |
| 3928 | for arg in split_top_args(&inner) { |
| 3929 | let a = arg.trim(); |
| 3930 | if let Some((key, val)) = named_arg(a) { |
| 3931 | if key == "size" { |
| 3932 | return parse_length(&val).map(length_pt); |
| 3933 | } |
| 3934 | } else if a.ends_with("pt") || a.ends_with("em") { |
| 3935 | // The first positional length is the size, as in `text(7pt)[...]`. |
| 3936 | return parse_length(a).map(length_pt); |
| 3937 | } |
| 3938 | } |
| 3939 | None |
| 3940 | } |
| 3941 | |
| 3942 | /// Parses a `#figure(...)` call (its buffer, a trailing `<label>` and all) into an [`Item::Figure`]. The |
| 3943 | /// positional argument is the body -- a wrapped `#table(...)` set in full, or an image call stood in for |
| 3944 | /// by a placeholder; `caption:` sets the caption, `supplement:`/`kind:` the "Figure" or "Table" label. |
| 3945 | fn parse_figure(buf: &str, arrays: &HashMap<String, Vec<Vec<Inline>>>) -> Option<Item> { |
| 3946 | let (body_src, label) = strip_trailing_label(buf); |
| 3947 | let inner = call_inner(&body_src, "figure")?; |
| 3948 | |
| 3949 | let mut caption: Option<Vec<Inline>> = None; |
| 3950 | let mut supplement: Option<String> = None; |
| 3951 | let mut kind: Option<String> = None; |
| 3952 | let mut positional: Option<String> = None; |
| 3953 | let mut placement: Option<FloatPlacement> = None; |
| 3954 | let mut scope = FloatScope::Column; |
| 3955 | for arg in split_top_args(&inner) { |
| 3956 | let a = arg.trim(); |
| 3957 | if a.is_empty() { |
| 3958 | continue; |
| 3959 | } |
| 3960 | if let Some((key, val)) = named_arg(a) { |
| 3961 | match key.as_str() { |
| 3962 | "caption" => caption = Some(caption_inlines(&val)), |
| 3963 | "supplement" => supplement = Some(unquote(&val)), |
| 3964 | "kind" => kind = Some(unquote(&val)), |
| 3965 | "placement" => placement = parse_placement(&val), |
| 3966 | "scope" => scope = parse_scope(&val), |
| 3967 | _ => {}, // the rest do not affect the set figure |
| 3968 | } |
| 3969 | continue; |
| 3970 | } |
| 3971 | if positional.is_none() { |
| 3972 | positional = Some(a.to_string()); // the first positional argument is the figure body |
| 3973 | } |
| 3974 | } |
| 3975 | |
| 3976 | let body_text = positional.unwrap_or_default(); |
| 3977 | let body = figure_body(&body_text, arrays); |
| 3978 | let supplement = supplement.unwrap_or_else(|| match kind.as_deref() { |
| 3979 | Some("table") => "Table".to_string(), |
| 3980 | _ => "Figure".to_string(), |
| 3981 | }); |
| 3982 | // A scope matters only to a float: Typst accepts `scope: "parent"` on a floating figure alone. |
| 3983 | let placement = placement.map(|side| Floating { side, scope }); |
| 3984 | Some(Item::Figure { body, caption, supplement, label, placement, span: Span::new(0, 0) }) |
| 3985 | } |
| 3986 | |
| 3987 | /// Reads a float's `scope:` value: `"parent"` spans every column of the page, anything else -- `"column"`, |
| 3988 | /// Typst's default -- keeps it within its column. |
| 3989 | fn parse_scope(val: &str) -> FloatScope { |
| 3990 | match unquote(val).as_str() { |
| 3991 | "parent" => FloatScope::Parent, |
| 3992 | _ => FloatScope::Column, |
| 3993 | } |
| 3994 | } |
| 3995 | |
| 3996 | /// Reads a `#figure` `placement:` value into a float placement. `auto` and `top` float to the page top, |
| 3997 | /// `bottom` to the foot; `none` (and anything unrecognised) leaves the figure in the flow. Typst's own |
| 3998 | /// default for a figure is `none`, so a figure that names no placement is not a float. |
| 3999 | fn parse_placement(val: &str) -> Option<FloatPlacement> { |
| 4000 | match val.trim() { |
| 4001 | "auto" => Some(FloatPlacement::Auto), |
| 4002 | "top" => Some(FloatPlacement::Top), |
| 4003 | "bottom" => Some(FloatPlacement::Bottom), |
| 4004 | _ => None, |
| 4005 | } |
| 4006 | } |
| 4007 | |
| 4008 | /// Decides a figure's body from its positional text: a wrapped `#table(...)` if one is present and |
| 4009 | /// parses, otherwise an image carrying the path and any declared sizing (empty path when none is found). |
| 4010 | fn figure_body(text: &str, arrays: &HashMap<String, Vec<Vec<Inline>>>) -> FigureBody { |
| 4011 | if let Some(inner) = call_inner(text, "table") { |
| 4012 | if let Some(spec) = parse_table_spec(&inner, arrays, outer_text_size(text)) { |
| 4013 | return FigureBody::Table(spec); |
| 4014 | } |
| 4015 | } |
| 4016 | // A CeTZ/Fletcher diagram, bar chart or line plot drawn inline is read into a builder that draws it |
| 4017 | // for real; only when the body is none of these does it fall through to the image/placeholder path. |
| 4018 | if let Some(cf) = super::codefig::parse_code_figure(text) { |
| 4019 | return FigureBody::Code(cf); |
| 4020 | } |
| 4021 | let (path, width, height, scale) = image_call(text); |
| 4022 | FigureBody::Image { path, width, height, scale } |
| 4023 | } |
| 4024 | |
| 4025 | /// The path and sizing of a `padded-image("...")` or `image("...")` call in `text`. The custom wrapper is |
| 4026 | /// tried first, since `image` is a word boundary within it only after the hyphen. The first positional |
| 4027 | /// argument is the path; `width`/`height` size an `image(...)`, `scale` a `padded-image(...)`. A path |
| 4028 | /// that is not found gives an empty string, which the block layer stands in for with a placeholder. |
| 4029 | fn image_call(text: &str) -> (String, Option<Length>, Option<Length>, Option<f64>) { |
| 4030 | for name in ["padded-image", "image"] { |
| 4031 | if let Some(inner) = call_inner(text, name) { |
| 4032 | let mut path: Option<String> = None; |
| 4033 | let mut width: Option<Length> = None; |
| 4034 | let mut height: Option<Length> = None; |
| 4035 | let mut scale: Option<f64> = None; |
| 4036 | for arg in split_top_args(&inner) { |
| 4037 | let a = arg.trim(); |
| 4038 | if a.is_empty() { |
| 4039 | continue; |
| 4040 | } |
| 4041 | if let Some((key, val)) = named_arg(a) { |
| 4042 | match key.as_str() { |
| 4043 | "width" => width = parse_length(&val), |
| 4044 | "height" => height = parse_length(&val), |
| 4045 | "scale" => scale = parse_percent(&val), |
| 4046 | _ => {}, // padding and the rest do not size the set image |
| 4047 | } |
| 4048 | continue; |
| 4049 | } |
| 4050 | if path.is_none() { |
| 4051 | path = first_string(a); |
| 4052 | } |
| 4053 | } |
| 4054 | if let Some(p) = path { |
| 4055 | return (p, width, height, scale); |
| 4056 | } |
| 4057 | } |
| 4058 | } |
| 4059 | (String::new(), None, None, None) |
| 4060 | } |
| 4061 | |
| 4062 | /// Reads a Typst length argument into a [`Length`]: a percentage as a fraction of the measure, a `pt`, |
| 4063 | /// `mm`, `cm` or `in` length as absolute points, a bare number as points. `auto` and anything unreadable |
| 4064 | /// give `None`, so the figure falls back to filling the measure. |
| 4065 | pub(crate) fn parse_length(val: &str) -> Option<Length> { |
| 4066 | let v = val.trim(); |
| 4067 | if let Some(pct) = v.strip_suffix('%') { |
| 4068 | return pct.trim().parse::<f64>().ok().map(|n| Length::Rel(n / 100.0)); |
| 4069 | } |
| 4070 | for (unit, per_pt) in [("pt", 1.0), ("mm", 72.0 / 25.4), ("cm", 72.0 / 2.54), ("in", 72.0)] { |
| 4071 | if let Some(num) = v.strip_suffix(unit) { |
| 4072 | return num.trim().parse::<f64>().ok().map(|n| Length::Abs(n * per_pt)); |
| 4073 | } |
| 4074 | } |
| 4075 | v.parse::<f64>().ok().map(Length::Abs) |
| 4076 | } |
| 4077 | |
| 4078 | /// Parses a standalone `#line(length:.., stroke:..)` into an [`Item::Rule`]. The length is a fraction of |
| 4079 | /// the measure (`100%`) or an absolute length; the stroke gives the rule's thickness (a `pt` length) and |
| 4080 | /// its grey (a `luma(N)` component). A missing length fills the measure; a missing thickness is a hairline |
| 4081 | /// half-point; a missing colour is black, Typst's default stroke. |
| 4082 | fn parse_line_rule(trimmed: &str) -> Option<Item> { |
| 4083 | let inner = call_inner(trimmed, "line")?; |
| 4084 | let mut width = Length::Rel(1.0); |
| 4085 | let mut thickness = 0.5; |
| 4086 | let mut grey = 0u8; |
| 4087 | for arg in split_top_args(&inner) { |
| 4088 | let a = arg.trim(); |
| 4089 | if let Some((key, val)) = named_arg(a) { |
| 4090 | match key.as_str() { |
| 4091 | "length" => if let Some(l) = parse_length(&val) { width = l; }, |
| 4092 | "stroke" => { |
| 4093 | let (t, g) = parse_stroke(&val); |
| 4094 | if let Some(t) = t { thickness = t; } |
| 4095 | if let Some(g) = g { grey = g; } |
| 4096 | }, |
| 4097 | _ => {}, // start, end, angle and the rest do not affect a horizontal divider |
| 4098 | } |
| 4099 | } |
| 4100 | } |
| 4101 | Some(Item::Rule { width, thickness, grey, span: Span::new(0, 0) }) |
| 4102 | } |
| 4103 | |
| 4104 | /// The thickness (a `pt` length) and grey (a `luma(N)` value, 0-255) of a `stroke:` value such as |
| 4105 | /// `0.5pt + luma(180)`; either component may be absent. A `luma` is read as a grey level; a bare colour |
| 4106 | /// name or an `rgb(...)` is not modelled and leaves the grey unset. |
| 4107 | fn parse_stroke(val: &str) -> (Option<f64>, Option<u8>) { |
| 4108 | let mut thickness: Option<f64> = None; |
| 4109 | let mut grey: Option<u8> = None; |
| 4110 | for part in val.split('+') { |
| 4111 | let p = part.trim(); |
| 4112 | if let Some(inner) = call_inner(p, "luma") { |
| 4113 | if let Ok(n) = inner.trim().trim_end_matches('%').trim().parse::<f64>() { |
| 4114 | grey = Some(n.clamp(0.0, 255.0) as u8); |
| 4115 | } |
| 4116 | } else if let Some(Length::Abs(pt)) = parse_length(p) { |
| 4117 | thickness = Some(pt); |
| 4118 | } |
| 4119 | } |
| 4120 | (thickness, grey) |
| 4121 | } |
| 4122 | |
| 4123 | /// Reads a percentage argument (`100%`) into a fraction (`1.0`), or `None` when it is not a percentage. |
| 4124 | fn parse_percent(val: &str) -> Option<f64> { |
| 4125 | val.trim().strip_suffix('%').and_then(|p| p.trim().parse::<f64>().ok()).map(|n| n / 100.0) |
| 4126 | } |
| 4127 | |
| 4128 | /// The content of the first `name(...)` call in `text`, balanced across nesting and strings, or `None`. |
| 4129 | /// `name` must sit at a word boundary, so a short name does not match inside a longer identifier. |
| 4130 | pub(crate) fn call_inner(text: &str, name: &str) -> Option<String> { |
| 4131 | let chars: Vec<char> = text.chars().collect(); |
| 4132 | let namev: Vec<char> = name.chars().collect(); |
| 4133 | let paren = find_call(&chars, &namev, 0)?; |
| 4134 | read_group(&chars, paren).map(|(inner, _)| inner) |
| 4135 | } |
| 4136 | |
| 4137 | /// The index of the `(` of the first `name(` at a word boundary at or after `from`, or `None`. |
| 4138 | fn find_call(chars: &[char], name: &[char], from: usize) -> Option<usize> { |
| 4139 | if name.is_empty() { |
| 4140 | return None; |
| 4141 | } |
| 4142 | let mut i = from; |
| 4143 | while i + name.len() < chars.len() { |
| 4144 | if chars[i..].starts_with(name) && chars.get(i + name.len()) == Some(&'(') { |
| 4145 | let boundary = i == 0 || !is_call_ident(chars[i - 1]); |
| 4146 | if boundary { |
| 4147 | return Some(i + name.len()); |
| 4148 | } |
| 4149 | } |
| 4150 | i += 1; |
| 4151 | } |
| 4152 | None |
| 4153 | } |
| 4154 | |
| 4155 | /// A character that continues a Typst identifier, for the word-boundary test in [`find_call`]. |
| 4156 | fn is_call_ident(c: char) -> bool { |
| 4157 | c.is_alphanumeric() || c == '-' || c == '_' |
| 4158 | } |
| 4159 | |
| 4160 | /// The first `"..."` string literal's content in `text`, or `None`. |
| 4161 | pub(crate) fn first_string(text: &str) -> Option<String> { |
| 4162 | let chars: Vec<char> = text.chars().collect(); |
| 4163 | let start = chars.iter().position(|&c| c == '"')?; |
| 4164 | let end = (start + 1..chars.len()).find(|&j| chars[j] == '"')?; |
| 4165 | Some(chars[start + 1..end].iter().collect()) |
| 4166 | } |
| 4167 | |
| 4168 | /// Splits the inner text of a call by its top-level commas, respecting `()[]{}` nesting and `"..."` |
| 4169 | /// strings, so a comma inside a nested group or a string does not part an argument. |
| 4170 | pub(crate) fn split_top_args(inner: &str) -> Vec<String> { |
| 4171 | let chars: Vec<char> = inner.chars().collect(); |
| 4172 | let mut args: Vec<String> = Vec::new(); |
| 4173 | let mut cur = String::new(); |
| 4174 | let mut state = SkipState::new(); |
| 4175 | let mut i = 0; |
| 4176 | while i < chars.len() { |
| 4177 | // A comma parts the arguments only at the top level; inside any frame -- a nested group, a string, |
| 4178 | // a maths span or a content block -- it is literal and joins the current argument. |
| 4179 | if !state.is_open() && chars[i] == ',' { |
| 4180 | args.push(std::mem::take(&mut cur)); |
| 4181 | i += 1; |
| 4182 | continue; |
| 4183 | } |
| 4184 | let consumed = state.step(&chars, i); |
| 4185 | for k in i..i + consumed { |
| 4186 | cur.push(chars[k]); |
| 4187 | } |
| 4188 | i += consumed; |
| 4189 | } |
| 4190 | if !cur.trim().is_empty() { |
| 4191 | args.push(cur); |
| 4192 | } |
| 4193 | args |
| 4194 | } |
| 4195 | |
| 4196 | /// Splits a `key: value` argument at its top-level colon, returning the key and the trimmed value, or |
| 4197 | /// `None` when there is no top-level colon or the key is not a bare identifier -- so a positional cell |
| 4198 | /// or a spread is not mistaken for a named argument. |
| 4199 | pub(crate) fn named_arg(arg: &str) -> Option<(String, String)> { |
| 4200 | let chars: Vec<char> = arg.chars().collect(); |
| 4201 | let mut state = SkipState::new(); |
| 4202 | let mut i = 0; |
| 4203 | while i < chars.len() { |
| 4204 | // A colon names the argument only at the top level; inside any frame it is part of the value (an |
| 4205 | // alignment `align: (col, row) => ...`, a ratio in a caption, a dictionary key in code). |
| 4206 | if !state.is_open() && chars[i] == ':' { |
| 4207 | let key: String = chars[..i].iter().collect(); |
| 4208 | let key = key.trim().to_string(); |
| 4209 | if !key.is_empty() && key.chars().all(is_call_ident) { |
| 4210 | let val: String = chars[i + 1..].iter().collect(); |
| 4211 | return Some((key, val.trim().to_string())); |
| 4212 | } |
| 4213 | return None; |
| 4214 | } |
| 4215 | i += state.step(&chars, i); |
| 4216 | } |
| 4217 | None |
| 4218 | } |
| 4219 | |
| 4220 | /// The array name of a `..name` or `..name.flatten()` spread argument, or `None`. |
| 4221 | fn spread_name(arg: &str) -> Option<String> { |
| 4222 | let rest = arg.trim().strip_prefix("..")?; |
| 4223 | let name: String = rest.chars().take_while(|&c| is_call_ident(c)).collect(); |
| 4224 | if name.is_empty() { |
| 4225 | None |
| 4226 | } else { |
| 4227 | Some(name) |
| 4228 | } |
| 4229 | } |
| 4230 | |
| 4231 | /// Parses a `columns:` value into a column count: an integer as itself, a track list `(a, b, c)` as its |
| 4232 | /// entry count, anything else as one column. |
| 4233 | fn parse_columns(val: &str) -> usize { |
| 4234 | let v = val.trim(); |
| 4235 | if let Ok(n) = v.parse::<usize>() { |
| 4236 | return n.max(1); |
| 4237 | } |
| 4238 | if v.starts_with('(') { |
| 4239 | let ch: Vec<char> = v.chars().collect(); |
| 4240 | if let Some((inner, _)) = read_group(&ch, 0) { |
| 4241 | let cnt = split_top_args(&inner).iter().filter(|p| !p.trim().is_empty()).count(); |
| 4242 | return cnt.max(1); |
| 4243 | } |
| 4244 | } |
| 4245 | 1 |
| 4246 | } |
| 4247 | |
| 4248 | /// The per-column fractional weights of a `columns:` track list: a track `Nfr` (or a bare `fr`, weight 1) |
| 4249 | /// contributes its weight, an `auto` or a fixed length contributes `0.0` so the column is sized to its |
| 4250 | /// content. A bare `columns: N` gives no weights (an empty vector), leaving every column content-sized. |
| 4251 | /// The weights let [`table::lower`](crate::table) reproduce Typst's fractional column sizing rather than |
| 4252 | /// sizing every column from its widest cell. |
| 4253 | fn parse_column_weights(val: &str) -> Vec<f64> { |
| 4254 | let v = val.trim(); |
| 4255 | if !v.starts_with('(') { |
| 4256 | return Vec::new(); |
| 4257 | } |
| 4258 | let ch: Vec<char> = v.chars().collect(); |
| 4259 | let inner = match read_group(&ch, 0) { |
| 4260 | Some((inner, _)) => inner, |
| 4261 | None => return Vec::new(), |
| 4262 | }; |
| 4263 | let mut out = Vec::new(); |
| 4264 | for track in split_top_args(&inner) { |
| 4265 | let t = track.trim(); |
| 4266 | if t.is_empty() { |
| 4267 | continue; |
| 4268 | } |
| 4269 | out.push(track_weight(t)); |
| 4270 | } |
| 4271 | out |
| 4272 | } |
| 4273 | |
| 4274 | /// The fractional weight of one `columns:` track: `Nfr` reads as `N`, a bare `fr` as `1`, and any other |
| 4275 | /// track -- `auto`, `3cm`, `40pt`, `20%` -- as `0.0`, which marks the column content-sized. |
| 4276 | fn track_weight(track: &str) -> f64 { |
| 4277 | match track.strip_suffix("fr") { |
| 4278 | Some(num) => { |
| 4279 | let n = num.trim(); |
| 4280 | if n.is_empty() { 1.0 } else { n.parse::<f64>().unwrap_or(0.0) } |
| 4281 | }, |
| 4282 | None => 0.0, |
| 4283 | } |
| 4284 | } |
| 4285 | |
| 4286 | /// Parses an `align:` value: a `(col, row) => ...` closure as [`AlignSpec::Closure`] (its parameter |
| 4287 | /// names and body captured for per-cell evaluation), a tuple of column alignments as |
| 4288 | /// [`AlignSpec::PerColumn`], a single alignment word as [`AlignSpec::Uniform`]. |
| 4289 | fn parse_align(val: &str) -> AlignSpec { |
| 4290 | let v = val.trim(); |
| 4291 | if let Some(arrow) = v.find("=>") { |
| 4292 | let params = v[..arrow].trim(); |
| 4293 | let body = v[arrow + 2..].trim().to_string(); |
| 4294 | // The parameter list `(col, row)`; a bare single parameter has no parentheses. The first names the |
| 4295 | // column, the second the row, matching Typst's `(col, row)` order. |
| 4296 | let names: Vec<String> = { |
| 4297 | let pch: Vec<char> = params.chars().collect(); |
| 4298 | match read_group(&pch, 0) { |
| 4299 | Some((inner, _)) => split_top_args(&inner).iter().map(|s| s.trim().to_string()).collect(), |
| 4300 | None => vec![params.trim().to_string()], |
| 4301 | } |
| 4302 | }; |
| 4303 | let col_var = names.first().cloned().filter(|s| !s.is_empty()).unwrap_or_else(|| "col".to_string()); |
| 4304 | let row_var = names.get(1).cloned().filter(|s| !s.is_empty()).unwrap_or_else(|| "row".to_string()); |
| 4305 | return AlignSpec::Closure(ClosureAlign { col_var, row_var, body }); |
| 4306 | } |
| 4307 | if v.starts_with('(') { |
| 4308 | let ch: Vec<char> = v.chars().collect(); |
| 4309 | if let Some((inner, _)) = read_group(&ch, 0) { |
| 4310 | let cols: Vec<Align> = split_top_args(&inner).iter().map(|p| word_align(p)).collect(); |
| 4311 | if !cols.is_empty() { |
| 4312 | return AlignSpec::PerColumn(cols); |
| 4313 | } |
| 4314 | } |
| 4315 | } |
| 4316 | AlignSpec::Uniform(word_align(v)) |
| 4317 | } |
| 4318 | |
| 4319 | /// Maps a Typst alignment word to an [`Align`], ignoring a `+ horizon`/`+ top` vertical component and |
| 4320 | /// treating `start`/`end` as left/right. An unknown word is left-aligned. |
| 4321 | fn word_align(s: &str) -> Align { |
| 4322 | let first = s.trim().split(|c: char| c.is_whitespace() || c == '+').next().unwrap_or("").trim(); |
| 4323 | match first { |
| 4324 | "center" | "centre" => Align::Centre, |
| 4325 | "right" | "end" => Align::Right, |
| 4326 | _ => Align::Left, |
| 4327 | } |
| 4328 | } |
| 4329 | |
| 4330 | /// Does a `fill:` value key on the first row, marking a header? A `fill: (col, row) => ...` closure whose |
| 4331 | /// body tests the row index against zero (`row == 0` or the common `y == 0`) fills the first row, which is |
| 4332 | /// the books' header idiom; a `y < n` band likewise begins at the first row. Written with or without |
| 4333 | /// spaces, and matching either name the closure gives its second (row) parameter. |
| 4334 | fn fill_marks_header(val: &str) -> bool { |
| 4335 | let compact: String = val.chars().filter(|c| !c.is_whitespace()).collect(); |
| 4336 | compact.contains("row==0") |
| 4337 | || compact.contains("y==0") |
| 4338 | || compact.contains("row<") |
| 4339 | || compact.contains("y<") |
| 4340 | } |
| 4341 | |
| 4342 | /// Parses a `caption: [...]` value into its inline runs: the bracket content scanned for markup, or the |
| 4343 | /// whole value scanned when it is not a bracket group, so a caption's emphasis, superscript or in-caption |
| 4344 | /// maths sets with its own face rather than flattening to upright text. |
| 4345 | fn caption_inlines(val: &str) -> Vec<Inline> { |
| 4346 | let v = val.trim(); |
| 4347 | let ch: Vec<char> = v.chars().collect(); |
| 4348 | if ch.first() == Some(&'[') { |
| 4349 | if let Some((content, _)) = read_group(&ch, 0) { |
| 4350 | return parse_inlines(&content); |
| 4351 | } |
| 4352 | } |
| 4353 | parse_inlines(v) |
| 4354 | } |
| 4355 | |
| 4356 | /// Strips a trailing `<label>` from a captured call, returning the call text without it and the label. |
| 4357 | /// A `<name>` with no inner whitespace at the very end labels the figure; anything else keeps the text. |
| 4358 | fn strip_trailing_label(buf: &str) -> (String, Option<String>) { |
| 4359 | let t = buf.trim_end(); |
| 4360 | if let Some(inner) = t.strip_suffix('>') { |
| 4361 | if let Some(p) = inner.rfind('<') { |
| 4362 | let label = &inner[p + 1..]; |
| 4363 | if !label.is_empty() && !label.contains(char::is_whitespace) { |
| 4364 | return (inner[..p].to_string(), Some(label.to_string())); |
| 4365 | } |
| 4366 | } |
| 4367 | } |
| 4368 | (buf.to_string(), None) |
| 4369 | } |
| 4370 | |
| 4371 | /// Strips a surrounding `"..."` from a string-literal argument value, leaving anything else unchanged. |
| 4372 | fn unquote(val: &str) -> String { |
| 4373 | let t = val.trim(); |
| 4374 | if t.len() >= 2 && t.starts_with('"') && t.ends_with('"') { |
| 4375 | return t[1..t.len() - 1].to_string(); |
| 4376 | } |
| 4377 | t.to_string() |
| 4378 | } |
| 4379 | |
| 4380 | #[cfg(test)] |
| 4381 | mod tests { |
| 4382 | use super::*; |
| 4383 | use crate::math::{Atom, MatKind}; |
| 4384 | |
| 4385 | /// A `#claim-label` emits a `MarginNote` carrying its compressed code as the margin display and its raw |
| 4386 | /// code for the reverse index, setting nothing in the body column; a `#claim-refs` emits a `MarginNote` |
| 4387 | /// with an empty display (nothing drawn) carrying its raw reference codes. The body prose on either side |
| 4388 | /// closes over the gap when flattened, so no raw markup leaks. |
| 4389 | #[test] |
| 4390 | fn claim_label_emits_a_margin_note_and_sets_nothing_inline() { |
| 4391 | let runs = parse_inlines("clinical authority#claim-label(<LS8>) bites hardest."); |
| 4392 | assert_eq!(runs.len(), 3, "text, margin note, text: got {:?}", runs); |
| 4393 | assert!(matches!(&runs[0], Inline::Text(t) if t == "clinical authority")); |
| 4394 | assert!(matches!(&runs[1], Inline::MarginNote { display, codes } if display == "LS8" && codes == &vec!["LS8".to_string()]), |
| 4395 | "the claim code rides in a margin note and registers for the reverse index: {:?}", runs[1]); |
| 4396 | assert!(matches!(&runs[2], Inline::Text(t) if t == " bites hardest.")); |
| 4397 | // A `#claim-refs` emits a margin note with an empty display carrying its reference codes. |
| 4398 | let refs = parse_inlines("formalised#claim-refs(<A1>, <A2>)."); |
| 4399 | assert!(refs.iter().any(|r| matches!(r, Inline::MarginNote { display, codes } |
| 4400 | if display.is_empty() && codes == &vec!["A1".to_string(), "A2".to_string()])), |
| 4401 | "a claim-refs registers its raw codes with no margin ink: {:?}", refs); |
| 4402 | // Neither the margin code nor a reference is part of the flattened body text. |
| 4403 | assert_eq!( |
| 4404 | flatten_markup("margins#claim-label(<CD14>, <CD15>, <CD4>) formalised#claim-refs(<A1>)."), |
| 4405 | "margins formalised."); |
| 4406 | } |
| 4407 | |
| 4408 | /// A bare `#name` naming a scalar `#let` value binding substitutes its display text: a string's contents |
| 4409 | /// unquoted, and a number/length literal's own source text, wherever it stands in running text -- at the |
| 4410 | /// start, mid-sentence, or immediately before punctuation. |
| 4411 | #[test] |
| 4412 | fn substitute_scalars_replaces_a_bound_scalar_in_prose() { |
| 4413 | let mut sfns = crate::lang::rules::ScalarFns::new(); |
| 4414 | sfns.insert("version".to_string(), crate::lang::rules::ScalarValue::Str("1.2".to_string())); |
| 4415 | sfns.insert("edition".to_string(), crate::lang::rules::ScalarValue::Number("3".to_string())); |
| 4416 | assert_eq!( |
| 4417 | substitute_scalars("This is version #version, edition #edition, of the guide.", &sfns), |
| 4418 | "This is version 1.2, edition 3, of the guide."); |
| 4419 | // A reference at the very start of the text (the shape a heading title reads). |
| 4420 | assert_eq!(substitute_scalars("#version Notes", &sfns), "1.2 Notes"); |
| 4421 | } |
| 4422 | |
| 4423 | /// A `#name` inside a raw `` `...` `` code span or a `$...$` maths span is left untouched -- neither is |
| 4424 | /// this reader's word, matching how [`parse_inlines_in`] treats them as literal/foreign territory -- and |
| 4425 | /// an escaped `\#name` is left alone too, so the backslash still reaches the inline scanner to produce a |
| 4426 | /// literal `#`. An unbound name, or a bound name immediately followed by `(`/`[` (a call, not a bare |
| 4427 | /// reference), is also untouched. |
| 4428 | #[test] |
| 4429 | fn substitute_scalars_leaves_raw_math_and_escaped_references_alone() { |
| 4430 | let mut sfns = crate::lang::rules::ScalarFns::new(); |
| 4431 | sfns.insert("version".to_string(), crate::lang::rules::ScalarValue::Str("1.2".to_string())); |
| 4432 | assert_eq!(substitute_scalars("see `#version` in code", &sfns), "see `#version` in code"); |
| 4433 | assert_eq!(substitute_scalars("the constant $#version$ here", &sfns), "the constant $#version$ here"); |
| 4434 | assert_eq!(substitute_scalars("literal \\#version stays", &sfns), "literal \\#version stays"); |
| 4435 | assert_eq!(substitute_scalars("#unknown-name here", &sfns), "#unknown-name here"); |
| 4436 | assert_eq!(substitute_scalars("#version(1) call-shaped", &sfns), "#version(1) call-shaped"); |
| 4437 | } |
| 4438 | |
| 4439 | /// A scalar `#let` value binding substitutes at a bare `#name` reference in both a heading title and |
| 4440 | /// running prose, through the whole [`document_with_templates`] pipeline (a [`Bindings::with_scalars`] |
| 4441 | /// scope, as [`crate::book::Scope::bindings`] builds for a real compile) -- not just in the |
| 4442 | /// [`substitute_scalars`] unit above. |
| 4443 | #[test] |
| 4444 | fn scalar_substitution_applies_in_a_heading_and_in_prose() { |
| 4445 | let tfns = crate::lang::rules::TemplateFns::new(); |
| 4446 | let cfns = crate::lang::rules::ContentFns::new(); |
| 4447 | let mut sfns = crate::lang::rules::ScalarFns::new(); |
| 4448 | sfns.insert("title".to_string(), crate::lang::rules::ScalarValue::Str("Guide".to_string())); |
| 4449 | let binds = crate::lang::rules::Bindings::with_scalars(&tfns, &cfns, &sfns); |
| 4450 | let src = "= #title\n\nThe #title is version one.\n"; |
| 4451 | let (items, _skips) = document_with_templates(src, binds).expect("parses"); |
| 4452 | match &items[0] { |
| 4453 | Item::Heading { runs, .. } => assert!( |
| 4454 | matches!(&runs[0], Inline::Text(t) if t.contains("Guide")), |
| 4455 | "the heading substitutes the bound scalar: {:?}", runs), |
| 4456 | other => panic!("expected a heading first, got {:?}", other), |
| 4457 | } |
| 4458 | match &items[1] { |
| 4459 | Item::Paragraph { runs, .. } => { |
| 4460 | let text: String = runs.iter().map(|r| match r { |
| 4461 | Inline::Text(t) => t.clone(), |
| 4462 | _ => String::new(), |
| 4463 | }).collect(); |
| 4464 | assert!(text.contains("Guide"), "the paragraph substitutes the bound scalar: {:?}", text); |
| 4465 | }, |
| 4466 | other => panic!("expected a paragraph second, got {:?}", other), |
| 4467 | } |
| 4468 | } |
| 4469 | |
| 4470 | /// A bare `#name` standing alone on its own line -- bound to no scalar, content or template binding -- |
| 4471 | /// is still a visible refusal, exactly as before scalar bindings existed: [`document_with_templates`] |
| 4472 | /// records it in the returned [`Refusals`] rather than leaking its raw source as a paragraph, whether or |
| 4473 | /// not the document also carries a scalar scope. Scalar substitution only ever fires for a NAME the |
| 4474 | /// scope actually binds ([`substitute_scalars`] is a no-op otherwise), so adding it never turns this |
| 4475 | /// existing refusal into a silent drop. |
| 4476 | #[test] |
| 4477 | fn unbound_standalone_reference_stays_a_visible_refusal() { |
| 4478 | let mut sfns = crate::lang::rules::ScalarFns::new(); |
| 4479 | sfns.insert("title".to_string(), crate::lang::rules::ScalarValue::Str("Guide".to_string())); |
| 4480 | let tfns = crate::lang::rules::TemplateFns::new(); |
| 4481 | let cfns = crate::lang::rules::ContentFns::new(); |
| 4482 | let binds = crate::lang::rules::Bindings::with_scalars(&tfns, &cfns, &sfns); |
| 4483 | let src = "Some prose above.\n\n#nosuchname\n\nSome prose below.\n"; |
| 4484 | let (items, skips) = document_with_templates(src, binds).expect("parses"); |
| 4485 | assert!(items.iter().all(|it| !matches!(it, Item::Paragraph { runs, .. } |
| 4486 | if runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("#nosuchname"))))), |
| 4487 | "an unbound standalone reference must not leak as paragraph text: {:?}", items); |
| 4488 | assert!(skips.report().is_some_and(|r| r.contains("nosuchname")), |
| 4489 | "an unbound standalone reference is a visible refusal, not a silent drop: {:?}", skips.report()); |
| 4490 | } |
| 4491 | |
| 4492 | /// A scalar `#let` binding referenced by a bare `#name` standing ALONE on its own line substitutes its |
| 4493 | /// value, through the whole [`document_with_templates`] pipeline -- not only mid-prose (which item 2 already |
| 4494 | /// handled) but where the reference is the line's only content, the case the standalone code-reference skip |
| 4495 | /// path swallowed before. The gate is self-non-vacuous: with the `names_scalar_alone` exemption reverted the |
| 4496 | /// standalone `#edition` is recorded as a `#pagebreak`-style skip and never becomes a paragraph, so the |
| 4497 | /// substituted value is absent and a skip is reported -- both assertions below then red. The companion |
| 4498 | /// [`unbound_standalone_reference_stays_a_visible_refusal`] proves an UNBOUND standalone name still refuses. |
| 4499 | #[test] |
| 4500 | fn standalone_line_scalar_reference_substitutes() { |
| 4501 | let tfns = crate::lang::rules::TemplateFns::new(); |
| 4502 | let cfns = crate::lang::rules::ContentFns::new(); |
| 4503 | let mut sfns = crate::lang::rules::ScalarFns::new(); |
| 4504 | sfns.insert("edition".to_string(), crate::lang::rules::ScalarValue::Number("3".to_string())); |
| 4505 | let binds = crate::lang::rules::Bindings::with_scalars(&tfns, &cfns, &sfns); |
| 4506 | let src = "Some prose above.\n\n#edition\n\nSome prose below.\n"; |
| 4507 | let (items, skips) = document_with_templates(src, binds).expect("parses"); |
| 4508 | assert!(items.iter().any(|it| matches!(it, Item::Paragraph { runs, .. } |
| 4509 | if matches!(runs.as_slice(), [Inline::Text(t)] if t == "3"))), |
| 4510 | "a standalone-line scalar reference substitutes its value as its own paragraph: {:?}", items); |
| 4511 | assert!(!skips.report().is_some_and(|r| r.contains("edition")), |
| 4512 | "a bound standalone scalar reference is substituted, not recorded as a skip: {:?}", skips.report()); |
| 4513 | } |
| 4514 | |
| 4515 | /// `lorem_words` reproduces `typst 0.15.1`'s `#lorem(n)` verbatim: the classic opening for small counts, |
| 4516 | /// the last word's trailing punctuation replaced by a full stop, a zero count empty, and a huge count |
| 4517 | /// capped at the embedded corpus rather than looping. The pinned strings are the exact oracle output |
| 4518 | /// (checked against the installed typst), so a regression in the corpus or the join reds here. |
| 4519 | #[test] |
| 4520 | fn lorem_matches_the_typst_oracle() { |
| 4521 | assert_eq!(lorem_words(1), "Lorem."); |
| 4522 | assert_eq!(lorem_words(5), "Lorem ipsum dolor sit amet."); |
| 4523 | // Word 20 ("quaerat") carries no source punctuation; the full stop is appended. |
| 4524 | assert_eq!(lorem_words(20), |
| 4525 | "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magnam aliquam quaerat."); |
| 4526 | // Word 49 ("voluptatem,") carries a comma in the source, replaced by the terminal full stop. |
| 4527 | assert!(lorem_words(49).ends_with("transferre in voluptatem."), |
| 4528 | "the terminal comma must become a full stop: {:?}", lorem_words(49)); |
| 4529 | assert_eq!(lorem_words(0), ""); |
| 4530 | // A count past the corpus is capped, not looped: it returns the whole embedded passage, ending in a |
| 4531 | // full stop, with exactly the corpus's word count. |
| 4532 | let corpus_len = LOREM_CORPUS.split_whitespace().count(); |
| 4533 | assert_eq!(lorem_words(100_000).split_whitespace().count(), corpus_len); |
| 4534 | assert!(lorem_words(100_000).ends_with('.')); |
| 4535 | } |
| 4536 | |
| 4537 | /// An own-line `#pagebreak()`, `#lorem(n)` and `#v(<abs len>)` are set on the block path rather than |
| 4538 | /// tallied as skipped constructs: the page break lowers to an `Item::PageBreak`, the lorem call to a |
| 4539 | /// plain paragraph of the oracle text, and the absolute vertical space to an `Item::Space`. A relative |
| 4540 | /// `#v(2em)` has no running size here, so it stays a visible refusal. |
| 4541 | #[test] |
| 4542 | fn block_builtins_are_set_not_skipped() -> Outcome<()> { |
| 4543 | let src = "Opening prose.\n\n#lorem(5)\n\n#v(12pt)\n\n#pagebreak()\n\nAfter.\n"; |
| 4544 | let (items, skips) = res!(document_with_refusals(src)); |
| 4545 | assert!(skips.report().is_none(), "no builtin should be tallied as skipped: {:?}", skips.report()); |
| 4546 | assert!(items.iter().any(|it| matches!(it, Item::PageBreak { .. })), "the page break is set: {:?}", items); |
| 4547 | assert!(items.iter().any(|it| matches!(it, Item::Space { .. })), "the absolute #v is set: {:?}", items); |
| 4548 | assert!(items.iter().any(|it| matches!(it, |
| 4549 | Item::Paragraph { runs, .. } if matches!(runs.as_slice(), [Inline::Text(t)] if t == "Lorem ipsum dolor sit amet."))), |
| 4550 | "the #lorem call sets its oracle text as a paragraph: {:?}", items); |
| 4551 | // A relative `#v(2em)` cannot be resolved here, so it stays a visible refusal rather than a wrong space. |
| 4552 | let (items2, skips2) = res!(document_with_refusals("#v(2em)\n")); |
| 4553 | assert!(!items2.iter().any(|it| matches!(it, Item::Space { .. })), "a relative #v must not be set: {:?}", items2); |
| 4554 | assert!(skips2.report().is_some(), "a relative #v is refused visibly"); |
| 4555 | Ok(()) |
| 4556 | } |
| 4557 | |
| 4558 | /// Gathers the plain text of every top-level paragraph, for the trailing-prose and continuation checks. |
| 4559 | #[cfg(test)] |
| 4560 | fn paragraph_text(items: &[Item]) -> String { |
| 4561 | items.iter().filter_map(|it| match it { |
| 4562 | Item::Paragraph { runs, .. } => Some(runs.iter().filter_map(|r| match r { |
| 4563 | Inline::Text(t) => Some(t.clone()), |
| 4564 | _ => None, |
| 4565 | }).collect::<String>()), |
| 4566 | _ => None, |
| 4567 | }).collect::<Vec<_>>().join(" ") |
| 4568 | } |
| 4569 | |
| 4570 | /// A balanced builtin call with prose trailing its `)` is NOT own-line: it falls through to the existing |
| 4571 | /// visible refusal, which keeps the trailing prose. This is the silent-loss regression the milestone audit |
| 4572 | /// flagged -- `#pagebreak() text`, `#lorem(5) text`, `#v(12pt) text` must not drop the trailing words nor |
| 4573 | /// set the builtin as a block. |
| 4574 | #[test] |
| 4575 | fn trailing_prose_after_a_builtin_is_kept() -> Outcome<()> { |
| 4576 | for src in [ |
| 4577 | "#pagebreak() and then more prose.\n", |
| 4578 | "#lorem(5) and then more prose.\n", |
| 4579 | "#v(12pt) and then more prose.\n", |
| 4580 | ] { |
| 4581 | let (items, _skips) = res!(document_with_refusals(src)); |
| 4582 | assert!(!items.iter().any(|it| matches!(it, Item::PageBreak { .. } | Item::Space { .. })), |
| 4583 | "a builtin with trailing prose must not be set as a block: {:?} -> {:?}", src, items); |
| 4584 | let body = paragraph_text(&items); |
| 4585 | assert!(body.contains("and then more prose"), |
| 4586 | "the trailing prose must survive for {:?}: body {:?}", src, body); |
| 4587 | assert!(!body.contains("Lorem ipsum dolor sit amet"), |
| 4588 | "a trailing-prose #lorem must not expand as a block for {:?}", src); |
| 4589 | } |
| 4590 | Ok(()) |
| 4591 | } |
| 4592 | |
| 4593 | /// An own-line builtin on the line directly after prose, with no blank line between, closes the paragraph |
| 4594 | /// and sets its own block -- exactly as `= heading\n#pagebreak()` and `#section-banner` already do, and as |
| 4595 | /// Typst 0.15.1 renders it. A blank line before the builtin is not required: the earlier deferral silently |
| 4596 | /// swallowed a `#pagebreak()`/`#v()` set directly beneath a paragraph, which is the render bug the live |
| 4597 | /// drive found (a paragraph before the break collapsed the page count to one; a bare heading before it did |
| 4598 | /// not, because a heading line opens no paragraph). The preceding prose survives as its own paragraph. A |
| 4599 | /// builtin with prose on the SAME line (`#lorem(5) more`) is still not own-line -- see |
| 4600 | /// [`trailing_prose_after_a_builtin_is_kept`] -- so inline mid-prose support remains a later unit. |
| 4601 | #[test] |
| 4602 | fn own_line_builtin_beneath_prose_flushes_and_sets() -> Outcome<()> { |
| 4603 | // `#lorem` beneath prose expands its oracle text as a fresh paragraph, and the prose above it is kept. |
| 4604 | let (items, skips) = res!(document_with_refusals("Some opening prose here.\n#lorem(5)\n")); |
| 4605 | assert!(items.iter().any(|it| matches!(it, |
| 4606 | Item::Paragraph { runs, .. } if matches!(runs.as_slice(), [Inline::Text(t)] if t == "Lorem ipsum dolor sit amet."))), |
| 4607 | "a #lorem directly beneath prose sets its oracle text as a block: {:?}", items); |
| 4608 | assert!(paragraph_text(&items).contains("Some opening prose here"), |
| 4609 | "the preceding prose is kept: {:?}", items); |
| 4610 | assert!(skips.report().is_none(), "an own-line builtin beneath prose is set, not refused: {:?}", skips.report()); |
| 4611 | |
| 4612 | // `#pagebreak` beneath prose sets the break; the prose is kept. |
| 4613 | let (items, skips) = res!(document_with_refusals("Some opening prose here.\n#pagebreak()\n")); |
| 4614 | assert!(items.iter().any(|it| matches!(it, Item::PageBreak { .. })), |
| 4615 | "a #pagebreak directly beneath prose is set: {:?}", items); |
| 4616 | assert!(paragraph_text(&items).contains("Some opening prose here"), "the preceding prose is kept: {:?}", items); |
| 4617 | assert!(skips.report().is_none(), "the break is set, not refused: {:?}", skips.report()); |
| 4618 | |
| 4619 | // `#v(<abs>)` beneath prose sets the space; the prose is kept. |
| 4620 | let (items, skips) = res!(document_with_refusals("Some opening prose here.\n#v(12pt)\n")); |
| 4621 | assert!(items.iter().any(|it| matches!(it, Item::Space { .. })), |
| 4622 | "an absolute #v directly beneath prose is set: {:?}", items); |
| 4623 | assert!(paragraph_text(&items).contains("Some opening prose here"), "the preceding prose is kept: {:?}", items); |
| 4624 | assert!(skips.report().is_none(), "the space is set, not refused: {:?}", skips.report()); |
| 4625 | Ok(()) |
| 4626 | } |
| 4627 | |
| 4628 | /// A `#pagebreak()` nested in a `#styled-box[ ... ]` callout body cannot be honoured -- the box is one keep |
| 4629 | /// unit -- so it is a visible refusal, not a silent drop, and no `Item::PageBreak` survives inside the box. |
| 4630 | #[test] |
| 4631 | fn page_break_inside_a_box_is_refused_not_dropped() -> Outcome<()> { |
| 4632 | let src = "#styled-box[\nInside the callout.\n\n#pagebreak()\n\nStill inside.\n]\n"; |
| 4633 | let (items, skips) = res!(document_with_refusals(src)); |
| 4634 | assert!(skips.report().map_or(false, |r| r.contains("#pagebreak")), |
| 4635 | "the boxed page break is refused visibly: {:?}", skips.report()); |
| 4636 | fn has_page_break(items: &[Item]) -> bool { |
| 4637 | items.iter().any(|it| match it { |
| 4638 | Item::PageBreak { .. } => true, |
| 4639 | Item::Box { items, .. } => has_page_break(items), |
| 4640 | Item::Scoped { items, .. } => has_page_break(items), |
| 4641 | _ => false, |
| 4642 | }) |
| 4643 | } |
| 4644 | assert!(!has_page_break(&items), "no page break may survive inside the box: {:?}", items); |
| 4645 | Ok(()) |
| 4646 | } |
| 4647 | |
| 4648 | /// An ignored keyword argument is refused, not silently set as the wrong thing: `#pagebreak(to: "odd")` |
| 4649 | /// selects a parity target the reader does not model, and `#v(24pt, weak: true)` asks for a collapsing |
| 4650 | /// space it does not model -- each stays a visible refusal rather than a plain break or a fixed space. |
| 4651 | #[test] |
| 4652 | fn unsupported_keyword_args_are_refused() -> Outcome<()> { |
| 4653 | let (items, skips) = res!(document_with_refusals("#pagebreak(to: \"odd\")\n")); |
| 4654 | assert!(!items.iter().any(|it| matches!(it, Item::PageBreak { .. })), "a `to:` pagebreak is not set: {:?}", items); |
| 4655 | assert!(skips.report().map_or(false, |r| r.contains("#pagebreak")), "a `to:` pagebreak is refused: {:?}", skips.report()); |
| 4656 | let (items2, skips2) = res!(document_with_refusals("#v(24pt, weak: true)\n")); |
| 4657 | assert!(!items2.iter().any(|it| matches!(it, Item::Space { .. })), "a weak #v is not set: {:?}", items2); |
| 4658 | assert!(skips2.report().map_or(false, |r| r.contains("#v")), "a weak #v is refused: {:?}", skips2.report()); |
| 4659 | Ok(()) |
| 4660 | } |
| 4661 | |
| 4662 | /// `_compress-codes`: a run of three or more consecutive same-prefix codes collapses to an en-dash |
| 4663 | /// range, a pair stays expanded, two or fewer codes join unchanged, and an unparseable run passes through. |
| 4664 | #[test] |
| 4665 | fn claim_codes_compress_consecutive_runs() { |
| 4666 | let display = |s: &str| parse_inlines(s).into_iter() |
| 4667 | .find_map(|r| match r { Inline::MarginNote { display, .. } => Some(display), _ => None }) |
| 4668 | .unwrap_or_default(); |
| 4669 | assert_eq!(display("x#claim-label(<B1>, <B2>, <B3>, <B4>)"), "B1\u{2013}4"); // B1–4 |
| 4670 | assert_eq!(display("x#claim-label(<A1>, <A2>)"), "A1 A2"); |
| 4671 | assert_eq!(display("x#claim-label(<CD14>, <CD15>, <CD4>)"), "CD14 CD15 CD4"); |
| 4672 | assert_eq!(display("x#claim-label(<LS8>)"), "LS8"); |
| 4673 | } |
| 4674 | |
| 4675 | /// A line that opens with a claim marker is prose, not a standalone call the line scanner skips, so |
| 4676 | /// the sentence that follows the marker is set rather than dropped with it. |
| 4677 | #[test] |
| 4678 | fn line_leading_claim_marker_is_prose() { |
| 4679 | assert!(is_inline_call("claim-label")); |
| 4680 | assert!(is_inline_call("claim-refs")); |
| 4681 | assert!(code_skip("#claim-label(<CD18>). Equilibrium appropriation follows.").is_none()); |
| 4682 | } |
| 4683 | |
| 4684 | /// A claim reference in a context the layout does not gather into the reverse claim index -- here a |
| 4685 | /// heading title -- is recorded as a refusal rather than dropped silently, while the same reference in a |
| 4686 | /// body paragraph (which IS gathered) draws no refusal. Guards the silent-loss path the audit flagged. |
| 4687 | #[test] |
| 4688 | fn claim_ref_in_a_non_body_context_is_refused_not_dropped() { |
| 4689 | let (_items, skips) = document_with_refusals("= Heading #claim-refs(<Z9>) here\n\nBody text follows.\n").expect("parse"); |
| 4690 | assert!(skips.sites().iter().any(|s| s.name.contains("claim reference") && s.name.contains("heading")), |
| 4691 | "a claim reference in a heading title must be a refusal: {:?}", skips.sites()); |
| 4692 | // A claim reference in a body paragraph is gathered into the index, so it is NOT refused. |
| 4693 | let (_i2, skips2) = document_with_refusals("Body carrying a reference#claim-refs(<Z9>) here.\n").expect("parse"); |
| 4694 | assert!(!skips2.sites().iter().any(|s| s.name.contains("claim reference")), |
| 4695 | "a body claim reference is indexed, not refused: {:?}", skips2.sites()); |
| 4696 | } |
| 4697 | |
| 4698 | /// A line-leading `#padded-image(...)` (a section opener's logo) is set as an [`Item::Image`] carrying |
| 4699 | /// its path and scale, not skipped as a template call and not wrapped in a numbered figure. |
| 4700 | #[test] |
| 4701 | fn line_leading_padded_image_reads_as_image() -> Outcome<()> { |
| 4702 | let (items, _skips) = res!(document_with_refusals( |
| 4703 | "= Pearl\n\n#padded-image(\"assets/svg/pearlite_logo_text_right.svg\", scale: 45%)\n\nPearl is the format.\n")); |
| 4704 | let img = res!(items.iter().find_map(|it| match it { |
| 4705 | Item::Image { path, scale, .. } => Some((path.clone(), *scale)), |
| 4706 | _ => None, |
| 4707 | }).ok_or_else(|| err!("no Item::Image was produced for the standalone padded-image"; Test, Bug))); |
| 4708 | assert_eq!(img.0, "assets/svg/pearlite_logo_text_right.svg", "the image path is read"); |
| 4709 | assert_eq!(img.1, Some(0.45), "the padded-image scale is read as a fraction"); |
| 4710 | // The line must not have been swallowed as a skipped construct, nor turned into a figure. |
| 4711 | assert!(!items.iter().any(|it| matches!(it, Item::Figure { .. })), "a section logo is not a figure"); |
| 4712 | Ok(()) |
| 4713 | } |
| 4714 | |
| 4715 | /// A line-leading `#section-banner("logo")` (a documentation section opener) is read as an |
| 4716 | /// [`Item::SectionBanner`] carrying its logo path, not skipped as a template call and not confused with a |
| 4717 | /// plain `#image`. |
| 4718 | #[test] |
| 4719 | fn line_leading_section_banner_reads_as_banner() -> Outcome<()> { |
| 4720 | let (items, skips) = res!(document_with_refusals( |
| 4721 | "#section-banner(\"assets/svg/fe2o3_logo_text_right.svg\")\n\n= Steel Server\n\nSteel is the server.\n")); |
| 4722 | let path = res!(items.iter().find_map(|it| match it { |
| 4723 | Item::SectionBanner { path, .. } => Some(path.clone()), |
| 4724 | _ => None, |
| 4725 | }).ok_or_else(|| err!("no Item::SectionBanner was produced for the standalone section-banner"; Test, Bug))); |
| 4726 | assert_eq!(path, "assets/svg/fe2o3_logo_text_right.svg", "the banner logo path is read"); |
| 4727 | // The call must not have been swallowed as a skipped construct, nor read as a plain centred image. |
| 4728 | assert!(!skips.entries().iter().any(|(name, _)| name == "#section-banner"), |
| 4729 | "a section banner must not be reported as a skipped construct"); |
| 4730 | assert!(!items.iter().any(|it| matches!(it, Item::Image { .. })), "a section banner is not a plain image"); |
| 4731 | Ok(()) |
| 4732 | } |
| 4733 | |
| 4734 | /// A line-leading `#styled-box[...]` callout, its body opening on the marker line and closing on a later |
| 4735 | /// one, is gathered whole and read as an [`Item::Box`] holding the re-parsed body -- not skipped as an |
| 4736 | /// unbalanced standalone call, which would drop the callout's text. The construct is set, so it is not |
| 4737 | /// reported as a skipped construct. |
| 4738 | #[test] |
| 4739 | fn line_leading_styled_box_reads_as_box() -> Outcome<()> { |
| 4740 | let (items, skips) = res!(document_with_refusals( |
| 4741 | "Lead prose.\n\n#styled-box[\n*Principle.* Every participant is accountable.\n]\n\nTrailing prose.\n")); |
| 4742 | let inner = res!(items.iter().find_map(|it| match it { |
| 4743 | Item::Box { items, .. } => Some(items.clone()), |
| 4744 | _ => None, |
| 4745 | }).ok_or_else(|| err!("no Item::Box was produced for the standalone styled-box"; Test, Bug))); |
| 4746 | // The body re-parses to a paragraph, and its lead-in bold survives as a strong run. |
| 4747 | let has_para = inner.iter().any(|it| matches!(it, Item::Paragraph { .. })); |
| 4748 | assert!(has_para, "the styled-box body must re-parse to a paragraph, got: {:?}", inner); |
| 4749 | let has_strong = inner.iter().any(|it| matches!(it, |
| 4750 | Item::Paragraph { runs, .. } if runs.iter().any(|r| matches!(r, Inline::Strong(t) if t == "Principle.")))); |
| 4751 | assert!(has_strong, "the body's bold lead-in must survive, got: {:?}", inner); |
| 4752 | // The callout must not have been swallowed as a skipped construct. |
| 4753 | assert!(!skips.entries().iter().any(|(name, _)| name == "#styled-box"), |
| 4754 | "a styled-box must not be reported as a skipped construct"); |
| 4755 | // The prose either side of the callout still sets. |
| 4756 | assert!(items.iter().any(|it| matches!(it, Item::Paragraph { runs, .. } |
| 4757 | if runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Lead prose."))))), |
| 4758 | "prose before the callout is dropped"); |
| 4759 | Ok(()) |
| 4760 | } |
| 4761 | |
| 4762 | /// A call to a bound `#let` furniture function -- `#pr-note[ ... ]` -- is gathered whole and expanded |
| 4763 | /// into an [`Item::Box`] carrying the definition's lowered patch (a transparent-wash, asymmetric-inset |
| 4764 | /// block), its `[ ... ]` body re-parsed into the box. The call is set, not skipped, so it is not tallied; |
| 4765 | /// an unbound `#name[...]` (no definition in scope) still falls through to be reported as a skip. |
| 4766 | #[test] |
| 4767 | fn pr_note_call_expands_to_a_box_and_is_not_skipped() -> Outcome<()> { |
| 4768 | let def = "#let pr-note(body) = block(inset: (left: 1.2em, right: 0.6em), above: 0.9em, below: 1.1em, \ |
| 4769 | { set text(size: 0.88em); set par(spacing: 0.55em, first-line-indent: 0em); body })\n"; |
| 4770 | let mut tfns = crate::lang::rules::TemplateFns::new(); |
| 4771 | crate::lang::rules::collect_template_fns(def, crate::ir::Sp::from_pt(10.0), |
| 4772 | &crate::lang::rules::Palette::new(), &mut tfns); |
| 4773 | assert!(tfns.contains_key("pr-note"), "the definition is collected"); |
| 4774 | |
| 4775 | let src = "Lead prose.\n\n#pr-note[\n*Baseline:* one measure.\n\nA second paragraph.\n]\n\nTrailing prose.\n"; |
| 4776 | let (items, skips) = res!(document_with_templates(src, crate::lang::rules::Bindings::new(&tfns, &crate::lang::rules::ContentFns::new()))); |
| 4777 | let (inner, patch) = res!(items.iter().find_map(|it| match it { |
| 4778 | Item::Box { items, patch, .. } => Some((items.clone(), patch.clone())), |
| 4779 | _ => None, |
| 4780 | }).ok_or_else(|| err!("no Item::Box was produced for the pr-note call"; Test, Bug))); |
| 4781 | // The box carries the pr-note geometry: an asymmetric inset and a transparent (no-wash) fill. |
| 4782 | assert_eq!(patch.callout.inset_left, Some(crate::ir::Sp::from_pt(12.0)), "left inset resolved at 10pt body"); |
| 4783 | assert_eq!(patch.callout.fill.map(|c| c.a), Some(0), "no fill -- a plain indented block"); |
| 4784 | assert_eq!(patch.text.body_size, Some(crate::ir::Sp::from_pt(8.8)), "the body sets at 0.88em"); |
| 4785 | // The body re-parses to paragraphs, its bold lead-in surviving as a strong run. |
| 4786 | assert!(inner.iter().any(|it| matches!(it, |
| 4787 | Item::Paragraph { runs, .. } if runs.iter().any(|r| matches!(r, Inline::Strong(t) if t == "Baseline:")))), |
| 4788 | "the body's bold lead-in must survive, got: {:?}", inner); |
| 4789 | // The call is not tallied as a skipped construct, and the surrounding prose still sets. |
| 4790 | assert!(!skips.entries().iter().any(|(name, _)| name == "#pr-note"), |
| 4791 | "a bound furniture call must not be reported as a skip"); |
| 4792 | assert!(items.iter().any(|it| matches!(it, Item::Paragraph { runs, .. } |
| 4793 | if runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Lead prose."))))), |
| 4794 | "prose before the call is dropped"); |
| 4795 | |
| 4796 | // With no definition in scope, the same call is left to be tallied as a skip, unchanged. |
| 4797 | let (_it2, skips2) = res!(document_with_refusals(src)); |
| 4798 | assert!(skips2.entries().iter().any(|(name, _)| name == "#pr-note"), |
| 4799 | "an unbound furniture call still reports as a skip"); |
| 4800 | Ok(()) |
| 4801 | } |
| 4802 | |
| 4803 | /// A `#aside-box(title: [...])[ ... ]` call expands into a washed box carrying the definition's fill and |
| 4804 | /// left stroke, its `title:` argument set as a leading bold paragraph ahead of the body, and it is not |
| 4805 | /// tallied as a skip. A bound call with no `[ ... ]` body (an argument-only `#aside-box(...)`) is TALLIED |
| 4806 | /// as a skip rather than dropped silently. |
| 4807 | #[test] |
| 4808 | fn aside_box_call_expands_with_title_and_stroke() -> Outcome<()> { |
| 4809 | let def = "#let aside-box(title: none, body) = box(width: 100%, inset: 8pt, \ |
| 4810 | fill: colours.yellow.lighten(50%), radius: 4pt, stroke: (left: 2pt + colours.yellow.darken(20%)), \ |
| 4811 | [#text(weight: \"bold\", size: 0.85em)[#title] #text(size: 0.85em)[#body]])\n"; |
| 4812 | let mut palette = crate::lang::rules::Palette::new(); |
| 4813 | crate::lang::rules::collect_palette("#let colours = (yellow: rgb(\"#f0f600\"),)\n", &mut palette); |
| 4814 | let mut tfns = crate::lang::rules::TemplateFns::new(); |
| 4815 | crate::lang::rules::collect_template_fns(def, crate::ir::Sp::from_pt(11.0), &palette, &mut tfns); |
| 4816 | let tf = res!(tfns.get("aside-box").ok_or_else(|| err!("aside-box collected"; Test, Bug))); |
| 4817 | assert!(tf.patch.callout.fill.map(|c| c.a) == Some(255), "the yellow fill resolved (opaque)"); |
| 4818 | assert!(tf.patch.callout.stroke_left_w.is_some(), "the left stroke width is set"); |
| 4819 | |
| 4820 | let src = "Lead.\n\n#aside-box(title: [The welfare theorems])[\nMarket efficiency proved.\n]\n\nTail.\n"; |
| 4821 | let (items, skips) = res!(document_with_templates(src, crate::lang::rules::Bindings::new(&tfns, &crate::lang::rules::ContentFns::new()))); |
| 4822 | let inner = res!(items.iter().find_map(|it| match it { |
| 4823 | Item::Box { items, .. } => Some(items.clone()), |
| 4824 | _ => None, |
| 4825 | }).ok_or_else(|| err!("no Item::Box for the aside-box call"; Test, Bug))); |
| 4826 | // The first item is the bold title (in a size scope), then the body. |
| 4827 | let has_title = inner.iter().any(|it| match it { |
| 4828 | Item::Scoped { items, .. } => items.iter().any(|p| matches!(p, |
| 4829 | Item::Paragraph { runs, .. } if runs.iter().any(|r| matches!(r, Inline::Strong(t) if t.contains("welfare"))))), |
| 4830 | Item::Paragraph { runs, .. } => runs.iter().any(|r| matches!(r, Inline::Strong(t) if t.contains("welfare"))), |
| 4831 | _ => false, |
| 4832 | }); |
| 4833 | assert!(has_title, "the title is set as a leading bold paragraph, got: {:?}", inner); |
| 4834 | assert!(!skips.entries().iter().any(|(n, _)| n == "#aside-box"), "the call is not a skip"); |
| 4835 | |
| 4836 | // A bound call with no `[body]` is tallied as a skip, not dropped. |
| 4837 | let (_it, skips2) = res!(document_with_templates("#aside-box(title: [X])\n", crate::lang::rules::Bindings::new(&tfns, &crate::lang::rules::ContentFns::new()))); |
| 4838 | assert!(skips2.entries().iter().any(|(n, _)| n == "#aside-box"), |
| 4839 | "an argument-only bound call with no body is tallied as a skip"); |
| 4840 | Ok(()) |
| 4841 | } |
| 4842 | |
| 4843 | /// A `#columns[...]` body's own top-level `#set` declarations scope to the spliced subtree (H1's |
| 4844 | /// flat-splice sibling): the reader nests the spliced items inside one `Item::Scoped` carrying the |
| 4845 | /// lowered patch. A columns body that declares nothing splices in flat, with no scope. |
| 4846 | #[test] |
| 4847 | fn columns_body_set_scopes_the_spliced_subtree() -> Outcome<()> { |
| 4848 | let (items, _skips) = res!(document_with_refusals( |
| 4849 | "#columns(2)[\n#set text(size: 20pt)\n\nScoped body.\n]\n")); |
| 4850 | let (patch, inner) = res!(items.iter().find_map(|it| match it { |
| 4851 | Item::Scoped { patch, items } => Some((patch.clone(), items)), |
| 4852 | _ => None, |
| 4853 | }).ok_or_else(|| err!("no Item::Scoped was produced for a columns body with a #set"; Test, Bug))); |
| 4854 | assert_eq!(patch.text.body_size, Some(crate::ir::Sp::from_pt(20.0)), |
| 4855 | "the columns body's #set text(size:) did not lower into the scope patch"); |
| 4856 | assert!(!inner.is_empty(), "the scope must carry the body's items"); |
| 4857 | |
| 4858 | // A columns body that declares nothing splices in flat, with no scope. |
| 4859 | let (plain, _) = res!(document_with_refusals("#columns(2)[\nPlain body.\n]\n")); |
| 4860 | assert!(!plain.iter().any(|it| matches!(it, Item::Scoped { .. })), |
| 4861 | "a columns body with no #set must not be wrapped in a scope"); |
| 4862 | Ok(()) |
| 4863 | } |
| 4864 | |
| 4865 | /// H2: a lowerable `#set` that lowers to nothing (an unrecognised argument) is recorded as a visible |
| 4866 | /// refusal named for the set, rather than silently dropped, while one that fully lowers records none. |
| 4867 | #[test] |
| 4868 | fn unconsumed_set_is_recorded_as_a_refusal() -> Outcome<()> { |
| 4869 | let (_items, skips) = res!(document_with_refusals("#set text(lang: \"de\")\n\nBody.\n")); |
| 4870 | assert!(skips.entries().iter().any(|(name, _)| name == "#set text"), |
| 4871 | "an unconsumed #set must be recorded as a refusal, got: {:?}", skips.entries()); |
| 4872 | |
| 4873 | let (_items2, skips2) = res!(document_with_refusals("#set text(size: 12pt)\n\nBody.\n")); |
| 4874 | assert!(!skips2.entries().iter().any(|(name, _)| name == "#set text"), |
| 4875 | "a fully-lowered #set must not be recorded as a refusal, got: {:?}", skips2.entries()); |
| 4876 | Ok(()) |
| 4877 | } |
| 4878 | |
| 4879 | /// `#emph[...]` is the call form of `_..._`: it yields an [`Inline::Emph`] run with the same inner |
| 4880 | /// text, and nothing raw leaks. |
| 4881 | #[test] |
| 4882 | fn emph_call_reads_as_emphasis() { |
| 4883 | let runs = parse_inlines("The cube asks #emph[who] does the extracting."); |
| 4884 | assert!(runs.iter().any(|r| matches!(r, Inline::Emph(t) if t == "who")), |
| 4885 | "emph run missing: {:?}", runs); |
| 4886 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#emph"))), |
| 4887 | "raw #emph leaked: {:?}", runs); |
| 4888 | // A call carrying its own markup expands the same way `_..._` does. |
| 4889 | let nested = parse_inlines("#emph[the #idx[Harvard Business Review] weekly]"); |
| 4890 | assert!(nested.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#emph") || t.contains("#idx"))), |
| 4891 | "raw markup leaked from nested emph: {:?}", nested); |
| 4892 | } |
| 4893 | |
| 4894 | /// An index marker carries a sort key distinct from its display: `#idx-as[Abbott, Andrew][Andrew Abbott]` |
| 4895 | /// files under the surname but shows "Andrew Abbott" (in the body and on the index page), and `#idx[_x_]` |
| 4896 | /// keeps its emphasis as an [`Inline::Emph`] display run while its sort key is the flattened plain text -- |
| 4897 | /// so the index sets the display, not the sort key, and an emphasised entry italicises rather than printing |
| 4898 | /// literal underscores. |
| 4899 | #[test] |
| 4900 | fn index_marker_splits_sort_key_from_styled_display() { |
| 4901 | let runs = parse_inlines("The sociologist #idx-as[Abbott, Andrew][Andrew Abbott] wrote widely."); |
| 4902 | let mut found = false; |
| 4903 | for r in &runs { |
| 4904 | if let Inline::Index { term, sub, display, .. } = r { |
| 4905 | assert_eq!(term, "Abbott, Andrew", "sort key wrong: {:?}", runs); |
| 4906 | assert!(sub.is_none()); |
| 4907 | assert!(matches!(display.as_slice(), [Inline::Text(t)] if t == "Andrew Abbott"), |
| 4908 | "display wrong: {:?}", display); |
| 4909 | found = true; |
| 4910 | } |
| 4911 | } |
| 4912 | assert!(found, "no index marker found: {:?}", runs); |
| 4913 | // The visible display "Andrew Abbott" is set in the body beside the marker. |
| 4914 | assert!(runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Andrew Abbott"))), |
| 4915 | "body display missing: {:?}", runs); |
| 4916 | |
| 4917 | // An emphasised entry: the sort key is flattened, the display keeps the emphasis. |
| 4918 | let ital = parse_inlines("the ruling #idx[_Browder v. Gayle_] held."); |
| 4919 | let mut seen = false; |
| 4920 | for r in &ital { |
| 4921 | if let Inline::Index { term, display, .. } = r { |
| 4922 | assert_eq!(term, "Browder v. Gayle", "italic sort key not flattened: {:?}", ital); |
| 4923 | assert!(matches!(display.as_slice(), [Inline::Emph(t)] if t == "Browder v. Gayle"), |
| 4924 | "italic display lost its emphasis: {:?}", display); |
| 4925 | seen = true; |
| 4926 | } |
| 4927 | } |
| 4928 | assert!(seen, "no italic index marker found: {:?}", ital); |
| 4929 | } |
| 4930 | |
| 4931 | /// `#strong[...]` and `#strong("...")` are the call forms of `*...*`: both yield an [`Inline::Strong`] |
| 4932 | /// run rather than a skip, and nothing raw leaks. A paren argument the reader cannot evaluate -- a |
| 4933 | /// bare identifier here -- keeps the current refusal rather than guessing at its text. |
| 4934 | #[test] |
| 4935 | fn strong_call_reads_as_bold() -> Outcome<()> { |
| 4936 | let bracket = parse_inlines("You want: #strong[the short version] first."); |
| 4937 | assert!(bracket.iter().any(|r| matches!(r, Inline::Strong(t) if t == "the short version")), |
| 4938 | "strong run missing from bracket form: {:?}", bracket); |
| 4939 | assert!(bracket.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#strong"))), |
| 4940 | "raw #strong leaked: {:?}", bracket); |
| 4941 | |
| 4942 | let paren = parse_inlines("#strong(\"the short version\") first."); |
| 4943 | assert!(paren.iter().any(|r| matches!(r, Inline::Strong(t) if t == "the short version")), |
| 4944 | "strong run missing from paren form: {:?}", paren); |
| 4945 | |
| 4946 | // A call carrying its own markup expands the same way `*...*` does. |
| 4947 | let nested = parse_inlines("#strong[the #idx[Harvard Business Review] weekly]"); |
| 4948 | assert!(nested.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#strong") || t.contains("#idx"))), |
| 4949 | "raw markup leaked from nested strong: {:?}", nested); |
| 4950 | |
| 4951 | // A paren argument that is not a plain string is left for the generic call handler, so it is |
| 4952 | // still tallied as a skip rather than silently guessed at. |
| 4953 | let (_it, skips) = res!(document_with_refusals("#strong(ident)\n\nBody.\n")); |
| 4954 | assert!(skips.entries().iter().any(|(n, _)| n == "#strong"), |
| 4955 | "an unresolvable #strong(...) argument must still be recorded as a refusal, got: {:?}", |
| 4956 | skips.entries()); |
| 4957 | Ok(()) |
| 4958 | } |
| 4959 | |
| 4960 | /// Emphasis nested one level -- `*_x_*` or `_*x*_` -- collapses to a single [`Inline::BoldItalic`] |
| 4961 | /// run rather than dropping the outer face, so a bold-italic term sets in the bold-italic face in |
| 4962 | /// prose, a footnote or a table cell alike. |
| 4963 | #[test] |
| 4964 | fn nested_emphasis_reads_as_bold_italic() { |
| 4965 | for src in ["here *_both_* faces", "here _*both*_ faces"] { |
| 4966 | let runs = parse_inlines(src); |
| 4967 | assert!(runs.iter().any(|r| matches!(r, Inline::BoldItalic(t) if t == "both")), |
| 4968 | "expected a bold-italic run from {:?}, got {:?}", src, runs); |
| 4969 | assert!(runs.iter().all(|r| !matches!(r, Inline::Emph(t) if t == "both")), |
| 4970 | "outer face dropped to plain emph in {:?}: {:?}", src, runs); |
| 4971 | } |
| 4972 | // A lone `*strong*` or `_emph_` still yields its single face, unchanged. |
| 4973 | assert!(parse_inlines("just *strong* here").iter().any(|r| matches!(r, Inline::Strong(t) if t == "strong"))); |
| 4974 | assert!(parse_inlines("just _emph_ here").iter().any(|r| matches!(r, Inline::Emph(t) if t == "emph"))); |
| 4975 | } |
| 4976 | |
| 4977 | /// A `(col, row) => ...` alignment closure is evaluated per cell: a row-keyed closure centres the |
| 4978 | /// header and flushes the body left; a per-column closure honours each column's own alignment, |
| 4979 | /// including an `or` over several columns. |
| 4980 | #[test] |
| 4981 | fn align_closure_evaluates_per_cell() { |
| 4982 | let row_keyed = parse_align("(col, row) => { if row == 0 { center } else { left } }"); |
| 4983 | match row_keyed { |
| 4984 | AlignSpec::Closure(cl) => { |
| 4985 | assert_eq!(cl.align_at(0, 0), Align::Centre); // header, column 0 |
| 4986 | assert_eq!(cl.align_at(0, 1), Align::Left); // body, column 0 -- was wrongly centred before |
| 4987 | assert_eq!(cl.align_at(2, 3), Align::Left); |
| 4988 | }, |
| 4989 | other => panic!("expected a closure, got {:?}", other), |
| 4990 | } |
| 4991 | let per_col = parse_align("(col, row) => { if row == 0 { center } else if col == 0 or col == 4 { left } else { center } }"); |
| 4992 | match per_col { |
| 4993 | AlignSpec::Closure(cl) => { |
| 4994 | assert_eq!(cl.align_at(0, 1), Align::Left); |
| 4995 | assert_eq!(cl.align_at(4, 1), Align::Left); |
| 4996 | assert_eq!(cl.align_at(1, 1), Align::Centre); |
| 4997 | assert_eq!(cl.align_at(3, 2), Align::Centre); |
| 4998 | }, |
| 4999 | other => panic!("expected a closure, got {:?}", other), |
| 5000 | } |
| 5001 | // A tuple pick indexed by the column parameter. |
| 5002 | let tuple = parse_align("(x, y) => (left, center, right).at(x)"); |
| 5003 | match tuple { |
| 5004 | AlignSpec::Closure(cl) => { |
| 5005 | assert_eq!(cl.align_at(0, 5), Align::Left); |
| 5006 | assert_eq!(cl.align_at(1, 5), Align::Centre); |
| 5007 | assert_eq!(cl.align_at(2, 5), Align::Right); |
| 5008 | }, |
| 5009 | other => panic!("expected a closure, got {:?}", other), |
| 5010 | } |
| 5011 | } |
| 5012 | |
| 5013 | /// A floating `#place` reads its side, scope, clearance and body; one that does not float, or carries an |
| 5014 | /// offset, is not a float and is refused at dispatch. |
| 5015 | #[test] |
| 5016 | fn place_reads_as_a_float_or_not_at_all() { |
| 5017 | match place_float_call("#place(top + center, scope: \"parent\", float: true, clearance: 2em)[Wide.]") { |
| 5018 | Some((f, c, body)) => { |
| 5019 | assert_eq!(f, Floating { side: FloatPlacement::Top, scope: FloatScope::Parent }); |
| 5020 | assert_eq!(c, Some(Spacing::Em(2.0))); |
| 5021 | assert_eq!(body, "Wide."); |
| 5022 | }, |
| 5023 | None => panic!("a floating place must read"), |
| 5024 | } |
| 5025 | assert!(matches!(place_float_call("#place(bottom, float: true)[Low.]"), |
| 5026 | Some((Floating { side: FloatPlacement::Bottom, scope: FloatScope::Column }, None, _)))); |
| 5027 | assert!(place_float_call("#place(top)[Overlay.]").is_none(), "a non-floating place is an overlay"); |
| 5028 | assert!(place_float_call("#place(top, float: true, dx: 2pt)[Moved.]").is_none(), "an offset is not set"); |
| 5029 | } |
| 5030 | |
| 5031 | /// `#smallcaps[...]` yields an [`Inline::SmallCaps`] run of its content, in both argument forms, and |
| 5032 | /// never leaves raw source behind. |
| 5033 | #[test] |
| 5034 | fn smallcaps_call_reads_as_small_caps() { |
| 5035 | let runs = parse_inlines("The #smallcaps[Nato] treaty, and #smallcaps(\"un\") too."); |
| 5036 | let sc: Vec<&String> = runs.iter().filter_map(|r| match r { |
| 5037 | Inline::SmallCaps(t) => Some(t), |
| 5038 | _ => None, |
| 5039 | }).collect(); |
| 5040 | assert_eq!(sc, vec!["Nato", "un"], "unexpected small-caps runs: {:?}", runs); |
| 5041 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#smallcaps"))), |
| 5042 | "raw #smallcaps leaked: {:?}", runs); |
| 5043 | } |
| 5044 | |
| 5045 | /// `#super[...]` yields an [`Inline::Super`] run of its content, in both the bracket and the string |
| 5046 | /// argument forms, and never leaves raw source behind. |
| 5047 | #[test] |
| 5048 | fn super_call_reads_as_superscript() { |
| 5049 | let runs = parse_inlines("The area is 10#super[6] units, split#super[†] on the case."); |
| 5050 | let sups: Vec<&String> = runs.iter().filter_map(|r| match r { |
| 5051 | Inline::Super(t) => Some(t), |
| 5052 | _ => None, |
| 5053 | }).collect(); |
| 5054 | assert_eq!(sups, vec!["6", "†"], "unexpected superscript runs: {:?}", runs); |
| 5055 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#super"))), |
| 5056 | "raw #super leaked: {:?}", runs); |
| 5057 | // The string-argument form reads the same, its quotes stripped. |
| 5058 | let quoted = parse_inlines("x#super(\"2\")"); |
| 5059 | assert!(quoted.iter().any(|r| matches!(r, Inline::Super(t) if t == "2")), |
| 5060 | "quoted super run missing: {:?}", quoted); |
| 5061 | } |
| 5062 | |
| 5063 | /// `#sub[...]` yields an [`Inline::Sub`] run of its content, as `CO#sub[2]` sets it in Lucronics, and |
| 5064 | /// never leaves raw source behind. |
| 5065 | #[test] |
| 5066 | fn sub_call_reads_as_subscript() { |
| 5067 | let runs = parse_inlines("CO#sub[2] scrubber and H#sub[2]O#sub[2] both drop."); |
| 5068 | let subs: Vec<&String> = runs.iter().filter_map(|r| match r { |
| 5069 | Inline::Sub(t) => Some(t), |
| 5070 | _ => None, |
| 5071 | }).collect(); |
| 5072 | assert_eq!(subs, vec!["2", "2", "2"], "unexpected subscript runs: {:?}", runs); |
| 5073 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#sub"))), |
| 5074 | "raw #sub leaked: {:?}", runs); |
| 5075 | // The string-argument form reads the same, its quotes stripped. |
| 5076 | let quoted = parse_inlines("x#sub(\"2\")"); |
| 5077 | assert!(quoted.iter().any(|r| matches!(r, Inline::Sub(t) if t == "2")), |
| 5078 | "quoted sub run missing: {:?}", quoted); |
| 5079 | } |
| 5080 | |
| 5081 | /// The `.enumerate().map(((idx, row)) => { if COND { row } else { table.cell(colspan: n)[...] } }) |
| 5082 | /// .flatten()` row-remap idiom -- Lucronics' E. coli comparison uses it to merge a section-heading row |
| 5083 | /// into one bold spanning cell while an ordinary row passes through -- resolves through the spread, not |
| 5084 | /// just the untransformed array. This is the reduced shape of the real table (3 columns, one heading |
| 5085 | /// row) rather than the full 7-column original. |
| 5086 | #[test] |
| 5087 | fn table_spread_map_merges_heading_rows_into_bold_spanning_cells() { |
| 5088 | let let_src = "#let data = (\n ([Head A], [Head B], [Head C]),\n ([Section], [], []),\n ([Item one], [x], [y]),\n)\n"; |
| 5089 | let mut arrays = HashMap::new(); |
| 5090 | arrays.insert("data".to_string(), parse_let_array(let_src)); |
| 5091 | |
| 5092 | let table_src = "\n columns: 3,\n ..data.enumerate().map(((idx, row)) => {\n if idx == 0 or row.at(0) != [Section] {\n row\n } else {\n table.cell(colspan: 3)[#strong(row.at(0))]\n }\n }).flatten()\n"; |
| 5093 | let spec = parse_table_spec(table_src, &arrays, None).expect("the spread must resolve to a table"); |
| 5094 | assert_eq!(spec.cells.len(), 9, "3 rows x 3 columns, the merged row padded to width"); |
| 5095 | // Row 0 (the header) and row 2 (an ordinary item) pass through unchanged. |
| 5096 | assert!(matches!(spec.cells[0].as_slice(), [Inline::Text(t)] if t == "Head A")); |
| 5097 | assert!(matches!(spec.cells[6].as_slice(), [Inline::Text(t)] if t == "Item one")); |
| 5098 | // Row 1 (the section heading) is merged into one bold cell, padded to the column count. |
| 5099 | assert!(matches!(spec.cells[3].as_slice(), [Inline::Strong(t)] if t == "Section"), |
| 5100 | "expected the merged heading cell, got {:?}", spec.cells[3]); |
| 5101 | assert!(matches!(spec.cells[4].as_slice(), [Inline::Text(t)] if t.is_empty()), "padding cell after the span"); |
| 5102 | assert!(matches!(spec.cells[5].as_slice(), [Inline::Text(t)] if t.is_empty()), "padding cell after the span"); |
| 5103 | } |
| 5104 | |
| 5105 | /// A citation nested in emphasis keeps its own [`Inline::Cite`] run rather than leaking its source, |
| 5106 | /// while the surrounding words take the emphasis face. |
| 5107 | #[test] |
| 5108 | fn cite_survives_inside_emphasis() { |
| 5109 | let runs = parse_inlines("the classic _early work #cite(<coase1937nature>) here_."); |
| 5110 | assert!(runs.iter().any(|r| matches!(r, Inline::Cite(k) if k == &vec!["coase1937nature".to_string()])), |
| 5111 | "cite run missing: {:?}", runs); |
| 5112 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#cite"))), |
| 5113 | "raw #cite leaked: {:?}", runs); |
| 5114 | } |
| 5115 | |
| 5116 | /// `#link("url")[text]` sets the link text in the running line and drops the URL, so no raw `#link` |
| 5117 | /// leaks; a bare `#link("url")` with no bracket sets the URL as its own text. |
| 5118 | #[test] |
| 5119 | fn link_call_renders_text_not_markup() { |
| 5120 | let runs = parse_inlines("See #link(\"https://aistatement.com\")[Centre for AI Safety, May 2023] on risk."); |
| 5121 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#link") || t.contains("http"))), |
| 5122 | "raw link markup leaked: {:?}", runs); |
| 5123 | let joined = flatten_markup("See #link(\"https://aistatement.com\")[Centre for AI Safety, May 2023] on risk."); |
| 5124 | assert_eq!(joined, "See Centre for AI Safety, May 2023 on risk."); |
| 5125 | // The label-destination form and the bare no-body form both set text without leaking. |
| 5126 | assert_eq!(flatten_markup("read #link(<intro>)[the opening]"), "read the opening"); |
| 5127 | assert_eq!(flatten_markup("at #link(\"elearnity.io\")"), "at elearnity.io"); |
| 5128 | // A line-leading link is prose the inline scanner reads, not a standalone call the line scanner drops. |
| 5129 | assert!(is_inline_call("link")); |
| 5130 | assert!(code_skip("#link(\"https://x.io\")[click]").is_none()); |
| 5131 | } |
| 5132 | |
| 5133 | /// Typst's smartypants: `--` becomes an en dash and `---` an em dash in ordinary prose, matched |
| 5134 | /// longest-first, but a raw code span, a maths span and an escaped hyphen are all left untouched. |
| 5135 | #[test] |
| 5136 | fn dash_runs_become_en_and_em_dashes_in_prose() { |
| 5137 | assert_eq!(flatten_markup("a--b"), "a\u{2013}b"); |
| 5138 | assert_eq!(flatten_markup("a---b"), "a\u{2014}b"); |
| 5139 | // Four hyphens is an em dash plus a literal hyphen, greedy left to right, not two en dashes. |
| 5140 | assert_eq!(flatten_markup("a----b"), "a\u{2014}-b"); |
| 5141 | assert_eq!(flatten_markup("Council of Trent (1545--1563)"), "Council of Trent (1545\u{2013}1563)"); |
| 5142 | // A raw code span keeps its `--` literal: the inline scanner claims `` ` `` before this fallback |
| 5143 | // is ever reached, so the substitution never sees the span's characters. |
| 5144 | let runs = parse_inlines("see `a--b` here"); |
| 5145 | assert!(runs.iter().any(|r| matches!(r, Inline::Code(t) if t == "a--b")), |
| 5146 | "raw span lost or converted: {:?}", runs); |
| 5147 | // A maths span keeps its `-` as subtraction, for the same reason. |
| 5148 | let runs = parse_inlines("$a-b$"); |
| 5149 | assert!(matches!(runs.as_slice(), [Inline::Math(_)]), "maths span not read as maths: {:?}", runs); |
| 5150 | // A `\-` escape is consumed on its own, so it never joins the hyphen after it into a run: the |
| 5151 | // pair stays two literal ASCII hyphens rather than folding to the single en-dash glyph `a--b` |
| 5152 | // (unescaped) becomes. |
| 5153 | assert_eq!(flatten_markup("a\\--b"), "a--b"); |
| 5154 | // `#link`'s destination is dropped entirely (austenite renders no clickable URL), so a `--` inside |
| 5155 | // one is never seen at all -- the strictest match to Typst leaving an auto-linked URL unconverted. |
| 5156 | assert_eq!(flatten_markup("see #link(\"https://x--y.com\")[the site]"), "see the site"); |
| 5157 | } |
| 5158 | |
| 5159 | /// The same fallback trivially covers Typst's other markup substitution on dots: three or more become |
| 5160 | /// an ellipsis, fewer stay literal, and the rule is still skipped inside code and maths. |
| 5161 | #[test] |
| 5162 | fn dot_runs_become_ellipsis_in_prose() { |
| 5163 | assert_eq!(flatten_markup("wait..."), "wait\u{2026}"); |
| 5164 | assert_eq!(flatten_markup("a....b"), "a\u{2026}.b"); |
| 5165 | assert_eq!(flatten_markup("a..b"), "a..b"); // two dots: no defined symbol, left alone |
| 5166 | assert_eq!(flatten_markup("v1.2.3"), "v1.2.3"); // scattered single dots: unaffected |
| 5167 | let runs = parse_inlines("`a...b`"); |
| 5168 | assert!(matches!(runs.as_slice(), [Inline::Code(t)] if t == "a...b"), "raw span converted: {:?}", runs); |
| 5169 | } |
| 5170 | |
| 5171 | /// The term-dictionary aliases set their argument text with the styling of their `gs` siblings: `g`/`gi` |
| 5172 | /// a first-use glossary term, `t`/`tcap` plain text, and none of them leak raw markup. A line opening |
| 5173 | /// with one is prose, not a skipped standalone call. |
| 5174 | /// Installs a fixed term dictionary so the term-dictionary family resolves deterministically. The map |
| 5175 | /// is a process-global shared across the parallel tests, so every term-dependent test installs the |
| 5176 | /// same one and their order cannot matter. |
| 5177 | fn install_test_terms() { |
| 5178 | let mut m = HashMap::new(); |
| 5179 | m.insert("org".to_string(), "Elearnity Pty Ltd".to_string()); |
| 5180 | m.insert("org_short".to_string(), "Elearnity".to_string()); |
| 5181 | m.insert("website".to_string(), "elearnity.oxegen.io".to_string()); |
| 5182 | m.insert("iniverse".to_string(), "iniverse".to_string()); |
| 5183 | set_term_dict(m).expect("install test term dict"); |
| 5184 | } |
| 5185 | |
| 5186 | #[test] |
| 5187 | fn term_dict_aliases_render_without_leaking() { |
| 5188 | install_test_terms(); |
| 5189 | // `iniverse` translates to itself, so the display equals the key here whether or not a map is set. |
| 5190 | let runs = parse_inlines("Call it the #g[iniverse], your inner universe."); |
| 5191 | assert!(runs.iter().any(|r| matches!(r, Inline::Glossary { term, display } if term == "iniverse" && display == "iniverse")), |
| 5192 | "glossary alias missing: {:?}", runs); |
| 5193 | // A key that differs from its value now sets the value, not the key. |
| 5194 | assert_eq!(flatten_markup("Visit #t[website] today"), "Visit elearnity.oxegen.io today"); |
| 5195 | // A key absent from the dictionary falls back to its own text, capitalised for the `-cap` form. |
| 5196 | assert_eq!(flatten_markup("#tcap[donate] to help"), "Donate to help"); |
| 5197 | assert!(is_inline_call("g") && is_inline_call("t") && is_inline_call("graw")); |
| 5198 | assert!(code_skip("#t[website]").is_none()); |
| 5199 | } |
| 5200 | |
| 5201 | /// The term-dictionary family sets a key's value: `t`/`graw` plain, `tcap` capitalised, `g` bold-italic |
| 5202 | /// on first use keyed by the key; an unknown key falls back to the key text and is recorded, never a |
| 5203 | /// panic (the template panics on a miss, the reader must not). |
| 5204 | #[test] |
| 5205 | fn term_dict_resolves_key_to_value() { |
| 5206 | install_test_terms(); |
| 5207 | assert_eq!(flatten_markup("Visit #t[website]"), "Visit elearnity.oxegen.io"); |
| 5208 | assert_eq!(flatten_markup("#graw[org]"), "Elearnity Pty Ltd"); |
| 5209 | let runs = parse_inlines("The #g[org] view."); |
| 5210 | assert!(runs.iter().any(|r| matches!(r, Inline::Glossary { term, display } |
| 5211 | if term == "org" && display == "Elearnity Pty Ltd")), |
| 5212 | "g did not translate the key to its value: {:?}", runs); |
| 5213 | // An unknown key: the key text stands and the miss is recorded on the skip tally. |
| 5214 | let mut skips = Refusals::default(); |
| 5215 | let runs = parse_inlines_in("A #t[nonesuch] term.", Span::new(0, 0), &mut skips); |
| 5216 | assert!(runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("nonesuch"))), |
| 5217 | "unknown term-dict key did not fall back to its text: {:?}", runs); |
| 5218 | assert_eq!(skips.total(), 1, "an unknown term-dict key was not recorded"); |
| 5219 | } |
| 5220 | |
| 5221 | /// A term-dictionary lookup inside a `#let` content-function body, keyed on the function's own parameter |
| 5222 | /// (`#let cite-term(w) = [Learn about #t(w).]`), resolves against the argument the caller passed -- |
| 5223 | /// `#cite-term("website")` must look up "website", not the literal parameter name "w" -- so substitution |
| 5224 | /// must run before the nested `#t` call is read. Reverting the ordering fix (substituting only bare |
| 5225 | /// `#param` markup references, never a bare identifier inside a call's argument list) reproduces exactly |
| 5226 | /// the reported failure: the key "w" is looked up, is not in the dictionary, and the fallback plus a |
| 5227 | /// recorded skip fire instead of the resolved value. |
| 5228 | #[test] |
| 5229 | fn term_dict_resolves_inside_a_content_fn_body_after_param_substitution() { |
| 5230 | install_test_terms(); |
| 5231 | let mut cfns = crate::lang::rules::ContentFns::new(); |
| 5232 | cfns.insert("cite-term".to_string(), crate::lang::rules::ContentFn { |
| 5233 | params: vec!["w".to_string()], |
| 5234 | body: "Learn about #t(w).".to_string(), |
| 5235 | wrapper: None, |
| 5236 | }); |
| 5237 | let tfns = crate::lang::rules::TemplateFns::new(); |
| 5238 | let binds = crate::lang::rules::Bindings::new(&tfns, &cfns); |
| 5239 | let (items, skips) = document_with_templates("#cite-term(\"website\")\n", binds).expect("parse"); |
| 5240 | let resolved = items.iter().any(|it| matches!(it, |
| 5241 | Item::Paragraph { runs, .. } if runs.iter().any(|r| |
| 5242 | matches!(r, Inline::Text(t) if t.contains("elearnity.oxegen.io"))))); |
| 5243 | assert!(resolved, |
| 5244 | "the content-fn's own parameter did not resolve as the term-dict key: {:?}", items); |
| 5245 | assert_eq!(skips.total(), 0, "a resolved key must not be recorded as an unknown term-dict miss: {:?}", skips); |
| 5246 | } |
| 5247 | |
| 5248 | /// The ordering fix is scoped to a call's argument list, not to ordinary prose: a parameter's name |
| 5249 | /// appearing as a genuine word in the body's running text (outside any call) is left exactly as written, |
| 5250 | /// since a bare word in markup mode is prose, never a code-mode variable reference. |
| 5251 | #[test] |
| 5252 | fn call_arg_substitution_does_not_touch_ordinary_prose() { |
| 5253 | let cf = crate::lang::rules::ContentFn { |
| 5254 | params: vec!["w".to_string()], |
| 5255 | body: "The word w on its own is prose, not #t(w).".to_string(), |
| 5256 | wrapper: None, |
| 5257 | }; |
| 5258 | let expanded = expand_content_body(&cf, &["website".to_string()]); |
| 5259 | assert_eq!(expanded, "The word w on its own is prose, not #t(\"website\").", |
| 5260 | "prose text was wrongly substituted, or the call argument was not: {:?}", expanded); |
| 5261 | } |
| 5262 | |
| 5263 | /// A blank line between numbered items does not restart the enum: Typst continues the numbering across |
| 5264 | /// the gap, so the items form one list. Real content between two lists still starts a fresh one. |
| 5265 | #[test] |
| 5266 | fn blank_line_between_enum_items_continues_one_list() { |
| 5267 | let src = "+ first item\n\n+ second item\n\n+ third item\n"; |
| 5268 | let (items, _) = document_with_refusals(src).expect("parse"); |
| 5269 | let lists: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::List { .. })).collect(); |
| 5270 | assert_eq!(lists.len(), 1, "blank lines split the enum: {:?}", items); |
| 5271 | match lists[0] { |
| 5272 | Item::List { ordered, items, .. } => { |
| 5273 | assert!(*ordered, "the continued list lost its ordered kind"); |
| 5274 | assert_eq!(items.len(), 3, "the enum dropped items across the blanks: {:?}", items); |
| 5275 | }, |
| 5276 | _ => unreachable!(), |
| 5277 | } |
| 5278 | // A paragraph between two lists still restarts, so genuinely separate lists are not merged. |
| 5279 | let src2 = "+ a\n\n+ b\n\nA paragraph between.\n\n+ c\n"; |
| 5280 | let (items2, _) = document_with_refusals(src2).expect("parse"); |
| 5281 | let lists2 = items2.iter().filter(|it| matches!(it, Item::List { .. })).count(); |
| 5282 | assert_eq!(lists2, 2, "prose between two lists did not restart them: {:?}", items2); |
| 5283 | } |
| 5284 | |
| 5285 | /// An indented `-` sub-bullet between two `+` steps nests under the step it follows rather than closing |
| 5286 | /// the enum: the ordered list stays one list of three items, and the sub-bullet hangs under the first. |
| 5287 | #[test] |
| 5288 | fn indented_sub_bullet_nests_and_enum_continues() { |
| 5289 | let src = "+ step one\n - a sub point\n - another sub point\n+ step two\n+ step three\n"; |
| 5290 | let (items, _) = document_with_refusals(src).expect("parse"); |
| 5291 | let lists: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::List { .. })).collect(); |
| 5292 | assert_eq!(lists.len(), 1, "the sub-bullet split the enum into several lists: {:?}", items); |
| 5293 | match lists[0] { |
| 5294 | Item::List { ordered, items, .. } => { |
| 5295 | assert!(*ordered, "the parent list lost its ordered kind"); |
| 5296 | assert_eq!(items.len(), 3, "the enum did not keep three steps: {:?}", items); |
| 5297 | // The sub-bullets hang under the first step, as an unordered child list of two items. |
| 5298 | assert_eq!(items[0].children.len(), 1, "the first step lost its sub-list: {:?}", items[0]); |
| 5299 | match &items[0].children[0] { |
| 5300 | Item::List { ordered: cord, items: citems, .. } => { |
| 5301 | assert!(!*cord, "the sub-list should be unordered"); |
| 5302 | assert_eq!(citems.len(), 2, "the sub-list dropped an item: {:?}", citems); |
| 5303 | }, |
| 5304 | other => panic!("the child was not a nested list: {:?}", other), |
| 5305 | } |
| 5306 | assert!(items[1].children.is_empty(), "step two wrongly gained children"); |
| 5307 | }, |
| 5308 | _ => unreachable!(), |
| 5309 | } |
| 5310 | } |
| 5311 | |
| 5312 | /// Two levels of indentation parse to two levels of nesting: a `-` under a `+`, and a deeper `-` under |
| 5313 | /// that `-`, so the tree is enum -> bullet -> bullet. |
| 5314 | #[test] |
| 5315 | fn two_level_nesting_parses_to_two_levels() { |
| 5316 | let src = "+ outer step\n - middle bullet\n - inner bullet\n+ next step\n"; |
| 5317 | let (items, _) = document_with_refusals(src).expect("parse"); |
| 5318 | let lists: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::List { .. })).collect(); |
| 5319 | assert_eq!(lists.len(), 1, "the deep nesting split the list: {:?}", items); |
| 5320 | match lists[0] { |
| 5321 | Item::List { items, .. } => { |
| 5322 | assert_eq!(items.len(), 2, "the outer enum did not keep two steps: {:?}", items); |
| 5323 | let mid = &items[0].children; |
| 5324 | assert_eq!(mid.len(), 1, "the middle level is missing: {:?}", items[0]); |
| 5325 | match &mid[0] { |
| 5326 | Item::List { items: mid_items, .. } => { |
| 5327 | assert_eq!(mid_items.len(), 1, "the middle list should hold one bullet"); |
| 5328 | let inner = &mid_items[0].children; |
| 5329 | assert_eq!(inner.len(), 1, "the inner level is missing: {:?}", mid_items[0]); |
| 5330 | match &inner[0] { |
| 5331 | Item::List { items: inner_items, .. } => |
| 5332 | assert_eq!(inner_items.len(), 1, "the inner list should hold one bullet"), |
| 5333 | other => panic!("the inner child was not a list: {:?}", other), |
| 5334 | } |
| 5335 | }, |
| 5336 | other => panic!("the middle child was not a list: {:?}", other), |
| 5337 | } |
| 5338 | }, |
| 5339 | _ => unreachable!(), |
| 5340 | } |
| 5341 | } |
| 5342 | |
| 5343 | /// An unhandled inline `#func[...]` is consumed and recorded rather than left as raw markup, its |
| 5344 | /// bracketed body folded in so its words survive; a paren-only call sets nothing where it stood. |
| 5345 | #[test] |
| 5346 | fn unknown_inline_call_is_recorded_not_leaked() { |
| 5347 | let mut skips = Refusals::default(); |
| 5348 | let runs = parse_inlines_in("a #overline[Nato] treaty and a #v(2pt) gap", Span::new(0, 0), &mut skips); |
| 5349 | assert!(runs.iter().all(|r| !matches!(r, Inline::Text(t) if t.contains("#overline") || t.contains("#v("))), |
| 5350 | "raw unknown call leaked: {:?}", runs); |
| 5351 | assert!(runs.iter().any(|r| matches!(r, Inline::Text(t) if t.contains("Nato"))), |
| 5352 | "overline body dropped: {:?}", runs); |
| 5353 | assert_eq!(skips.total(), 2); |
| 5354 | let names: Vec<String> = skips.entries().into_iter().map(|(n, _)| n).collect(); |
| 5355 | assert!(names.contains(&"#overline".to_string()) && names.contains(&"#v".to_string()), |
| 5356 | "unexpected skip names: {:?}", names); |
| 5357 | } |
| 5358 | |
| 5359 | /// The reader tallies the code lines and unknown calls it skips, and reports them one line, so a |
| 5360 | /// dropped construct is visible rather than silent. Handled inline calls do not appear in the tally. |
| 5361 | #[test] |
| 5362 | fn skip_summary_reports_skipped_constructs() { |
| 5363 | install_test_terms(); // so `#g[iniverse]` resolves and adds no term-dict miss to the tally |
| 5364 | // `#import` and `#set rect` (which names no theme element) stay reader refusals. A per-element |
| 5365 | // `#show <selector>: ...` line is a rule the engine collects and applies (or refuses in its own |
| 5366 | // diagnostic), so the reader captures it as a declarative-styling construct and no longer tallies it |
| 5367 | // -- see `show_selector_rule_is_not_a_reader_skip` and the L0 double-report fix. |
| 5368 | let src = "#import \"x.typ\": *\n#set rect(stroke: 1pt)\n\nBody with #g[iniverse] and a #footnote[note].\n\n#show heading: it => it\n"; |
| 5369 | let (_, skips) = document_with_refusals(src).expect("parse"); |
| 5370 | assert_eq!(skips.total(), 2, "the selector show rule is captured for the engine, not tallied: {:?}", skips.sites()); |
| 5371 | let report = skips.report().expect("a report"); |
| 5372 | assert!(report.starts_with("skipped 2 unsupported constructs:"), "report was {:?}", report); |
| 5373 | for name in ["#import", "#set"] { |
| 5374 | assert!(report.contains(name), "{} missing from {:?}", name, report); |
| 5375 | } |
| 5376 | assert!(!report.contains("heading"), |
| 5377 | "a selector show rule must not appear in the reader's skip tally (it is the engine's to apply/refuse): {:?}", report); |
| 5378 | // A source the reader sets whole has nothing to report. |
| 5379 | let (_, clean) = document_with_refusals("Just prose with #g[iniverse].\n").expect("parse"); |
| 5380 | assert!(clean.is_empty() && clean.report().is_none()); |
| 5381 | } |
| 5382 | |
| 5383 | /// A `#show <selector>: <transform>` line is a per-element rule the engine collects and applies, so the |
| 5384 | /// reader captures it as a declarative-styling construct rather than tallying it as a skipped one -- the |
| 5385 | /// L0 double-report fix. Both the lowerable set-fields form and the refused introspective form are |
| 5386 | /// captured; the refused one surfaces through the rule engine's own diagnostic, not the reader's tally. |
| 5387 | #[test] |
| 5388 | fn show_selector_rule_is_not_a_reader_skip() { |
| 5389 | let (_, lowerable) = document_with_refusals("#show par: set text(size: 9pt)\n\nBody.\n").expect("parse"); |
| 5390 | assert_eq!(lowerable.total(), 0, "a lowerable selector rule is not a reader skip: {:?}", lowerable.sites()); |
| 5391 | let (_, introspective) = document_with_refusals("#show heading: it => it\n\nBody.\n").expect("parse"); |
| 5392 | assert_eq!(introspective.total(), 0, |
| 5393 | "a refused selector rule surfaces via the engine, not the reader tally: {:?}", introspective.sites()); |
| 5394 | } |
| 5395 | |
| 5396 | /// A `#show: <template>.with(...)` application and a lowerable top-level `#set` are captured rather |
| 5397 | /// than refused -- their styling lowers onto the theme -- so neither adds to the refusal tally, while |
| 5398 | /// an introspective `#show` and a `#set` on an unsupported target still do. |
| 5399 | #[test] |
| 5400 | fn lowerable_set_and_doc_with_are_captured_not_refused() { |
| 5401 | // A multi-line `#show: doc.with(...)` and a lowerable `#set text(...)`: both captured, no refusal. |
| 5402 | let lowered = "#show: doc.with(\n title: [X],\n heading-font: \"Graystroke\",\n)\n\n#set text(size: 11pt)\n\n= Heading\n\nBody.\n"; |
| 5403 | let (_, skips) = document_with_refusals(lowered).expect("parse"); |
| 5404 | assert_eq!(skips.total(), 0, "lowerable declarations should not be refused: {:?}", skips.sites()); |
| 5405 | |
| 5406 | // The unsupported `#set` still refuses; a `#show <selector>:` rule is now the engine's to apply or |
| 5407 | // refuse, so it is captured here rather than tallied by the reader (see |
| 5408 | // `show_selector_rule_is_not_a_reader_skip`). |
| 5409 | let refused = "#set rect(stroke: 1pt)\n\n#show heading: it => it\n"; |
| 5410 | let (_, skips) = document_with_refusals(refused).expect("parse"); |
| 5411 | assert_eq!(skips.total(), 1, "the unsupported #set still refuses; the show rule is captured for the engine: {:?}", skips.sites()); |
| 5412 | } |
| 5413 | |
| 5414 | /// A line-leading `#context[...]` -- Typst's self-observation entry point -- is refused as exactly |
| 5415 | /// one site, classed `Introspective`. It is now gathered as a capture (so its body can be inspected for |
| 5416 | /// the reverse-claim-index signature) and refused when that signature is absent, so its span is the |
| 5417 | /// zero-width caret at the construct's opening offset, as the reader's other capture refusals record. |
| 5418 | #[test] |
| 5419 | fn context_call_is_one_introspective_refusal() { |
| 5420 | let src = "#context[whatever]\n"; |
| 5421 | let (_, refusals) = document_with_refusals(src).expect("parse"); |
| 5422 | assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites()); |
| 5423 | let site = &refusals.sites()[0]; |
| 5424 | assert_eq!(site.name, "#context"); |
| 5425 | assert_eq!(site.class, RefusalClass::Introspective); |
| 5426 | assert_eq!(site.span, Span::new(0, 0), "a captured-construct refusal records the caret at its opening offset"); |
| 5427 | } |
| 5428 | |
| 5429 | /// The brace twin of the above -- a line-leading `#context{ ... }` code-block call, the shape a book's |
| 5430 | /// reverse-reference index is written with (Lucronics ch29.8) -- is recognised and refused the same |
| 5431 | /// way, not set as prose. Both a one-line block and a multi-line one are refused as one introspective |
| 5432 | /// site, and neither leaks its source into the body: the reader must produce no paragraph at all. |
| 5433 | #[test] |
| 5434 | fn context_brace_block_is_refused_not_set_as_prose() { |
| 5435 | // One line, brace immediately after the keyword. |
| 5436 | let one = "#context{ let x = 1 }\n"; |
| 5437 | let (items, refusals) = document_with_refusals(one).expect("parse"); |
| 5438 | assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites()); |
| 5439 | assert_eq!(refusals.sites()[0].name, "#context"); |
| 5440 | assert_eq!(refusals.sites()[0].class, RefusalClass::Introspective); |
| 5441 | assert!(!items.iter().any(|it| matches!(it, Item::Paragraph { .. })), |
| 5442 | "a #context{{}} block must not survive as body text: {:?}", items); |
| 5443 | |
| 5444 | // Several lines, with a space before the brace -- a multi-line `#context {` block that does NOT call |
| 5445 | // `collect-claim-refs(` (so it is a plain introspective refusal, not the reverse claim index). The |
| 5446 | // whole block, including its own `[...]` and nested `{...}`, is consumed by the capture, not one line |
| 5447 | // of it set as prose. |
| 5448 | let many = "Before.\n\n#context {\n let by = (:)\n if by.len() == 0 [\n _None._\n ] else {\n let n = 1\n }\n}\n\nAfter.\n"; |
| 5449 | let (items, refusals) = document_with_refusals(many).expect("parse"); |
| 5450 | assert_eq!(refusals.total(), 1, "the multi-line brace block is one refusal: {:?}", refusals.sites()); |
| 5451 | assert_eq!(refusals.sites()[0].name, "#context"); |
| 5452 | let bodies: Vec<String> = items.iter().filter_map(|it| match it { |
| 5453 | Item::Paragraph { runs, .. } => Some(fmt!("{:?}", runs)), |
| 5454 | _ => None, |
| 5455 | }).collect(); |
| 5456 | assert!(!bodies.iter().any(|b| b.contains("None") || b.contains("let by") || b.contains("by-code")), |
| 5457 | "no line of the #context{{}} block may leak into a paragraph: {:?}", bodies); |
| 5458 | // The two real paragraphs around it still set. |
| 5459 | assert_eq!(bodies.len(), 2, "the prose on either side of the block must still set: {:?}", bodies); |
| 5460 | } |
| 5461 | |
| 5462 | /// A `//` or `/* ... */` comment inside a `#context { ... }` body must not fold its own `}`/`]` into the |
| 5463 | /// skip scanner's bracket balance -- the G3 fix. Before it, `let c = 1 // }` popped the outer brace early, |
| 5464 | /// so the tail of the block (the `if`/`else` and the closing `}`) leaked into the body as raw prose; this |
| 5465 | /// reds on a reverted `step` exactly the way the earlier form's leak did. The body calls a non-claim query |
| 5466 | /// (`counter(page).display()`, not `collect-claim-refs(`), so it stays a plain introspective refusal and |
| 5467 | /// does not trip the reverse-claim-index recognition covered by |
| 5468 | /// `context_collect_claim_refs_lowers_to_a_claim_index` below. |
| 5469 | #[test] |
| 5470 | fn context_brace_block_comment_does_not_close_early() { |
| 5471 | let many = "Before.\n\n#context {\n let refs = counter(page).display()\n let c = 1 // }\n /* a note about } */\n if refs.len() == 0 [\n _None._\n ] else {\n let by = (:)\n }\n}\n\nAfter.\n"; |
| 5472 | let (items, refusals) = document_with_refusals(many).expect("parse"); |
| 5473 | assert_eq!(refusals.total(), 1, "the whole commented block is still one refusal: {:?}", refusals.sites()); |
| 5474 | assert_eq!(refusals.sites()[0].name, "#context"); |
| 5475 | let bodies: Vec<String> = items.iter().filter_map(|it| match it { |
| 5476 | Item::Paragraph { runs, .. } => Some(fmt!("{:?}", runs)), |
| 5477 | _ => None, |
| 5478 | }).collect(); |
| 5479 | assert!(!bodies.iter().any(|b| b.contains("counter") || b.contains("let refs") || b.contains("let by")), |
| 5480 | "a comment's `}}` must not close the guard early and leak its tail: {:?}", bodies); |
| 5481 | assert_eq!(bodies.len(), 2, "the prose on either side of the block must still set: {:?}", bodies); |
| 5482 | } |
| 5483 | |
| 5484 | /// The one `#context { ... }` block the reader does not refuse: the Logic appendix's reverse claim index, |
| 5485 | /// recognised by the `collect-claim-refs(` signature in its body (never by evaluating the `#context`). It |
| 5486 | /// lowers to a single `Item::ClaimIndex`, records no refusal, and leaks no line of its source as prose. |
| 5487 | #[test] |
| 5488 | fn context_collect_claim_refs_lowers_to_a_claim_index() { |
| 5489 | let src = "Before.\n\n#context {\n let refs = collect-claim-refs()\n if refs.len() == 0 [\n _None._\n ] else {\n let by = (:)\n }\n}\n\nAfter.\n"; |
| 5490 | let (items, refusals) = document_with_refusals(src).expect("parse"); |
| 5491 | assert_eq!(refusals.total(), 0, "the reverse claim index is recognised, not refused: {:?}", refusals.sites()); |
| 5492 | assert_eq!(items.iter().filter(|it| matches!(it, Item::ClaimIndex { .. })).count(), 1, |
| 5493 | "the collect-claim-refs block lowers to exactly one ClaimIndex: {:?}", items); |
| 5494 | let bodies: Vec<String> = items.iter().filter_map(|it| match it { |
| 5495 | Item::Paragraph { runs, .. } => Some(fmt!("{:?}", runs)), |
| 5496 | _ => None, |
| 5497 | }).collect(); |
| 5498 | assert!(!bodies.iter().any(|b| b.contains("collect") || b.contains("let refs") || b.contains("None")), |
| 5499 | "no line of the block may leak into a paragraph: {:?}", bodies); |
| 5500 | assert_eq!(bodies.len(), 2, "the prose on either side of the block must still set: {:?}", bodies); |
| 5501 | } |
| 5502 | |
| 5503 | /// A line-leading `#query(...)` -- reading the document's own resolved structure back -- is refused |
| 5504 | /// as exactly one site, classed `Introspective`. |
| 5505 | #[test] |
| 5506 | fn query_call_is_one_introspective_refusal() { |
| 5507 | let src = "#query(heading)\n"; |
| 5508 | let (_, refusals) = document_with_refusals(src).expect("parse"); |
| 5509 | assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites()); |
| 5510 | let site = &refusals.sites()[0]; |
| 5511 | assert_eq!(site.name, "#query"); |
| 5512 | assert_eq!(site.class, RefusalClass::Introspective); |
| 5513 | assert_eq!(site.span, Span::new(0, src.len() as u32 - 1)); |
| 5514 | } |
| 5515 | |
| 5516 | /// A line-leading `#state(...)` call -- a state read/write that only resolves against Typst's own |
| 5517 | /// layout observation -- is refused as exactly one site, classed `Introspective`. |
| 5518 | #[test] |
| 5519 | fn state_call_is_one_introspective_refusal() { |
| 5520 | let src = "#state(\"count\", 0)\n"; |
| 5521 | let (_, refusals) = document_with_refusals(src).expect("parse"); |
| 5522 | assert_eq!(refusals.total(), 1, "expected exactly one refusal: {:?}", refusals.sites()); |
| 5523 | let site = &refusals.sites()[0]; |
| 5524 | assert_eq!(site.name, "#state"); |
| 5525 | assert_eq!(site.class, RefusalClass::Introspective); |
| 5526 | assert_eq!(site.span, Span::new(0, src.len() as u32 - 1)); |
| 5527 | } |
| 5528 | |
| 5529 | /// The three classes land where the classifier's own doc comment says they should: Typst's general |
| 5530 | /// evaluation primitives are `FixedPoint`, an unrecognised call or wrapper is `Unsupported`. |
| 5531 | #[test] |
| 5532 | fn refusal_class_sorts_fixed_point_and_unsupported_correctly() { |
| 5533 | assert_eq!(RefusalClass::classify("#let"), RefusalClass::FixedPoint); |
| 5534 | assert_eq!(RefusalClass::classify("#show"), RefusalClass::FixedPoint); |
| 5535 | assert_eq!(RefusalClass::classify("#columns"), RefusalClass::Unsupported); |
| 5536 | assert_eq!(RefusalClass::classify("#overline"), RefusalClass::Unsupported); |
| 5537 | } |
| 5538 | |
| 5539 | /// A `#columns(n)[ ... ]` wrapper is recorded as skipped and its body set single-column, so the words |
| 5540 | /// survive and no raw wrapper leaks into the block stream. |
| 5541 | #[test] |
| 5542 | fn columns_wrapper_flattens_to_single_column() { |
| 5543 | let src = "#columns(2)[\nFirst paragraph here.\n\nSecond paragraph here.\n]\n"; |
| 5544 | let (items, skips) = document_with_refusals(src).expect("parse"); |
| 5545 | let paras = items.iter().filter(|it| matches!(it, Item::Paragraph { .. })).count(); |
| 5546 | assert_eq!(paras, 2, "column body not set as paragraphs: {:?}", items); |
| 5547 | assert_eq!(skips.entries(), vec![("#columns".to_string(), 1)]); |
| 5548 | } |
| 5549 | |
| 5550 | /// Reads a single [`Item::Paragraph`]'s runs out of a parse, failing loudly with the whole item list |
| 5551 | /// when the source did not yield exactly one paragraph -- the shape every `math_open` regression test |
| 5552 | /// below expects, since a display block that leaked a line would instead split the source into a |
| 5553 | /// paragraph plus a stray heading or list. |
| 5554 | fn one_paragraph(items: &[Item]) -> Outcome<(Vec<Inline>, Option<String>)> { |
| 5555 | let paras: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::Paragraph { .. })).collect(); |
| 5556 | match paras.as_slice() { |
| 5557 | [Item::Paragraph { runs, label, .. }] => Ok((runs.clone(), label.clone())), |
| 5558 | _ => Err(err!("expected exactly one paragraph, got: {:?}", items; Test, Bug)), |
| 5559 | } |
| 5560 | } |
| 5561 | |
| 5562 | /// A source line inside an open `$...$` display block that begins `=` (an alignment row such as |
| 5563 | /// `=> 2N &= ...`, Oxegen TechSpec `app_maths.typ:533-534`) must stay in the equation, not be read as |
| 5564 | /// a heading -- the heading branch is gated off by `math_open` for exactly this shape. |
| 5565 | #[test] |
| 5566 | fn equals_lead_row_inside_display_math_stays_in_block() -> Outcome<()> { |
| 5567 | let src = "Total hashes.\n\n$\nN &= sum_(j=1)^J n_j \\\n=> 2N &= sum_(j=2)^(J+1) 2^(j-1) \\\n=> 2N - N &= 2^J - 1 \\\n$\n\nEnd of block.\n"; |
| 5568 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5569 | assert!(!items.iter().any(|it| matches!(it, Item::Heading { .. })), |
| 5570 | "a `=`-lead row inside the block must not become a heading: {:?}", items); |
| 5571 | let has_align = items.iter().any(|it| matches!(it, |
| 5572 | Item::Paragraph { runs, .. } if matches!(runs.as_slice(), |
| 5573 | [Inline::Math(Atom::Matrix { kind: MatKind::Align, .. })]))); |
| 5574 | assert!(has_align, "no single-run display alignment paragraph was produced: {:?}", items); |
| 5575 | Ok(()) |
| 5576 | } |
| 5577 | |
| 5578 | /// A source line inside an open `$...$` display block that begins `-` (a row such as |
| 5579 | /// `- tilde(N)_(1 0) u (...) = 0`, Oxegen TechSpec `ch04_nodes.typ:1657`) must stay in the equation, |
| 5580 | /// not be read as a bullet -- the list-marker branch is gated off by `math_open` for exactly this shape. |
| 5581 | #[test] |
| 5582 | fn dash_lead_row_inside_display_math_stays_in_block() -> Outcome<()> { |
| 5583 | let src = "$\na &= b \\\n- tilde(N)_(1 0) u (x) = 0 \\\nc &= d\n$\n"; |
| 5584 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5585 | assert!(!items.iter().any(|it| matches!(it, Item::List { .. })), |
| 5586 | "a `-`-lead row inside the block must not become a list: {:?}", items); |
| 5587 | let (runs, _label) = res!(one_paragraph(&items)); |
| 5588 | assert!(matches!(runs.as_slice(), [Inline::Math(Atom::Matrix { kind: MatKind::Align, .. })]), |
| 5589 | "expected one display alignment run, got: {:?}", runs); |
| 5590 | Ok(()) |
| 5591 | } |
| 5592 | |
| 5593 | /// A blank source line inside an open `$...$` display block (Oxegen TechSpec `ch04_nodes.typ:1661-1666`) |
| 5594 | /// must stay in the equation rather than flush the paragraph early, and a closing `$ <label>` still |
| 5595 | /// labels the resulting equation. |
| 5596 | #[test] |
| 5597 | fn blank_line_inside_display_math_stays_in_block_and_label_still_attaches() -> Outcome<()> { |
| 5598 | let src = "$\na &= b \\\n\nc &= d\n$ <eq_test>\n"; |
| 5599 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5600 | let (runs, label) = res!(one_paragraph(&items)); |
| 5601 | assert!(matches!(runs.as_slice(), [Inline::Math(Atom::Matrix { kind: MatKind::Align, .. })]), |
| 5602 | "expected one display alignment run, got: {:?}", runs); |
| 5603 | assert_eq!(label, Some("eq_test".to_string()), "the closing label must still attach"); |
| 5604 | Ok(()) |
| 5605 | } |
| 5606 | |
| 5607 | /// The plain text of a run of inline markup, for asserting a caption or paragraph's words without |
| 5608 | /// caring how they were split into text, emphasis, glossary or maths runs. |
| 5609 | fn plain(runs: &[Inline]) -> String { |
| 5610 | let mut s = String::new(); |
| 5611 | for r in runs { |
| 5612 | match r { |
| 5613 | Inline::Text(t) | Inline::Strong(t) | Inline::Emph(t) |
| 5614 | | Inline::BoldItalic(t) | Inline::Super(t) | Inline::Sub(t) | Inline::Code(t) => s.push_str(t), |
| 5615 | Inline::Glossary { display, .. } => s.push_str(display), |
| 5616 | _ => {}, |
| 5617 | } |
| 5618 | } |
| 5619 | s |
| 5620 | } |
| 5621 | |
| 5622 | /// The one [`Item::Figure`] in a parse, failing loudly with the item list when there is not exactly one. |
| 5623 | fn one_figure(items: &[Item]) -> Outcome<(Option<Vec<Inline>>, String, Option<String>)> { |
| 5624 | let figs: Vec<&Item> = items.iter().filter(|it| matches!(it, Item::Figure { .. })).collect(); |
| 5625 | match figs.as_slice() { |
| 5626 | [Item::Figure { caption, supplement, label, .. }] => |
| 5627 | Ok((caption.clone(), supplement.clone(), label.clone())), |
| 5628 | _ => Err(err!("expected exactly one figure, got: {:?}", items; Test, Bug)), |
| 5629 | } |
| 5630 | } |
| 5631 | |
| 5632 | /// A `#figure(...)` whose caption prose carries an author's unbalanced `(` (Oxegen TechSpec |
| 5633 | /// `app_maths.typ:550-566`: "...network messages (latency $100 unit(\"ms\")$ ... chunk sizes $d$.") |
| 5634 | /// must still close at its own `)`: content mode treats the stray paren as literal prose, where the old |
| 5635 | /// flat depth counter stuck open to end of source and swallowed both the figure and the tail after it. |
| 5636 | #[test] |
| 5637 | fn figure_caption_with_unbalanced_paren_closes_and_tail_survives() -> Outcome<()> { |
| 5638 | let src = "\ |
| 5639 | #figure( |
| 5640 | block(width: 70%)[ |
| 5641 | #table( |
| 5642 | columns: (10fr, 10fr), |
| 5643 | [a], [b], |
| 5644 | ) |
| 5645 | ], |
| 5646 | caption: [Representative processing times for hashing and network messages (latency $100 unit(\"ms\")$ per message for a Merkle tree of $1 unit(\"GiB\")$ of datastate with various chunk sizes $d$.], |
| 5647 | kind: \"table\", |
| 5648 | supplement: \"Table\", |
| 5649 | ) <merkle_tree_chunk_size> |
| 5650 | |
| 5651 | Trailing prose. |
| 5652 | "; |
| 5653 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5654 | let (caption, supplement, label) = res!(one_figure(&items)); |
| 5655 | let caption = res!(caption.ok_or_else(|| err!("the figure lost its caption"; Test, Bug))); |
| 5656 | assert!(plain(&caption).contains("Representative processing times"), |
| 5657 | "the caption prose was lost: {:?}", caption); |
| 5658 | assert_eq!(supplement, "Table", "the supplement was not read from the figure"); |
| 5659 | assert_eq!(label, Some("merkle_tree_chunk_size".to_string()), "the figure label was lost"); |
| 5660 | // The tail after the figure must survive rather than be swallowed by a stuck bracket count. |
| 5661 | let tail = items.iter().any(|it| matches!(it, |
| 5662 | Item::Paragraph { runs, .. } if plain(runs).contains("Trailing prose"))); |
| 5663 | assert!(tail, "the prose after the figure was swallowed: {:?}", items); |
| 5664 | Ok(()) |
| 5665 | } |
| 5666 | |
| 5667 | /// A `(`, a `)`, a `[` or a `]` inside a `$...$` maths span in a caption is literal maths, never a |
| 5668 | /// structural bracket, so an interval or a parenthesised function does not miscount and close the |
| 5669 | /// caption early. Without the maths frame, the `]` in `$[a, b]$` would pop the caption content block. |
| 5670 | #[test] |
| 5671 | fn caption_maths_parens_do_not_count() -> Outcome<()> { |
| 5672 | let src = "\ |
| 5673 | #figure( |
| 5674 | image(\"fig.png\"), |
| 5675 | caption: [see $f(x)$ over $[a, b]$ where it holds.], |
| 5676 | ) <fig_maths> |
| 5677 | |
| 5678 | After the figure. |
| 5679 | "; |
| 5680 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5681 | let (caption, _supplement, label) = res!(one_figure(&items)); |
| 5682 | let caption = res!(caption.ok_or_else(|| err!("the maths caption was lost"; Test, Bug))); |
| 5683 | assert!(plain(&caption).contains("see") && plain(&caption).contains("where it holds"), |
| 5684 | "the caption prose around the maths was truncated: {:?}", caption); |
| 5685 | assert!(caption.iter().any(|r| matches!(r, Inline::Math(_))), |
| 5686 | "the caption maths span was not parsed as maths: {:?}", caption); |
| 5687 | assert_eq!(label, Some("fig_maths".to_string()), "the label after a maths caption was lost"); |
| 5688 | assert!(items.iter().any(|it| matches!(it, |
| 5689 | Item::Paragraph { runs, .. } if plain(runs).contains("After the figure"))), |
| 5690 | "the tail after a maths caption was swallowed: {:?}", items); |
| 5691 | Ok(()) |
| 5692 | } |
| 5693 | |
| 5694 | /// A `(` or `)` inside a `"..."` string argument -- here an image path `image("a(b).png")` -- is literal |
| 5695 | /// string content, not a structural paren, so a single-line figure carrying such a path closes on its |
| 5696 | /// line rather than opening a run-away capture. |
| 5697 | #[test] |
| 5698 | fn code_string_paren_does_not_count() -> Outcome<()> { |
| 5699 | let src = "#figure(image(\"a(b).png\"), caption: [c])\n\nNext paragraph.\n"; |
| 5700 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5701 | let (caption, _supplement, _label) = res!(one_figure(&items)); |
| 5702 | let caption = res!(caption.ok_or_else(|| err!("the figure lost its caption"; Test, Bug))); |
| 5703 | assert_eq!(plain(&caption), "c", "the caption was misread past the string paren: {:?}", caption); |
| 5704 | let path = items.iter().find_map(|it| match it { |
| 5705 | Item::Figure { body: FigureBody::Image { path, .. }, .. } => Some(path.clone()), |
| 5706 | _ => None, |
| 5707 | }); |
| 5708 | assert_eq!(path, Some("a(b).png".to_string()), "the image path with parens was misread: {:?}", path); |
| 5709 | assert!(items.iter().any(|it| matches!(it, |
| 5710 | Item::Paragraph { runs, .. } if plain(runs).contains("Next paragraph"))), |
| 5711 | "the paragraph after a single-line figure was swallowed: {:?}", items); |
| 5712 | Ok(()) |
| 5713 | } |
| 5714 | |
| 5715 | /// A `#name(...)` call met inside a caption's content re-enters code mode, so the string inside it is |
| 5716 | /// protected and a following unbalanced `(` in the surrounding prose is still literal: the caption |
| 5717 | /// closes at its own `]` and the tail survives. |
| 5718 | #[test] |
| 5719 | fn content_call_inside_caption_reenters_code() -> Outcome<()> { |
| 5720 | let src = "\ |
| 5721 | #figure( |
| 5722 | image(\"g.png\"), |
| 5723 | caption: [see #link(\"http://x\")[y] and note (z], |
| 5724 | ) <fig_call> |
| 5725 | |
| 5726 | Following text. |
| 5727 | "; |
| 5728 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5729 | let (caption, _supplement, label) = res!(one_figure(&items)); |
| 5730 | let caption = res!(caption.ok_or_else(|| err!("the call caption was lost"; Test, Bug))); |
| 5731 | assert!(plain(&caption).contains("see") && plain(&caption).contains("note (z"), |
| 5732 | "the caption prose around the call was truncated: {:?}", caption); |
| 5733 | assert_eq!(label, Some("fig_call".to_string()), "the label after a call caption was lost"); |
| 5734 | assert!(items.iter().any(|it| matches!(it, |
| 5735 | Item::Paragraph { runs, .. } if plain(runs).contains("Following text"))), |
| 5736 | "the tail after a call caption was swallowed: {:?}", items); |
| 5737 | Ok(()) |
| 5738 | } |
| 5739 | |
| 5740 | /// [`read_group`] and [`split_top_args`] on a caption argument list directly: an unbalanced `(` in the |
| 5741 | /// caption prose is literal, so the group closes at its own `]` and the top-level commas still part the |
| 5742 | /// arguments, with the maths span and its string protected throughout. |
| 5743 | #[test] |
| 5744 | fn read_group_and_split_handle_caption_prose_paren() -> Outcome<()> { |
| 5745 | // read_group on a caption bracket whose prose carries an unbalanced `(` and a `$...$` span with a |
| 5746 | // quoted `unit("ms")` inside: the group must end at the caption's own `]`, keeping its prose and |
| 5747 | // stopping before the trailing text. |
| 5748 | let s: Vec<char> = "[messages (latency $100 unit(\"ms\")$ per $d$.] more".chars().collect(); |
| 5749 | let (inner, next) = res!(read_group(&s, 0) |
| 5750 | .ok_or_else(|| err!("the caption group did not close"; Test, Bug))); |
| 5751 | assert!(inner.contains("(latency"), "the caption prose was lost: {:?}", inner); |
| 5752 | assert!(!inner.contains("more"), "the caption group over-ran its closing bracket: {:?}", inner); |
| 5753 | assert_eq!(s[next..].iter().collect::<String>(), " more", "the index past the closer is wrong"); |
| 5754 | |
| 5755 | // split_top_args across the same shape: three arguments, the middle a caption whose unbalanced paren |
| 5756 | // and maths span do not part it, and named_arg reads the caption key back off it. |
| 5757 | let args = split_top_args("image(\"p.png\"), caption: [x (y $z(w)$.], kind: \"table\""); |
| 5758 | assert_eq!(args.len(), 3, "the unbalanced caption paren split the arg list wrong: {:?}", args); |
| 5759 | let named = res!(named_arg(args[1].trim()) |
| 5760 | .ok_or_else(|| err!("the caption argument did not read as named: {:?}", args[1]; Test, Bug))); |
| 5761 | assert_eq!(named.0, "caption", "the caption key was misread: {:?}", named); |
| 5762 | assert!(named.1.contains("(y"), "the caption value lost its unbalanced paren: {:?}", named); |
| 5763 | Ok(()) |
| 5764 | } |
| 5765 | |
| 5766 | /// A prose line that opens with an inline `#raw("...")` and carries a stray `"` from an author's |
| 5767 | /// quotation split across the line break must be set as prose, not read as the start of a multi-line |
| 5768 | /// code skip: a dangling quote is a character, not an open code string. Only an unclosed structural |
| 5769 | /// bracket keeps a skip open (Hematite `sec_net.typ:679-681`, where `express "no\nbound".` split a |
| 5770 | /// quotation over two lines after an inline `#raw`). |
| 5771 | #[test] |
| 5772 | fn prose_line_with_inline_raw_and_dangling_quote_is_not_skipped() -> Outcome<()> { |
| 5773 | let src = "passes its own pair to #raw(\"read_message\"), or to\n\ |
| 5774 | #raw(\"WebSocket::with_limits\"). There is deliberately no way to express \"no\n\ |
| 5775 | bound\".\n"; |
| 5776 | let (items, _skips) = res!(document_with_refusals(src)); |
| 5777 | let text: String = items.iter() |
| 5778 | .filter_map(|it| match it { |
| 5779 | Item::Paragraph { runs, .. } => Some(plain(runs)), |
| 5780 | _ => None, |
| 5781 | }) |
| 5782 | .collect::<Vec<_>>() |
| 5783 | .join(" "); |
| 5784 | assert!(text.contains("no way to express"), |
| 5785 | "the prose after an inline #raw was swallowed as a skip: {:?}", items); |
| 5786 | assert!(text.contains("bound"), "the continuation line was swallowed: {:?}", items); |
| 5787 | Ok(()) |
| 5788 | } |
| 5789 | |
| 5790 | /// A `` `//` `` and `` `/* */` `` inside a backtick code span, in a prose caption naming the operators |
| 5791 | /// themselves, must not be read as comment openers: [`Frame::Raw`] keeps the span literal, so the group |
| 5792 | /// still closes at its own `]` and the raw span survives verbatim in the inner text. |
| 5793 | #[test] |
| 5794 | fn read_group_protects_backtick_span_with_comment_markers() -> Outcome<()> { |
| 5795 | let s: Vec<char> = "[the `//` and `/* */` operators]".chars().collect(); |
| 5796 | let (inner, next) = res!(read_group(&s, 0) |
| 5797 | .ok_or_else(|| err!("the backtick span made the group mis-scan as a comment"; Test, Bug))); |
| 5798 | assert_eq!(inner, "the `//` and `/* */` operators", |
| 5799 | "the raw span was not kept literal: {:?}", inner); |
| 5800 | assert_eq!(next, s.len(), "the group did not close at its own bracket"); |
| 5801 | Ok(()) |
| 5802 | } |
| 5803 | |
| 5804 | /// A `//` on one line of a multi-line capture buffer must be bounded to that line's own end, not eat |
| 5805 | /// every line after it: the buffer's real closing bracket, on the following line, must still be found. |
| 5806 | #[test] |
| 5807 | fn read_group_line_comment_does_not_eat_the_next_line() -> Outcome<()> { |
| 5808 | let s: Vec<char> = "[first line // trailing\nsecond line]".chars().collect(); |
| 5809 | let (inner, next) = res!(read_group(&s, 0) |
| 5810 | .ok_or_else(|| err!("the // comment ate past its own line and swallowed the closer"; Test, Bug))); |
| 5811 | assert_eq!(inner, "first line // trailing\nsecond line", |
| 5812 | "the buffer content around the line comment was misread: {:?}", inner); |
| 5813 | assert_eq!(next, s.len(), "the group did not close at its own bracket"); |
| 5814 | Ok(()) |
| 5815 | } |
| 5816 | } |