oxedyne/fe2o3/fe2o3_austenite/src/lang/mathparse.rs
39.1 KiB, 90 runs
created by r1870400018:36567, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! A parser for Typst's inline maths syntax into the engine's [`Atom`](crate::math::Atom) tree. |
| 2 | //! |
| 3 | //! The engine sets a rich mathematics already; this is the front end that reads `$...$` the way an |
| 4 | //! author writes it in Typst and hands the layout an `Atom`. The subset covered is the common one: a |
| 5 | //! variable is a letter, a number a run of digits; `^` and `_` attach a superscript and a subscript, |
| 6 | //! grouped with `(...)` or `{...}`; `/` and `frac(a, b)` make a fraction; `sqrt(x)` a radical; `(...)` |
| 7 | //! grows a fence; a run of letters is looked up in a table of Greek letters and common operators |
| 8 | //! (`alpha`, `sum`, `times`, `->`) or, failing that, set as an identifier. Beyond that: the style |
| 9 | //! alphabets `cal`/`bold`/`bb`/`frak`/`sans`/`mono`/`upright`; the accents `dot`/`hat`/`tilde`/`bar` |
| 10 | //! and the rules `overline`/`underline`, and the explicit `accent(base, mark)`; a quoted `"..."` and a |
| 11 | //! `text("...")` as an upright run; the spacing words `thin`/`med`/`thick`/`quad`/`wide`/`space` and an |
| 12 | //! `#h(..em)`; a dotted symbol name (`arrow.r`, `plus.minus`, `dot.op`); the grids `mat`, `cases` and |
| 13 | //! `vec`; and a multi-line alignment block broken at `\` and aligned at `&`. An unknown control word is |
| 14 | //! set as itself rather than rejected, so a document still sets. |
| 15 | |
| 16 | use crate::math::{ |
| 17 | Accent, |
| 18 | Atom, |
| 19 | Class, |
| 20 | MatKind, |
| 21 | }; |
| 22 | |
| 23 | use oxedyne_fe2o3_core::prelude::*; |
| 24 | |
| 25 | /// Parses one maths expression (the text between the `$` delimiters) into an [`Atom`]. |
| 26 | pub fn parse(src: &str) -> Outcome<Atom> { |
| 27 | let mut p = Parser { chars: src.chars().collect(), i: 0 }; |
| 28 | let atom = res!(p.aligned()); |
| 29 | Ok(atom) |
| 30 | } |
| 31 | |
| 32 | struct Parser { |
| 33 | chars: Vec<char>, |
| 34 | i: usize, |
| 35 | } |
| 36 | |
| 37 | impl Parser { |
| 38 | fn peek(&self) -> Option<char> { self.chars.get(self.i).copied() } |
| 39 | fn peek2(&self) -> Option<char> { self.chars.get(self.i + 1).copied() } |
| 40 | fn at(&self, k: usize) -> Option<char> { self.chars.get(k).copied() } |
| 41 | |
| 42 | fn skip_ws(&mut self) { |
| 43 | while matches!(self.peek(), Some(c) if c.is_whitespace()) { |
| 44 | self.i += 1; |
| 45 | } |
| 46 | } |
| 47 | |
| 48 | fn eat(&mut self, c: char) -> bool { |
| 49 | if self.peek() == Some(c) { |
| 50 | self.i += 1; |
| 51 | true |
| 52 | } else { |
| 53 | false |
| 54 | } |
| 55 | } |
| 56 | |
| 57 | /// Is the cursor at a `\` line break -- a backslash at the end of input or before whitespace, as |
| 58 | /// distinct from a `\,` or `\/` escape whose backslash is followed by the character it escapes? |
| 59 | fn is_linebreak(&self) -> bool { |
| 60 | if self.peek() != Some('\\') { |
| 61 | return false; |
| 62 | } |
| 63 | match self.peek2() { |
| 64 | None => true, |
| 65 | Some(c) => c.is_whitespace(), |
| 66 | } |
| 67 | } |
| 68 | |
| 69 | /// The whole expression, as a sequence of cells separated by `&` and rows separated by a `\` line |
| 70 | /// break. A single cell in a single row unwraps to that atom -- the common case of one expression -- |
| 71 | /// while any `&` or break yields an alignment [`Atom::Matrix`] the display layout stacks and aligns. |
| 72 | fn aligned(&mut self) -> Outcome<Atom> { |
| 73 | let mut rows: Vec<Vec<Atom>> = Vec::new(); |
| 74 | let mut cells: Vec<Atom> = Vec::new(); |
| 75 | loop { |
| 76 | let cell = res!(self.row(&['&'])); |
| 77 | cells.push(cell); |
| 78 | self.skip_ws(); |
| 79 | match self.peek() { |
| 80 | Some('&') => { self.i += 1; }, |
| 81 | _ if self.is_linebreak() => { |
| 82 | self.i += 1; // consume the backslash |
| 83 | rows.push(std::mem::take(&mut cells)); |
| 84 | self.skip_ws(); |
| 85 | if self.peek().is_none() { |
| 86 | break; // a trailing break: no empty final row |
| 87 | } |
| 88 | }, |
| 89 | _ => { rows.push(std::mem::take(&mut cells)); break; }, |
| 90 | } |
| 91 | } |
| 92 | if rows.len() == 1 && rows[0].len() == 1 { |
| 93 | return Ok(rows.pop().and_then(|mut r| r.pop()).unwrap_or_else(|| Atom::row(Vec::new()))); |
| 94 | } |
| 95 | Ok(Atom::matrix(rows, None, None, MatKind::Align)) |
| 96 | } |
| 97 | |
| 98 | /// A sequence of factors up to end of input or a stop character, folding `/` into fractions. A row of |
| 99 | /// one element unwraps to that element, so a bare `x` is a symbol, not a one-item row. A `\` line |
| 100 | /// break stops the row, left for the caller to consume. |
| 101 | fn row(&mut self, stop: &[char]) -> Outcome<Atom> { |
| 102 | let mut items: Vec<Atom> = Vec::new(); |
| 103 | loop { |
| 104 | self.skip_ws(); |
| 105 | if self.is_linebreak() { |
| 106 | break; |
| 107 | } |
| 108 | match self.peek() { |
| 109 | None => break, |
| 110 | Some(c) if stop.contains(&c) => break, |
| 111 | _ => {}, |
| 112 | } |
| 113 | if self.peek() == Some('/') { |
| 114 | // A fraction of the preceding factor and the next one. Parentheses directly around either |
| 115 | // part are grouping, not literal, so Typst hides them -- `(a + b)/(c)` stacks without |
| 116 | // parentheses -- and they are unwrapped here. |
| 117 | self.i += 1; |
| 118 | self.skip_ws(); |
| 119 | let den = ungroup(res!(self.factor(stop))); |
| 120 | // A postfix that binds tighter than the fraction -- a factorial or a prime -- stays with the |
| 121 | // denominator: `e^x / k!` sets `k!` under the bar, not the `!` beside it. |
| 122 | let den = self.attach_postfix(den); |
| 123 | let num = ungroup(items.pop().unwrap_or_else(|| Atom::row(Vec::new()))); |
| 124 | items.push(Atom::frac(num, den)); |
| 125 | continue; |
| 126 | } |
| 127 | items.push(res!(self.factor(stop))); |
| 128 | } |
| 129 | if items.len() == 1 { |
| 130 | Ok(items.pop().unwrap_or_else(|| Atom::row(Vec::new()))) |
| 131 | } else { |
| 132 | Ok(Atom::row(items)) |
| 133 | } |
| 134 | } |
| 135 | |
| 136 | /// A base atom with any superscript and subscript that follow it, in either order. |
| 137 | fn factor(&mut self, stop: &[char]) -> Outcome<Atom> { |
| 138 | let base = res!(self.atom(stop)); |
| 139 | let mut sup: Option<Atom> = None; |
| 140 | let mut sub: Option<Atom> = None; |
| 141 | loop { |
| 142 | self.skip_ws(); |
| 143 | match self.peek() { |
| 144 | Some('^') => { self.i += 1; sup = Some(res!(self.script_arg(stop))); }, |
| 145 | Some('_') => { self.i += 1; sub = Some(res!(self.script_arg(stop))); }, |
| 146 | _ => break, |
| 147 | } |
| 148 | } |
| 149 | Ok(match (sub, sup) { |
| 150 | (None, None) => base, |
| 151 | (Some(sb), None) => Atom::sub(base, sb), |
| 152 | (None, Some(sp)) => Atom::sup(base, sp), |
| 153 | (Some(sb), Some(sp)) => Atom::subsup(base, sb, sp), |
| 154 | }) |
| 155 | } |
| 156 | |
| 157 | /// Appends any immediately following postfix marks -- a factorial `!` or a prime `'` -- to an atom, so |
| 158 | /// a fraction denominator keeps a `k!` whole. Whitespace ends the run. |
| 159 | fn attach_postfix(&mut self, atom: Atom) -> Atom { |
| 160 | let mut items = vec![atom]; |
| 161 | loop { |
| 162 | match self.peek() { |
| 163 | Some('!') => { self.i += 1; items.push(Atom::sym("!", Class::Close)); }, |
| 164 | Some('\'') => { self.i += 1; items.push(Atom::sym("\u{2032}", Class::Ord)); }, |
| 165 | _ => break, |
| 166 | } |
| 167 | } |
| 168 | if items.len() == 1 { |
| 169 | items.pop().unwrap_or_else(|| Atom::row(Vec::new())) |
| 170 | } else { |
| 171 | Atom::row(items) |
| 172 | } |
| 173 | } |
| 174 | |
| 175 | /// The operand of a `^` or `_`: a parenthesised or braced group, or a single atom. |
| 176 | fn script_arg(&mut self, stop: &[char]) -> Outcome<Atom> { |
| 177 | self.skip_ws(); |
| 178 | match self.peek() { |
| 179 | Some('(') => { self.i += 1; let r = res!(self.row(&[')'])); self.eat(')'); Ok(r) }, |
| 180 | Some('{') => { self.i += 1; let r = res!(self.row(&['}'])); self.eat('}'); Ok(r) }, |
| 181 | _ => self.atom(stop), |
| 182 | } |
| 183 | } |
| 184 | |
| 185 | fn atom(&mut self, _stop: &[char]) -> Outcome<Atom> { |
| 186 | self.skip_ws(); |
| 187 | let c = match self.peek() { |
| 188 | Some(c) => c, |
| 189 | None => return Ok(Atom::row(Vec::new())), |
| 190 | }; |
| 191 | |
| 192 | // A backslash escapes the next character, set literally: `\,` a comma, `\/` a plain slash (not a |
| 193 | // fraction). A backslash before whitespace is a line break, handled by the row, not here. |
| 194 | if c == '\\' { |
| 195 | self.i += 1; |
| 196 | return Ok(match self.peek() { |
| 197 | Some(',') => { self.i += 1; Atom::sym(",", Class::Punct) }, |
| 198 | Some('/') => { self.i += 1; Atom::sym("/", Class::Ord) }, |
| 199 | Some(ch) => { self.i += 1; Atom::sym(ch.to_string(), Class::Ord) }, |
| 200 | None => Atom::row(Vec::new()), |
| 201 | }); |
| 202 | } |
| 203 | |
| 204 | // A code escape inside maths: `#h(..)` is explicit spacing, anything else (`#150%`, `#move(..)`) |
| 205 | // is dropped to nothing so a stray call does not derail the row. |
| 206 | if c == '#' { |
| 207 | return Ok(self.hash()); |
| 208 | } |
| 209 | |
| 210 | // A quoted string is an upright roman run. Typst keeps a space between such a run and a following |
| 211 | // word (`"if" x`), where it drops one between bare symbols, so a space that separates the run from |
| 212 | // a following letter or digit is preserved as a thin space. |
| 213 | if c == '"' { |
| 214 | self.i += 1; |
| 215 | let s = self.take_while(|c| c != '"'); |
| 216 | self.eat('"'); |
| 217 | let spaced = self.peek() == Some(' ') |
| 218 | && matches!(self.peek_nonws(), Some(n) if n.is_alphanumeric()); |
| 219 | if spaced { |
| 220 | return Ok(Atom::row(vec![Atom::text(s), Atom::space(220)])); |
| 221 | } |
| 222 | return Ok(Atom::text(s)); |
| 223 | } |
| 224 | |
| 225 | // A parenthesised group grows a fence around its row. |
| 226 | if c == '(' { |
| 227 | self.i += 1; |
| 228 | let body = res!(self.row(&[')'])); |
| 229 | self.eat(')'); |
| 230 | return Ok(Atom::fence('(', body, ')')); |
| 231 | } |
| 232 | if c == '{' { |
| 233 | self.i += 1; |
| 234 | let body = res!(self.row(&['}'])); |
| 235 | self.eat('}'); |
| 236 | return Ok(body); |
| 237 | } |
| 238 | |
| 239 | // A number: a run of digits and dots. |
| 240 | if c.is_ascii_digit() { |
| 241 | let n = self.take_while(|c| c.is_ascii_digit() || c == '.'); |
| 242 | return Ok(Atom::num(n)); |
| 243 | } |
| 244 | |
| 245 | // A word: a function, a named symbol (possibly dotted), or an identifier. |
| 246 | if c.is_alphabetic() { |
| 247 | let word = self.take_while(|c| c.is_alphabetic()); |
| 248 | return self.word(&word); |
| 249 | } |
| 250 | |
| 251 | // An operator, possibly two characters (`<=`, `->`). |
| 252 | Ok(self.operator()) |
| 253 | } |
| 254 | |
| 255 | /// A `#...` code escape inside maths. `#h(1em)` or `#h(0.5em)` becomes a fixed space; `#h(..)` in |
| 256 | /// other units, and any other call or literal, is dropped to an empty row so it takes no space. |
| 257 | fn hash(&mut self) -> Atom { |
| 258 | self.i += 1; // the '#' |
| 259 | // `#h(<number>em)` -> a space of that many em. |
| 260 | if self.peek() == Some('h') && self.peek2() == Some('(') { |
| 261 | self.i += 2; |
| 262 | let num = self.take_while(|c| c.is_ascii_digit() || c == '.' || c == '-'); |
| 263 | let unit = self.take_while(|c| c.is_alphabetic() || c == '%'); |
| 264 | // consume to the closing ')' |
| 265 | while matches!(self.peek(), Some(c) if c != ')') { |
| 266 | self.i += 1; |
| 267 | } |
| 268 | self.eat(')'); |
| 269 | if unit == "em" { |
| 270 | let em: f64 = num.parse().unwrap_or(0.0); |
| 271 | return Atom::space((em * 1000.0) as i32); |
| 272 | } |
| 273 | return Atom::row(Vec::new()); |
| 274 | } |
| 275 | // Any other `#name`, `#name(...)`, `#123%`: consume the token and drop it. |
| 276 | let _ = self.take_while(|c| c.is_alphanumeric() || c == '.' || c == '%'); |
| 277 | if self.peek() == Some('(') { |
| 278 | self.skip_balanced(); |
| 279 | } |
| 280 | Atom::row(Vec::new()) |
| 281 | } |
| 282 | |
| 283 | /// Consumes a balanced `(...)` from the opening parenthesis, so a dropped call takes its whole |
| 284 | /// argument list with it. |
| 285 | fn skip_balanced(&mut self) { |
| 286 | if !self.eat('(') { |
| 287 | return; |
| 288 | } |
| 289 | let mut depth = 1usize; |
| 290 | while depth > 0 { |
| 291 | match self.peek() { |
| 292 | Some('(') => depth += 1, |
| 293 | Some(')') => depth -= 1, |
| 294 | None => break, |
| 295 | _ => {}, |
| 296 | } |
| 297 | self.i += 1; |
| 298 | } |
| 299 | } |
| 300 | |
| 301 | /// A run of letters, resolved as a function, a named symbol or an identifier. Functions read their |
| 302 | /// bracketed arguments; a name in the symbol table (perhaps extended by a dotted tail) becomes that |
| 303 | /// symbol; anything else is set as an identifier, a single letter as an italic variable. |
| 304 | fn word(&mut self, word: &str) -> Outcome<Atom> { |
| 305 | // A style alphabet: restyle the argument's letters. |
| 306 | if let Some(alpha) = alphabet(word) { |
| 307 | let arg = res!(self.one_arg()); |
| 308 | return Ok(restyle(&arg, alpha)); |
| 309 | } |
| 310 | |
| 311 | match word { |
| 312 | "frac" => { |
| 313 | let (a, b) = res!(self.two_args()); |
| 314 | return Ok(Atom::frac(a, b)); |
| 315 | }, |
| 316 | "sqrt" => { |
| 317 | let a = res!(self.one_arg()); |
| 318 | return Ok(Atom::sqrt(a)); |
| 319 | }, |
| 320 | "root" => { |
| 321 | // root(index, radicand): the index is not yet drawn, so the radicand alone is set. |
| 322 | let (_, b) = res!(self.two_args()); |
| 323 | return Ok(Atom::sqrt(b)); |
| 324 | }, |
| 325 | // ceil/floor: the argument between ceiling or floor corners, grown to it. `one_arg` returns the |
| 326 | // row without a paren fence, so there is no double bracket. |
| 327 | "ceil" => return Ok(Atom::fence('\u{2308}', res!(self.one_arg()), '\u{2309}')), |
| 328 | "floor" => return Ok(Atom::fence('\u{230A}', res!(self.one_arg()), '\u{230B}')), |
| 329 | "binom" => { |
| 330 | let (a, b) = res!(self.two_args()); |
| 331 | return Ok(Atom::binom(a, b)); |
| 332 | }, |
| 333 | "display" => return Ok(Atom::display(res!(self.one_arg()))), |
| 334 | "text" => { |
| 335 | // text("..."): the quoted argument as an upright run. |
| 336 | let a = res!(self.one_arg()); |
| 337 | return Ok(a); // one_arg parses the "..." to an Atom::Text already |
| 338 | }, |
| 339 | "lr" => { |
| 340 | // lr(content, size: ..): the auto-sized delimiters are already the content's own fence, so |
| 341 | // the content is kept and the size argument dropped. |
| 342 | if self.peek_nonws() == Some('(') { |
| 343 | self.skip_ws(); |
| 344 | self.i += 1; |
| 345 | let inner = res!(self.row(&[',', ')'])); |
| 346 | self.skip_commas_to_close(); |
| 347 | return Ok(inner); |
| 348 | } |
| 349 | }, |
| 350 | "mat" => return self.grid('(', Some(')'), ';', MatKind::Matrix), |
| 351 | "vec" => return self.grid('(', Some(')'), ',', MatKind::Matrix), |
| 352 | "cases" => return self.grid('{', None, ',', MatKind::Cases), |
| 353 | "overline" => return Ok(Atom::accent(res!(self.one_arg()), Accent::OverRule)), |
| 354 | "underline" => return Ok(Atom::accent(res!(self.one_arg()), Accent::UnderRule)), |
| 355 | "accent" => { |
| 356 | // accent(base, mark): the second argument names the accent, read as a bare word. |
| 357 | self.skip_ws(); |
| 358 | if self.eat('(') { |
| 359 | let base = res!(self.row(&[',', ')'])); |
| 360 | self.eat(','); |
| 361 | self.skip_ws(); |
| 362 | let mark = self.take_while(|c| c.is_alphabetic() || c == '.'); |
| 363 | self.skip_commas_to_close(); |
| 364 | let a = accent_of(&mark).unwrap_or(Accent::Over("\u{02D9}".to_string())); |
| 365 | return Ok(Atom::accent(base, a)); |
| 366 | } |
| 367 | return Ok(symbol("accent")); |
| 368 | }, |
| 369 | // The `dif` of a differential: a thin space and an upright `d`. |
| 370 | "dif" => return Ok(Atom::row(vec![Atom::space(167), Atom::text("d")])), |
| 371 | // The custom tight scientific-notation cross reads as a multiplication sign. |
| 372 | "ttimes" => return Ok(Atom::sym("\u{00D7}", Class::Bin)), |
| 373 | _ => {}, |
| 374 | } |
| 375 | |
| 376 | // A glyph accent taken as a function when an argument follows: `dot(x)`, `hat(y)`, `bar(z)`. |
| 377 | if self.peek_nonws() == Some('(') { |
| 378 | if let Some(mark) = accent_of(word) { |
| 379 | return Ok(Atom::accent(res!(self.one_arg()), mark)); |
| 380 | } |
| 381 | } |
| 382 | |
| 383 | // A spacing word. |
| 384 | if let Some(mem) = spacing(word) { |
| 385 | return Ok(Atom::space(mem)); |
| 386 | } |
| 387 | |
| 388 | // A dotted symbol name, `arrow.r`, `plus.minus`, resolved to the longest matching name. |
| 389 | if self.peek() == Some('.') { |
| 390 | if let Some(atom) = self.dotted_word(word) { |
| 391 | return Ok(atom); |
| 392 | } |
| 393 | } |
| 394 | |
| 395 | Ok(symbol(word)) |
| 396 | } |
| 397 | |
| 398 | /// The next non-whitespace character, without consuming it. |
| 399 | fn peek_nonws(&self) -> Option<char> { |
| 400 | let mut k = self.i; |
| 401 | while matches!(self.at(k), Some(c) if c.is_whitespace()) { |
| 402 | k += 1; |
| 403 | } |
| 404 | self.at(k) |
| 405 | } |
| 406 | |
| 407 | /// Resolves a dotted symbol name from a base word already read and a `.` at the cursor. Reads the |
| 408 | /// dotted segments without committing, then matches the longest `word.seg.seg` in the dotted table, |
| 409 | /// consuming exactly that; returns `None` (consuming nothing) when no dotted name matches, so the |
| 410 | /// base word falls through to ordinary handling and the `.` stays as punctuation. |
| 411 | fn dotted_word(&mut self, word: &str) -> Option<Atom> { |
| 412 | // Probe the extension segments and the index just past each. |
| 413 | let mut ends: Vec<usize> = Vec::new(); // index after segment k |
| 414 | let mut names: Vec<String> = Vec::new(); // "word.seg1...segk" |
| 415 | let mut probe = self.i; |
| 416 | let mut acc = word.to_string(); |
| 417 | while self.at(probe) == Some('.') { |
| 418 | let mut j = probe + 1; |
| 419 | while matches!(self.at(j), Some(c) if c.is_alphabetic()) { |
| 420 | j += 1; |
| 421 | } |
| 422 | if j == probe + 1 { |
| 423 | break; // a lone dot, no segment: punctuation |
| 424 | } |
| 425 | let seg: String = self.chars[probe + 1..j].iter().collect(); |
| 426 | acc.push('.'); |
| 427 | acc.push_str(&seg); |
| 428 | names.push(acc.clone()); |
| 429 | ends.push(j); |
| 430 | probe = j; |
| 431 | } |
| 432 | // Longest match first. |
| 433 | for k in (0..names.len()).rev() { |
| 434 | if let Some((s, class)) = dotted(&names[k]) { |
| 435 | self.i = ends[k]; |
| 436 | return Some(Atom::sym(s, class)); |
| 437 | } |
| 438 | } |
| 439 | None |
| 440 | } |
| 441 | |
| 442 | /// A grid function's body: `mat`/`vec` (rows by `;` or `,`), `cases` (rows by `,`, cells by `&`). |
| 443 | /// The row separator is given; a cell is separated by `&` for a `cases`, else the grid has one cell |
| 444 | /// per row. A trailing separator before the close does not open an empty final row. |
| 445 | fn grid(&mut self, left: char, right: Option<char>, row_sep: char, kind: MatKind) -> Outcome<Atom> { |
| 446 | if self.peek_nonws() != Some('(') { |
| 447 | return Ok(symbol("mat")); // not a call after all |
| 448 | } |
| 449 | self.skip_ws(); |
| 450 | self.i += 1; // the '(' |
| 451 | let cell_sep = if kind == MatKind::Cases { '&' } else { row_sep }; |
| 452 | let stops = [cell_sep, row_sep, ')']; |
| 453 | let mut rows: Vec<Vec<Atom>> = Vec::new(); |
| 454 | let mut cells: Vec<Atom> = Vec::new(); |
| 455 | loop { |
| 456 | self.skip_ws(); |
| 457 | if self.peek() == Some(')') || self.peek().is_none() { |
| 458 | break; |
| 459 | } |
| 460 | let cell = res!(self.row(&stops)); |
| 461 | cells.push(cell); |
| 462 | self.skip_ws(); |
| 463 | match self.peek() { |
| 464 | Some(c) if c == cell_sep && cell_sep != row_sep => { self.i += 1; }, |
| 465 | Some(c) if c == row_sep => { |
| 466 | self.i += 1; |
| 467 | rows.push(std::mem::take(&mut cells)); |
| 468 | }, |
| 469 | _ => break, |
| 470 | } |
| 471 | } |
| 472 | if !cells.is_empty() { |
| 473 | rows.push(cells); |
| 474 | } |
| 475 | self.skip_ws(); |
| 476 | self.eat(')'); |
| 477 | Ok(Atom::matrix(rows, Some(left), right, kind)) |
| 478 | } |
| 479 | |
| 480 | /// The single bracketed argument of a function: `(row)`. |
| 481 | fn one_arg(&mut self) -> Outcome<Atom> { |
| 482 | self.skip_ws(); |
| 483 | if !self.eat('(') { |
| 484 | return Ok(Atom::row(Vec::new())); |
| 485 | } |
| 486 | let a = res!(self.row(&[')', ','])); |
| 487 | self.skip_commas_to_close(); |
| 488 | Ok(a) |
| 489 | } |
| 490 | |
| 491 | /// The two comma-separated bracketed arguments of a function: `(row, row)`. |
| 492 | fn two_args(&mut self) -> Outcome<(Atom, Atom)> { |
| 493 | self.skip_ws(); |
| 494 | if !self.eat('(') { |
| 495 | return Ok((Atom::row(Vec::new()), Atom::row(Vec::new()))); |
| 496 | } |
| 497 | let a = res!(self.row(&[',', ')'])); |
| 498 | self.eat(','); |
| 499 | let b = res!(self.row(&[')', ','])); |
| 500 | self.skip_commas_to_close(); |
| 501 | Ok((a, b)) |
| 502 | } |
| 503 | |
| 504 | /// Consumes any trailing arguments and the closing `)`, so an extra argument does not derail the row. |
| 505 | fn skip_commas_to_close(&mut self) { |
| 506 | loop { |
| 507 | self.skip_ws(); |
| 508 | match self.peek() { |
| 509 | Some(')') => { self.i += 1; break; }, |
| 510 | Some(',') => { self.i += 1; let _ = self.row(&[',', ')']); }, |
| 511 | None => break, |
| 512 | _ => break, |
| 513 | } |
| 514 | } |
| 515 | } |
| 516 | |
| 517 | /// One or two operator characters, mapped to a symbol of the right spacing class. |
| 518 | fn operator(&mut self) -> Atom { |
| 519 | let a = self.peek().unwrap_or(' '); |
| 520 | let b = self.peek2(); |
| 521 | // Two-character operators first. |
| 522 | if let Some(b) = b { |
| 523 | let pair = [a, b]; |
| 524 | let two: Option<(&str, Class)> = match pair { |
| 525 | ['<', '='] => Some(("\u{2264}", Class::Rel)), // <= |
| 526 | ['>', '='] => Some(("\u{2265}", Class::Rel)), // >= |
| 527 | ['!', '='] => Some(("\u{2260}", Class::Rel)), // != |
| 528 | ['-', '>'] => Some(("\u{2192}", Class::Rel)), // -> |
| 529 | ['=', '>'] => Some(("\u{21D2}", Class::Rel)), // => |
| 530 | ['<', '-'] => Some(("\u{2190}", Class::Rel)), // <- |
| 531 | _ => None, |
| 532 | }; |
| 533 | if let Some((s, class)) = two { |
| 534 | self.i += 2; |
| 535 | return Atom::sym(s, class); |
| 536 | } |
| 537 | } |
| 538 | self.i += 1; |
| 539 | let (s, class): (String, Class) = match a { |
| 540 | '+' => ("+".to_string(), Class::Bin), |
| 541 | '-' => ("\u{2212}".to_string(), Class::Bin), // a true minus sign |
| 542 | '*' => ("\u{22C5}".to_string(), Class::Bin), // a centred dot for multiplication |
| 543 | '=' => ("=".to_string(), Class::Rel), |
| 544 | '<' => ("<".to_string(), Class::Rel), |
| 545 | '>' => (">".to_string(), Class::Rel), |
| 546 | ',' => (",".to_string(), Class::Punct), |
| 547 | '.' => (".".to_string(), Class::Punct), |
| 548 | '!' => ("!".to_string(), Class::Close), |
| 549 | _ => (a.to_string(), Class::Ord), |
| 550 | }; |
| 551 | Atom::sym(s, class) |
| 552 | } |
| 553 | |
| 554 | fn take_while(&mut self, pred: impl Fn(char) -> bool) -> String { |
| 555 | let start = self.i; |
| 556 | while matches!(self.peek(), Some(c) if pred(c)) { |
| 557 | self.i += 1; |
| 558 | } |
| 559 | self.chars[start..self.i].iter().collect() |
| 560 | } |
| 561 | } |
| 562 | |
| 563 | /// Unwraps a round-parenthesis fence to its body, used on a fraction's parts where Typst treats such |
| 564 | /// parentheses as grouping and hides them. Anything else is returned unchanged. |
| 565 | fn ungroup(atom: Atom) -> Atom { |
| 566 | match atom { |
| 567 | Atom::Fence { left: '(', body, right: ')' } => *body, |
| 568 | other => other, |
| 569 | } |
| 570 | } |
| 571 | |
| 572 | /// Resolves a control word to a symbol: a Greek letter, a common operator or relation by name, or -- |
| 573 | /// when the name is unknown -- an identifier set as itself (a single letter as an italic variable, a |
| 574 | /// longer word upright, the way Typst sets a known multi-letter operator like `sin`). |
| 575 | fn symbol(word: &str) -> Atom { |
| 576 | if let Some((s, class)) = named(word) { |
| 577 | return Atom::sym(s, class); |
| 578 | } |
| 579 | if word.chars().count() == 1 { |
| 580 | Atom::var(word) // a lone letter is an italic variable |
| 581 | } else { |
| 582 | Atom::op(word) // an unknown multi-letter word, set upright like an operator name |
| 583 | } |
| 584 | } |
| 585 | |
| 586 | /// A style-alphabet function name mapped to its alphabet, or `None` when the word is not one. |
| 587 | fn alphabet(word: &str) -> Option<Alphabet> { |
| 588 | Some(match word { |
| 589 | "cal" => Alphabet::Cal, |
| 590 | "bold" => Alphabet::BoldItalic, |
| 591 | "bb" => Alphabet::Bb, |
| 592 | "frak" => Alphabet::Frak, |
| 593 | "sans" => Alphabet::Sans, |
| 594 | "mono" => Alphabet::Mono, |
| 595 | "upright" => Alphabet::Upright, |
| 596 | _ => return None, |
| 597 | }) |
| 598 | } |
| 599 | |
| 600 | /// The maths style alphabets Typst offers, each a remapping of the Latin letters (and, for some, the |
| 601 | /// digits) to a Unicode mathematical-alphanumeric block. |
| 602 | #[derive(Clone, Copy)] |
| 603 | enum Alphabet { |
| 604 | Cal, // script / calligraphic |
| 605 | BoldItalic, // Typst's `bold` of a variable is bold italic |
| 606 | Bb, // blackboard bold |
| 607 | Frak, // Fraktur |
| 608 | Sans, // sans serif |
| 609 | Mono, // monospace |
| 610 | Upright, // roman upright (drops the default italic) |
| 611 | } |
| 612 | |
| 613 | /// Restyles an atom's letters to a maths style alphabet, recursing through rows, scripts, fractions, |
| 614 | /// fences and accents. A single ASCII letter (or, for the alphabets that carry them, a digit) is |
| 615 | /// remapped to the alphabet's codepoint and set as an upright run so the layout shapes it as drawn |
| 616 | /// rather than remapping it a second time to the maths italic. Symbols the alphabet does not cover -- |
| 617 | /// Greek, operators -- are left as they are. |
| 618 | fn restyle(atom: &Atom, alpha: Alphabet) -> Atom { |
| 619 | match atom { |
| 620 | Atom::Sym(s, Class::Ord) => { |
| 621 | let mut it = s.chars(); |
| 622 | if let (Some(c), None) = (it.next(), it.next()) { |
| 623 | if let Some(m) = styled_char(c, alpha) { |
| 624 | return Atom::text(m.to_string()); |
| 625 | } |
| 626 | } |
| 627 | atom.clone() |
| 628 | }, |
| 629 | Atom::Row(items) => Atom::row(items.iter().map(|a| restyle(a, alpha)).collect()), |
| 630 | Atom::Frac { num, den } => Atom::frac(restyle(num, alpha), restyle(den, alpha)), |
| 631 | Atom::Script { base, sup, sub } => Atom::Script { |
| 632 | base: Box::new(restyle(base, alpha)), |
| 633 | sup: sup.as_ref().map(|a| Box::new(restyle(a, alpha))), |
| 634 | sub: sub.as_ref().map(|a| Box::new(restyle(a, alpha))), |
| 635 | }, |
| 636 | Atom::Fence { left, body, right } => Atom::fence(*left, restyle(body, alpha), *right), |
| 637 | Atom::Accent { base, mark } => Atom::accent(restyle(base, alpha), mark.clone()), |
| 638 | Atom::Binom { top, bottom } => Atom::binom(restyle(top, alpha), restyle(bottom, alpha)), |
| 639 | Atom::Display(inner) => Atom::display(restyle(inner, alpha)), |
| 640 | other => other.clone(), |
| 641 | } |
| 642 | } |
| 643 | |
| 644 | /// The mathematical-alphanumeric codepoint of an ASCII letter (or digit) in a style alphabet, or `None` |
| 645 | /// where the alphabet does not cover the character. The script, Fraktur and blackboard alphabets have |
| 646 | /// holes in their Unicode blocks that the letterlike block fills; those are given explicitly. |
| 647 | fn styled_char(c: char, alpha: Alphabet) -> Option<char> { |
| 648 | let up = c.is_ascii_uppercase(); |
| 649 | let lo = c.is_ascii_lowercase(); |
| 650 | let digit = c.is_ascii_digit(); |
| 651 | let i = |base: u32, first: char| base + (c as u32 - first as u32); |
| 652 | let cp = match alpha { |
| 653 | Alphabet::Upright => { |
| 654 | // Roman upright: the letter itself, shaped upright rather than in the maths italic. |
| 655 | if up || lo { c as u32 } else { return None } |
| 656 | }, |
| 657 | Alphabet::BoldItalic => { |
| 658 | if up { i(0x1D468, 'A') } else if lo { i(0x1D482, 'a') } |
| 659 | else if digit { i(0x1D7CE, '0') } else { return None } |
| 660 | }, |
| 661 | Alphabet::Sans => { |
| 662 | if up { i(0x1D5A0, 'A') } else if lo { i(0x1D5BA, 'a') } |
| 663 | else if digit { i(0x1D7E2, '0') } else { return None } |
| 664 | }, |
| 665 | Alphabet::Mono => { |
| 666 | if up { i(0x1D670, 'A') } else if lo { i(0x1D68A, 'a') } |
| 667 | else if digit { i(0x1D7F6, '0') } else { return None } |
| 668 | }, |
| 669 | Alphabet::Cal => { |
| 670 | return script_char(c); |
| 671 | }, |
| 672 | Alphabet::Frak => { |
| 673 | return frak_char(c); |
| 674 | }, |
| 675 | Alphabet::Bb => { |
| 676 | return bb_char(c); |
| 677 | }, |
| 678 | }; |
| 679 | char::from_u32(cp) |
| 680 | } |
| 681 | |
| 682 | /// The script (calligraphic) codepoint of a letter, filling the block's holes from the letterlike set. |
| 683 | fn script_char(c: char) -> Option<char> { |
| 684 | let cp = match c { |
| 685 | 'B' => 0x212C, 'E' => 0x2130, 'F' => 0x2131, 'H' => 0x210B, 'I' => 0x2110, |
| 686 | 'L' => 0x2112, 'M' => 0x2133, 'R' => 0x211B, |
| 687 | 'e' => 0x212F, 'g' => 0x210A, 'o' => 0x2134, |
| 688 | 'A'..='Z' => 0x1D49C + (c as u32 - 'A' as u32), |
| 689 | 'a'..='z' => 0x1D4B6 + (c as u32 - 'a' as u32), |
| 690 | _ => return None, |
| 691 | }; |
| 692 | char::from_u32(cp) |
| 693 | } |
| 694 | |
| 695 | /// The Fraktur codepoint of a letter, filling the block's holes from the letterlike set. |
| 696 | fn frak_char(c: char) -> Option<char> { |
| 697 | let cp = match c { |
| 698 | 'C' => 0x212D, 'H' => 0x210C, 'I' => 0x2111, 'R' => 0x211C, 'Z' => 0x2128, |
| 699 | 'A'..='Z' => 0x1D504 + (c as u32 - 'A' as u32), |
| 700 | 'a'..='z' => 0x1D51E + (c as u32 - 'a' as u32), |
| 701 | _ => return None, |
| 702 | }; |
| 703 | char::from_u32(cp) |
| 704 | } |
| 705 | |
| 706 | /// The blackboard-bold codepoint of a letter or digit, filling the block's holes from the letterlike set. |
| 707 | fn bb_char(c: char) -> Option<char> { |
| 708 | let cp = match c { |
| 709 | 'C' => 0x2102, 'H' => 0x210D, 'N' => 0x2115, 'P' => 0x2119, 'Q' => 0x211A, |
| 710 | 'R' => 0x211D, 'Z' => 0x2124, |
| 711 | 'A'..='Z' => 0x1D538 + (c as u32 - 'A' as u32), |
| 712 | 'a'..='z' => 0x1D552 + (c as u32 - 'a' as u32), |
| 713 | '0'..='9' => 0x1D7D8 + (c as u32 - '0' as u32), |
| 714 | _ => return None, |
| 715 | }; |
| 716 | char::from_u32(cp) |
| 717 | } |
| 718 | |
| 719 | /// The accent a name denotes -- for a `dot(x)` glyph function or the second argument of `accent(x, .)`. |
| 720 | /// A glyph accent is a spacing modifier the maths font carries; `overline`/`underline` are rules. |
| 721 | fn accent_of(name: &str) -> Option<Accent> { |
| 722 | Some(match name { |
| 723 | "dot" => Accent::Over("\u{02D9}".to_string()), // dot above |
| 724 | "dot.double" | "ddot" => Accent::Over("\u{00A8}".to_string()), // diaeresis |
| 725 | "hat" => Accent::Over("\u{02C6}".to_string()), // circumflex |
| 726 | "tilde" => Accent::Over("\u{02DC}".to_string()), // small tilde |
| 727 | "bar" | "macron" => Accent::Over("\u{00AF}".to_string()), // macron |
| 728 | "breve" => Accent::Over("\u{02D8}".to_string()), |
| 729 | "check" | "caron" => Accent::Over("\u{02C7}".to_string()), |
| 730 | "acute" => Accent::Over("\u{00B4}".to_string()), |
| 731 | "grave" => Accent::Over("\u{02CB}".to_string()), |
| 732 | "arrow" | "vec" => Accent::Over("\u{2192}".to_string()), // a right arrow over the base |
| 733 | "overline" => Accent::OverRule, |
| 734 | "underline" => Accent::UnderRule, |
| 735 | _ => return None, |
| 736 | }) |
| 737 | } |
| 738 | |
| 739 | /// The width of a spacing word, in thousandths of an em, or `None` when the word is not a space. Thin, |
| 740 | /// medium and thick match TeX's three-, four- and five-`mu` spaces; `quad` is an em and `wide` two. |
| 741 | fn spacing(word: &str) -> Option<i32> { |
| 742 | Some(match word { |
| 743 | "thin" => 167, // 3 mu |
| 744 | "med" => 222, // 4 mu |
| 745 | "thick" => 278, // 5 mu |
| 746 | "quad" => 1000, // 1 em |
| 747 | "wide" => 2000, // 2 em |
| 748 | "space" => 250, // a normal interword space |
| 749 | _ => return None, |
| 750 | }) |
| 751 | } |
| 752 | |
| 753 | /// The table of named symbols Typst authors reach for: the Greek alphabet and the common operators and |
| 754 | /// relations. Returned as the symbol's characters and its spacing class. |
| 755 | fn named(word: &str) -> Option<(&'static str, Class)> { |
| 756 | use Class::*; |
| 757 | let out = match word { |
| 758 | // Greek lower case. |
| 759 | "alpha" => ("\u{03B1}", Ord), "beta" => ("\u{03B2}", Ord), |
| 760 | "gamma" => ("\u{03B3}", Ord), "delta" => ("\u{03B4}", Ord), |
| 761 | "epsilon" => ("\u{03B5}", Ord), "zeta" => ("\u{03B6}", Ord), |
| 762 | "eta" => ("\u{03B7}", Ord), "theta" => ("\u{03B8}", Ord), |
| 763 | "iota" => ("\u{03B9}", Ord), "kappa" => ("\u{03BA}", Ord), |
| 764 | "lambda" => ("\u{03BB}", Ord), "mu" => ("\u{03BC}", Ord), |
| 765 | "nu" => ("\u{03BD}", Ord), "xi" => ("\u{03BE}", Ord), |
| 766 | "omicron" => ("\u{03BF}", Ord), |
| 767 | "pi" => ("\u{03C0}", Ord), "rho" => ("\u{03C1}", Ord), |
| 768 | "sigma" => ("\u{03C3}", Ord), "tau" => ("\u{03C4}", Ord), |
| 769 | "upsilon" => ("\u{03C5}", Ord), |
| 770 | "phi" => ("\u{03C6}", Ord), "chi" => ("\u{03C7}", Ord), |
| 771 | "psi" => ("\u{03C8}", Ord), "omega" => ("\u{03C9}", Ord), |
| 772 | // Greek upper case, the ones with a distinct glyph. |
| 773 | "Gamma" => ("\u{0393}", Ord), "Delta" => ("\u{0394}", Ord), |
| 774 | "Theta" => ("\u{0398}", Ord), "Lambda" => ("\u{039B}", Ord), |
| 775 | "Xi" => ("\u{039E}", Ord), "Pi" => ("\u{03A0}", Ord), |
| 776 | "Sigma" => ("\u{03A3}", Ord), "Upsilon" => ("\u{03A5}", Ord), |
| 777 | "Phi" => ("\u{03A6}", Ord), "Psi" => ("\u{03A8}", Ord), |
| 778 | "Omega" => ("\u{03A9}", Ord), |
| 779 | // Big operators. |
| 780 | "sum" => ("\u{2211}", Op), "prod" => ("\u{220F}", Op), |
| 781 | "product" => ("\u{220F}", Op), "coprod" => ("\u{2210}", Op), |
| 782 | "int" => ("\u{222B}", Op), "integral" => ("\u{222B}", Op), |
| 783 | "oint" => ("\u{222E}", Op), |
| 784 | "union" => ("\u{22C3}", Op), "sect" => ("\u{22C2}", Op), |
| 785 | // Binary operators. |
| 786 | "times" => ("\u{00D7}", Bin), "div" => ("\u{00F7}", Bin), |
| 787 | "cdot" => ("\u{22C5}", Bin), "pm" => ("\u{00B1}", Bin), |
| 788 | "mp" => ("\u{2213}", Bin), "ast" => ("\u{2217}", Bin), |
| 789 | "star" => ("\u{22C6}", Bin), "circ" => ("\u{2218}", Bin), |
| 790 | "bullet" => ("\u{2219}", Bin), "dot" => ("\u{22C5}", Bin), |
| 791 | "plus" => ("+", Bin), "minus" => ("\u{2212}", Bin), |
| 792 | "cup" => ("\u{222A}", Bin), "cap" => ("\u{2229}", Bin), |
| 793 | "slash" => ("/", Ord), |
| 794 | // Relations. |
| 795 | "leq" => ("\u{2264}", Rel), "geq" => ("\u{2265}", Rel), |
| 796 | "neq" => ("\u{2260}", Rel), "approx" => ("\u{2248}", Rel), |
| 797 | "equiv" => ("\u{2261}", Rel), "sim" => ("\u{223C}", Rel), |
| 798 | "simeq" => ("\u{2243}", Rel), "cong" => ("\u{2245}", Rel), |
| 799 | "propto" => ("\u{221D}", Rel), "prop" => ("\u{221D}", Rel), |
| 800 | "in" => ("\u{2208}", Rel), "notin" => ("\u{2209}", Rel), |
| 801 | "ni" => ("\u{220B}", Rel), "mid" => ("\u{2223}", Rel), |
| 802 | "subset" => ("\u{2282}", Rel), "subseteq" => ("\u{2286}", Rel), |
| 803 | "supset" => ("\u{2283}", Rel), "supseteq" => ("\u{2287}", Rel), |
| 804 | "ll" => ("\u{226A}", Rel), "gg" => ("\u{226B}", Rel), |
| 805 | "parallel" => ("\u{2225}", Rel), |
| 806 | // Arrows and miscellany. |
| 807 | "arrow" => ("\u{2192}", Rel), "to" => ("\u{2192}", Rel), |
| 808 | "mapsto" => ("\u{21A6}", Rel), |
| 809 | "infinity" => ("\u{221E}", Ord), "oo" => ("\u{221E}", Ord), |
| 810 | "partial" => ("\u{2202}", Ord), "nabla" => ("\u{2207}", Ord), |
| 811 | "dots" => ("\u{2026}", Ord), "ldots" => ("\u{2026}", Ord), |
| 812 | "cdots" => ("\u{22EF}", Ord), "emptyset" => ("\u{2205}", Ord), |
| 813 | "angle" => ("\u{2220}", Ord), "degree" => ("\u{00B0}", Ord), |
| 814 | "forall" => ("\u{2200}", Ord), "exists" => ("\u{2203}", Ord), |
| 815 | "neg" => ("\u{00AC}", Ord), "ell" => ("\u{2113}", Ord), |
| 816 | "and" => ("\u{2227}", Bin), "or" => ("\u{2228}", Bin), |
| 817 | // Blackboard sets. |
| 818 | "NN" => ("\u{2115}", Ord), "RR" => ("\u{211D}", Ord), |
| 819 | "ZZ" => ("\u{2124}", Ord), "QQ" => ("\u{211A}", Ord), |
| 820 | "GG" => ("\u{1D53E}", Ord), |
| 821 | _ => return None, |
| 822 | }; |
| 823 | Some(out) |
| 824 | } |
| 825 | |
| 826 | /// The table of dotted symbol names -- Typst's namespaced glyphs, `arrow.r`, `plus.minus`, `dot.op` -- |
| 827 | /// mapped to the glyph and its spacing class. The longest matching name wins at the call site. |
| 828 | fn dotted(name: &str) -> Option<(&'static str, Class)> { |
| 829 | use Class::*; |
| 830 | let out = match name { |
| 831 | "arrow.r" => ("\u{2192}", Rel), |
| 832 | "arrow.l" => ("\u{2190}", Rel), |
| 833 | "arrow.t" => ("\u{2191}", Rel), |
| 834 | "arrow.b" => ("\u{2193}", Rel), |
| 835 | "arrow.l.r" => ("\u{2194}", Rel), |
| 836 | "arrow.r.double" => ("\u{21D2}", Rel), |
| 837 | "arrow.l.double" => ("\u{21D0}", Rel), |
| 838 | "arrow.l.r.double" => ("\u{21D4}", Rel), |
| 839 | "arrow.r.long" => ("\u{27F6}", Rel), |
| 840 | "arrow.l.long" => ("\u{27F5}", Rel), |
| 841 | "arrow.r.bar" => ("\u{21A6}", Rel), |
| 842 | "arrow.r.long.bar" => ("\u{27FC}", Rel), |
| 843 | "plus.minus" => ("\u{00B1}", Bin), |
| 844 | "minus.plus" => ("\u{2213}", Bin), |
| 845 | "dot.op" => ("\u{22C5}", Bin), |
| 846 | // Circled operators. |
| 847 | "plus.circle" => ("\u{2295}", Bin), |
| 848 | "times.circle" => ("\u{2297}", Bin), |
| 849 | "dot.circle" => ("\u{2299}", Bin), |
| 850 | "dots.h" => ("\u{2026}", Ord), |
| 851 | "dots.h.c" => ("\u{22EF}", Ord), |
| 852 | "dots.v" => ("\u{22EE}", Ord), |
| 853 | "dots.down" => ("\u{22F1}", Ord), |
| 854 | "eq.not" => ("\u{2260}", Rel), |
| 855 | "lt.eq" => ("\u{2264}", Rel), |
| 856 | "gt.eq" => ("\u{2265}", Rel), |
| 857 | "lt.eq.not" => ("\u{2270}", Rel), |
| 858 | "gt.eq.not" => ("\u{2271}", Rel), |
| 859 | "dash.em" => ("\u{2014}", Ord), |
| 860 | "dash.en" => ("\u{2013}", Ord), |
| 861 | "angle.l" => ("\u{27E8}", Open), |
| 862 | "angle.r" => ("\u{27E9}", Close), |
| 863 | // Greek variant forms. |
| 864 | "pi.alt" => ("\u{03D6}", Ord), |
| 865 | "phi.alt" => ("\u{03D5}", Ord), |
| 866 | "epsilon.alt" => ("\u{03F5}", Ord), |
| 867 | "theta.alt" => ("\u{03D1}", Ord), |
| 868 | "rho.alt" => ("\u{03F1}", Ord), |
| 869 | "sigma.alt" => ("\u{03C2}", Ord), |
| 870 | "beta.alt" => ("\u{03D0}", Ord), |
| 871 | _ => return None, |
| 872 | }; |
| 873 | Some(out) |
| 874 | } |
| 875 | |
| 876 | #[cfg(test)] |
| 877 | mod tests { |
| 878 | use super::*; |
| 879 | |
| 880 | // The `cal(E)` script letter resolves to the mathematical script E (U+2130). |
| 881 | #[test] |
| 882 | fn cal_script() { |
| 883 | match parse("cal(E)") { |
| 884 | Ok(Atom::Text(s)) => assert_eq!(s, "\u{2130}"), |
| 885 | other => panic!("cal(E) -> {:?}", other), |
| 886 | } |
| 887 | } |
| 888 | |
| 889 | // `bold(D)` restyles to the mathematical bold-italic D (U+1D46B = U+1D468 + 3). |
| 890 | #[test] |
| 891 | fn bold_letter() { |
| 892 | match parse("bold(D)") { |
| 893 | Ok(Atom::Text(s)) => assert_eq!(s, "\u{1D46B}"), |
| 894 | other => panic!("bold(D) -> {:?}", other), |
| 895 | } |
| 896 | } |
| 897 | |
| 898 | // A quoted run is an upright text atom, never a maths italic. |
| 899 | #[test] |
| 900 | fn quoted_text() { |
| 901 | match parse("\"Pr\"") { |
| 902 | Ok(Atom::Text(s)) => assert_eq!(s, "Pr"), |
| 903 | other => panic!("quoted -> {:?}", other), |
| 904 | } |
| 905 | } |
| 906 | |
| 907 | // `dot(H)` is an over-accent on its base; `overline(x)` a rule over it. |
| 908 | #[test] |
| 909 | fn accents() { |
| 910 | assert!(matches!(parse("dot(H)"), Ok(Atom::Accent { mark: Accent::Over(_), .. }))); |
| 911 | assert!(matches!(parse("overline(x)"), Ok(Atom::Accent { mark: Accent::OverRule, .. }))); |
| 912 | } |
| 913 | |
| 914 | // A spacing word becomes a fixed space. |
| 915 | #[test] |
| 916 | fn spacing_word() { |
| 917 | assert!(matches!(parse("quad"), Ok(Atom::Space(1000)))); |
| 918 | } |
| 919 | |
| 920 | // A dotted symbol name resolves through its longest match. |
| 921 | #[test] |
| 922 | fn dotted_arrow() { |
| 923 | match parse("arrow.r.double") { |
| 924 | Ok(Atom::Sym(s, _)) => assert_eq!(s, "\u{21D2}"), |
| 925 | other => panic!("arrow.r.double -> {:?}", other), |
| 926 | } |
| 927 | } |
| 928 | |
| 929 | // `cases` yields a two-row grid with a left brace and no right delimiter. |
| 930 | #[test] |
| 931 | fn cases_grid() { |
| 932 | match parse("cases(1\\, & y, x & z)") { |
| 933 | Ok(Atom::Matrix { rows, left, right, kind }) => { |
| 934 | assert_eq!(rows.len(), 2); |
| 935 | assert_eq!(left, Some('{')); |
| 936 | assert_eq!(right, None); |
| 937 | assert_eq!(kind, MatKind::Cases); |
| 938 | }, |
| 939 | other => panic!("cases -> {:?}", other), |
| 940 | } |
| 941 | } |
| 942 | |
| 943 | // `mat` splits its rows on `;`. |
| 944 | #[test] |
| 945 | fn mat_grid() { |
| 946 | match parse("mat(1; 2; 3)") { |
| 947 | Ok(Atom::Matrix { rows, kind, .. }) => { |
| 948 | assert_eq!(rows.len(), 3); |
| 949 | assert_eq!(kind, MatKind::Matrix); |
| 950 | }, |
| 951 | other => panic!("mat -> {:?}", other), |
| 952 | } |
| 953 | } |
| 954 | |
| 955 | // A `\`-broken, `&`-aligned expression yields an alignment grid. |
| 956 | #[test] |
| 957 | fn align_block() { |
| 958 | match parse("a &= b \\ &= c") { |
| 959 | Ok(Atom::Matrix { rows, kind, .. }) => { |
| 960 | assert_eq!(rows.len(), 2); |
| 961 | assert_eq!(kind, MatKind::Align); |
| 962 | }, |
| 963 | other => panic!("align -> {:?}", other), |
| 964 | } |
| 965 | } |
| 966 | |
| 967 | // Parentheses around a fraction's parts are grouping, dropped, so no fence survives. |
| 968 | #[test] |
| 969 | fn frac_ungroups_parens() { |
| 970 | match parse("(a)/(b)") { |
| 971 | Ok(Atom::Frac { num, den }) => { |
| 972 | assert!(!matches!(*num, Atom::Fence { .. })); |
| 973 | assert!(!matches!(*den, Atom::Fence { .. })); |
| 974 | }, |
| 975 | other => panic!("(a)/(b) -> {:?}", other), |
| 976 | } |
| 977 | } |
| 978 | |
| 979 | // `dif` sets a thin space and an upright d. |
| 980 | #[test] |
| 981 | fn differential() { |
| 982 | match parse("dif") { |
| 983 | Ok(Atom::Row(items)) => { |
| 984 | assert!(matches!(items.first(), Some(Atom::Space(_)))); |
| 985 | assert!(matches!(items.get(1), Some(Atom::Text(s)) if s == "d")); |
| 986 | }, |
| 987 | other => panic!("dif -> {:?}", other), |
| 988 | } |
| 989 | } |
| 990 | |
| 991 | /// The class of the first `Sym` carrying `text` anywhere in the tree, or `None`. Used to assert a symbol |
| 992 | /// resolved (and to what spacing class) without depending on how the parser wrapped it in rows. |
| 993 | fn find_sym(a: &Atom, text: &str) -> Option<Class> { |
| 994 | match a { |
| 995 | Atom::Sym(s, c) if s == text => Some(*c), |
| 996 | Atom::Sym(..) => None, |
| 997 | Atom::Text(_) | Atom::Space(_) => None, |
| 998 | Atom::Row(items) => items.iter().find_map(|x| find_sym(x, text)), |
| 999 | Atom::Frac { num, den } => find_sym(num, text).or_else(|| find_sym(den, text)), |
| 1000 | Atom::Script { base, sup, sub } => find_sym(base, text) |
| 1001 | .or_else(|| sup.as_deref().and_then(|x| find_sym(x, text))) |
| 1002 | .or_else(|| sub.as_deref().and_then(|x| find_sym(x, text))), |
| 1003 | Atom::Sqrt(r) => find_sym(r, text), |
| 1004 | Atom::Fence { body, .. } => find_sym(body, text), |
| 1005 | Atom::Accent { base, .. } => find_sym(base, text), |
| 1006 | Atom::Matrix { rows, .. } => rows.iter().flatten().find_map(|x| find_sym(x, text)), |
| 1007 | Atom::Binom { top, bottom } => find_sym(top, text).or_else(|| find_sym(bottom, text)), |
| 1008 | Atom::Display(inner) => find_sym(inner, text), |
| 1009 | } |
| 1010 | } |
| 1011 | |
| 1012 | // The circled operators ⊗ ⊕ ⊙, as dotted names and in a subscript (`M_times.circle`, ch04:1950). |
| 1013 | #[test] |
| 1014 | fn circled_operators() { |
| 1015 | assert!(matches!(parse("times.circle"), Ok(Atom::Sym(ref s, Class::Bin)) if s == "\u{2297}"), |
| 1016 | "times.circle -> {:?}", parse("times.circle")); |
| 1017 | assert!(matches!(parse("plus.circle"), Ok(Atom::Sym(ref s, Class::Bin)) if s == "\u{2295}"), |
| 1018 | "plus.circle -> {:?}", parse("plus.circle")); |
| 1019 | assert!(matches!(parse("dot.circle"), Ok(Atom::Sym(ref s, Class::Bin)) if s == "\u{2299}"), |
| 1020 | "dot.circle -> {:?}", parse("dot.circle")); |
| 1021 | match parse("M_times.circle") { |
| 1022 | Ok(a @ Atom::Script { .. }) => assert_eq!(find_sym(&a, "\u{2297}"), Some(Class::Bin), |
| 1023 | "the subscript ⊗ was lost: {:?}", a), |
| 1024 | other => panic!("M_times.circle -> {:?}", other), |
| 1025 | } |
| 1026 | } |
| 1027 | |
| 1028 | // Blackboard sets NN RR ZZ QQ GG resolve; CC (and UU) deliberately do not, staying upright operators. |
| 1029 | #[test] |
| 1030 | fn blackboard_sets_and_cc_exclusion() -> Outcome<()> { |
| 1031 | assert!(matches!(parse("NN"), Ok(Atom::Sym(ref s, Class::Ord)) if s == "\u{2115}"), |
| 1032 | "NN -> {:?}", parse("NN")); |
| 1033 | assert!(matches!(parse("RR"), Ok(Atom::Sym(ref s, Class::Ord)) if s == "\u{211D}")); |
| 1034 | assert!(matches!(parse("ZZ"), Ok(Atom::Sym(ref s, Class::Ord)) if s == "\u{2124}")); |
| 1035 | assert!(matches!(parse("QQ"), Ok(Atom::Sym(ref s, Class::Ord)) if s == "\u{211A}")); |
| 1036 | // GG is the astral blackboard G (U+1D53E, no BMP form), here under a subscript. |
| 1037 | match parse("GG_2") { |
| 1038 | Ok(a @ Atom::Script { .. }) => { |
| 1039 | assert_eq!(find_sym(&a, "\u{1D53E}"), Some(Class::Ord), |
| 1040 | "GG did not resolve to the astral blackboard G: {:?}", a); |
| 1041 | assert!(find_sym(&a, "2").is_some(), "GG_2 lost its subscript: {:?}", a); |
| 1042 | }, |
| 1043 | other => panic!("GG_2 -> {:?}", other), |
| 1044 | } |
| 1045 | // CC is not a blackboard set: a lone `parse("CC")` and the subscript `k'_(CC)` both keep it upright. |
| 1046 | assert!(matches!(parse("CC"), Ok(Atom::Sym(ref s, Class::Op)) if s == "CC"), |
| 1047 | "CC must stay an upright operator word, got {:?}", parse("CC")); |
| 1048 | let sub = res!(parse("k'_(CC)")); |
| 1049 | assert_eq!(find_sym(&sub, "CC"), Some(Class::Op), |
| 1050 | "CC in a subscript must stay an upright operator, got {:?}", sub); |
| 1051 | Ok(()) |
| 1052 | } |
| 1053 | |
| 1054 | // ceil/floor wrap the argument in the corner delimiters, with no extra paren fence (no double bracket). |
| 1055 | #[test] |
| 1056 | fn ceil_floor_fences() { |
| 1057 | match parse("ceil(x)") { |
| 1058 | Ok(Atom::Fence { left, right, body }) => { |
| 1059 | assert_eq!(left, '\u{2308}'); |
| 1060 | assert_eq!(right, '\u{2309}'); |
| 1061 | assert!(!matches!(*body, Atom::Fence { left: '(', .. }), "double bracket: {:?}", body); |
| 1062 | }, |
| 1063 | other => panic!("ceil -> {:?}", other), |
| 1064 | } |
| 1065 | match parse("floor(y)") { |
| 1066 | Ok(Atom::Fence { left, right, .. }) => { |
| 1067 | assert_eq!(left, '\u{230A}'); |
| 1068 | assert_eq!(right, '\u{230B}'); |
| 1069 | }, |
| 1070 | other => panic!("floor -> {:?}", other), |
| 1071 | } |
| 1072 | // ceil(log_2 m) (app_maths:487): the argument is a row, still fenced by the corners. |
| 1073 | assert!(matches!(parse("ceil(log_2 m)"), Ok(Atom::Fence { left: '\u{2308}', .. })), |
| 1074 | "ceil(log_2 m) -> {:?}", parse("ceil(log_2 m)")); |
| 1075 | } |
| 1076 | |
| 1077 | // binom(n, k) is a binomial atom carrying both terms (ch04:1234/1251). |
| 1078 | #[test] |
| 1079 | fn binom_atom() { |
| 1080 | match parse("binom(n, k)") { |
| 1081 | Ok(a @ Atom::Binom { .. }) => { |
| 1082 | assert!(find_sym(&a, "n").is_some(), "binom lost its top term: {:?}", a); |
| 1083 | assert!(find_sym(&a, "k").is_some(), "binom lost its bottom term: {:?}", a); |
| 1084 | }, |
| 1085 | other => panic!("binom -> {:?}", other), |
| 1086 | } |
| 1087 | } |
| 1088 | |
| 1089 | // display(x) forces display style, an atom wrapping its argument. |
| 1090 | #[test] |
| 1091 | fn display_atom() { |
| 1092 | match parse("display(x)") { |
| 1093 | Ok(a @ Atom::Display(_)) => assert!(find_sym(&a, "x").is_some(), |
| 1094 | "display lost its argument: {:?}", a), |
| 1095 | other => panic!("display -> {:?}", other), |
| 1096 | } |
| 1097 | } |
| 1098 | } |