oxedyne/fe2o3/fe2o3_text/src/doc/djot/block.rs
36.1 KiB, 1 run
created by r1870400018:14698, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Block structure: the pass that divides Djot text into headings, paragraphs, lists, code, |
| 2 | //! quotations, divisions, tables and breaks. |
| 3 | //! |
| 4 | //! Djot is read in two passes, as Markdown is, because its two levels are decided by different things. |
| 5 | //! A block is decided by how a line *starts* and by the blank lines around it; an inline run is |
| 6 | //! decided by delimiters within a line. This module does the first pass and hands each block's text to |
| 7 | //! [`crate::doc::djot::inline`] for the second. |
| 8 | //! |
| 9 | //! Before either pass, a sweep gathers the document's reference definitions -- the `[ref]: url` lines |
| 10 | //! by which a `[text][ref]` finds its destination -- so that a reference may be defined anywhere and |
| 11 | //! still resolve a link that stands above it. |
| 12 | |
| 13 | use crate::doc::{ |
| 14 | Align, |
| 15 | Attrs, |
| 16 | Block, |
| 17 | Cell, |
| 18 | Row, |
| 19 | djot::inline, |
| 20 | }; |
| 21 | |
| 22 | use std::collections::HashMap; |
| 23 | |
| 24 | use oxedyne_fe2o3_core::prelude::*; |
| 25 | |
| 26 | /// How deep a document may nest its blocks before the parser refuses it. |
| 27 | /// |
| 28 | /// A division inside a list inside a quotation is legitimate; a thousand of them is a document built |
| 29 | /// to exhaust the stack of whatever reads it. The limit is generous beside anything an author writes |
| 30 | /// and far below what would trouble the machine. |
| 31 | pub const DEPTH_LIMIT: usize = 32; |
| 32 | |
| 33 | /// The columns a tab advances by, counting from a tab stop. |
| 34 | const TAB: usize = 4; |
| 35 | |
| 36 | /// The column past which whitespace after a list marker is more than the marker's own. |
| 37 | const WIDE: usize = 4; |
| 38 | |
| 39 | /// Divides Djot text into its blocks. |
| 40 | pub fn parse(src: &str) -> Outcome<Vec<Block>> { |
| 41 | let lines: Vec<&str> = src.lines().collect(); |
| 42 | let refs = collect_refs(&lines); |
| 43 | blocks(&lines, 0, &refs) |
| 44 | } |
| 45 | |
| 46 | /// What a line begins, judged by how it starts. |
| 47 | enum Start { |
| 48 | /// Nothing but whitespace. |
| 49 | Blank, |
| 50 | /// An ATX heading: its level, and its text. |
| 51 | Atx(u8, String), |
| 52 | /// A code fence. |
| 53 | Fence { |
| 54 | /// The character the fence is made of. |
| 55 | ch: u8, |
| 56 | /// How many characters the fence runs to. |
| 57 | len: usize, |
| 58 | /// The column the fence starts at, which its content is stripped back to. |
| 59 | ind: usize, |
| 60 | /// The language the info string named, if it named one. |
| 61 | lang: Option<String>, |
| 62 | }, |
| 63 | /// A division fence: a run of three or more colons. |
| 64 | Div { |
| 65 | /// How many colons the fence runs to, which its closing fence must match or exceed. |
| 66 | len: usize, |
| 67 | /// The attributes the opening fence named, empty where it named none. |
| 68 | attrs: Attrs, |
| 69 | /// Whether the fence opens a division rather than closing one. |
| 70 | open: bool, |
| 71 | }, |
| 72 | /// A thematic break. |
| 73 | Rule, |
| 74 | /// A block quotation marker. |
| 75 | Quote, |
| 76 | /// A standalone attributes line, which attaches to the block that follows it. |
| 77 | Attr(Attrs), |
| 78 | /// A list item marker. |
| 79 | Item { |
| 80 | /// Whether the marker is a number. |
| 81 | ord: bool, |
| 82 | /// The bullet, or the delimiter that follows a number. |
| 83 | mark: u8, |
| 84 | /// The column the item's content starts at. |
| 85 | col: usize, |
| 86 | /// The number an ordered marker gave, if it gave one. |
| 87 | num: Option<u64>, |
| 88 | /// Whether the marker line carries no content of its own. |
| 89 | bare: bool, |
| 90 | /// The item's first line, with the marker taken off. |
| 91 | head: String, |
| 92 | }, |
| 93 | /// Anything else, which is paragraph text. |
| 94 | Text, |
| 95 | } |
| 96 | |
| 97 | /// Reads a run of lines into the blocks they are, at the given nesting depth. |
| 98 | fn blocks(lines: &[&str], depth: usize, refs: &HashMap<String, String>) -> Outcome<Vec<Block>> { |
| 99 | if depth > DEPTH_LIMIT { |
| 100 | return Err(err!( |
| 101 | "Djot blocks nest more than {} deep, which no document written to be read \ |
| 102 | does.", DEPTH_LIMIT; |
| 103 | Excessive, Input)); |
| 104 | } |
| 105 | let mut out = Vec::new(); |
| 106 | let mut pending: Option<Attrs> = None; // A standalone attributes line, awaiting its block. |
| 107 | let mut i = 0; |
| 108 | while i < lines.len() { |
| 109 | // A reference definition was gathered in the first pass and is not itself a block. |
| 110 | if ref_def_of(lines[i]).is_some() { |
| 111 | i += 1; |
| 112 | continue; |
| 113 | } |
| 114 | match classify(lines[i]) { |
| 115 | Start::Blank => i += 1, |
| 116 | Start::Rule => { |
| 117 | emit(&mut out, &mut pending, Block::Rule); |
| 118 | i += 1; |
| 119 | } |
| 120 | Start::Atx(level, text) => { |
| 121 | let (content, n) = res!(heading(&lines[i..], text, refs)); |
| 122 | emit(&mut out, &mut pending, Block::Heading { level, content }); |
| 123 | i += n; |
| 124 | } |
| 125 | Start::Fence { ch, len, ind, lang } => { |
| 126 | let (b, n) = code_fenced(&lines[i..], ch, len, ind, lang); |
| 127 | emit(&mut out, &mut pending, b); |
| 128 | i += n; |
| 129 | } |
| 130 | Start::Div { len, attrs, .. } => { |
| 131 | let (b, n) = res!(div(&lines[i..], depth, len, attrs, refs)); |
| 132 | emit(&mut out, &mut pending, b); |
| 133 | i += n; |
| 134 | } |
| 135 | Start::Quote => { |
| 136 | let (b, n) = res!(quote(&lines[i..], depth, refs)); |
| 137 | emit(&mut out, &mut pending, b); |
| 138 | i += n; |
| 139 | } |
| 140 | Start::Attr(a) => { |
| 141 | // A second attributes line before the block merges into the first. |
| 142 | pending = Some(match pending.take() { |
| 143 | Some(p) => merge_attrs(p, a), |
| 144 | None => a, |
| 145 | }); |
| 146 | i += 1; |
| 147 | } |
| 148 | Start::Item { .. } => { |
| 149 | let (b, n) = res!(list(&lines[i..], depth, refs)); |
| 150 | emit(&mut out, &mut pending, b); |
| 151 | i += n; |
| 152 | } |
| 153 | Start::Text => { |
| 154 | let (b, n) = res!(para(&lines[i..], refs)); |
| 155 | emit(&mut out, &mut pending, b); |
| 156 | i += n; |
| 157 | } |
| 158 | } |
| 159 | } |
| 160 | Ok(out) |
| 161 | } |
| 162 | |
| 163 | /// Adds a block, wrapping it in a division where a standalone attributes line stood above it. |
| 164 | /// |
| 165 | /// The attributes have nowhere else to go: this tree carries block attributes only on a division, so a |
| 166 | /// `{.note}` above a paragraph makes a division of the paragraph. Above a division of its own, the |
| 167 | /// attributes merge into it rather than wrap it in a second, so `{#a}` over `::: note` names the one |
| 168 | /// box. |
| 169 | fn emit(out: &mut Vec<Block>, pending: &mut Option<Attrs>, b: Block) { |
| 170 | match pending.take() { |
| 171 | None => out.push(b), |
| 172 | Some(attrs) => match b { |
| 173 | Block::Div { attrs: inner, content } => { |
| 174 | out.push(Block::Div { attrs: merge_attrs(attrs, inner), content }); |
| 175 | } |
| 176 | other => { |
| 177 | out.push(Block::Div { attrs, content: vec![other] }); |
| 178 | } |
| 179 | }, |
| 180 | } |
| 181 | } |
| 182 | |
| 183 | /// Combines an outer set of attributes, written first, with an inner set the block carried of its own. |
| 184 | /// |
| 185 | /// The inner set is written second, so where the two both name an id the inner wins, as the last id |
| 186 | /// written wins within a single brace group. Classes and pairs of both are kept, the outer's first. |
| 187 | fn merge_attrs(outer: Attrs, inner: Attrs) -> Attrs { |
| 188 | let mut classes = outer.classes; |
| 189 | classes.extend(inner.classes); |
| 190 | let mut pairs = outer.pairs; |
| 191 | pairs.extend(inner.pairs); |
| 192 | Attrs { |
| 193 | id: inner.id.or(outer.id), |
| 194 | classes, |
| 195 | pairs, |
| 196 | } |
| 197 | } |
| 198 | |
| 199 | // ── Classification ─────────────────────────────────────────────── |
| 200 | |
| 201 | /// Judges what a line begins. |
| 202 | fn classify(line: &str) -> Start { |
| 203 | if is_blank(line) { |
| 204 | return Start::Blank; |
| 205 | } |
| 206 | let ind = indent_of(line); |
| 207 | let off = ws_end(line); |
| 208 | let rest = &line[off..]; |
| 209 | let b = rest.as_bytes(); |
| 210 | |
| 211 | // A thematic break is judged first, so that `***` is a break and not a list of nothing. |
| 212 | if is_rule(rest) { |
| 213 | return Start::Rule; |
| 214 | } |
| 215 | |
| 216 | // An ATX heading: one to six hashes, set apart from the text by a space. |
| 217 | if b[0] == b'#' { |
| 218 | let n = rest.bytes().take_while(|c| *c == b'#').count(); |
| 219 | let after = &rest[n..]; |
| 220 | if n <= 6 && (after.is_empty() || after.starts_with(' ') || after.starts_with('\t')) { |
| 221 | return Start::Atx(n as u8, atx_text(after)); |
| 222 | } |
| 223 | } |
| 224 | |
| 225 | // A division fence: three or more colons, and the class or attributes an opener names. |
| 226 | if b[0] == b':' { |
| 227 | let n = rest.bytes().take_while(|c| *c == b':').count(); |
| 228 | if n >= 3 { |
| 229 | let info = rest[n..].trim(); |
| 230 | let open = !info.is_empty(); |
| 231 | let attrs = div_attrs(info); |
| 232 | return Start::Div { len: n, attrs, open }; |
| 233 | } |
| 234 | } |
| 235 | |
| 236 | // A code fence: three or more backticks or tildes, and what they say about the code. |
| 237 | if b[0] == b'`' || b[0] == b'~' { |
| 238 | let ch = b[0]; |
| 239 | let len = rest.bytes().take_while(|c| *c == ch).count(); |
| 240 | let info = rest[len..].trim(); |
| 241 | // A backtick fence's info string may hold no backtick, or `a ` b` would open one. |
| 242 | if len >= 3 && (ch != b'`' || !info.contains('`')) { |
| 243 | let lang = info.split_whitespace().next().map(|w| w.to_string()); |
| 244 | return Start::Fence { ch, len, ind, lang }; |
| 245 | } |
| 246 | } |
| 247 | |
| 248 | if b[0] == b'>' { |
| 249 | return Start::Quote; |
| 250 | } |
| 251 | |
| 252 | // A line that is nothing but a brace group attaches its attributes to the next block. |
| 253 | if b[0] == b'{' { |
| 254 | let t = rest.trim_end(); |
| 255 | if let Some((a, used)) = inline::attrs_of(t) { |
| 256 | if used == t.len() { |
| 257 | return Start::Attr(a); |
| 258 | } |
| 259 | } |
| 260 | } |
| 261 | |
| 262 | match item_start(line, off, ind) { |
| 263 | Some(s) => s, |
| 264 | None => Start::Text, |
| 265 | } |
| 266 | } |
| 267 | |
| 268 | /// The attributes a division's opening fence named: a bare class name, or a full brace group. |
| 269 | fn div_attrs(info: &str) -> Attrs { |
| 270 | if info.is_empty() { |
| 271 | Attrs::default() |
| 272 | } else if info.starts_with('{') { |
| 273 | match inline::attrs_of(info) { |
| 274 | Some((a, _)) => a, |
| 275 | None => Attrs::default(), |
| 276 | } |
| 277 | } else { |
| 278 | // A single word after the colons is a class, its leading dot optional. |
| 279 | let mut a = Attrs::default(); |
| 280 | if let Some(tok) = info.split_whitespace().next() { |
| 281 | a.classes.push(tok.trim_start_matches('.').to_string()); |
| 282 | } |
| 283 | a |
| 284 | } |
| 285 | } |
| 286 | |
| 287 | /// Recognises a list item marker, and what it says about the item it begins. |
| 288 | fn item_start(line: &str, off: usize, ind: usize) -> Option<Start> { |
| 289 | let b = line.as_bytes(); |
| 290 | let mut p = off; // Byte offset. |
| 291 | let mut c = ind; // Column. |
| 292 | let mut ord = false; |
| 293 | let mut num = None; |
| 294 | let mark; |
| 295 | match b[p] { |
| 296 | b'-' | b'*' | b'+' => { |
| 297 | mark = b[p]; |
| 298 | p += 1; |
| 299 | c += 1; |
| 300 | } |
| 301 | d if d.is_ascii_digit() => { |
| 302 | let s = p; |
| 303 | // Nine digits is as long a number as a list may count to. |
| 304 | while p < b.len() && b[p].is_ascii_digit() && p - s < 9 { |
| 305 | p += 1; |
| 306 | } |
| 307 | if p >= b.len() || (b[p] != b'.' && b[p] != b')') { |
| 308 | return None; |
| 309 | } |
| 310 | num = line[s..p].parse::<u64>().ok(); |
| 311 | mark = b[p]; |
| 312 | ord = true; |
| 313 | c += p - s + 1; |
| 314 | p += 1; |
| 315 | } |
| 316 | _ => return None, |
| 317 | } |
| 318 | let rest = &line[p..]; |
| 319 | // A marker with nothing after it is an empty item. |
| 320 | if rest.is_empty() || is_blank(rest) { |
| 321 | return Some(Start::Item { |
| 322 | ord, |
| 323 | mark, |
| 324 | col: c + 1, |
| 325 | num, |
| 326 | bare: true, |
| 327 | head: String::new(), |
| 328 | }); |
| 329 | } |
| 330 | let sp = indent_of(rest); // Columns of whitespace between marker and content. |
| 331 | if sp == 0 { |
| 332 | return None; // `-foo` is a word beginning with a dash, not a list. |
| 333 | } |
| 334 | // Whitespace far past the marker is content indented within the item, not part of the marker. |
| 335 | let take = if sp > WIDE { 1 } else { sp }; |
| 336 | Some(Start::Item { |
| 337 | ord, |
| 338 | mark, |
| 339 | col: c + take, |
| 340 | num, |
| 341 | bare: false, |
| 342 | head: strip_cols(rest, take), |
| 343 | }) |
| 344 | } |
| 345 | |
| 346 | /// Whether the line, past its indent, is a thematic break. |
| 347 | fn is_rule(rest: &str) -> bool { |
| 348 | let b = rest.as_bytes(); |
| 349 | if b.is_empty() { |
| 350 | return false; |
| 351 | } |
| 352 | let ch = b[0]; |
| 353 | if ch != b'-' && ch != b'*' && ch != b'_' { |
| 354 | return false; |
| 355 | } |
| 356 | let mut n = 0; |
| 357 | for &c in b { |
| 358 | if c == ch { |
| 359 | n += 1; |
| 360 | } else if c != b' ' && c != b'\t' { |
| 361 | return false; |
| 362 | } |
| 363 | } |
| 364 | n >= 3 |
| 365 | } |
| 366 | |
| 367 | /// The alignments a delimiter row gives, if the line is one and divides into as many cells as the |
| 368 | /// header above it did. |
| 369 | fn delim_of(line: &str, n: usize) -> Option<Vec<Align>> { |
| 370 | let cells = cells_of(line); |
| 371 | if cells.is_empty() || cells.len() != n { |
| 372 | return None; |
| 373 | } |
| 374 | let mut cols = Vec::with_capacity(cells.len()); |
| 375 | for cell in &cells { |
| 376 | match align_of(cell) { |
| 377 | Some(align) => cols.push(align), |
| 378 | None => return None, |
| 379 | } |
| 380 | } |
| 381 | Some(cols) |
| 382 | } |
| 383 | |
| 384 | /// The alignment one delimiter cell gives: a colon at the side the column is aligned to, and dashes |
| 385 | /// between. |
| 386 | fn align_of(cell: &str) -> Option<Align> { |
| 387 | let b = cell.as_bytes(); |
| 388 | if b.is_empty() { |
| 389 | return None; |
| 390 | } |
| 391 | let start = b[0] == b':'; |
| 392 | let end = b[b.len() - 1] == b':'; |
| 393 | let s = if start { 1 } else { 0 }; // Where the dashes begin. |
| 394 | let e = b.len() - if end { 1 } else { 0 }; // Where they end. |
| 395 | if s >= e || !b[s..e].iter().all(|c| *c == b'-') { |
| 396 | return None; |
| 397 | } |
| 398 | Some(match (start, end) { |
| 399 | (true, true) => Align::Centre, |
| 400 | (true, false) => Align::Start, |
| 401 | (false, true) => Align::End, |
| 402 | (false, false) => Align::None, |
| 403 | }) |
| 404 | } |
| 405 | |
| 406 | /// The cells a table row divides into, split on every pipe the author did not escape. |
| 407 | /// |
| 408 | /// A pipe at either end of the row is the row's own edge and not an empty cell, so an author may draw |
| 409 | /// the edges or leave them off. A `\|` divides nothing: it stays in the cell for [`inline::parse`] to |
| 410 | /// resolve as the escape it is. |
| 411 | fn cells_of(line: &str) -> Vec<&str> { |
| 412 | let t = trim(line); |
| 413 | let b = t.as_bytes(); |
| 414 | let mut out = Vec::new(); |
| 415 | let mut i = if b.first() == Some(&b'|') { 1 } else { 0 }; |
| 416 | let mut s = i; // Where the cell being read begins. |
| 417 | while i < b.len() { |
| 418 | match b[i] { |
| 419 | b'\\' => i += 2, // A backslash and whatever it escapes divide nothing. |
| 420 | b'|' => { |
| 421 | out.push(trim(&t[s..i])); |
| 422 | i += 1; |
| 423 | s = i; |
| 424 | } |
| 425 | _ => i += 1, |
| 426 | } |
| 427 | } |
| 428 | if s < t.len() { |
| 429 | out.push(trim(&t[s..])); |
| 430 | } |
| 431 | out |
| 432 | } |
| 433 | |
| 434 | /// A heading's text, with the whitespace around it taken off. |
| 435 | /// |
| 436 | /// Djot heads a section with hashes at the front alone, and keeps no closing run of them, so a `#` at |
| 437 | /// the end of the line is part of what the heading says. |
| 438 | fn atx_text(after: &str) -> String { |
| 439 | after.trim_matches(|c| c == ' ' || c == '\t').to_string() |
| 440 | } |
| 441 | |
| 442 | /// Whether a line begins something that ends the paragraph above it. |
| 443 | fn interrupts(line: &str) -> bool { |
| 444 | match classify(line) { |
| 445 | Start::Blank |
| 446 | | Start::Atx(..) |
| 447 | | Start::Fence { .. } |
| 448 | | Start::Div { .. } |
| 449 | | Start::Rule |
| 450 | | Start::Attr(_) |
| 451 | | Start::Quote => true, |
| 452 | // A list ends a paragraph only where it plainly begins one, so that a year opening a |
| 453 | // sentence, or a full stop wrapping onto its own line, is prose and not a list. |
| 454 | Start::Item { ord, num, bare, .. } => !bare && (!ord || num == Some(1)), |
| 455 | Start::Text => false, |
| 456 | } |
| 457 | } |
| 458 | |
| 459 | /// The reference label and destination a `[ref]: url` line defines, if it is one. |
| 460 | fn ref_def_of(line: &str) -> Option<(String, String)> { |
| 461 | let rest = &line[ws_end(line)..]; |
| 462 | let b = rest.as_bytes(); |
| 463 | if b.is_empty() || b[0] != b'[' { |
| 464 | return None; |
| 465 | } |
| 466 | let mut j = 1; |
| 467 | while j < b.len() { |
| 468 | match b[j] { |
| 469 | b'\\' => j += 2, |
| 470 | b']' => break, |
| 471 | _ => j += 1, |
| 472 | } |
| 473 | } |
| 474 | if j >= b.len() || b[j] != b']' { |
| 475 | return None; |
| 476 | } |
| 477 | let label = &rest[1..j]; |
| 478 | if label.is_empty() { |
| 479 | return None; |
| 480 | } |
| 481 | let k = j + 1; |
| 482 | if k >= b.len() || b[k] != b':' { |
| 483 | return None; |
| 484 | } |
| 485 | Some((label.to_string(), rest[k + 1..].trim().to_string())) |
| 486 | } |
| 487 | |
| 488 | /// Gathers the document's reference definitions, folding each label the way a reference to it will be. |
| 489 | /// |
| 490 | /// The sweep steps over fenced code, so that a line that looks like a definition within a fence is the |
| 491 | /// code it is and defines nothing. The first definition of a label wins, as Djot has it. |
| 492 | fn collect_refs(lines: &[&str]) -> HashMap<String, String> { |
| 493 | let mut map = HashMap::new(); |
| 494 | let mut i = 0; |
| 495 | while i < lines.len() { |
| 496 | match classify(lines[i]) { |
| 497 | Start::Fence { ch, len, .. } => { |
| 498 | i += 1; |
| 499 | while i < lines.len() && !fence_closes(lines[i], ch, len) { |
| 500 | i += 1; |
| 501 | } |
| 502 | if i < lines.len() { |
| 503 | i += 1; |
| 504 | } |
| 505 | } |
| 506 | _ => { |
| 507 | if let Some((label, url)) = ref_def_of(lines[i]) { |
| 508 | map.entry(inline::normalise(&label)).or_insert(url); |
| 509 | } |
| 510 | i += 1; |
| 511 | } |
| 512 | } |
| 513 | } |
| 514 | map |
| 515 | } |
| 516 | |
| 517 | // ── The blocks themselves ──────────────────────────────────────── |
| 518 | |
| 519 | /// Reads a heading and the lazy lines that carry its text on. |
| 520 | /// |
| 521 | /// A heading's text runs across the non-blank lines that follow it, to a blank line or to whatever |
| 522 | /// begins a block of its own, so an author may wrap a long heading. The lines are joined and read as |
| 523 | /// one run, where the soft breaks between them say spaces. |
| 524 | fn heading(lines: &[&str], text: String, refs: &HashMap<String, String>) |
| 525 | -> Outcome<(Vec<crate::doc::Inline>, usize)> |
| 526 | { |
| 527 | let mut acc = vec![text]; |
| 528 | let mut i = 1; |
| 529 | while i < lines.len() { |
| 530 | if is_blank(lines[i]) || ref_def_of(lines[i]).is_some() { |
| 531 | break; |
| 532 | } |
| 533 | match classify(lines[i]) { |
| 534 | Start::Text => { |
| 535 | acc.push(strip_leading(lines[i]).to_string()); |
| 536 | i += 1; |
| 537 | } |
| 538 | _ => break, |
| 539 | } |
| 540 | } |
| 541 | let content = res!(inline::parse(&acc.join("\n"), refs)); |
| 542 | Ok((content, i)) |
| 543 | } |
| 544 | |
| 545 | /// Reads a paragraph, or the table a delimiter row turns its last line into. |
| 546 | fn para(lines: &[&str], refs: &HashMap<String, String>) -> Outcome<(Block, usize)> { |
| 547 | let mut acc = vec![strip_leading(lines[0]).to_string()]; |
| 548 | let mut i = 1; |
| 549 | while i < lines.len() { |
| 550 | // A table's header row is a paragraph line until the delimiter row beneath it says otherwise. |
| 551 | if let Some(cols) = delim_of(lines[i], cells_of(&acc[i - 1]).len()) { |
| 552 | // The lines above the header are a paragraph of their own, and they end here. The header |
| 553 | // goes back to the document to be read again, as the table's first line. |
| 554 | if i > 1 { |
| 555 | acc.pop(); |
| 556 | return Ok(( |
| 557 | Block::Para(res!(inline::parse(acc.join("\n").trim_end(), refs))), |
| 558 | i - 1, |
| 559 | )); |
| 560 | } |
| 561 | return table(lines, cols, refs); |
| 562 | } |
| 563 | if ref_def_of(lines[i]).is_some() || interrupts(lines[i]) { |
| 564 | break; |
| 565 | } |
| 566 | acc.push(strip_leading(lines[i]).to_string()); |
| 567 | i += 1; |
| 568 | } |
| 569 | Ok((Block::Para(res!(inline::parse(acc.join("\n").trim_end(), refs))), i)) |
| 570 | } |
| 571 | |
| 572 | /// Reads a table: the header row, the delimiter row that made it one, and the body beneath them. |
| 573 | fn table(lines: &[&str], cols: Vec<Align>, refs: &HashMap<String, String>) |
| 574 | -> Outcome<(Block, usize)> |
| 575 | { |
| 576 | let head = res!(row(lines[0], cols.len(), refs)); |
| 577 | let mut rows = Vec::new(); |
| 578 | let mut i = 2; |
| 579 | // The table runs to the first line that begins a block of its own, a blank line among them. |
| 580 | while i < lines.len() && !interrupts(lines[i]) { |
| 581 | rows.push(res!(row(lines[i], cols.len(), refs))); |
| 582 | i += 1; |
| 583 | } |
| 584 | Ok((Block::Table { head: Some(head), rows, cols }, i)) |
| 585 | } |
| 586 | |
| 587 | /// Reads one row of a table, held to the width the header set. |
| 588 | fn row(line: &str, n: usize, refs: &HashMap<String, String>) -> Outcome<Row> { |
| 589 | let mut cells = Vec::with_capacity(n); |
| 590 | for text in cells_of(line).iter().take(n) { |
| 591 | cells.push(Cell(res!(inline::parse(text, refs)))); |
| 592 | } |
| 593 | // A row of fewer cells than the header named columns is short of the grid rather than wrong, and |
| 594 | // a row of more has said something the header made no column for, and what it has no column for is |
| 595 | // dropped. |
| 596 | while cells.len() < n { |
| 597 | cells.push(Cell(Vec::new())); |
| 598 | } |
| 599 | Ok(Row(cells)) |
| 600 | } |
| 601 | |
| 602 | /// Reads a fenced code block, which runs to its closing fence or to the end of the input. |
| 603 | fn code_fenced(lines: &[&str], ch: u8, len: usize, ind: usize, lang: Option<String>) |
| 604 | -> (Block, usize) |
| 605 | { |
| 606 | let mut text = String::new(); |
| 607 | let mut i = 1; |
| 608 | while i < lines.len() { |
| 609 | if fence_closes(lines[i], ch, len) { |
| 610 | return (Block::Code { lang, text }, i + 1); |
| 611 | } |
| 612 | text.push_str(&strip_cols(lines[i], ind)); |
| 613 | text.push('\n'); |
| 614 | i += 1; |
| 615 | } |
| 616 | (Block::Code { lang, text }, i) |
| 617 | } |
| 618 | |
| 619 | /// Whether the line closes a code fence of the given character and length. |
| 620 | fn fence_closes(line: &str, ch: u8, len: usize) -> bool { |
| 621 | let rest = &line[ws_end(line)..]; |
| 622 | let n = rest.bytes().take_while(|c| *c == ch).count(); |
| 623 | // A closing fence is at least as long as the one it closes, and says nothing else. |
| 624 | n >= len && is_blank(&rest[n..]) |
| 625 | } |
| 626 | |
| 627 | /// Reads a division and the blocks within it, which run to its closing fence or to the end of input. |
| 628 | /// |
| 629 | /// The fences nest, so a division holds divisions. An opening fence names a class or attributes and an |
| 630 | /// inner opening fence deepens the nesting; a bare fence of colons closes the innermost division still |
| 631 | /// open. A division nobody closed runs to the end, as an unclosed code fence does. |
| 632 | fn div(lines: &[&str], depth: usize, len: usize, attrs: Attrs, refs: &HashMap<String, String>) |
| 633 | -> Outcome<(Block, usize)> |
| 634 | { |
| 635 | let mut nest = 1; |
| 636 | let mut close = None; |
| 637 | let mut i = 1; |
| 638 | while i < lines.len() { |
| 639 | if let Some((flen, open)) = div_fence(lines[i]) { |
| 640 | if open { |
| 641 | nest += 1; |
| 642 | } else if flen >= len { |
| 643 | nest -= 1; |
| 644 | if nest == 0 { |
| 645 | close = Some(i); |
| 646 | break; |
| 647 | } |
| 648 | } |
| 649 | } |
| 650 | i += 1; |
| 651 | } |
| 652 | let end = match close { |
| 653 | Some(c) => c, |
| 654 | None => lines.len(), |
| 655 | }; |
| 656 | let content = res!(blocks(&lines[1..end], depth + 1, refs)); |
| 657 | let consumed = match close { |
| 658 | Some(c) => c + 1, |
| 659 | None => lines.len(), |
| 660 | }; |
| 661 | Ok((Block::Div { attrs, content }, consumed)) |
| 662 | } |
| 663 | |
| 664 | /// Whether the line is a division fence, how many colons it runs to, and whether it opens. |
| 665 | /// |
| 666 | /// A fence that names a class or attributes opens a division; a bare run of colons closes one. An |
| 667 | /// anonymous division opened by a bare fence is read where it does not nest inside another anonymous |
| 668 | /// one, which no prose written to be read does. |
| 669 | fn div_fence(line: &str) -> Option<(usize, bool)> { |
| 670 | let rest = &line[ws_end(line)..]; |
| 671 | let b = rest.as_bytes(); |
| 672 | if b.is_empty() || b[0] != b':' { |
| 673 | return None; |
| 674 | } |
| 675 | let n = rest.bytes().take_while(|c| *c == b':').count(); |
| 676 | if n < 3 { |
| 677 | return None; |
| 678 | } |
| 679 | Some((n, !rest[n..].trim().is_empty())) |
| 680 | } |
| 681 | |
| 682 | /// Reads a block quotation and the blocks within it. |
| 683 | fn quote(lines: &[&str], depth: usize, refs: &HashMap<String, String>) -> Outcome<(Block, usize)> { |
| 684 | let mut body: Vec<String> = Vec::new(); |
| 685 | let mut i = 0; |
| 686 | while i < lines.len() { |
| 687 | if let Some(rest) = quote_strip(lines[i]) { |
| 688 | body.push(rest); |
| 689 | i += 1; |
| 690 | continue; |
| 691 | } |
| 692 | if is_blank(lines[i]) || !lazy(&body, lines[i]) { |
| 693 | break; |
| 694 | } |
| 695 | body.push(lines[i].trim_start_matches(|c| c == ' ' || c == '\t').to_string()); |
| 696 | i += 1; |
| 697 | } |
| 698 | let refs_in: Vec<&str> = body.iter().map(|s| s.as_str()).collect(); |
| 699 | Ok((Block::Quote(res!(blocks(&refs_in, depth + 1, refs))), i)) |
| 700 | } |
| 701 | |
| 702 | /// The line with its quotation marker taken off, if it carries one. |
| 703 | fn quote_strip(line: &str) -> Option<String> { |
| 704 | match line[ws_end(line)..].strip_prefix('>') { |
| 705 | // One space after the marker is the marker's own, and no more. |
| 706 | Some(rest) => Some(strip_cols(rest, 1)), |
| 707 | None => None, |
| 708 | } |
| 709 | } |
| 710 | |
| 711 | /// Reads a list: its first item, and every item that follows it at the same level with the same |
| 712 | /// marker. |
| 713 | fn list(lines: &[&str], depth: usize, refs: &HashMap<String, String>) -> Outcome<(Block, usize)> { |
| 714 | let (ord, mark) = match classify(lines[0]) { |
| 715 | Start::Item { ord, mark, .. } => (ord, mark), |
| 716 | _ => return Err(err!( |
| 717 | "A list was read from a line that does not begin one."; Bug)), |
| 718 | }; |
| 719 | let mut items = Vec::new(); |
| 720 | let mut i = 0; |
| 721 | loop { |
| 722 | // Blank lines may sit between items, but they are the list's only if an item follows. |
| 723 | let mut j = i; |
| 724 | while j < lines.len() && is_blank(lines[j]) { |
| 725 | j += 1; |
| 726 | } |
| 727 | if j >= lines.len() { |
| 728 | break; |
| 729 | } |
| 730 | // A different marker begins a different list, which is how an author separates two. |
| 731 | let (col, head) = match classify(lines[j]) { |
| 732 | Start::Item { ord: o, mark: m, col, head, .. } if o == ord && m == mark |
| 733 | => (col, head), |
| 734 | _ => break, |
| 735 | }; |
| 736 | let (body, n) = item_body(&lines[j..], head, col); |
| 737 | let refs_in: Vec<&str> = body.iter().map(|s| s.as_str()).collect(); |
| 738 | items.push(res!(blocks(&refs_in, depth + 1, refs))); |
| 739 | i = j + n; |
| 740 | } |
| 741 | Ok((Block::List { ordered: ord, items }, i)) |
| 742 | } |
| 743 | |
| 744 | /// Reads one list item's lines, with the marker and the indentation that stands for it taken off. |
| 745 | fn item_body(lines: &[&str], head: String, col: usize) -> (Vec<String>, usize) { |
| 746 | let mut body = vec![head]; |
| 747 | let mut i = 1; |
| 748 | while i < lines.len() { |
| 749 | if is_blank(lines[i]) { |
| 750 | // A blank line is the item's only if indented content follows it. |
| 751 | let mut k = i; |
| 752 | while k < lines.len() && is_blank(lines[k]) { |
| 753 | k += 1; |
| 754 | } |
| 755 | if k >= lines.len() || indent_of(lines[k]) < col { |
| 756 | break; |
| 757 | } |
| 758 | for _ in i..k { |
| 759 | body.push(String::new()); |
| 760 | } |
| 761 | i = k; |
| 762 | continue; |
| 763 | } |
| 764 | if indent_of(lines[i]) >= col { |
| 765 | body.push(strip_cols(lines[i], col)); |
| 766 | i += 1; |
| 767 | continue; |
| 768 | } |
| 769 | if !lazy(&body, lines[i]) { |
| 770 | break; |
| 771 | } |
| 772 | body.push(lines[i].trim_start_matches(|c| c == ' ' || c == '\t').to_string()); |
| 773 | i += 1; |
| 774 | } |
| 775 | (body, i) |
| 776 | } |
| 777 | |
| 778 | /// Whether an under-indented line carries on the paragraph a container was in the middle of. |
| 779 | /// |
| 780 | /// This is the laziness an author leans on when wrapping a quoted or listed paragraph: a line that is |
| 781 | /// only text goes on belonging to the paragraph above. |
| 782 | fn lazy(body: &[String], line: &str) -> bool { |
| 783 | match body.last() { |
| 784 | Some(l) if !is_blank(l) => matches!(classify(line), Start::Text), |
| 785 | _ => false, |
| 786 | } |
| 787 | } |
| 788 | |
| 789 | // ── Whitespace ─────────────────────────────────────────────────── |
| 790 | |
| 791 | /// The text with the spaces and tabs at either end of it taken off. |
| 792 | fn trim(s: &str) -> &str { |
| 793 | s.trim_matches(|c| c == ' ' || c == '\t') |
| 794 | } |
| 795 | |
| 796 | /// The line with the leading spaces and tabs taken off. |
| 797 | fn strip_leading(line: &str) -> &str { |
| 798 | line.trim_start_matches(|c| c == ' ' || c == '\t') |
| 799 | } |
| 800 | |
| 801 | /// Whether the line holds nothing but whitespace. |
| 802 | fn is_blank(line: &str) -> bool { |
| 803 | line.bytes().all(|b| b == b' ' || b == b'\t') |
| 804 | } |
| 805 | |
| 806 | /// The column the line's first non-whitespace character sits at, a tab advancing to the next tab |
| 807 | /// stop. |
| 808 | fn indent_of(line: &str) -> usize { |
| 809 | let mut c = 0; |
| 810 | for b in line.bytes() { |
| 811 | match b { |
| 812 | b' ' => c += 1, |
| 813 | b'\t' => c += TAB - (c % TAB), |
| 814 | _ => break, |
| 815 | } |
| 816 | } |
| 817 | c |
| 818 | } |
| 819 | |
| 820 | /// The byte offset of the line's first non-whitespace character. |
| 821 | fn ws_end(line: &str) -> usize { |
| 822 | let mut i = 0; |
| 823 | for b in line.bytes() { |
| 824 | if b != b' ' && b != b'\t' { |
| 825 | break; |
| 826 | } |
| 827 | i += 1; |
| 828 | } |
| 829 | i |
| 830 | } |
| 831 | |
| 832 | /// The line with `n` columns of leading whitespace taken off. |
| 833 | /// |
| 834 | /// A tab that straddles the cut gives up what it spans past it as spaces, which is the only way to |
| 835 | /// take a fixed number of columns from a line a tab has indented. |
| 836 | fn strip_cols(line: &str, n: usize) -> String { |
| 837 | let mut c = 0; // Column. |
| 838 | let mut i = 0; // Byte offset. |
| 839 | for b in line.bytes() { |
| 840 | if c >= n { |
| 841 | break; |
| 842 | } |
| 843 | match b { |
| 844 | b' ' => { |
| 845 | c += 1; |
| 846 | i += 1; |
| 847 | } |
| 848 | b'\t' => { |
| 849 | let next = c + TAB - (c % TAB); |
| 850 | if next > n { |
| 851 | let mut s = " ".repeat(next - n); |
| 852 | s.push_str(&line[i + 1..]); |
| 853 | return s; |
| 854 | } |
| 855 | c = next; |
| 856 | i += 1; |
| 857 | } |
| 858 | _ => break, |
| 859 | } |
| 860 | } |
| 861 | line[i..].to_string() |
| 862 | } |
| 863 | |
| 864 | #[cfg(test)] |
| 865 | mod tests { |
| 866 | use super::*; |
| 867 | use crate::doc::Inline; |
| 868 | |
| 869 | /// The text of a block's inlines, for tests that care what a block says and not how. |
| 870 | fn said(blocks: &[Block]) -> Vec<String> { |
| 871 | blocks.iter().map(|b| match b { |
| 872 | Block::Para(c) => crate::doc::text_of(c), |
| 873 | Block::Heading { content, .. } => crate::doc::text_of(content), |
| 874 | Block::Code { text, .. } => text.clone(), |
| 875 | _ => String::new(), |
| 876 | }).collect() |
| 877 | } |
| 878 | |
| 879 | /// What a table's rows say, cell by cell, the header first. |
| 880 | fn grid(b: &Block) -> Vec<Vec<String>> { |
| 881 | match b { |
| 882 | Block::Table { head, rows, .. } => { |
| 883 | let mut out = Vec::new(); |
| 884 | if let Some(head) = head { |
| 885 | out.push(head.0.iter().map(|c| c.text_of()).collect()); |
| 886 | } |
| 887 | for row in rows { |
| 888 | out.push(row.0.iter().map(|c| c.text_of()).collect()); |
| 889 | } |
| 890 | out |
| 891 | } |
| 892 | other => panic!("expected a table, got {:?}", other), |
| 893 | } |
| 894 | } |
| 895 | |
| 896 | /// A run of hashes and a space open a heading, at each of the six levels. |
| 897 | #[test] |
| 898 | fn test_hashes_open_a_heading_at_every_level_00() -> Outcome<()> { |
| 899 | for n in 1..=6u8 { |
| 900 | let src = format!("{} Heading\n", "#".repeat(n as usize)); |
| 901 | let b = res!(parse(&src)); |
| 902 | assert_eq!(b, vec![Block::Heading { level: n, content: vec![Inline::Text("Heading".into())] }]); |
| 903 | } |
| 904 | // Seven hashes is no heading, since there is no seventh level to give it. |
| 905 | assert_eq!(said(&res!(parse("####### Seven\n"))), vec!["####### Seven"]); |
| 906 | Ok(()) |
| 907 | } |
| 908 | |
| 909 | /// A heading's text runs across the lazy lines that follow it, to a blank line. |
| 910 | #[test] |
| 911 | fn test_a_heading_carries_on_lazily_01() -> Outcome<()> { |
| 912 | let b = res!(parse("# A heading that\nwraps a long way\n\nA paragraph.\n")); |
| 913 | assert_eq!(b.len(), 2); |
| 914 | assert_eq!(said(&b[..1]), vec!["A heading that wraps a long way"]); |
| 915 | assert_eq!(said(&b[1..]), vec!["A paragraph."]); |
| 916 | Ok(()) |
| 917 | } |
| 918 | |
| 919 | /// A soft line ending within a paragraph says a space, so that prose reflows. |
| 920 | #[test] |
| 921 | fn test_a_soft_break_says_a_space_02() -> Outcome<()> { |
| 922 | let b = res!(parse("A paragraph the author\nhard wrapped at a\nnarrow width.\n")); |
| 923 | assert_eq!(b, vec![Block::Para(vec![ |
| 924 | Inline::Text("A paragraph the author hard wrapped at a narrow width.".into()), |
| 925 | ])]); |
| 926 | Ok(()) |
| 927 | } |
| 928 | |
| 929 | /// A backslash at the end of a line is a break the author asked for. |
| 930 | #[test] |
| 931 | fn test_a_backslash_makes_a_hard_break_03() -> Outcome<()> { |
| 932 | let b = res!(parse("one\\\ntwo\n")); |
| 933 | assert_eq!(b, vec![Block::Para(vec![ |
| 934 | Inline::Text("one".into()), |
| 935 | Inline::Break, |
| 936 | Inline::Text("two".into()), |
| 937 | ])]); |
| 938 | Ok(()) |
| 939 | } |
| 940 | |
| 941 | /// A fence holds code exactly as written, with a language or without, by backticks or by tildes. |
| 942 | #[test] |
| 943 | fn test_a_fence_holds_code_04() -> Outcome<()> { |
| 944 | let b = res!(parse("```rust\nlet x = *y;\n```\n")); |
| 945 | assert_eq!(b, vec![Block::Code { lang: Some("rust".into()), text: "let x = *y;\n".into() }]); |
| 946 | let b = res!(parse("~~~\nplain\n~~~\n")); |
| 947 | assert_eq!(b, vec![Block::Code { lang: None, text: "plain\n".into() }]); |
| 948 | // A fence nobody closed runs to the end. |
| 949 | let b = res!(parse("```\nstill code\n")); |
| 950 | assert_eq!(b, vec![Block::Code { lang: None, text: "still code\n".into() }]); |
| 951 | Ok(()) |
| 952 | } |
| 953 | |
| 954 | /// Three or more of a break's characters make a thematic break. |
| 955 | #[test] |
| 956 | fn test_a_thematic_break_05() -> Outcome<()> { |
| 957 | for src in ["---\n", "***\n", "___\n", "* * *\n"] { |
| 958 | assert_eq!(res!(parse(src)), vec![Block::Rule], "for {:?}", src); |
| 959 | } |
| 960 | Ok(()) |
| 961 | } |
| 962 | |
| 963 | /// A quotation holds blocks, and reads them as a document of its own. |
| 964 | #[test] |
| 965 | fn test_a_quotation_holds_blocks_06() -> Outcome<()> { |
| 966 | let b = res!(parse("> # Heading\n>\n> A paragraph.\n")); |
| 967 | match &b[0] { |
| 968 | Block::Quote(inner) => assert_eq!(said(inner), vec!["Heading", "A paragraph."]), |
| 969 | other => panic!("expected a quotation, got {:?}", other), |
| 970 | } |
| 971 | Ok(()) |
| 972 | } |
| 973 | |
| 974 | /// Bullets make an unordered list, and numbers an ordered one, nesting by indentation. |
| 975 | #[test] |
| 976 | fn test_lists_nest_by_indentation_07() -> Outcome<()> { |
| 977 | let b = res!(parse("- a\n - inner\n- b\n")); |
| 978 | match &b[0] { |
| 979 | Block::List { ordered, items } => { |
| 980 | assert!(!ordered); |
| 981 | assert_eq!(items.len(), 2); |
| 982 | assert_eq!(said(&items[0][..1]), vec!["a"]); |
| 983 | match &items[0][1] { |
| 984 | Block::List { items: inner, .. } => assert_eq!(said(&inner[0]), vec!["inner"]), |
| 985 | other => panic!("expected a nested list, got {:?}", other), |
| 986 | } |
| 987 | } |
| 988 | other => panic!("expected a list, got {:?}", other), |
| 989 | } |
| 990 | // A number and a delimiter make an ordered list. |
| 991 | let b = res!(parse("1. one\n2. two\n")); |
| 992 | match &b[0] { |
| 993 | Block::List { ordered, items } => { |
| 994 | assert!(ordered); |
| 995 | assert_eq!(items.len(), 2); |
| 996 | } |
| 997 | other => panic!("expected a list, got {:?}", other), |
| 998 | } |
| 999 | Ok(()) |
| 1000 | } |
| 1001 | |
| 1002 | /// A header row and a delimiter row make a table, whose colons align the columns. |
| 1003 | #[test] |
| 1004 | fn test_a_pipe_table_08() -> Outcome<()> { |
| 1005 | let b = res!(parse("| w | x | y | z |\n| --- | :-- | :-: | --: |\n| 1 | 2 | 3 | 4 |\n")); |
| 1006 | assert_eq!(b.len(), 1); |
| 1007 | match &b[0] { |
| 1008 | Block::Table { cols, .. } => assert_eq!( |
| 1009 | cols, |
| 1010 | &vec![Align::None, Align::Start, Align::Centre, Align::End], |
| 1011 | ), |
| 1012 | other => panic!("expected a table, got {:?}", other), |
| 1013 | } |
| 1014 | assert_eq!(grid(&b[0]), vec![ |
| 1015 | vec!["w", "x", "y", "z"], |
| 1016 | vec!["1", "2", "3", "4"], |
| 1017 | ]); |
| 1018 | Ok(()) |
| 1019 | } |
| 1020 | |
| 1021 | /// A colon fence names a division, its opening word a class. |
| 1022 | #[test] |
| 1023 | fn test_a_colon_fence_names_a_division_09() -> Outcome<()> { |
| 1024 | let b = res!(parse("::: warning\nMind the step.\n:::\n")); |
| 1025 | assert_eq!(b.len(), 1); |
| 1026 | match &b[0] { |
| 1027 | Block::Div { attrs, content } => { |
| 1028 | assert_eq!(attrs.classes, vec!["warning".to_string()]); |
| 1029 | assert!(attrs.id.is_none()); |
| 1030 | assert_eq!(said(content), vec!["Mind the step."]); |
| 1031 | } |
| 1032 | other => panic!("expected a division, got {:?}", other), |
| 1033 | } |
| 1034 | Ok(()) |
| 1035 | } |
| 1036 | |
| 1037 | /// Divisions nest, one within another. |
| 1038 | #[test] |
| 1039 | fn test_divisions_nest_10() -> Outcome<()> { |
| 1040 | let b = res!(parse("::: outer\nbefore\n::: inner\nwithin\n:::\nafter\n:::\n")); |
| 1041 | assert_eq!(b.len(), 1); |
| 1042 | match &b[0] { |
| 1043 | Block::Div { attrs, content } => { |
| 1044 | assert_eq!(attrs.classes, vec!["outer".to_string()]); |
| 1045 | // A paragraph, then the inner division, then a paragraph. |
| 1046 | assert_eq!(content.len(), 3); |
| 1047 | assert_eq!(said(&content[..1]), vec!["before"]); |
| 1048 | match &content[1] { |
| 1049 | Block::Div { attrs, content } => { |
| 1050 | assert_eq!(attrs.classes, vec!["inner".to_string()]); |
| 1051 | assert_eq!(said(content), vec!["within"]); |
| 1052 | } |
| 1053 | other => panic!("expected an inner division, got {:?}", other), |
| 1054 | } |
| 1055 | assert_eq!(said(&content[2..]), vec!["after"]); |
| 1056 | } |
| 1057 | other => panic!("expected a division, got {:?}", other), |
| 1058 | } |
| 1059 | Ok(()) |
| 1060 | } |
| 1061 | |
| 1062 | /// A full brace group after the colons names the division's id, classes and pairs. |
| 1063 | #[test] |
| 1064 | fn test_a_division_takes_a_brace_group_11() -> Outcome<()> { |
| 1065 | let b = res!(parse("::: {#box .note key=val}\nInside.\n:::\n")); |
| 1066 | match &b[0] { |
| 1067 | Block::Div { attrs, .. } => { |
| 1068 | assert_eq!(attrs.id, Some("box".to_string())); |
| 1069 | assert_eq!(attrs.classes, vec!["note".to_string()]); |
| 1070 | assert_eq!(attrs.pairs, vec![("key".to_string(), "val".to_string())]); |
| 1071 | } |
| 1072 | other => panic!("expected a division, got {:?}", other), |
| 1073 | } |
| 1074 | Ok(()) |
| 1075 | } |
| 1076 | |
| 1077 | /// A standalone attributes line attaches its attributes to the block that follows it. |
| 1078 | #[test] |
| 1079 | fn test_a_standalone_attributes_line_attaches_12() -> Outcome<()> { |
| 1080 | let b = res!(parse("{.note #x}\nA plain paragraph.\n")); |
| 1081 | assert_eq!(b.len(), 1); |
| 1082 | match &b[0] { |
| 1083 | Block::Div { attrs, content } => { |
| 1084 | assert_eq!(attrs.classes, vec!["note".to_string()]); |
| 1085 | assert_eq!(attrs.id, Some("x".to_string())); |
| 1086 | assert_eq!(content.len(), 1); |
| 1087 | assert!(matches!(content[0], Block::Para(_))); |
| 1088 | assert_eq!(said(content), vec!["A plain paragraph."]); |
| 1089 | } |
| 1090 | other => panic!("expected a division, got {:?}", other), |
| 1091 | } |
| 1092 | Ok(()) |
| 1093 | } |
| 1094 | |
| 1095 | /// A standalone attributes line above a division merges into it rather than wrapping it. |
| 1096 | #[test] |
| 1097 | fn test_a_standalone_line_merges_into_a_division_13() -> Outcome<()> { |
| 1098 | let b = res!(parse("{#a}\n::: note\nInside.\n:::\n")); |
| 1099 | assert_eq!(b.len(), 1); |
| 1100 | match &b[0] { |
| 1101 | Block::Div { attrs, content } => { |
| 1102 | // The one box carries both the line's id and the fence's class, not two boxes. |
| 1103 | assert_eq!(attrs.id, Some("a".to_string())); |
| 1104 | assert_eq!(attrs.classes, vec!["note".to_string()]); |
| 1105 | assert_eq!(said(content), vec!["Inside."]); |
| 1106 | } |
| 1107 | other => panic!("expected a division, got {:?}", other), |
| 1108 | } |
| 1109 | Ok(()) |
| 1110 | } |
| 1111 | |
| 1112 | /// A reference definition names a link that stands above it, and is no block of its own. |
| 1113 | #[test] |
| 1114 | fn test_a_reference_definition_resolves_a_link_14() -> Outcome<()> { |
| 1115 | let b = res!(parse("See [the site][ref].\n\n[ref]: https://a.b\n")); |
| 1116 | // The definition line is not a block, so only the paragraph remains. |
| 1117 | assert_eq!(b.len(), 1); |
| 1118 | match &b[0] { |
| 1119 | Block::Para(content) => assert_eq!(content, &vec![ |
| 1120 | Inline::Text("See ".into()), |
| 1121 | Inline::Link { to: "https://a.b".into(), content: vec![Inline::Text("the site".into())] }, |
| 1122 | Inline::Text(".".into()), |
| 1123 | ]), |
| 1124 | other => panic!("expected a paragraph, got {:?}", other), |
| 1125 | } |
| 1126 | Ok(()) |
| 1127 | } |
| 1128 | |
| 1129 | /// A backslash escape makes a literal of the punctuation it stands before. |
| 1130 | #[test] |
| 1131 | fn test_a_backslash_escape_15() -> Outcome<()> { |
| 1132 | let b = res!(parse("A \\*literal\\* star.\n")); |
| 1133 | assert_eq!(b, vec![Block::Para(vec![Inline::Text("A *literal* star.".into())])]); |
| 1134 | Ok(()) |
| 1135 | } |
| 1136 | |
| 1137 | /// Syntax the reader does not yet read survives as the text it is, without crashing. |
| 1138 | #[test] |
| 1139 | fn test_deferred_syntax_survives_16() -> Outcome<()> { |
| 1140 | // Display maths, a definition list and a task list are all not yet read. |
| 1141 | let b = res!(parse("$$x^2$$\n")); |
| 1142 | assert_eq!(said(&b), vec!["$$x^2$$"]); |
| 1143 | let b = res!(parse("A smile :smile: and a note[^1].\n")); |
| 1144 | assert_eq!(said(&b), vec!["A smile :smile: and a note[^1]."]); |
| 1145 | Ok(()) |
| 1146 | } |
| 1147 | |
| 1148 | /// Nesting past the limit is refused, which is the parser's one refusal. |
| 1149 | #[test] |
| 1150 | fn test_nesting_past_the_limit_is_refused_17() -> Outcome<()> { |
| 1151 | let ok = format!("{} deep\n", ">".repeat(DEPTH_LIMIT)); |
| 1152 | assert!(parse(&ok).is_ok()); |
| 1153 | let deep = format!("{} deep\n", ">".repeat(DEPTH_LIMIT + 8)); |
| 1154 | assert!(parse(&deep).is_err()); |
| 1155 | Ok(()) |
| 1156 | } |
| 1157 | |
| 1158 | /// An empty document holds nothing, and is not a failure. |
| 1159 | #[test] |
| 1160 | fn test_an_empty_document_holds_nothing_18() -> Outcome<()> { |
| 1161 | assert_eq!(res!(parse("")), Vec::<Block>::new()); |
| 1162 | assert_eq!(res!(parse("\n\n \n")), Vec::<Block>::new()); |
| 1163 | Ok(()) |
| 1164 | } |
| 1165 | |
| 1166 | /// A whole document of every block reads as the document it is. |
| 1167 | #[test] |
| 1168 | fn test_a_document_of_every_block_19() -> Outcome<()> { |
| 1169 | let src = "\ |
| 1170 | # Title |
| 1171 | |
| 1172 | An opening paragraph with _emphasis_ and *strength*. |
| 1173 | |
| 1174 | ::: note |
| 1175 | A boxed aside. |
| 1176 | ::: |
| 1177 | |
| 1178 | - one |
| 1179 | - two |
| 1180 | - nested |
| 1181 | |
| 1182 | > A quotation. |
| 1183 | |
| 1184 | ```sh |
| 1185 | echo hi |
| 1186 | ``` |
| 1187 | |
| 1188 | | a | b | |
| 1189 | | --- | --- | |
| 1190 | | 1 | 2 | |
| 1191 | |
| 1192 | --- |
| 1193 | |
| 1194 | The end. |
| 1195 | "; |
| 1196 | let b = res!(parse(src)); |
| 1197 | assert!(matches!(b[0], Block::Heading { level: 1, .. })); |
| 1198 | assert!(matches!(b[1], Block::Para(_))); |
| 1199 | assert!(matches!(b[2], Block::Div { .. })); |
| 1200 | assert!(matches!(b[3], Block::List { ordered: false, .. })); |
| 1201 | assert!(matches!(b[4], Block::Quote(_))); |
| 1202 | assert!(matches!(b[5], Block::Code { .. })); |
| 1203 | assert!(matches!(b[6], Block::Table { .. })); |
| 1204 | assert!(matches!(b[7], Block::Rule)); |
| 1205 | assert!(matches!(b[8], Block::Para(_))); |
| 1206 | assert_eq!(b.len(), 9); |
| 1207 | Ok(()) |
| 1208 | } |
| 1209 | } |