oxedyne/fe2o3/fe2o3_text/src/doc/markdown/block.rs
43.0 KiB, 31 runs
created by r1870400018:14262, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Block structure: the pass that divides Markdown text into headings, paragraphs, lists, code, |
| 2 | //! quotations and breaks. |
| 3 | //! |
| 4 | //! Markdown is read in two passes, because its two levels are decided by different things. A block is |
| 5 | //! decided by how a line *starts* and by the blank lines around it; an inline run is decided by |
| 6 | //! delimiters within a line. This module does the first pass and hands each block's text to |
| 7 | //! [`crate::doc::markdown::inline`] for the second. |
| 8 | |
| 9 | use crate::doc::{ |
| 10 | Align, |
| 11 | Block, |
| 12 | Cell, |
| 13 | Row, |
| 14 | markdown::inline, |
| 15 | }; |
| 16 | |
| 17 | use oxedyne_fe2o3_core::prelude::*; |
| 18 | |
| 19 | /// How deep a document may nest its blocks before the parser refuses it. |
| 20 | /// |
| 21 | /// A quotation inside a list inside a quotation is legitimate; a thousand of them is a document built |
| 22 | /// to exhaust the stack of whatever reads it. The limit is generous beside anything an author writes |
| 23 | /// and far below what would trouble the machine. |
| 24 | pub const DEPTH_LIMIT: usize = 32; |
| 25 | |
| 26 | /// The columns a tab advances by, counting from a tab stop. |
| 27 | const TAB: usize = 4; |
| 28 | |
| 29 | /// The column at which a line is code by its indentation alone. |
| 30 | const CODE_COL: usize = 4; |
| 31 | |
| 32 | /// Divides Markdown text into its blocks. |
| 33 | pub fn parse(src: &str) -> Outcome<Vec<Block>> { |
| 34 | let lines: Vec<&str> = src.lines().collect(); |
| 35 | blocks(&lines, 0) |
| 36 | } |
| 37 | |
| 38 | /// What a line begins, judged by how it starts. |
| 39 | /// |
| 40 | /// A line's opening is nearly all a block parser needs: the two things it does not settle are a |
| 41 | /// setext underline, which only means anything under a paragraph (see [`under_of`]), and lazy |
| 42 | /// continuation, which the reader of each container decides. |
| 43 | enum Start { |
| 44 | /// Nothing but whitespace. |
| 45 | Blank, |
| 46 | /// An ATX heading: its level, and its text. |
| 47 | Atx(u8, String), |
| 48 | /// A code fence. |
| 49 | Fence { |
| 50 | /// The character the fence is made of. |
| 51 | ch: u8, |
| 52 | /// How many characters the fence runs to. |
| 53 | len: usize, |
| 54 | /// The column the fence starts at, which its content is stripped back to. |
| 55 | ind: usize, |
| 56 | /// The language the info string named, if it named one. |
| 57 | lang: Option<String>, |
| 58 | }, |
| 59 | /// A thematic break. |
| 60 | Rule, |
| 61 | /// A block quotation marker. |
| 62 | Quote, |
| 63 | /// A list item marker. |
| 64 | Item { |
| 65 | /// Whether the marker is a number. |
| 66 | ord: bool, |
| 67 | /// The bullet, or the delimiter that follows a number. |
| 68 | mark: u8, |
| 69 | /// The column the item's content starts at. |
| 70 | col: usize, |
| 71 | /// The number an ordered marker gave, if it gave one. |
| 72 | num: Option<u64>, |
| 73 | /// Whether the marker line carries no content of its own. |
| 74 | bare: bool, |
| 75 | /// The item's first line, with the marker taken off. |
| 76 | head: String, |
| 77 | }, |
| 78 | /// Code marked by its indentation. |
| 79 | Code, |
| 80 | /// Anything else, which is paragraph text. |
| 81 | Text, |
| 82 | } |
| 83 | |
| 84 | /// Reads a run of lines into the blocks they are, at the given nesting depth. |
| 85 | fn blocks(lines: &[&str], depth: usize) -> Outcome<Vec<Block>> { |
| 86 | if depth > DEPTH_LIMIT { |
| 87 | return Err(err!( |
| 88 | "Markdown blocks nest more than {} deep, which no document written to be read \ |
| 89 | does.", DEPTH_LIMIT; |
| 90 | Excessive, Input)); |
| 91 | } |
| 92 | let mut out = Vec::new(); |
| 93 | let mut i = 0; |
| 94 | while i < lines.len() { |
| 95 | match classify(lines[i]) { |
| 96 | Start::Blank => i += 1, |
| 97 | Start::Rule => { |
| 98 | out.push(Block::Rule); |
| 99 | i += 1; |
| 100 | } |
| 101 | Start::Atx(level, text) => { |
| 102 | out.push(Block::Heading { |
| 103 | level, |
| 104 | content: res!(inline::parse(&text)), |
| 105 | }); |
| 106 | i += 1; |
| 107 | } |
| 108 | Start::Fence { ch, len, ind, lang } => { |
| 109 | let (b, n) = code_fenced(&lines[i..], ch, len, ind, lang); |
| 110 | out.push(b); |
| 111 | i += n; |
| 112 | } |
| 113 | Start::Code => { |
| 114 | let (b, n) = code_indented(&lines[i..]); |
| 115 | out.push(b); |
| 116 | i += n; |
| 117 | } |
| 118 | Start::Quote => { |
| 119 | let (b, n) = res!(quote(&lines[i..], depth)); |
| 120 | out.push(b); |
| 121 | i += n; |
| 122 | } |
| 123 | Start::Item { .. } => { |
| 124 | let (b, n) = res!(list(&lines[i..], depth)); |
| 125 | out.push(b); |
| 126 | i += n; |
| 127 | } |
| 128 | Start::Text => { |
| 129 | let (b, n) = res!(para(&lines[i..])); |
| 130 | out.push(b); |
| 131 | i += n; |
| 132 | } |
| 133 | } |
| 134 | } |
| 135 | Ok(out) |
| 136 | } |
| 137 | |
| 138 | // ── Classification ─────────────────────────────────────────────── |
| 139 | |
| 140 | /// Judges what a line begins. |
| 141 | fn classify(line: &str) -> Start { |
| 142 | if is_blank(line) { |
| 143 | return Start::Blank; |
| 144 | } |
| 145 | let ind = indent_of(line); |
| 146 | if ind >= CODE_COL { |
| 147 | return Start::Code; |
| 148 | } |
| 149 | let off = ws_end(line); |
| 150 | let rest = &line[off..]; |
| 151 | |
| 152 | // A thematic break is judged first, so that `***` is a break and not a list of nothing. |
| 153 | if is_rule(rest) { |
| 154 | return Start::Rule; |
| 155 | } |
| 156 | |
| 157 | // An ATX heading: one to six hashes, set apart from the text by a space. |
| 158 | if rest.starts_with('#') { |
| 159 | let n = rest.bytes().take_while(|c| *c == b'#').count(); |
| 160 | let after = &rest[n..]; |
| 161 | if n <= 6 && (after.is_empty() || after.starts_with(' ') || after.starts_with('\t')) { |
| 162 | return Start::Atx(n as u8, atx_text(after)); |
| 163 | } |
| 164 | } |
| 165 | |
| 166 | // A code fence: three or more backticks or tildes, and what they say about the code. |
| 167 | let b = rest.as_bytes(); |
| 168 | if b[0] == b'`' || b[0] == b'~' { |
| 169 | let ch = b[0]; |
| 170 | let len = rest.bytes().take_while(|c| *c == ch).count(); |
| 171 | let info = rest[len..].trim(); |
| 172 | // A backtick fence's info string may hold no backtick, or `a ` b` would open one. |
| 173 | if len >= 3 && (ch != b'`' || !info.contains('`')) { |
| 174 | let lang = match info.split_whitespace().next() { |
| 175 | Some(w) => Some(w.to_string()), |
| 176 | None => None, |
| 177 | }; |
| 178 | return Start::Fence { ch, len, ind, lang }; |
| 179 | } |
| 180 | } |
| 181 | |
| 182 | if b[0] == b'>' { |
| 183 | return Start::Quote; |
| 184 | } |
| 185 | |
| 186 | match item_start(line, off, ind) { |
| 187 | Some(s) => s, |
| 188 | None => Start::Text, |
| 189 | } |
| 190 | } |
| 191 | |
| 192 | /// Recognises a list item marker, and what it says about the item it begins. |
| 193 | fn item_start(line: &str, off: usize, ind: usize) -> Option<Start> { |
| 194 | let b = line.as_bytes(); |
| 195 | let mut p = off; // Byte offset. |
| 196 | let mut c = ind; // Column. |
| 197 | let mut ord = false; |
| 198 | let mut num = None; |
| 199 | let mark; |
| 200 | match b[p] { |
| 201 | b'-' | b'*' | b'+' => { |
| 202 | mark = b[p]; |
| 203 | p += 1; |
| 204 | c += 1; |
| 205 | } |
| 206 | d if d.is_ascii_digit() => { |
| 207 | let s = p; |
| 208 | // Nine digits is as long a number as a list may count to. |
| 209 | while p < b.len() && b[p].is_ascii_digit() && p - s < 9 { |
| 210 | p += 1; |
| 211 | } |
| 212 | if p >= b.len() || (b[p] != b'.' && b[p] != b')') { |
| 213 | return None; |
| 214 | } |
| 215 | num = line[s..p].parse::<u64>().ok(); |
| 216 | mark = b[p]; |
| 217 | ord = true; |
| 218 | c += p - s + 1; |
| 219 | p += 1; |
| 220 | } |
| 221 | _ => return None, |
| 222 | } |
| 223 | let rest = &line[p..]; |
| 224 | // A marker with nothing after it is an empty item. |
| 225 | if rest.is_empty() || is_blank(rest) { |
| 226 | return Some(Start::Item { |
| 227 | ord, |
| 228 | mark, |
| 229 | col: c + 1, |
| 230 | num, |
| 231 | bare: true, |
| 232 | head: String::new(), |
| 233 | }); |
| 234 | } |
| 235 | let sp = indent_of(rest); // Columns of whitespace between marker and content. |
| 236 | if sp == 0 { |
| 237 | return None; // `-foo` is a word beginning with a dash, not a list. |
| 238 | } |
| 239 | // Whitespace past the fourth column is code within the item, not part of the marker. |
| 240 | let take = if sp > CODE_COL { 1 } else { sp }; |
| 241 | Some(Start::Item { |
| 242 | ord, |
| 243 | mark, |
| 244 | col: c + take, |
| 245 | num, |
| 246 | bare: false, |
| 247 | head: strip_cols(rest, take), |
| 248 | }) |
| 249 | } |
| 250 | |
| 251 | /// Whether the line, past its indent, is a thematic break. |
| 252 | fn is_rule(rest: &str) -> bool { |
| 253 | let b = rest.as_bytes(); |
| 254 | if b.is_empty() { |
| 255 | return false; |
| 256 | } |
| 257 | let ch = b[0]; |
| 258 | if ch != b'-' && ch != b'*' && ch != b'_' { |
| 259 | return false; |
| 260 | } |
| 261 | let mut n = 0; |
| 262 | for &c in b { |
| 263 | if c == ch { |
| 264 | n += 1; |
| 265 | } else if c != b' ' && c != b'\t' { |
| 266 | return false; |
| 267 | } |
| 268 | } |
| 269 | n >= 3 |
| 270 | } |
| 271 | |
| 272 | /// The heading level a setext underline gives, if the line is one. |
| 273 | fn under_of(line: &str) -> Option<u8> { |
| 274 | if indent_of(line) >= CODE_COL { |
| 275 | return None; |
| 276 | } |
| 277 | let t = line.trim_matches(|c| c == ' ' || c == '\t'); |
| 278 | if t.is_empty() { |
| 279 | return None; |
| 280 | } |
| 281 | let ch = t.as_bytes()[0]; |
| 282 | if (ch != b'=' && ch != b'-') || !t.bytes().all(|c| c == ch) { |
| 283 | return None; |
| 284 | } |
| 285 | Some(if ch == b'=' { 1 } else { 2 }) |
| 286 | } |
| 287 | |
| 288 | /// The alignments a delimiter row gives, if the line is one and divides into as many cells as the |
| 289 | /// header above it did. |
| 290 | /// |
| 291 | /// The matching count is the whole of the test, and it is what tells a table from a paragraph that |
| 292 | /// happens to hold pipes and dashes. A row of `---` under a header of three columns is not a table |
| 293 | /// with two columns missing, and it is not a table the parser should repair: it is prose, and prose |
| 294 | /// is what it is read as. Nothing here guesses at a row the author did not write. |
| 295 | fn delim_of(line: &str, n: usize) -> Option<Vec<Align>> { |
| 296 | // Indentation past the fourth column is code, as it is under a setext underline. |
| 297 | if indent_of(line) >= CODE_COL { |
| 298 | return None; |
| 299 | } |
| 300 | let cells = cells_of(line); |
| 301 | if cells.is_empty() || cells.len() != n { |
| 302 | return None; |
| 303 | } |
| 304 | let mut cols = Vec::with_capacity(cells.len()); |
| 305 | for cell in &cells { |
| 306 | match align_of(cell) { |
| 307 | Some(align) => cols.push(align), |
| 308 | None => return None, |
| 309 | } |
| 310 | } |
| 311 | Some(cols) |
| 312 | } |
| 313 | |
| 314 | /// The alignment one delimiter cell gives: a colon at the side the column is aligned to, and dashes |
| 315 | /// between. `None` where the cell is no delimiter cell at all, which is what makes the row no |
| 316 | /// delimiter row. |
| 317 | fn align_of(cell: &str) -> Option<Align> { |
| 318 | let b = cell.as_bytes(); |
| 319 | if b.is_empty() { |
| 320 | return None; |
| 321 | } |
| 322 | let start = b[0] == b':'; |
| 323 | let end = b[b.len() - 1] == b':'; |
| 324 | let s = if start { 1 } else { 0 }; // Where the dashes begin. |
| 325 | let e = b.len() - if end { 1 } else { 0 }; // Where they end. |
| 326 | // A cell of colons alone marks nothing out, and a cell of anything but dashes between them is |
| 327 | // not a delimiter cell. |
| 328 | if s >= e || !b[s..e].iter().all(|c| *c == b'-') { |
| 329 | return None; |
| 330 | } |
| 331 | Some(match (start, end) { |
| 332 | (true, true) => Align::Centre, |
| 333 | (true, false) => Align::Start, |
| 334 | (false, true) => Align::End, |
| 335 | (false, false) => Align::None, |
| 336 | }) |
| 337 | } |
| 338 | |
| 339 | /// The cells a table row divides into, split on every pipe the author did not escape. |
| 340 | /// |
| 341 | /// A pipe at either end of the row is the row's own edge and not an empty cell, so an author may draw |
| 342 | /// the edges or leave them off. A `\|` divides nothing: it stays in the cell for [`inline::parse`] to |
| 343 | /// resolve as the escape it is, which is where every other escape is resolved. |
| 344 | /// |
| 345 | /// The split happens before the inline pass, so a pipe within a code span divides a cell like any |
| 346 | /// other, and `` `a|b` `` is two cells. That is GFM's wart and it is kept deliberately: what a row |
| 347 | /// divides into is settled by the pipes on the line and by nothing that must be parsed to be found, |
| 348 | /// which means an author can see the grid in the source. Reading the span first would make the count |
| 349 | /// of a row's cells depend on the inline pass, and a header and a delimiter row could then disagree |
| 350 | /// for reasons neither line shows. |
| 351 | fn cells_of(line: &str) -> Vec<&str> { |
| 352 | let t = trim(line); |
| 353 | let b = t.as_bytes(); |
| 354 | let mut out = Vec::new(); |
| 355 | // A pipe opening the row is its edge, and the first cell begins after it. |
| 356 | let mut i = if b.first() == Some(&b'|') { 1 } else { 0 }; |
| 357 | let mut s = i; // Where the cell being read begins. |
| 358 | while i < b.len() { |
| 359 | match b[i] { |
| 360 | b'\\' => i += 2, // A backslash and whatever it escapes divide nothing. |
| 361 | b'|' => { |
| 362 | out.push(trim(&t[s..i])); |
| 363 | i += 1; |
| 364 | s = i; |
| 365 | } |
| 366 | _ => i += 1, |
| 367 | } |
| 368 | } |
| 369 | // A pipe closing the row is its edge as well, and what follows it is no cell. |
| 370 | if s < t.len() { |
| 371 | out.push(trim(&t[s..])); |
| 372 | } |
| 373 | out |
| 374 | } |
| 375 | |
| 376 | /// A heading's text, with the closing run of hashes an author may have balanced it with taken off. |
| 377 | fn atx_text(after: &str) -> String { |
| 378 | let t = after.trim_matches(|c| c == ' ' || c == '\t'); |
| 379 | let n = t.bytes().rev().take_while(|c| *c == b'#').count(); |
| 380 | if n == 0 { |
| 381 | return t.to_string(); |
| 382 | } |
| 383 | let head = &t[..t.len() - n]; |
| 384 | // The closing run counts only when a space sets it apart, or when it is all there is. |
| 385 | if head.is_empty() || head.ends_with(' ') || head.ends_with('\t') { |
| 386 | head.trim_end_matches(|c| c == ' ' || c == '\t').to_string() |
| 387 | } else { |
| 388 | t.to_string() |
| 389 | } |
| 390 | } |
| 391 | |
| 392 | /// Whether a line begins something that ends the paragraph above it. |
| 393 | fn interrupts(line: &str) -> bool { |
| 394 | match classify(line) { |
| 395 | Start::Blank |
| 396 | | Start::Atx(..) |
| 397 | | Start::Fence { .. } |
| 398 | | Start::Rule |
| 399 | | Start::Quote => true, |
| 400 | // A list ends a paragraph only where it plainly begins one, so that a year opening a |
| 401 | // sentence, or a full stop wrapping onto its own line, is prose and not a list. |
| 402 | Start::Item { ord, num, bare, .. } => !bare && (!ord || num == Some(1)), |
| 403 | Start::Code |
| 404 | | Start::Text => false, |
| 405 | } |
| 406 | } |
| 407 | |
| 408 | // ── The blocks themselves ──────────────────────────────────────── |
| 409 | |
| 410 | /// Reads a paragraph, the heading a setext underline turns it into, or the table a delimiter row |
| 411 | /// turns its last line into. |
| 412 | fn para(lines: &[&str]) -> Outcome<(Block, usize)> { |
| 413 | let mut acc = vec![lines[0].trim_start_matches(|c| c == ' ' || c == '\t')]; |
| 414 | let mut i = 1; |
| 415 | while i < lines.len() { |
| 416 | // An underline under a paragraph is a heading, which is how `---` means a heading here |
| 417 | // and a thematic break anywhere else. It is judged before a delimiter row is, so that a |
| 418 | // line of dashes under a line of prose stays the heading it has always been. |
| 419 | if let Some(level) = under_of(lines[i]) { |
| 420 | return Ok((Block::Heading { |
| 421 | level, |
| 422 | content: res!(inline::parse(acc.join("\n").trim_end())), |
| 423 | }, i + 1)); |
| 424 | } |
| 425 | // A table's header row is a paragraph line until the delimiter row beneath it says |
| 426 | // otherwise, which is the ambiguity a setext underline has, one line lower down. |
| 427 | if let Some(cols) = delim_of(lines[i], cells_of(acc[i - 1]).len()) { |
| 428 | // The lines above the header are a paragraph of their own, and they end here. The |
| 429 | // header goes back to the document to be read again, as the table's first line. |
| 430 | if i > 1 { |
| 431 | acc.pop(); |
| 432 | return Ok(( |
| 433 | Block::Para(res!(inline::parse(acc.join("\n").trim_end()))), |
| 434 | i - 1, |
| 435 | )); |
| 436 | } |
| 437 | return table(lines, cols); |
| 438 | } |
| 439 | if interrupts(lines[i]) { |
| 440 | break; |
| 441 | } |
| 442 | acc.push(lines[i].trim_start_matches(|c| c == ' ' || c == '\t')); |
| 443 | i += 1; |
| 444 | } |
| 445 | Ok((Block::Para(res!(inline::parse(acc.join("\n").trim_end()))), i)) |
| 446 | } |
| 447 | |
| 448 | /// Reads a table: the header row, the delimiter row that made it one, and the body beneath them. |
| 449 | /// |
| 450 | /// The header is `lines[0]` and the delimiter row `lines[1]`, which is what [`delim_of`] has just |
| 451 | /// established of them, and the columns are what it read there. |
| 452 | fn table(lines: &[&str], cols: Vec<Align>) -> Outcome<(Block, usize)> { |
| 453 | let head = res!(row(lines[0], cols.len())); |
| 454 | let mut rows = Vec::new(); |
| 455 | let mut i = 2; |
| 456 | // The table runs to the first line that begins a block of its own, a blank line among them. A |
| 457 | // line of plain text is a row however little it looks like one, since a row need draw no pipes. |
| 458 | while i < lines.len() && !interrupts(lines[i]) { |
| 459 | rows.push(res!(row(lines[i], cols.len()))); |
| 460 | i += 1; |
| 461 | } |
| 462 | Ok((Block::Table { head: Some(head), rows, cols }, i)) |
| 463 | } |
| 464 | |
| 465 | /// Reads one row of a table, held to the width the header set. |
| 466 | fn row(line: &str, n: usize) -> Outcome<Row> { |
| 467 | let mut cells = Vec::with_capacity(n); |
| 468 | for text in cells_of(line).iter().take(n) { |
| 469 | cells.push(Cell(res!(inline::parse(text)))); |
| 470 | } |
| 471 | // A row of fewer cells than the header named columns is short of the grid rather than wrong, and |
| 472 | // the cells it did not write are empty. A row of more has said something the header made no |
| 473 | // column for, and what it has no column for is dropped. |
| 474 | while cells.len() < n { |
| 475 | cells.push(Cell(Vec::new())); |
| 476 | } |
| 477 | Ok(Row(cells)) |
| 478 | } |
| 479 | |
| 480 | /// Reads a fenced code block, which runs to its closing fence or to the end of the input. |
| 481 | fn code_fenced(lines: &[&str], ch: u8, len: usize, ind: usize, lang: Option<String>) |
| 482 | -> (Block, usize) |
| 483 | { |
| 484 | let mut text = String::new(); |
| 485 | let mut i = 1; |
| 486 | while i < lines.len() { |
| 487 | if fence_closes(lines[i], ch, len) { |
| 488 | return (Block::Code { lang, text }, i + 1); |
| 489 | } |
| 490 | text.push_str(&strip_cols(lines[i], ind)); |
| 491 | text.push('\n'); |
| 492 | i += 1; |
| 493 | } |
| 494 | (Block::Code { lang, text }, i) |
| 495 | } |
| 496 | |
| 497 | /// Whether the line closes a fence of the given character and length. |
| 498 | fn fence_closes(line: &str, ch: u8, len: usize) -> bool { |
| 499 | if indent_of(line) >= CODE_COL { |
| 500 | return false; |
| 501 | } |
| 502 | let rest = &line[ws_end(line)..]; |
| 503 | let n = rest.bytes().take_while(|c| *c == ch).count(); |
| 504 | // A closing fence is at least as long as the one it closes, and says nothing else. |
| 505 | n >= len && is_blank(&rest[n..]) |
| 506 | } |
| 507 | |
| 508 | /// Reads a run of code marked by its indentation. |
| 509 | fn code_indented(lines: &[&str]) -> (Block, usize) { |
| 510 | let mut out: Vec<String> = Vec::new(); |
| 511 | let mut pend: Vec<String> = Vec::new(); // Blank lines held back, in case the code ends here. |
| 512 | let mut i = 0; |
| 513 | while i < lines.len() { |
| 514 | if is_blank(lines[i]) { |
| 515 | pend.push(strip_cols(lines[i], CODE_COL)); |
| 516 | i += 1; |
| 517 | continue; |
| 518 | } |
| 519 | if indent_of(lines[i]) < CODE_COL { |
| 520 | break; |
| 521 | } |
| 522 | out.append(&mut pend); |
| 523 | out.push(strip_cols(lines[i], CODE_COL)); |
| 524 | i += 1; |
| 525 | } |
| 526 | let mut text = String::new(); |
| 527 | for l in &out { |
| 528 | text.push_str(l); |
| 529 | text.push('\n'); |
| 530 | } |
| 531 | // The blank lines that trail the code belong to whatever follows it. |
| 532 | (Block::Code { lang: None, text }, i - pend.len()) |
| 533 | } |
| 534 | |
| 535 | /// Reads a block quotation and the blocks within it. |
| 536 | fn quote(lines: &[&str], depth: usize) -> Outcome<(Block, usize)> { |
| 537 | let mut body: Vec<String> = Vec::new(); |
| 538 | let mut i = 0; |
| 539 | while i < lines.len() { |
| 540 | if let Some(rest) = quote_strip(lines[i]) { |
| 541 | body.push(rest); |
| 542 | i += 1; |
| 543 | continue; |
| 544 | } |
| 545 | if is_blank(lines[i]) || !lazy(&body, lines[i]) { |
| 546 | break; |
| 547 | } |
| 548 | body.push(lines[i].trim_start_matches(|c| c == ' ' || c == '\t').to_string()); |
| 549 | i += 1; |
| 550 | } |
| 551 | let refs: Vec<&str> = body.iter().map(|s| s.as_str()).collect(); |
| 552 | Ok((Block::Quote(res!(blocks(&refs, depth + 1))), i)) |
| 553 | } |
| 554 | |
| 555 | /// The line with its quotation marker taken off, if it carries one. |
| 556 | fn quote_strip(line: &str) -> Option<String> { |
| 557 | if indent_of(line) >= CODE_COL { |
| 558 | return None; |
| 559 | } |
| 560 | match line[ws_end(line)..].strip_prefix('>') { |
| 561 | // One space after the marker is the marker's own, and no more. |
| 562 | Some(rest) => Some(strip_cols(rest, 1)), |
| 563 | None => None, |
| 564 | } |
| 565 | } |
| 566 | |
| 567 | /// Reads a list: its first item, and every item that follows it at the same level with the same |
| 568 | /// marker. |
| 569 | fn list(lines: &[&str], depth: usize) -> Outcome<(Block, usize)> { |
| 570 | let (ord, mark) = match classify(lines[0]) { |
| 571 | Start::Item { ord, mark, .. } => (ord, mark), |
| 572 | _ => return Err(err!( |
| 573 | "A list was read from a line that does not begin one."; Bug)), |
| 574 | }; |
| 575 | let mut items = Vec::new(); |
| 576 | let mut i = 0; |
| 577 | loop { |
| 578 | // Blank lines may sit between items, but they are the list's only if an item follows: a |
| 579 | // blank line before a paragraph ends the list and belongs to the document. |
| 580 | let mut j = i; |
| 581 | while j < lines.len() && is_blank(lines[j]) { |
| 582 | j += 1; |
| 583 | } |
| 584 | if j >= lines.len() { |
| 585 | break; |
| 586 | } |
| 587 | // A different marker begins a different list, which is how an author separates two. |
| 588 | let (col, head) = match classify(lines[j]) { |
| 589 | Start::Item { ord: o, mark: m, col, head, .. } if o == ord && m == mark |
| 590 | => (col, head), |
| 591 | _ => break, |
| 592 | }; |
| 593 | let (body, n) = item_body(&lines[j..], head, col); |
| 594 | let refs: Vec<&str> = body.iter().map(|s| s.as_str()).collect(); |
| 595 | items.push(res!(blocks(&refs, depth + 1))); |
| 596 | i = j + n; |
| 597 | } |
| 598 | Ok((Block::List { ordered: ord, items }, i)) |
| 599 | } |
| 600 | |
| 601 | /// Reads one list item's lines, with the marker and the indentation that stands for it taken off. |
| 602 | fn item_body(lines: &[&str], head: String, col: usize) -> (Vec<String>, usize) { |
| 603 | let mut body = vec![head]; |
| 604 | let mut i = 1; |
| 605 | while i < lines.len() { |
| 606 | if is_blank(lines[i]) { |
| 607 | // A blank line is the item's only if indented content follows it. |
| 608 | let mut k = i; |
| 609 | while k < lines.len() && is_blank(lines[k]) { |
| 610 | k += 1; |
| 611 | } |
| 612 | if k >= lines.len() || indent_of(lines[k]) < col { |
| 613 | break; |
| 614 | } |
| 615 | for _ in i..k { |
| 616 | body.push(String::new()); |
| 617 | } |
| 618 | i = k; |
| 619 | continue; |
| 620 | } |
| 621 | if indent_of(lines[i]) >= col { |
| 622 | body.push(strip_cols(lines[i], col)); |
| 623 | i += 1; |
| 624 | continue; |
| 625 | } |
| 626 | if !lazy(&body, lines[i]) { |
| 627 | break; |
| 628 | } |
| 629 | body.push(lines[i].trim_start_matches(|c| c == ' ' || c == '\t').to_string()); |
| 630 | i += 1; |
| 631 | } |
| 632 | (body, i) |
| 633 | } |
| 634 | |
| 635 | /// Whether an under-indented line carries on the paragraph a container was in the middle of. |
| 636 | /// |
| 637 | /// This is Markdown's laziness: an author who wraps a quoted or listed paragraph need not mark every |
| 638 | /// line of it, so a line that is only text goes on belonging to the paragraph above. |
| 639 | fn lazy(body: &[String], line: &str) -> bool { |
| 640 | match body.last() { |
| 641 | Some(l) if !is_blank(l) => matches!(classify(line), Start::Text | Start::Code), |
| 642 | _ => false, |
| 643 | } |
| 644 | } |
| 645 | |
| 646 | // ── Whitespace ─────────────────────────────────────────────────── |
| 647 | |
| 648 | /// The text with the spaces and tabs at either end of it taken off. |
| 649 | fn trim(s: &str) -> &str { |
| 650 | s.trim_matches(|c| c == ' ' || c == '\t') |
| 651 | } |
| 652 | |
| 653 | /// Whether the line holds nothing but whitespace. |
| 654 | fn is_blank(line: &str) -> bool { |
| 655 | line.bytes().all(|b| b == b' ' || b == b'\t') |
| 656 | } |
| 657 | |
| 658 | /// The column the line's first non-whitespace character sits at, a tab advancing to the next tab |
| 659 | /// stop. |
| 660 | fn indent_of(line: &str) -> usize { |
| 661 | let mut c = 0; |
| 662 | for b in line.bytes() { |
| 663 | match b { |
| 664 | b' ' => c += 1, |
| 665 | b'\t' => c += TAB - (c % TAB), |
| 666 | _ => break, |
| 667 | } |
| 668 | } |
| 669 | c |
| 670 | } |
| 671 | |
| 672 | /// The byte offset of the line's first non-whitespace character. |
| 673 | fn ws_end(line: &str) -> usize { |
| 674 | let mut i = 0; |
| 675 | for b in line.bytes() { |
| 676 | if b != b' ' && b != b'\t' { |
| 677 | break; |
| 678 | } |
| 679 | i += 1; |
| 680 | } |
| 681 | i |
| 682 | } |
| 683 | |
| 684 | /// The line with `n` columns of leading whitespace taken off. |
| 685 | /// |
| 686 | /// A tab that straddles the cut gives up what it spans past it as spaces, which is the only way to |
| 687 | /// take a fixed number of columns from a line a tab has indented. |
| 688 | fn strip_cols(line: &str, n: usize) -> String { |
| 689 | let mut c = 0; // Column. |
| 690 | let mut i = 0; // Byte offset. |
| 691 | for b in line.bytes() { |
| 692 | if c >= n { |
| 693 | break; |
| 694 | } |
| 695 | match b { |
| 696 | b' ' => { |
| 697 | c += 1; |
| 698 | i += 1; |
| 699 | } |
| 700 | b'\t' => { |
| 701 | let next = c + TAB - (c % TAB); |
| 702 | if next > n { |
| 703 | let mut s = " ".repeat(next - n); |
| 704 | s.push_str(&line[i + 1..]); |
| 705 | return s; |
| 706 | } |
| 707 | c = next; |
| 708 | i += 1; |
| 709 | } |
| 710 | _ => break, |
| 711 | } |
| 712 | } |
| 713 | line[i..].to_string() |
| 714 | } |
| 715 | |
| 716 | #[cfg(test)] |
| 717 | mod tests { |
| 718 | use super::*; |
| 719 | use crate::doc::Inline; |
| 720 | |
| 721 | /// The text of a block's inlines, for tests that care what a block says and not how. |
| 722 | fn said(blocks: &[Block]) -> Vec<String> { |
| 723 | blocks.iter().map(|b| match b { |
| 724 | Block::Para(c) => crate::doc::text_of(c), |
| 725 | Block::Heading { content, .. } => crate::doc::text_of(content), |
| 726 | Block::Code { text, .. } => text.clone(), |
| 727 | _ => String::new(), |
| 728 | }).collect() |
| 729 | } |
| 730 | |
| 731 | /// What a table's rows say, cell by cell, the header first: for tests that care what a table holds |
| 732 | /// and not how it was marked up. |
| 733 | fn grid(b: &Block) -> Vec<Vec<String>> { |
| 734 | match b { |
| 735 | Block::Table { head, rows, .. } => { |
| 736 | let mut out = Vec::new(); |
| 737 | if let Some(head) = head { |
| 738 | out.push(head.0.iter().map(|c| c.text_of()).collect()); |
| 739 | } |
| 740 | for row in rows { |
| 741 | out.push(row.0.iter().map(|c| c.text_of()).collect()); |
| 742 | } |
| 743 | out |
| 744 | } |
| 745 | other => panic!("expected a table, got {:?}", other), |
| 746 | } |
| 747 | } |
| 748 | |
| 749 | /// A hash and a space open a heading, and the hashes count its level. |
| 750 | #[test] |
| 751 | fn test_a_run_of_hashes_opens_a_heading_00() -> Outcome<()> { |
| 752 | let b = res!(parse("# One\n\n### Three\n")); |
| 753 | assert_eq!(b.len(), 2); |
| 754 | assert_eq!(b[0], Block::Heading { level: 1, content: vec![Inline::Text("One".into())] }); |
| 755 | assert_eq!(b[1], Block::Heading { level: 3, content: vec![Inline::Text("Three".into())] }); |
| 756 | Ok(()) |
| 757 | } |
| 758 | |
| 759 | /// Seven hashes is not a heading, because there is no seventh level to give it. |
| 760 | #[test] |
| 761 | fn test_more_hashes_than_levels_is_not_a_heading_01() -> Outcome<()> { |
| 762 | let b = res!(parse("####### Seven\n")); |
| 763 | assert_eq!(b, vec![Block::Para(vec![Inline::Text("####### Seven".into())])]); |
| 764 | Ok(()) |
| 765 | } |
| 766 | |
| 767 | /// A run of hashes closing a heading is decoration, and is not part of what it says. |
| 768 | #[test] |
| 769 | fn test_a_closing_run_of_hashes_is_stripped_02() -> Outcome<()> { |
| 770 | assert_eq!(said(&res!(parse("## Two ##\n"))), vec!["Two"]); |
| 771 | assert_eq!(said(&res!(parse("## Two #########\n"))), vec!["Two"]); |
| 772 | // Without a space, the hash is part of the word and stays. |
| 773 | assert_eq!(said(&res!(parse("## Two#\n"))), vec!["Two#"]); |
| 774 | // A heading of nothing but its closing run says nothing. |
| 775 | assert_eq!(said(&res!(parse("# #\n"))), vec![""]); |
| 776 | assert_eq!(said(&res!(parse("#\n"))), vec![""]); |
| 777 | Ok(()) |
| 778 | } |
| 779 | |
| 780 | /// A hash with no space after it is a word, not a heading. |
| 781 | #[test] |
| 782 | fn test_a_hash_without_a_space_is_text_03() -> Outcome<()> { |
| 783 | let b = res!(parse("#hashtag\n")); |
| 784 | assert_eq!(b, vec![Block::Para(vec![Inline::Text("#hashtag".into())])]); |
| 785 | Ok(()) |
| 786 | } |
| 787 | |
| 788 | /// Blank lines divide paragraphs, and the lines between them are one paragraph. |
| 789 | #[test] |
| 790 | fn test_blank_lines_divide_paragraphs_04() -> Outcome<()> { |
| 791 | let b = res!(parse("One line\nand its second.\n\nA second paragraph.\n")); |
| 792 | assert_eq!(b.len(), 2); |
| 793 | assert_eq!(said(&b), vec!["One line and its second.", "A second paragraph."]); |
| 794 | Ok(()) |
| 795 | } |
| 796 | |
| 797 | /// A fence holds code exactly as written, and its info string names the language. |
| 798 | #[test] |
| 799 | fn test_a_fence_holds_code_and_names_its_language_05() -> Outcome<()> { |
| 800 | let b = res!(parse("```rust\nlet x = *y;\n```\n")); |
| 801 | assert_eq!(b, vec![Block::Code { |
| 802 | lang: Some("rust".into()), |
| 803 | text: "let x = *y;\n".into(), |
| 804 | }]); |
| 805 | // Tildes fence as well as backticks, and a fence may name nothing. |
| 806 | let b = res!(parse("~~~\nplain\n~~~\n")); |
| 807 | assert_eq!(b, vec![Block::Code { lang: None, text: "plain\n".into() }]); |
| 808 | Ok(()) |
| 809 | } |
| 810 | |
| 811 | /// A fence nobody closed runs to the end of the input rather than failing. |
| 812 | #[test] |
| 813 | fn test_an_unclosed_fence_runs_to_the_end_06() -> Outcome<()> { |
| 814 | let b = res!(parse("```\nstill code\nand more\n")); |
| 815 | assert_eq!(b, vec![Block::Code { lang: None, text: "still code\nand more\n".into() }]); |
| 816 | Ok(()) |
| 817 | } |
| 818 | |
| 819 | /// A fence swallows what would otherwise be markup, which is what a fence is for. |
| 820 | #[test] |
| 821 | fn test_a_fence_swallows_markup_07() -> Outcome<()> { |
| 822 | let b = res!(parse("```\n# not a heading\n- not a list\n```\n\nAfter.\n")); |
| 823 | assert_eq!(b.len(), 2); |
| 824 | assert_eq!(b[0], Block::Code { |
| 825 | lang: None, |
| 826 | text: "# not a heading\n- not a list\n".into(), |
| 827 | }); |
| 828 | Ok(()) |
| 829 | } |
| 830 | |
| 831 | /// Four spaces, or one tab, makes code of a line. |
| 832 | #[test] |
| 833 | fn test_indentation_makes_code_08() -> Outcome<()> { |
| 834 | let b = res!(parse(" let x = 1;\n let y = 2;\n")); |
| 835 | assert_eq!(b, vec![Block::Code { lang: None, text: "let x = 1;\nlet y = 2;\n".into() }]); |
| 836 | let b = res!(parse("\tby a tab\n")); |
| 837 | assert_eq!(b, vec![Block::Code { lang: None, text: "by a tab\n".into() }]); |
| 838 | Ok(()) |
| 839 | } |
| 840 | |
| 841 | /// Blank lines within indented code are kept, and those trailing it are not. |
| 842 | #[test] |
| 843 | fn test_indented_code_keeps_its_inner_blank_lines_09() -> Outcome<()> { |
| 844 | let b = res!(parse(" one\n\n two\n\nProse.\n")); |
| 845 | assert_eq!(b.len(), 2); |
| 846 | assert_eq!(b[0], Block::Code { lang: None, text: "one\n\ntwo\n".into() }); |
| 847 | assert_eq!(said(&b[1..]), vec!["Prose."]); |
| 848 | Ok(()) |
| 849 | } |
| 850 | |
| 851 | /// Indentation cannot make code of a line that carries a paragraph on. |
| 852 | #[test] |
| 853 | fn test_indentation_does_not_interrupt_a_paragraph_10() -> Outcome<()> { |
| 854 | let b = res!(parse("A paragraph\n and its wrapped line.\n")); |
| 855 | assert_eq!(b.len(), 1); |
| 856 | assert_eq!(said(&b), vec!["A paragraph and its wrapped line."]); |
| 857 | Ok(()) |
| 858 | } |
| 859 | |
| 860 | /// A quotation holds blocks, and reads them as a document of its own. |
| 861 | #[test] |
| 862 | fn test_a_quotation_holds_blocks_11() -> Outcome<()> { |
| 863 | let b = res!(parse("> # Heading\n>\n> A paragraph.\n")); |
| 864 | assert_eq!(b.len(), 1); |
| 865 | match &b[0] { |
| 866 | Block::Quote(inner) => { |
| 867 | assert_eq!(inner.len(), 2); |
| 868 | assert_eq!(said(inner), vec!["Heading", "A paragraph."]); |
| 869 | } |
| 870 | other => panic!("expected a quotation, got {:?}", other), |
| 871 | } |
| 872 | Ok(()) |
| 873 | } |
| 874 | |
| 875 | /// Quotations nest, and each `>` is a level. |
| 876 | #[test] |
| 877 | fn test_quotations_nest_12() -> Outcome<()> { |
| 878 | let b = res!(parse("> outer\n>> inner\n")); |
| 879 | match &b[0] { |
| 880 | Block::Quote(a) => match &a[1] { |
| 881 | Block::Quote(c) => assert_eq!(said(c), vec!["inner"]), |
| 882 | other => panic!("expected a nested quotation, got {:?}", other), |
| 883 | }, |
| 884 | other => panic!("expected a quotation, got {:?}", other), |
| 885 | } |
| 886 | Ok(()) |
| 887 | } |
| 888 | |
| 889 | /// An author who wraps a quoted paragraph need not mark every line of it. |
| 890 | #[test] |
| 891 | fn test_a_quotation_carries_on_lazily_13() -> Outcome<()> { |
| 892 | let b = res!(parse("> one\ntwo\n\nOut.\n")); |
| 893 | assert_eq!(b.len(), 2); |
| 894 | match &b[0] { |
| 895 | Block::Quote(inner) => assert_eq!(said(inner), vec!["one two"]), |
| 896 | other => panic!("expected a quotation, got {:?}", other), |
| 897 | } |
| 898 | Ok(()) |
| 899 | } |
| 900 | |
| 901 | /// Every bullet makes an unordered list. |
| 902 | #[test] |
| 903 | fn test_bullets_make_an_unordered_list_14() -> Outcome<()> { |
| 904 | for src in ["- a\n- b\n", "* a\n* b\n", "+ a\n+ b\n"] { |
| 905 | let b = res!(parse(src)); |
| 906 | match &b[0] { |
| 907 | Block::List { ordered, items } => { |
| 908 | assert!(!ordered); |
| 909 | assert_eq!(items.len(), 2, "for {:?}", src); |
| 910 | assert_eq!(said(&items[0]), vec!["a"]); |
| 911 | assert_eq!(said(&items[1]), vec!["b"]); |
| 912 | } |
| 913 | other => panic!("expected a list, got {:?}", other), |
| 914 | } |
| 915 | } |
| 916 | Ok(()) |
| 917 | } |
| 918 | |
| 919 | /// A number and a delimiter make an ordered list, whichever delimiter it is. |
| 920 | #[test] |
| 921 | fn test_numbers_make_an_ordered_list_15() -> Outcome<()> { |
| 922 | for src in ["1. a\n2. b\n", "1) a\n2) b\n"] { |
| 923 | let b = res!(parse(src)); |
| 924 | match &b[0] { |
| 925 | Block::List { ordered, items } => { |
| 926 | assert!(ordered, "for {:?}", src); |
| 927 | assert_eq!(items.len(), 2); |
| 928 | } |
| 929 | other => panic!("expected a list, got {:?}", other), |
| 930 | } |
| 931 | } |
| 932 | Ok(()) |
| 933 | } |
| 934 | |
| 935 | /// A change of marker begins a new list, which is how an author sets two lists apart. |
| 936 | #[test] |
| 937 | fn test_a_change_of_marker_begins_a_new_list_16() -> Outcome<()> { |
| 938 | let b = res!(parse("- a\n- b\n\n* c\n")); |
| 939 | assert_eq!(b.len(), 2); |
| 940 | match (&b[0], &b[1]) { |
| 941 | (Block::List { items: x, .. }, Block::List { items: y, .. }) => { |
| 942 | assert_eq!(x.len(), 2); |
| 943 | assert_eq!(y.len(), 1); |
| 944 | } |
| 945 | other => panic!("expected two lists, got {:?}", other), |
| 946 | } |
| 947 | Ok(()) |
| 948 | } |
| 949 | |
| 950 | /// Indentation nests one list inside another, within the item it is indented under. |
| 951 | #[test] |
| 952 | fn test_a_list_nests_within_a_list_17() -> Outcome<()> { |
| 953 | let b = res!(parse("- a\n - inner\n- b\n")); |
| 954 | match &b[0] { |
| 955 | Block::List { items, .. } => { |
| 956 | assert_eq!(items.len(), 2); |
| 957 | assert_eq!(items[0].len(), 2); // The paragraph, then the nested list. |
| 958 | assert_eq!(said(&items[0][..1]), vec!["a"]); |
| 959 | match &items[0][1] { |
| 960 | Block::List { items: inner, .. } => { |
| 961 | assert_eq!(said(&inner[0]), vec!["inner"]); |
| 962 | } |
| 963 | other => panic!("expected a nested list, got {:?}", other), |
| 964 | } |
| 965 | } |
| 966 | other => panic!("expected a list, got {:?}", other), |
| 967 | } |
| 968 | Ok(()) |
| 969 | } |
| 970 | |
| 971 | /// A quotation nests within a list item, which is nesting of one kind inside another. |
| 972 | #[test] |
| 973 | fn test_a_quotation_nests_within_a_list_item_18() -> Outcome<()> { |
| 974 | let b = res!(parse("- an item\n\n > quoted\n")); |
| 975 | match &b[0] { |
| 976 | Block::List { items, .. } => match &items[0][1] { |
| 977 | Block::Quote(inner) => assert_eq!(said(inner), vec!["quoted"]), |
| 978 | other => panic!("expected a quotation, got {:?}", other), |
| 979 | }, |
| 980 | other => panic!("expected a list, got {:?}", other), |
| 981 | } |
| 982 | Ok(()) |
| 983 | } |
| 984 | |
| 985 | /// A blank line between items leaves them items of the one list. |
| 986 | #[test] |
| 987 | fn test_a_blank_line_between_items_keeps_one_list_19() -> Outcome<()> { |
| 988 | let b = res!(parse("- a\n\n- b\n")); |
| 989 | assert_eq!(b.len(), 1); |
| 990 | match &b[0] { |
| 991 | Block::List { items, .. } => assert_eq!(items.len(), 2), |
| 992 | other => panic!("expected a list, got {:?}", other), |
| 993 | } |
| 994 | Ok(()) |
| 995 | } |
| 996 | |
| 997 | /// A list item holds every block written under it, not merely a line. |
| 998 | #[test] |
| 999 | fn test_a_list_item_holds_blocks_20() -> Outcome<()> { |
| 1000 | let b = res!(parse("- first\n\n second\n\n- next\n")); |
| 1001 | match &b[0] { |
| 1002 | Block::List { items, .. } => { |
| 1003 | assert_eq!(items.len(), 2); |
| 1004 | assert_eq!(said(&items[0]), vec!["first", "second"]); |
| 1005 | } |
| 1006 | other => panic!("expected a list, got {:?}", other), |
| 1007 | } |
| 1008 | Ok(()) |
| 1009 | } |
| 1010 | |
| 1011 | /// A list ends where a paragraph that is nobody's item begins. |
| 1012 | #[test] |
| 1013 | fn test_a_list_ends_at_an_unindented_paragraph_21() -> Outcome<()> { |
| 1014 | let b = res!(parse("- a\n- b\n\nAfter the list.\n")); |
| 1015 | assert_eq!(b.len(), 2); |
| 1016 | assert_eq!(said(&b[1..]), vec!["After the list."]); |
| 1017 | Ok(()) |
| 1018 | } |
| 1019 | |
| 1020 | /// Three or more of a break's characters make a thematic break. |
| 1021 | #[test] |
| 1022 | fn test_three_characters_make_a_thematic_break_22() -> Outcome<()> { |
| 1023 | for src in ["---\n", "***\n", "___\n", "- - -\n", "*****\n"] { |
| 1024 | assert_eq!(res!(parse(src)), vec![Block::Rule], "for {:?}", src); |
| 1025 | } |
| 1026 | // Two is not enough. |
| 1027 | assert_eq!(said(&res!(parse("--\n"))), vec!["--"]); |
| 1028 | Ok(()) |
| 1029 | } |
| 1030 | |
| 1031 | /// An underline under a paragraph is a heading, and the same characters alone are a break. |
| 1032 | #[test] |
| 1033 | fn test_an_underline_beats_a_thematic_break_23() -> Outcome<()> { |
| 1034 | let b = res!(parse("A title\n---\n")); |
| 1035 | assert_eq!(b, vec![Block::Heading { level: 2, content: vec![Inline::Text("A title".into())] }]); |
| 1036 | // With nothing above it to underline, the same line is a break. |
| 1037 | let b = res!(parse("\n---\n")); |
| 1038 | assert_eq!(b, vec![Block::Rule]); |
| 1039 | // And a break after a blank line is a break, not a heading. |
| 1040 | let b = res!(parse("A title\n\n---\n")); |
| 1041 | assert_eq!(b, vec![ |
| 1042 | Block::Para(vec![Inline::Text("A title".into())]), |
| 1043 | Block::Rule, |
| 1044 | ]); |
| 1045 | Ok(()) |
| 1046 | } |
| 1047 | |
| 1048 | /// Equals signs underline a first-level heading, which has no thematic break to argue with. |
| 1049 | #[test] |
| 1050 | fn test_equals_signs_underline_a_first_level_heading_24() -> Outcome<()> { |
| 1051 | let b = res!(parse("A title\n===\n")); |
| 1052 | assert_eq!(b, vec![Block::Heading { level: 1, content: vec![Inline::Text("A title".into())] }]); |
| 1053 | // Alone, they are only what they are. |
| 1054 | assert_eq!(said(&res!(parse("===\n"))), vec!["==="]); |
| 1055 | Ok(()) |
| 1056 | } |
| 1057 | |
| 1058 | /// A heading, a fence, a break or a quotation ends the paragraph above it without a blank line. |
| 1059 | #[test] |
| 1060 | fn test_a_block_may_interrupt_a_paragraph_25() -> Outcome<()> { |
| 1061 | let b = res!(parse("A paragraph\n# A heading\n")); |
| 1062 | assert_eq!(b.len(), 2); |
| 1063 | let b = res!(parse("A paragraph\n> quoted\n")); |
| 1064 | assert_eq!(b.len(), 2); |
| 1065 | let b = res!(parse("A paragraph\n***\n")); |
| 1066 | assert_eq!(b.len(), 2); |
| 1067 | let b = res!(parse("A paragraph\n```\ncode\n```\n")); |
| 1068 | assert_eq!(b.len(), 2); |
| 1069 | Ok(()) |
| 1070 | } |
| 1071 | |
| 1072 | /// A number that opens a sentence is prose, because a list that interrupts a paragraph counts |
| 1073 | /// from one. |
| 1074 | #[test] |
| 1075 | fn test_a_year_does_not_begin_a_list_26() -> Outcome<()> { |
| 1076 | let b = res!(parse("The year was\n2024. A good one.\n")); |
| 1077 | assert_eq!(b.len(), 1); |
| 1078 | assert_eq!(said(&b), vec!["The year was 2024. A good one."]); |
| 1079 | // But a list that counts from one does begin. |
| 1080 | let b = res!(parse("A paragraph\n1. an item\n")); |
| 1081 | assert_eq!(b.len(), 2); |
| 1082 | Ok(()) |
| 1083 | } |
| 1084 | |
| 1085 | /// A dash with no space after it is a word, not an item. |
| 1086 | #[test] |
| 1087 | fn test_a_dash_without_a_space_is_text_27() -> Outcome<()> { |
| 1088 | let b = res!(parse("-not-a-list\n")); |
| 1089 | assert_eq!(b, vec![Block::Para(vec![Inline::Text("-not-a-list".into())])]); |
| 1090 | Ok(()) |
| 1091 | } |
| 1092 | |
| 1093 | /// Nesting past the limit is refused, which is the parser's one refusal. |
| 1094 | #[test] |
| 1095 | fn test_nesting_past_the_limit_is_refused_28() -> Outcome<()> { |
| 1096 | // A quotation for every level the limit allows is read. |
| 1097 | let ok = format!("{} deep\n", ">".repeat(DEPTH_LIMIT)); |
| 1098 | assert!(parse(&ok).is_ok()); |
| 1099 | // One past it, and past by far, is not. |
| 1100 | let deep = format!("{} deep\n", ">".repeat(DEPTH_LIMIT + 8)); |
| 1101 | assert!(parse(&deep).is_err()); |
| 1102 | let very = format!("{} deep\n", ">".repeat(2000)); |
| 1103 | assert!(parse(&very).is_err()); |
| 1104 | Ok(()) |
| 1105 | } |
| 1106 | |
| 1107 | /// An empty document is a document with nothing in it, and not a failure. |
| 1108 | #[test] |
| 1109 | fn test_an_empty_document_holds_nothing_29() -> Outcome<()> { |
| 1110 | assert_eq!(res!(parse("")), Vec::<Block>::new()); |
| 1111 | assert_eq!(res!(parse("\n\n \n")), Vec::<Block>::new()); |
| 1112 | Ok(()) |
| 1113 | } |
| 1114 | |
| 1115 | /// Windows line endings are line endings. |
| 1116 | #[test] |
| 1117 | fn test_carriage_returns_are_not_text_30() -> Outcome<()> { |
| 1118 | let b = res!(parse("# A heading\r\n\r\nA paragraph.\r\n")); |
| 1119 | assert_eq!(said(&b), vec!["A heading", "A paragraph."]); |
| 1120 | Ok(()) |
| 1121 | } |
| 1122 | |
| 1123 | /// A tab indents a list item's content as spaces would. |
| 1124 | #[test] |
| 1125 | fn test_a_tab_indents_an_item_31() -> Outcome<()> { |
| 1126 | let b = res!(parse("-\tan item\n")); |
| 1127 | match &b[0] { |
| 1128 | Block::List { items, .. } => assert_eq!(said(&items[0]), vec!["an item"]), |
| 1129 | other => panic!("expected a list, got {:?}", other), |
| 1130 | } |
| 1131 | Ok(()) |
| 1132 | } |
| 1133 | |
| 1134 | /// A fence's indentation is taken off the code it holds, and no more. |
| 1135 | #[test] |
| 1136 | fn test_a_fence_strips_its_own_indent_32() -> Outcome<()> { |
| 1137 | let b = res!(parse(" ```\n code\n deeper\n ```\n")); |
| 1138 | assert_eq!(b, vec![Block::Code { lang: None, text: "code\n deeper\n".into() }]); |
| 1139 | Ok(()) |
| 1140 | } |
| 1141 | |
| 1142 | /// A paragraph the author hard wrapped is one run of prose, and reflows to whatever reads it. |
| 1143 | /// |
| 1144 | /// Prose arrives wrapped to the width its author wrote at. That width means nothing, so it is |
| 1145 | /// not kept: a paragraph is what it says, and where its lines fall is the reader's to decide. |
| 1146 | #[test] |
| 1147 | fn test_a_hard_wrapped_paragraph_reflows_34() -> Outcome<()> { |
| 1148 | let b = res!(parse("A paragraph that the author\nhard wrapped at a narrow width\nacross three lines.\n")); |
| 1149 | assert_eq!(b, vec![Block::Para(vec![ |
| 1150 | Inline::Text("A paragraph that the author hard wrapped at a narrow width across three lines.".into()), |
| 1151 | ])]); |
| 1152 | Ok(()) |
| 1153 | } |
| 1154 | |
| 1155 | /// A break the author did ask for survives the paragraph it is in. |
| 1156 | #[test] |
| 1157 | fn test_a_hard_break_survives_a_paragraph_35() -> Outcome<()> { |
| 1158 | let b = res!(parse("one \ntwo\n")); |
| 1159 | assert_eq!(b, vec![Block::Para(vec![ |
| 1160 | Inline::Text("one".into()), |
| 1161 | Inline::Break, |
| 1162 | Inline::Text("two".into()), |
| 1163 | ])]); |
| 1164 | // A backslash asks as plainly as two spaces do. |
| 1165 | let b = res!(parse("one\\\ntwo\n")); |
| 1166 | assert_eq!(b, vec![Block::Para(vec![ |
| 1167 | Inline::Text("one".into()), |
| 1168 | Inline::Break, |
| 1169 | Inline::Text("two".into()), |
| 1170 | ])]); |
| 1171 | Ok(()) |
| 1172 | } |
| 1173 | |
| 1174 | /// A header row and the delimiter row beneath it make a table, and the lines under them are its |
| 1175 | /// body. |
| 1176 | #[test] |
| 1177 | fn test_a_header_and_a_delimiter_row_make_a_table_36() -> Outcome<()> { |
| 1178 | let b = res!(parse("| Name | Age |\n| --- | --- |\n| Alice | 30 |\n| Bob | 25 |\n")); |
| 1179 | assert_eq!(b.len(), 1); |
| 1180 | assert_eq!(grid(&b[0]), vec![ |
| 1181 | vec!["Name", "Age"], |
| 1182 | vec!["Alice", "30"], |
| 1183 | vec!["Bob", "25"], |
| 1184 | ]); |
| 1185 | Ok(()) |
| 1186 | } |
| 1187 | |
| 1188 | /// The pipes at a row's edges draw nothing, so an author may leave them off. |
| 1189 | #[test] |
| 1190 | fn test_the_pipes_at_a_rows_edges_are_optional_37() -> Outcome<()> { |
| 1191 | let b = res!(parse("a | b\n--- | ---\n1 | 2\n")); |
| 1192 | assert_eq!(grid(&b[0]), vec![vec!["a", "b"], vec!["1", "2"]]); |
| 1193 | Ok(()) |
| 1194 | } |
| 1195 | |
| 1196 | /// The delimiter row's colons align the columns, and a column given no colon is aligned by |
| 1197 | /// nothing. |
| 1198 | #[test] |
| 1199 | fn test_the_delimiter_rows_colons_align_the_columns_38() -> Outcome<()> { |
| 1200 | let b = res!(parse("| w | x | y | z |\n| --- | :-- | :-: | --: |\n")); |
| 1201 | match &b[0] { |
| 1202 | // Start and End, never left and right: the tree does not know which way its text runs. |
| 1203 | Block::Table { cols, .. } => assert_eq!( |
| 1204 | cols, |
| 1205 | &vec![Align::None, Align::Start, Align::Centre, Align::End], |
| 1206 | ), |
| 1207 | other => panic!("expected a table, got {:?}", other), |
| 1208 | } |
| 1209 | Ok(()) |
| 1210 | } |
| 1211 | |
| 1212 | /// A pipe the author escaped is a pipe in the cell, and divides nothing. |
| 1213 | #[test] |
| 1214 | fn test_an_escaped_pipe_is_a_pipe_39() -> Outcome<()> { |
| 1215 | let b = res!(parse("| a | b |\n| --- | --- |\n| x \\| y | z |\n")); |
| 1216 | assert_eq!(grid(&b[0]), vec![vec!["a", "b"], vec!["x | y", "z"]]); |
| 1217 | Ok(()) |
| 1218 | } |
| 1219 | |
| 1220 | /// A delimiter row that gives a different number of cells than the header did is no delimiter |
| 1221 | /// row, and what is left is the paragraph it always was. |
| 1222 | #[test] |
| 1223 | fn test_a_delimiter_row_of_the_wrong_width_is_prose_40() -> Outcome<()> { |
| 1224 | let b = res!(parse("| a | b |\n| --- |\n")); |
| 1225 | assert_eq!(b.len(), 1); |
| 1226 | assert_eq!(said(&b), vec!["| a | b | | --- |"]); |
| 1227 | // And the other way about: a delimiter row wider than its header is no more a table. |
| 1228 | let b = res!(parse("| a |\n| --- | --- |\n| 1 |\n")); |
| 1229 | assert_eq!(b.len(), 1); |
| 1230 | assert!(matches!(b[0], Block::Para(_)), "expected a paragraph, got {:?}", b[0]); |
| 1231 | Ok(()) |
| 1232 | } |
| 1233 | |
| 1234 | /// A table of a header and nothing else is a table: the rows are what it has none of. |
| 1235 | #[test] |
| 1236 | fn test_a_table_may_have_no_rows_41() -> Outcome<()> { |
| 1237 | let b = res!(parse("| a | b |\n| --- | --- |\n")); |
| 1238 | match &b[0] { |
| 1239 | Block::Table { head, rows, cols } => { |
| 1240 | assert!(head.is_some(), "the table lost its header row"); |
| 1241 | assert!(rows.is_empty(), "the table invented a row: {:?}", rows); |
| 1242 | assert_eq!(cols.len(), 2); |
| 1243 | } |
| 1244 | other => panic!("expected a table, got {:?}", other), |
| 1245 | } |
| 1246 | Ok(()) |
| 1247 | } |
| 1248 | |
| 1249 | /// A cell holds an inline run like any other, since a cell's text is read by the inline pass. |
| 1250 | #[test] |
| 1251 | fn test_a_cell_holds_inline_markup_42() -> Outcome<()> { |
| 1252 | let b = res!(parse("| a | b |\n| --- | --- |\n| *loud* | [link](somewhere) |\n")); |
| 1253 | match &b[0] { |
| 1254 | Block::Table { rows, .. } => { |
| 1255 | assert_eq!(rows[0].0[0], Cell(vec![Inline::Emph { |
| 1256 | strong: false, |
| 1257 | content: vec![Inline::Text("loud".into())], |
| 1258 | }])); |
| 1259 | assert_eq!(rows[0].0[1], Cell(vec![Inline::Link { |
| 1260 | to: "somewhere".into(), |
| 1261 | content: vec![Inline::Text("link".into())], |
| 1262 | }])); |
| 1263 | } |
| 1264 | other => panic!("expected a table, got {:?}", other), |
| 1265 | } |
| 1266 | Ok(()) |
| 1267 | } |
| 1268 | |
| 1269 | /// A pipe within a code span divides the cell, because the row is divided before the inline pass |
| 1270 | /// ever sees it. |
| 1271 | /// |
| 1272 | /// This is GFM's wart, and it is asserted here because it is deliberate rather than an oversight: |
| 1273 | /// what a row divides into is settled by the pipes on the line and by nothing that has to be |
| 1274 | /// parsed to be found. |
| 1275 | #[test] |
| 1276 | fn test_a_pipe_in_a_code_span_divides_a_cell_43() -> Outcome<()> { |
| 1277 | let b = res!(parse("| a | b |\n| --- | --- |\n| `x | y` | z |\n")); |
| 1278 | // Three cells were written where the header named two, so the third is cut, and neither of |
| 1279 | // the two that remain holds a code span: each holds a backtick that closes nothing. |
| 1280 | assert_eq!(grid(&b[0]), vec![vec!["a", "b"], vec!["`x", "y`"]]); |
| 1281 | Ok(()) |
| 1282 | } |
| 1283 | |
| 1284 | /// A row of fewer cells than the header is filled out, and one of more is cut. |
| 1285 | #[test] |
| 1286 | fn test_a_row_is_held_to_the_headers_width_44() -> Outcome<()> { |
| 1287 | let b = res!(parse("| a | b |\n| --- | --- |\n| 1 |\n| 1 | 2 | 3 |\n")); |
| 1288 | assert_eq!(grid(&b[0]), vec![ |
| 1289 | vec!["a", "b"], |
| 1290 | vec!["1", ""], |
| 1291 | vec!["1", "2"], |
| 1292 | ]); |
| 1293 | Ok(()) |
| 1294 | } |
| 1295 | |
| 1296 | /// A table below a paragraph ends it, and the header is no part of what the paragraph said. |
| 1297 | #[test] |
| 1298 | fn test_a_table_ends_the_paragraph_above_it_45() -> Outcome<()> { |
| 1299 | let b = res!(parse("Some prose.\nand its second line.\n| a | b |\n| --- | --- |\n| 1 | 2 |\n")); |
| 1300 | assert_eq!(b.len(), 2); |
| 1301 | assert_eq!(said(&b[..1]), vec!["Some prose. and its second line."]); |
| 1302 | assert_eq!(grid(&b[1]), vec![vec!["a", "b"], vec!["1", "2"]]); |
| 1303 | Ok(()) |
| 1304 | } |
| 1305 | |
| 1306 | /// A table runs to the first blank line, or to whatever begins a block of its own. |
| 1307 | #[test] |
| 1308 | fn test_a_table_ends_where_the_next_block_begins_46() -> Outcome<()> { |
| 1309 | let b = res!(parse("| a |\n| --- |\n| 1 |\n\nAfter.\n")); |
| 1310 | assert_eq!(b.len(), 2); |
| 1311 | assert_eq!(grid(&b[0]), vec![vec!["a"], vec!["1"]]); |
| 1312 | assert_eq!(said(&b[1..]), vec!["After."]); |
| 1313 | let b = res!(parse("| a |\n| --- |\n| 1 |\n# A heading\n")); |
| 1314 | assert_eq!(b.len(), 2); |
| 1315 | assert!(matches!(b[1], Block::Heading { level: 1, .. })); |
| 1316 | Ok(()) |
| 1317 | } |
| 1318 | |
| 1319 | /// An underline still beats what would otherwise be a delimiter row, as it beats a thematic |
| 1320 | /// break: a line of dashes under a line of prose is the heading it has always been. |
| 1321 | #[test] |
| 1322 | fn test_an_underline_beats_a_delimiter_row_47() -> Outcome<()> { |
| 1323 | let b = res!(parse("A title\n---\n")); |
| 1324 | assert_eq!(b, vec![Block::Heading { level: 2, content: vec![Inline::Text("A title".into())] }]); |
| 1325 | Ok(()) |
| 1326 | } |
| 1327 | |
| 1328 | /// A whole document of every block reads as the document it is. |
| 1329 | #[test] |
| 1330 | fn test_a_document_of_every_block_33() -> Outcome<()> { |
| 1331 | let src = "\ |
| 1332 | # Title |
| 1333 | |
| 1334 | An opening paragraph. |
| 1335 | |
| 1336 | ## A section |
| 1337 | |
| 1338 | - one |
| 1339 | - two |
| 1340 | - nested |
| 1341 | |
| 1342 | > A quotation. |
| 1343 | |
| 1344 | ```sh |
| 1345 | echo hi |
| 1346 | ``` |
| 1347 | |
| 1348 | --- |
| 1349 | |
| 1350 | The end. |
| 1351 | "; |
| 1352 | let b = res!(parse(src)); |
| 1353 | assert!(matches!(b[0], Block::Heading { level: 1, .. })); |
| 1354 | assert!(matches!(b[1], Block::Para(_))); |
| 1355 | assert!(matches!(b[2], Block::Heading { level: 2, .. })); |
| 1356 | assert!(matches!(b[3], Block::List { ordered: false, .. })); |
| 1357 | assert!(matches!(b[4], Block::Quote(_))); |
| 1358 | assert!(matches!(b[5], Block::Code { .. })); |
| 1359 | assert!(matches!(b[6], Block::Rule)); |
| 1360 | assert!(matches!(b[7], Block::Para(_))); |
| 1361 | assert_eq!(b.len(), 8); |
| 1362 | Ok(()) |
| 1363 | } |
| 1364 | } |