oxedyne/fe2o3/fe2o3_text/src/doc/html/read.rs
51.3 KiB, 1 run
created by r1870400018:14324, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Reading HTML into the document tree: the pass that turns tags into blocks and inlines, and throws |
| 2 | //! the exporter's whitespace away. |
| 3 | //! |
| 4 | //! # The dialect |
| 5 | //! |
| 6 | //! The elements read are the ones the tree has a node for, and no others: the six headings, |
| 7 | //! paragraphs, lists, `pre`, quotations, thematic breaks and tables, and the inline run of emphasis, |
| 8 | //! links, images, code spans and breaks. Comments, `head`, `script` and `style` hold no prose and are |
| 9 | //! dropped, content and all. |
| 10 | //! |
| 11 | //! # Everything else is unwrapped |
| 12 | //! |
| 13 | //! A `div`, a `span`, and every element this has never heard of, contribute nothing themselves and are |
| 14 | //! erased: the tag goes and the content is read exactly where the tag stood. There is no generic |
| 15 | //! container in the tree to put them in, and inventing one would make every consumer learn HTML -- |
| 16 | //! which is the one thing the tree exists to prevent. |
| 17 | //! |
| 18 | //! Erasing rather than recursing is what makes this work in both directions at once. A `div` holding |
| 19 | //! paragraphs is read as those paragraphs; a `div` holding bare words is read as the paragraph those |
| 20 | //! words are; a `span` in the middle of a sentence stays in the middle of that sentence, with the text |
| 21 | //! either side of it joined across the hole the tag left. The rule is the same in each case, and no |
| 22 | //! prose is lost in any of them. |
| 23 | //! |
| 24 | //! # Robustness |
| 25 | //! |
| 26 | //! The input this is written for is a generator's output: well formed, and wrong in none of the ways a |
| 27 | //! browser must survive. So there is no error recovery here beyond the cheap kind -- a stray close tag |
| 28 | //! closes nothing and the document carries on, an element left open runs to the end of what encloses |
| 29 | //! it. The one refusal is [`DEPTH_LIMIT`]. |
| 30 | |
| 31 | use crate::doc::{ |
| 32 | Align, |
| 33 | Block, |
| 34 | Cell, |
| 35 | Inline, |
| 36 | Row, |
| 37 | }; |
| 38 | use crate::html::decode_entities; |
| 39 | |
| 40 | use oxedyne_fe2o3_core::prelude::*; |
| 41 | |
| 42 | use std::mem::take; |
| 43 | |
| 44 | /// How deep a document may nest its elements before the reader refuses it. |
| 45 | /// |
| 46 | /// A quotation inside a list inside a quotation is legitimate; a thousand of them is a document built |
| 47 | /// to exhaust the stack of whatever reads it. The limit is generous beside anything a generator |
| 48 | /// exports and far below what would trouble the machine. |
| 49 | /// |
| 50 | /// Only the elements the tree has a node for are counted, because only those are read by a recursive |
| 51 | /// walk. A `div` and an unknown element are erased where they stand and cost no stack at all, so a |
| 52 | /// document of a million nested `div`s is read as the flat prose it says rather than refused. |
| 53 | pub const DEPTH_LIMIT: usize = 32; |
| 54 | |
| 55 | /// Reads HTML into the blocks it is made of. |
| 56 | pub fn parse(src: &str) -> Outcome<Vec<Block>> { |
| 57 | let mut lex = Lex { src, i: 0, raw: None }; |
| 58 | blocks(&mut lex, None, 0) |
| 59 | } |
| 60 | |
| 61 | /// What an element is, which is what decides where its content goes. |
| 62 | #[derive(Clone, Copy, Debug, PartialEq)] |
| 63 | enum Kind { |
| 64 | /// A heading, of the level its name gives. |
| 65 | Heading(u8), |
| 66 | /// A paragraph. |
| 67 | Para, |
| 68 | /// A list, ordered where the flag says so. |
| 69 | List(bool), |
| 70 | /// A run of code, whose whitespace is what it says. |
| 71 | Code, |
| 72 | /// A quotation. |
| 73 | Quote, |
| 74 | /// A table. |
| 75 | Table, |
| 76 | /// A thematic break. |
| 77 | Rule, |
| 78 | /// Emphasis, strong where the flag says so. |
| 79 | Emph(bool), |
| 80 | /// A link. |
| 81 | Link, |
| 82 | /// An image. |
| 83 | Image, |
| 84 | /// A code span within a line. |
| 85 | Span, |
| 86 | /// A break the author asked for. |
| 87 | Break, |
| 88 | /// Content that is not prose, and goes nowhere. |
| 89 | Drop, |
| 90 | /// Anything else: the tag is erased and its content read in its place. |
| 91 | Bare, |
| 92 | } |
| 93 | |
| 94 | impl Kind { |
| 95 | |
| 96 | /// Whether the element belongs within a line rather than standing on its own. |
| 97 | fn is_inline(&self) -> bool { |
| 98 | matches!(self, Self::Emph(_) | Self::Link | Self::Image | Self::Span | Self::Break) |
| 99 | } |
| 100 | } |
| 101 | |
| 102 | /// What an element is, by the name it was written with. |
| 103 | /// |
| 104 | /// The name is lowered for the match, because HTML does not care how a tag is written and neither does |
| 105 | /// this. Anything not named here is [`Kind::Bare`]. |
| 106 | fn kind(name: &str) -> Kind { |
| 107 | match name.to_ascii_lowercase().as_str() { |
| 108 | "h1" => Kind::Heading(1), |
| 109 | "h2" => Kind::Heading(2), |
| 110 | "h3" => Kind::Heading(3), |
| 111 | "h4" => Kind::Heading(4), |
| 112 | "h5" => Kind::Heading(5), |
| 113 | "h6" => Kind::Heading(6), |
| 114 | "p" => Kind::Para, |
| 115 | "ul" => Kind::List(false), |
| 116 | "ol" => Kind::List(true), |
| 117 | "pre" => Kind::Code, |
| 118 | "blockquote" => Kind::Quote, |
| 119 | "table" => Kind::Table, |
| 120 | "hr" => Kind::Rule, |
| 121 | "em" | "i" => Kind::Emph(false), |
| 122 | "strong" | "b" => Kind::Emph(true), |
| 123 | "a" => Kind::Link, |
| 124 | "img" => Kind::Image, |
| 125 | "code" => Kind::Span, |
| 126 | "br" => Kind::Break, |
| 127 | "head" | "script" | "style" => Kind::Drop, |
| 128 | _ => Kind::Bare, |
| 129 | } |
| 130 | } |
| 131 | |
| 132 | /// The elements that hold nothing, and so need no closing tag. |
| 133 | const VOID: [&str; 14] = [ |
| 134 | "area", |
| 135 | "base", |
| 136 | "br", |
| 137 | "col", |
| 138 | "embed", |
| 139 | "hr", |
| 140 | "img", |
| 141 | "input", |
| 142 | "link", |
| 143 | "meta", |
| 144 | "param", |
| 145 | "source", |
| 146 | "track", |
| 147 | "wbr", |
| 148 | ]; |
| 149 | |
| 150 | /// The elements whose content is text and not markup, so that a `<` within them opens nothing. |
| 151 | const RAW: [&str; 2] = [ |
| 152 | "script", |
| 153 | "style", |
| 154 | ]; |
| 155 | |
| 156 | /// Whether a name is the given element's, without regard to the case it was written in. |
| 157 | fn is(name: &str, want: &str) -> bool { |
| 158 | name.eq_ignore_ascii_case(want) |
| 159 | } |
| 160 | |
| 161 | // ----------------------------------------------------------------------------------------------- |
| 162 | // The walk. |
| 163 | // ----------------------------------------------------------------------------------------------- |
| 164 | |
| 165 | /// Reads the children of an element into the blocks they are, until the given element closes or the |
| 166 | /// input ends. |
| 167 | /// |
| 168 | /// Loose inline content -- words that reach block level with no paragraph around them, which is what a |
| 169 | /// `div` full of prose leaves behind once it is erased -- is gathered into the paragraph it is. |
| 170 | fn blocks(lex: &mut Lex, until: Option<&str>, depth: usize) -> Outcome<Vec<Block>> { |
| 171 | if depth > DEPTH_LIMIT { |
| 172 | return Err(err!( |
| 173 | "HTML elements nest more than {} deep, which no document written to be read \ |
| 174 | does.", DEPTH_LIMIT; |
| 175 | Excessive, Input)); |
| 176 | } |
| 177 | let mut out = Vec::new(); |
| 178 | let mut run = Run::default(); // Loose inline content, awaiting the paragraph it makes. |
| 179 | while let Some(tok) = lex.next() { |
| 180 | match tok { |
| 181 | Tok::Text(t) => run.text(t), |
| 182 | Tok::Close(name) => { |
| 183 | // A close tag that answers nothing closes nothing, and the document carries on. |
| 184 | if let Some(until) = until { |
| 185 | if is(name, until) { |
| 186 | break; |
| 187 | } |
| 188 | } |
| 189 | } |
| 190 | Tok::Open { name, attrs, void } => { |
| 191 | let k = kind(name); |
| 192 | match k { |
| 193 | Kind::Drop => skip(lex, name, void), |
| 194 | // The tag is erased, and its content read as though it had never been there. |
| 195 | Kind::Bare => {} |
| 196 | _ if k.is_inline() => { |
| 197 | res!(inline(lex, &mut run, k, name, attrs, void, depth)); |
| 198 | } |
| 199 | _ => { |
| 200 | flush(&mut out, &mut run); |
| 201 | if let Some(b) = res!(block(lex, k, name, attrs, void, depth)) { |
| 202 | out.push(b); |
| 203 | } |
| 204 | } |
| 205 | } |
| 206 | } |
| 207 | } |
| 208 | } |
| 209 | flush(&mut out, &mut run); |
| 210 | Ok(out) |
| 211 | } |
| 212 | |
| 213 | /// Turns the loose inline content gathered so far into the paragraph it is, where it says anything. |
| 214 | /// |
| 215 | /// A run of nothing but the whitespace that lay between two blocks says nothing, and makes no |
| 216 | /// paragraph. |
| 217 | fn flush(out: &mut Vec<Block>, run: &mut Run) { |
| 218 | let content = take(run).end(); |
| 219 | if !content.is_empty() { |
| 220 | out.push(Block::Para(content)); |
| 221 | } |
| 222 | } |
| 223 | |
| 224 | /// Reads the block an opening tag begins, where the tree has one for it. |
| 225 | fn block(lex: &mut Lex, k: Kind, name: &str, attrs: &str, void: bool, depth: usize) |
| 226 | -> Outcome<Option<Block>> |
| 227 | { |
| 228 | let b = match k { |
| 229 | Kind::Rule => Block::Rule, |
| 230 | Kind::Code => code(lex, attrs, void), |
| 231 | Kind::List(ordered) => match void { |
| 232 | true => Block::List { ordered, items: Vec::new() }, |
| 233 | false => res!(list(lex, ordered, name, depth)), |
| 234 | }, |
| 235 | Kind::Table => match void { |
| 236 | true => Block::Table { head: None, rows: Vec::new(), cols: Vec::new() }, |
| 237 | false => res!(table(lex, depth)), |
| 238 | }, |
| 239 | Kind::Quote => match void { |
| 240 | true => Block::Quote(Vec::new()), |
| 241 | false => Block::Quote(res!(blocks(lex, Some(name), depth + 1))), |
| 242 | }, |
| 243 | Kind::Heading(level) => Block::Heading { |
| 244 | level, |
| 245 | content: match void { |
| 246 | true => Vec::new(), |
| 247 | false => res!(inlines(lex, name, depth + 1)), |
| 248 | }, |
| 249 | }, |
| 250 | Kind::Para => { |
| 251 | let content = match void { |
| 252 | true => Vec::new(), |
| 253 | false => res!(inlines(lex, name, depth + 1)), |
| 254 | }; |
| 255 | // A paragraph of nothing but the whitespace an exporter laid it out with says nothing, |
| 256 | // and the tree is better off without it. |
| 257 | if content.is_empty() { |
| 258 | return Ok(None); |
| 259 | } |
| 260 | Block::Para(content) |
| 261 | } |
| 262 | // Everything else here is a part of a list or a table that only its own reader sees. One that |
| 263 | // reaches this stood outside the thing it belongs to, where it says nothing: the tag is erased |
| 264 | // and the content within it is read where it stands. |
| 265 | _ => return Ok(None), |
| 266 | }; |
| 267 | Ok(Some(b)) |
| 268 | } |
| 269 | |
| 270 | /// Reads the children of an element into the run of inlines they are, until the given element closes |
| 271 | /// or the input ends. |
| 272 | fn inlines(lex: &mut Lex, until: &str, depth: usize) -> Outcome<Vec<Inline>> { |
| 273 | if depth > DEPTH_LIMIT { |
| 274 | return Err(err!( |
| 275 | "HTML elements nest more than {} deep, which no prose written to be read \ |
| 276 | does.", DEPTH_LIMIT; |
| 277 | Excessive, Input)); |
| 278 | } |
| 279 | let mut run = Run::default(); |
| 280 | while let Some(tok) = lex.next() { |
| 281 | match tok { |
| 282 | Tok::Text(t) => run.text(t), |
| 283 | Tok::Close(name) => { |
| 284 | if is(name, until) { |
| 285 | break; |
| 286 | } |
| 287 | } |
| 288 | Tok::Open { name, attrs, void } => { |
| 289 | let k = kind(name); |
| 290 | match k { |
| 291 | Kind::Drop => skip(lex, name, void), |
| 292 | Kind::Bare => {} |
| 293 | _ if k.is_inline() => { |
| 294 | res!(inline(lex, &mut run, k, name, attrs, void, depth)); |
| 295 | } |
| 296 | // A block within a line is a thing the tree cannot hold: a cell is given inlines |
| 297 | // and nothing else, deliberately. The tag is erased like any other it has no node |
| 298 | // for, but it stands as the boundary it is, so that the words either side of it |
| 299 | // stay words apart rather than running together into one. |
| 300 | _ => run.space(), |
| 301 | } |
| 302 | } |
| 303 | } |
| 304 | } |
| 305 | Ok(run.end()) |
| 306 | } |
| 307 | |
| 308 | /// Reads the inline element an opening tag begins into the run it belongs to. |
| 309 | fn inline(lex: &mut Lex, run: &mut Run, k: Kind, name: &str, attrs: &str, void: bool, depth: usize) |
| 310 | -> Outcome<()> |
| 311 | { |
| 312 | match k { |
| 313 | Kind::Break => run.push(Inline::Break), |
| 314 | Kind::Image => run.push(Inline::Image { |
| 315 | src: attr(attrs, "src").unwrap_or_default(), |
| 316 | alt: attr(attrs, "alt").unwrap_or_default(), |
| 317 | }), |
| 318 | Kind::Span => { |
| 319 | // A code span is not a `pre`: HTML collapses the whitespace within one exactly as it does |
| 320 | // anywhere else, and so does this. What is not trimmed is the space at either end, because |
| 321 | // a span sits in a line and the space beside it is the line's. |
| 322 | let text = match void { |
| 323 | true => String::new(), |
| 324 | false => text_in(lex, name), |
| 325 | }; |
| 326 | run.push(Inline::Code(collapse(&text))); |
| 327 | } |
| 328 | Kind::Emph(strong) => { |
| 329 | let content = match void { |
| 330 | true => Vec::new(), |
| 331 | false => res!(inlines(lex, name, depth + 1)), |
| 332 | }; |
| 333 | run.push(Inline::Emph { strong, content }); |
| 334 | } |
| 335 | Kind::Link => match attr(attrs, "href") { |
| 336 | Some(to) => { |
| 337 | let content = match void { |
| 338 | true => Vec::new(), |
| 339 | false => res!(inlines(lex, name, depth + 1)), |
| 340 | }; |
| 341 | run.push(Inline::Link { to, content }); |
| 342 | } |
| 343 | // An `a` with no destination is an anchor and not a link: there is nowhere for a reader to |
| 344 | // go. The tag is erased, and the words within it stay in the line they were in. |
| 345 | None => {} |
| 346 | }, |
| 347 | // Nothing else reaches here: this is called only where the kind is inline. |
| 348 | _ => {} |
| 349 | } |
| 350 | Ok(()) |
| 351 | } |
| 352 | |
| 353 | /// Reads a `ul` or an `ol` into the list it is. |
| 354 | fn list(lex: &mut Lex, ordered: bool, until: &str, depth: usize) -> Outcome<Block> { |
| 355 | let mut items: Vec<Vec<Block>> = Vec::new(); |
| 356 | while let Some(tok) = lex.next() { |
| 357 | match tok { |
| 358 | Tok::Text(t) => { |
| 359 | // The whitespace an exporter lays a list out with says nothing. Words that reach a |
| 360 | // list with no item to sit in are still words, and stand as an item of their own |
| 361 | // rather than being dropped. |
| 362 | let s = collapse(&decode_entities(t)); |
| 363 | let s = s.trim(); |
| 364 | if !s.is_empty() { |
| 365 | items.push(vec![Block::Para(vec![Inline::Text(s.to_string())])]); |
| 366 | } |
| 367 | } |
| 368 | Tok::Close(name) => { |
| 369 | if is(name, until) { |
| 370 | break; |
| 371 | } |
| 372 | } |
| 373 | Tok::Open { name, void, .. } => { |
| 374 | if is(name, "li") { |
| 375 | items.push(match void { |
| 376 | true => Vec::new(), |
| 377 | false => res!(blocks(lex, Some(name), depth + 1)), |
| 378 | }); |
| 379 | } else if kind(name) == Kind::Drop { |
| 380 | skip(lex, name, void); |
| 381 | } |
| 382 | // Anything else between the items is erased, so a list whose items are wrapped in |
| 383 | // something the tree does not know still finds them. |
| 384 | } |
| 385 | } |
| 386 | } |
| 387 | Ok(Block::List { ordered, items }) |
| 388 | } |
| 389 | |
| 390 | /// Reads a `table` into the grid it is. |
| 391 | fn table(lex: &mut Lex, depth: usize) -> Outcome<Block> { |
| 392 | let mut head: Option<Row> = None; |
| 393 | let mut rows: Vec<Row> = Vec::new(); |
| 394 | let mut cols: Vec<Align> = Vec::new(); |
| 395 | let mut in_head = false; // Whether the rows being read are the table's header. |
| 396 | while let Some(tok) = lex.next() { |
| 397 | match tok { |
| 398 | // The whitespace a table is laid out with says nothing, and a table has nowhere to put a |
| 399 | // word that reached it without a cell to sit in. |
| 400 | Tok::Text(_) => {} |
| 401 | Tok::Close(name) => { |
| 402 | if is(name, "table") { |
| 403 | break; |
| 404 | } |
| 405 | if is(name, "thead") { |
| 406 | in_head = false; |
| 407 | } |
| 408 | } |
| 409 | Tok::Open { name, void, .. } => { |
| 410 | if is(name, "thead") { |
| 411 | in_head = true; |
| 412 | } else if is(name, "tr") && !void { |
| 413 | let (r, all_th) = res!(row(lex, &mut cols, depth + 1)); |
| 414 | // A table names its columns in a `thead`, or in a first row of nothing but `th`. |
| 415 | // Both say the same thing, and an exporter picks whichever it likes. |
| 416 | if head.is_none() && rows.is_empty() && (in_head || all_th) { |
| 417 | head = Some(r); |
| 418 | } else { |
| 419 | rows.push(r); |
| 420 | } |
| 421 | } else if kind(name) == Kind::Drop { |
| 422 | skip(lex, name, void); |
| 423 | } |
| 424 | // A `tbody` or a `tfoot` groups rows and says nothing else, so it is erased and the |
| 425 | // rows within it are read where they stand. |
| 426 | } |
| 427 | } |
| 428 | } |
| 429 | Ok(Block::Table { head, rows, cols }) |
| 430 | } |
| 431 | |
| 432 | /// Reads a `tr` into the row it is, filling in the alignment its cells declare for their columns. |
| 433 | /// |
| 434 | /// Whether every cell was a `th` comes back with the row, because a first row of nothing but `th` is a |
| 435 | /// table naming its columns whether or not anyone wrapped it in a `thead`. |
| 436 | fn row(lex: &mut Lex, cols: &mut Vec<Align>, depth: usize) -> Outcome<(Row, bool)> { |
| 437 | let mut cells: Vec<Cell> = Vec::new(); |
| 438 | let mut all_th = true; |
| 439 | while let Some(tok) = lex.next() { |
| 440 | match tok { |
| 441 | Tok::Text(_) => {} |
| 442 | Tok::Close(name) => { |
| 443 | if is(name, "tr") { |
| 444 | break; |
| 445 | } |
| 446 | } |
| 447 | Tok::Open { name, attrs, void } => { |
| 448 | let th = is(name, "th"); |
| 449 | if th || is(name, "td") { |
| 450 | if !th { |
| 451 | all_th = false; |
| 452 | } |
| 453 | let i = cells.len(); |
| 454 | while cols.len() <= i { |
| 455 | cols.push(Align::None); |
| 456 | } |
| 457 | // A column takes its alignment from the first cell that declares one. |
| 458 | if cols[i] == Align::None { |
| 459 | cols[i] = align_of(attrs); |
| 460 | } |
| 461 | cells.push(Cell(match void { |
| 462 | true => Vec::new(), |
| 463 | false => res!(inlines(lex, name, depth + 1)), |
| 464 | })); |
| 465 | } else if kind(name) == Kind::Drop { |
| 466 | skip(lex, name, void); |
| 467 | } |
| 468 | } |
| 469 | } |
| 470 | } |
| 471 | // A row of no cells names no columns, whatever its cells were not. |
| 472 | let named = all_th && !cells.is_empty(); |
| 473 | Ok((Row(cells), named)) |
| 474 | } |
| 475 | |
| 476 | /// Reads a `pre` into the run of code it holds. |
| 477 | fn code(lex: &mut Lex, attrs: &str, void: bool) -> Block { |
| 478 | // A `pre` may name the language on itself, or on the `code` within it, which is where the |
| 479 | // convention every highlighter reads puts it. |
| 480 | let mut lang = lang_of(attrs); |
| 481 | let mut text = String::new(); |
| 482 | if !void { |
| 483 | while let Some(tok) = lex.next() { |
| 484 | match tok { |
| 485 | // Not collapsed, and not trimmed: within a `pre` the whitespace is the content, which |
| 486 | // is the whole reason the element exists. |
| 487 | Tok::Text(t) => text.push_str(&decode_entities(t)), |
| 488 | Tok::Close(name) => { |
| 489 | if is(name, "pre") { |
| 490 | break; |
| 491 | } |
| 492 | } |
| 493 | Tok::Open { name, attrs, void } => { |
| 494 | if kind(name) == Kind::Drop { |
| 495 | skip(lex, name, void); |
| 496 | } else if is(name, "code") && lang.is_none() { |
| 497 | lang = lang_of(attrs); |
| 498 | } |
| 499 | // Every other tag within is decoration -- a highlighter's own markup around a |
| 500 | // keyword -- and what the block says is the text under it. |
| 501 | } |
| 502 | } |
| 503 | } |
| 504 | } |
| 505 | // A line ending directly after the opening tag is the tag's own and not the code's. HTML's own |
| 506 | // parser drops it, and a reader that kept it would grow a blank first line on every block written |
| 507 | // the way most are. |
| 508 | let text = match text.strip_prefix("\r\n") { |
| 509 | Some(rest) => rest.to_string(), |
| 510 | None => match text.strip_prefix('\n') { |
| 511 | Some(rest) => rest.to_string(), |
| 512 | None => text, |
| 513 | }, |
| 514 | }; |
| 515 | Block::Code { lang, text } |
| 516 | } |
| 517 | |
| 518 | /// The text of an element and everything within it, its entities decoded and its whitespace left as it |
| 519 | /// was written. |
| 520 | fn text_in(lex: &mut Lex, until: &str) -> String { |
| 521 | let mut out = String::new(); |
| 522 | while let Some(tok) = lex.next() { |
| 523 | match tok { |
| 524 | Tok::Text(t) => out.push_str(&decode_entities(t)), |
| 525 | Tok::Close(name) => { |
| 526 | if is(name, until) { |
| 527 | break; |
| 528 | } |
| 529 | } |
| 530 | Tok::Open { name, void, .. } => { |
| 531 | if kind(name) == Kind::Drop { |
| 532 | skip(lex, name, void); |
| 533 | } |
| 534 | } |
| 535 | } |
| 536 | } |
| 537 | out |
| 538 | } |
| 539 | |
| 540 | /// Skips an element and everything within it, tags and all. |
| 541 | /// |
| 542 | /// The skip counts its own element's tags rather than recursing, so a document built to nest deeply |
| 543 | /// costs nothing here. |
| 544 | fn skip(lex: &mut Lex, name: &str, void: bool) { |
| 545 | if void { |
| 546 | return; |
| 547 | } |
| 548 | let mut depth = 0usize; |
| 549 | while let Some(tok) = lex.next() { |
| 550 | match tok { |
| 551 | Tok::Text(_) => {} |
| 552 | Tok::Open { name: n, void: v, .. } => { |
| 553 | if !v && is(n, name) { |
| 554 | depth += 1; |
| 555 | } |
| 556 | } |
| 557 | Tok::Close(n) => { |
| 558 | if is(n, name) { |
| 559 | if depth == 0 { |
| 560 | return; |
| 561 | } |
| 562 | depth -= 1; |
| 563 | } |
| 564 | } |
| 565 | } |
| 566 | } |
| 567 | } |
| 568 | |
| 569 | // ----------------------------------------------------------------------------------------------- |
| 570 | // Whitespace. |
| 571 | // ----------------------------------------------------------------------------------------------- |
| 572 | |
| 573 | /// A run of inline content, gathered with HTML's whitespace rules applied as it grows. |
| 574 | /// |
| 575 | /// This is where the rule the module documentation states is actually kept: a run of whitespace says |
| 576 | /// one space, a space at the start of a run says nothing, and the space at the end is taken off when |
| 577 | /// the run closes. Only a `pre` escapes it, and a `pre` is read by [`code`] and never reaches here. |
| 578 | #[derive(Default)] |
| 579 | struct Run { |
| 580 | /// The inlines gathered so far. |
| 581 | out: Vec<Inline>, |
| 582 | } |
| 583 | |
| 584 | impl Run { |
| 585 | |
| 586 | /// Whether a space would say nothing here: at the start of the run, where a space has been said |
| 587 | /// already, or after a break, there is nothing for one to separate. |
| 588 | fn open(&self) -> bool { |
| 589 | match self.out.last() { |
| 590 | None => true, |
| 591 | Some(Inline::Break) => true, |
| 592 | Some(Inline::Text(t)) => t.ends_with(' '), |
| 593 | Some(_) => false, |
| 594 | } |
| 595 | } |
| 596 | |
| 597 | /// Adds a run of text, its entities decoded and its whitespace collapsed. |
| 598 | fn text(&mut self, raw: &str) { |
| 599 | // Decoded first and collapsed second, which is the order that matters: a ` ` says a |
| 600 | // newline, and a newline is whitespace like any other. Only a no-break space survives, and it |
| 601 | // survives because it is not whitespace this collapses. |
| 602 | let s = collapse(&decode_entities(raw)); |
| 603 | let s = match self.open() { |
| 604 | true => s.strip_prefix(' ').unwrap_or(&s), |
| 605 | false => s.as_str(), |
| 606 | }; |
| 607 | if s.is_empty() { |
| 608 | return; |
| 609 | } |
| 610 | self.push(Inline::Text(s.to_string())); |
| 611 | } |
| 612 | |
| 613 | /// Says the space an erased block stands for, where the run does not say one already. |
| 614 | fn space(&mut self) { |
| 615 | if !self.open() { |
| 616 | self.push(Inline::Text(" ".to_string())); |
| 617 | } |
| 618 | } |
| 619 | |
| 620 | /// Adds a settled inline, joining it to the run before it where both are text. |
| 621 | fn push(&mut self, item: Inline) { |
| 622 | if let Inline::Text(t) = &item { |
| 623 | if let Some(Inline::Text(last)) = self.out.last_mut() { |
| 624 | last.push_str(t); |
| 625 | return; |
| 626 | } |
| 627 | } |
| 628 | self.out.push(item); |
| 629 | } |
| 630 | |
| 631 | /// The run, with the trailing space that HTML does not say taken off it. |
| 632 | fn end(mut self) -> Vec<Inline> { |
| 633 | if let Some(Inline::Text(t)) = self.out.last_mut() { |
| 634 | while t.ends_with(' ') { |
| 635 | t.pop(); |
| 636 | } |
| 637 | if t.is_empty() { |
| 638 | self.out.pop(); |
| 639 | } |
| 640 | } |
| 641 | self.out |
| 642 | } |
| 643 | } |
| 644 | |
| 645 | /// Whether a character is whitespace that HTML collapses. |
| 646 | /// |
| 647 | /// A no-break space is deliberately not among them. It is not this whitespace, it does not collapse, |
| 648 | /// and an author who wrote one meant it. |
| 649 | fn is_ws(c: char) -> bool { |
| 650 | matches!(c, ' ' | '\t' | '\n' | '\r' | '\u{c}') |
| 651 | } |
| 652 | |
| 653 | /// Collapses every run of whitespace in a run of text to the single space it says. |
| 654 | fn collapse(s: &str) -> String { |
| 655 | let mut out = String::with_capacity(s.len()); |
| 656 | let mut ws = false; // Whether whitespace has been passed over since the last character kept. |
| 657 | for c in s.chars() { |
| 658 | if is_ws(c) { |
| 659 | ws = true; |
| 660 | } else { |
| 661 | if ws { |
| 662 | out.push(' '); |
| 663 | ws = false; |
| 664 | } |
| 665 | out.push(c); |
| 666 | } |
| 667 | } |
| 668 | if ws { |
| 669 | out.push(' '); |
| 670 | } |
| 671 | out |
| 672 | } |
| 673 | |
| 674 | // ----------------------------------------------------------------------------------------------- |
| 675 | // Attributes. |
| 676 | // ----------------------------------------------------------------------------------------------- |
| 677 | |
| 678 | /// The value of one attribute from a tag's unparsed run of them, where the tag carries it. |
| 679 | /// |
| 680 | /// Names are matched without regard to case, and a value is taken from double quotes, single quotes or |
| 681 | /// no quotes at all, because all three are HTML and a generator picks whichever it likes. An attribute |
| 682 | /// written with no value at all is present, and says the empty string. What comes back has its entities |
| 683 | /// decoded, so a destination written `a&b` is read as the `a&b` it names. |
| 684 | fn attr(attrs: &str, want: &str) -> Option<String> { |
| 685 | let b = attrs.as_bytes(); |
| 686 | let mut i = 0; |
| 687 | while i < b.len() { |
| 688 | while i < b.len() && (b[i].is_ascii_whitespace() || b[i] == b'/') { |
| 689 | i += 1; |
| 690 | } |
| 691 | let ns = i; |
| 692 | while i < b.len() && !b[i].is_ascii_whitespace() && b[i] != b'=' && b[i] != b'/' { |
| 693 | i += 1; |
| 694 | } |
| 695 | let name = &attrs[ns..i]; |
| 696 | while i < b.len() && b[i].is_ascii_whitespace() { |
| 697 | i += 1; |
| 698 | } |
| 699 | let mut val = ""; |
| 700 | if i < b.len() && b[i] == b'=' { |
| 701 | i += 1; |
| 702 | while i < b.len() && b[i].is_ascii_whitespace() { |
| 703 | i += 1; |
| 704 | } |
| 705 | if i < b.len() && (b[i] == b'"' || b[i] == b'\'') { |
| 706 | let q = b[i]; |
| 707 | i += 1; |
| 708 | let vs = i; |
| 709 | while i < b.len() && b[i] != q { |
| 710 | i += 1; |
| 711 | } |
| 712 | val = &attrs[vs..i]; |
| 713 | if i < b.len() { |
| 714 | i += 1; |
| 715 | } |
| 716 | } else { |
| 717 | let vs = i; |
| 718 | while i < b.len() && !b[i].is_ascii_whitespace() { |
| 719 | i += 1; |
| 720 | } |
| 721 | val = &attrs[vs..i]; |
| 722 | } |
| 723 | } |
| 724 | if !name.is_empty() && name.eq_ignore_ascii_case(want) { |
| 725 | return Some(decode_entities(val)); |
| 726 | } |
| 727 | if name.is_empty() && val.is_empty() { |
| 728 | // Nothing was read, so nothing more will be: this is the end of the run. |
| 729 | break; |
| 730 | } |
| 731 | } |
| 732 | None |
| 733 | } |
| 734 | |
| 735 | /// The language a `class="language-x"` names, where the class names one. |
| 736 | fn lang_of(attrs: &str) -> Option<String> { |
| 737 | let class = match attr(attrs, "class") { |
| 738 | Some(c) => c, |
| 739 | None => return None, |
| 740 | }; |
| 741 | for word in class.split_ascii_whitespace() { |
| 742 | if let Some(lang) = word.strip_prefix("language-") { |
| 743 | if !lang.is_empty() { |
| 744 | return Some(lang.to_string()); |
| 745 | } |
| 746 | } |
| 747 | } |
| 748 | None |
| 749 | } |
| 750 | |
| 751 | /// The alignment a cell declares, where it declares one this can honour. |
| 752 | /// |
| 753 | /// Only the logical keywords are read: `start`, `end`, and `center`, which is unambiguous. CSS's |
| 754 | /// `left` and `right` are deliberately not mapped, for the reason [`Align`] gives at length -- the |
| 755 | /// tree does not know which way its text runs, so it cannot know which side `left` is on. A cell that |
| 756 | /// says `left` is read as a cell that says nothing, and the consumer's own default stands, which is |
| 757 | /// wrong for nobody. Mapping it would be wrong for half the world's prose, and silently. |
| 758 | fn align_of(attrs: &str) -> Align { |
| 759 | let style = match attr(attrs, "style") { |
| 760 | Some(s) => s, |
| 761 | None => return Align::None, |
| 762 | }; |
| 763 | // The lowered copy is only used to find the property, and an ASCII lowering does not move a byte, |
| 764 | // so the index it gives is an index into the original. |
| 765 | let at = match style.to_ascii_lowercase().find("text-align") { |
| 766 | Some(i) => i + "text-align".len(), |
| 767 | None => return Align::None, |
| 768 | }; |
| 769 | let val = match style[at..].trim_start().strip_prefix(':') { |
| 770 | Some(v) => v, |
| 771 | None => return Align::None, |
| 772 | }; |
| 773 | let val = val.split(';').next().unwrap_or("").trim(); |
| 774 | match val.to_ascii_lowercase().as_str() { |
| 775 | "start" => Align::Start, |
| 776 | "center" => Align::Centre, |
| 777 | "end" => Align::End, |
| 778 | _ => Align::None, |
| 779 | } |
| 780 | } |
| 781 | |
| 782 | // ----------------------------------------------------------------------------------------------- |
| 783 | // The lexer. |
| 784 | // ----------------------------------------------------------------------------------------------- |
| 785 | |
| 786 | /// One thing the reader takes from the source: a tag, or the text between tags. |
| 787 | enum Tok<'a> { |
| 788 | /// An opening tag. |
| 789 | Open { |
| 790 | /// The element's name, in whatever case it was written. |
| 791 | name: &'a str, |
| 792 | /// The attributes, unparsed: [`attr`] reads one out where something wants it. |
| 793 | attrs: &'a str, |
| 794 | /// Whether the tag holds nothing, either by being a void element or by closing itself. |
| 795 | void: bool, |
| 796 | }, |
| 797 | /// A closing tag, by its name. |
| 798 | Close(&'a str), |
| 799 | /// The text between two tags, its entities undecoded and its whitespace uncollapsed. |
| 800 | Text(&'a str), |
| 801 | } |
| 802 | |
| 803 | /// The source, and how far into it the reader has come. |
| 804 | struct Lex<'a> { |
| 805 | /// The HTML being read. |
| 806 | src: &'a str, |
| 807 | /// Where the next token begins. |
| 808 | i: usize, |
| 809 | /// The raw text element being read, where one is. |
| 810 | raw: Option<&'a str>, |
| 811 | } |
| 812 | |
| 813 | impl<'a> Lex<'a> { |
| 814 | |
| 815 | /// The next token, or nothing where the source is spent. |
| 816 | /// |
| 817 | /// Comments, doctypes and processing instructions are passed over here rather than being handed on, |
| 818 | /// because there is nothing above this that would do anything with them but drop them. |
| 819 | fn next(&mut self) -> Option<Tok<'a>> { |
| 820 | let b = self.src.as_bytes(); |
| 821 | loop { |
| 822 | if self.i >= b.len() { |
| 823 | return None; |
| 824 | } |
| 825 | // Within a script or a stylesheet everything up to the closing tag is text, so a `<` in |
| 826 | // the code opens nothing. The content is dropped above, but it must be read as what it is |
| 827 | // or a comparison in a script would be read as an element. |
| 828 | if let Some(name) = self.raw.take() { |
| 829 | let end = self.raw_end(name); |
| 830 | let t = &self.src[self.i..end]; |
| 831 | self.i = end; |
| 832 | return Some(Tok::Text(t)); |
| 833 | } |
| 834 | if b[self.i] == b'<' && opens_tag(b, self.i) { |
| 835 | if self.src[self.i..].starts_with("<!--") { |
| 836 | // A comment says nothing to a document tree. |
| 837 | self.i = match self.src[self.i..].find("-->") { |
| 838 | Some(k) => self.i + k + 3, |
| 839 | None => b.len(), |
| 840 | }; |
| 841 | continue; |
| 842 | } |
| 843 | if b[self.i + 1] == b'!' || b[self.i + 1] == b'?' { |
| 844 | // A doctype or a processing instruction says nothing either. |
| 845 | self.i = self.to_gt(self.i + 1); |
| 846 | continue; |
| 847 | } |
| 848 | if b[self.i + 1] == b'/' { |
| 849 | let ns = self.i + 2; |
| 850 | let ne = name_end(b, ns); |
| 851 | let name = &self.src[ns..ne]; |
| 852 | self.i = self.to_gt(ne); |
| 853 | return Some(Tok::Close(name)); |
| 854 | } |
| 855 | let ns = self.i + 1; |
| 856 | let ne = name_end(b, ns); |
| 857 | let name = &self.src[ns..ne]; |
| 858 | let (attrs, end) = self.attrs_of(ne); |
| 859 | self.i = end; |
| 860 | // A trailing slash closes the tag itself; a void element closes itself whether or not |
| 861 | // anyone wrote one. |
| 862 | let slash = attrs.ends_with('/'); |
| 863 | let attrs = match slash { |
| 864 | true => &attrs[..attrs.len() - 1], |
| 865 | false => attrs, |
| 866 | }; |
| 867 | let void = slash || VOID.iter().any(|v| is(name, v)); |
| 868 | if !void && RAW.iter().any(|r| is(name, r)) { |
| 869 | self.raw = Some(name); |
| 870 | } |
| 871 | return Some(Tok::Open { name, attrs, void }); |
| 872 | } |
| 873 | // Text, up to the next tag. A `<` that opens nothing -- a less-than in prose that nobody |
| 874 | // escaped -- is text like any other, so the scan steps over it and carries on. |
| 875 | let mut j = self.i; |
| 876 | let end = loop { |
| 877 | match self.src[j..].find('<') { |
| 878 | None => break b.len(), |
| 879 | Some(k) => { |
| 880 | let at = j + k; |
| 881 | if opens_tag(b, at) { |
| 882 | break at; |
| 883 | } |
| 884 | j = at + 1; |
| 885 | } |
| 886 | } |
| 887 | }; |
| 888 | let t = &self.src[self.i..end]; |
| 889 | self.i = end; |
| 890 | return Some(Tok::Text(t)); |
| 891 | } |
| 892 | } |
| 893 | |
| 894 | /// Where the current raw text element's content ends: at its closing tag, or at the end of the |
| 895 | /// source where it has none. |
| 896 | fn raw_end(&self, name: &str) -> usize { |
| 897 | let mut j = self.i; |
| 898 | loop { |
| 899 | match self.src[j..].find('<') { |
| 900 | None => return self.src.len(), |
| 901 | Some(k) => { |
| 902 | let at = j + k; |
| 903 | let rest = &self.src[at..]; |
| 904 | if rest.starts_with("</") && rest[2..].to_ascii_lowercase().starts_with(name) { |
| 905 | return at; |
| 906 | } |
| 907 | j = at + 1; |
| 908 | } |
| 909 | } |
| 910 | } |
| 911 | } |
| 912 | |
| 913 | /// The run of attributes a tag carries, and where the tag ends. |
| 914 | fn attrs_of(&self, from: usize) -> (&'a str, usize) { |
| 915 | let end = self.to_gt(from); |
| 916 | // `to_gt` steps past the `>`, which is not the tag's to give away. |
| 917 | let close = match end > from && self.src.as_bytes()[end - 1] == b'>' { |
| 918 | true => end - 1, |
| 919 | false => end, |
| 920 | }; |
| 921 | (self.src[from..close].trim(), end) |
| 922 | } |
| 923 | |
| 924 | /// Where the tag that is being read ends: just past its `>`, or at the end of the source where it |
| 925 | /// has none. A `>` within a quoted value ends nothing. |
| 926 | fn to_gt(&self, from: usize) -> usize { |
| 927 | let b = self.src.as_bytes(); |
| 928 | let mut k = from; |
| 929 | let mut q = 0u8; // The quote mark a value is sitting in, or zero for none. |
| 930 | while k < b.len() { |
| 931 | let c = b[k]; |
| 932 | if q != 0 { |
| 933 | if c == q { |
| 934 | q = 0; |
| 935 | } |
| 936 | } else if c == b'"' || c == b'\'' { |
| 937 | q = c; |
| 938 | } else if c == b'>' { |
| 939 | return k + 1; |
| 940 | } |
| 941 | k += 1; |
| 942 | } |
| 943 | b.len() |
| 944 | } |
| 945 | } |
| 946 | |
| 947 | /// Whether the `<` at the given index opens a tag, rather than being a less-than nobody escaped. |
| 948 | fn opens_tag(b: &[u8], i: usize) -> bool { |
| 949 | match b.get(i + 1) { |
| 950 | Some(c) => c.is_ascii_alphabetic() || *c == b'!' || *c == b'/' || *c == b'?', |
| 951 | None => false, |
| 952 | } |
| 953 | } |
| 954 | |
| 955 | /// Where the element name beginning at the given index ends. |
| 956 | fn name_end(b: &[u8], from: usize) -> usize { |
| 957 | let mut k = from; |
| 958 | while k < b.len() && (b[k].is_ascii_alphanumeric() || b[k] == b'-' || b[k] == b':') { |
| 959 | k += 1; |
| 960 | } |
| 961 | k |
| 962 | } |
| 963 | |
| 964 | #[cfg(test)] |
| 965 | mod tests { |
| 966 | use super::*; |
| 967 | |
| 968 | use crate::doc::text_of; |
| 969 | |
| 970 | /// A run of literal text, for the tests that expect one. |
| 971 | fn t(s: &str) -> Inline { |
| 972 | Inline::Text(s.to_string()) |
| 973 | } |
| 974 | |
| 975 | /// The text of a block's inlines, for tests that care what a block says and not how. |
| 976 | fn said(blocks: &[Block]) -> Vec<String> { |
| 977 | blocks.iter().map(|b| match b { |
| 978 | Block::Para(c) => text_of(c), |
| 979 | Block::Heading { content, .. } => text_of(content), |
| 980 | Block::Code { text, .. } => text.clone(), |
| 981 | _ => String::new(), |
| 982 | }).collect() |
| 983 | } |
| 984 | |
| 985 | /// What a table's rows say, cell by cell, the header first. |
| 986 | fn grid(b: &Block) -> Vec<Vec<String>> { |
| 987 | match b { |
| 988 | Block::Table { head, rows, .. } => { |
| 989 | let mut out = Vec::new(); |
| 990 | if let Some(head) = head { |
| 991 | out.push(head.0.iter().map(|c| c.text_of()).collect()); |
| 992 | } |
| 993 | for row in rows { |
| 994 | out.push(row.0.iter().map(|c| c.text_of()).collect()); |
| 995 | } |
| 996 | out |
| 997 | } |
| 998 | other => panic!("expected a table, got {:?}", other), |
| 999 | } |
| 1000 | } |
| 1001 | |
| 1002 | /// The six headings reach the six levels, whatever case they were written in. |
| 1003 | #[test] |
| 1004 | fn test_the_headings_give_their_levels_00() -> Outcome<()> { |
| 1005 | let b = res!(parse("<h1>One</h1><h3>Three</h3><H6>Six</H6>")); |
| 1006 | assert_eq!(b, vec![ |
| 1007 | Block::Heading { level: 1, content: vec![t("One")] }, |
| 1008 | Block::Heading { level: 3, content: vec![t("Three")] }, |
| 1009 | Block::Heading { level: 6, content: vec![t("Six")] }, |
| 1010 | ]); |
| 1011 | Ok(()) |
| 1012 | } |
| 1013 | |
| 1014 | /// A `p` is a paragraph, and the blank space an exporter laid it out with is not part of it. |
| 1015 | #[test] |
| 1016 | fn test_a_paragraph_is_a_paragraph_01() -> Outcome<()> { |
| 1017 | let b = res!(parse("<p>One.</p>\n\n<p>Two.</p>\n")); |
| 1018 | assert_eq!(b, vec![Block::Para(vec![t("One.")]), Block::Para(vec![t("Two.")])]); |
| 1019 | Ok(()) |
| 1020 | } |
| 1021 | |
| 1022 | /// THE RULE. A run of spaces, tabs and newlines between two words says one space, whatever the |
| 1023 | /// exporter wrote. A reader that kept them would freeze a book at the width it was exported at. |
| 1024 | #[test] |
| 1025 | fn test_a_run_of_whitespace_says_one_space_02() -> Outcome<()> { |
| 1026 | // A newline the exporter's line wrapping put there. |
| 1027 | assert_eq!(said(&res!(parse("<p>One line\nand its continuation.</p>"))), |
| 1028 | vec!["One line and its continuation."]); |
| 1029 | // Indentation, on its own line, as an exporter lays a document out. |
| 1030 | assert_eq!(said(&res!(parse("<p>\n\tOne line\n\tand its continuation.\n</p>"))), |
| 1031 | vec!["One line and its continuation."]); |
| 1032 | // Spaces, tabs and newlines together, in a run of any length. |
| 1033 | assert_eq!(said(&res!(parse("<p>a \t \n\r\n b</p>"))), vec!["a b"]); |
| 1034 | // And the break the exporter's wrapping made is never a break the author asked for. |
| 1035 | let b = res!(parse("<p>One line\nand its continuation.</p>")); |
| 1036 | assert_eq!(b, vec![Block::Para(vec![t("One line and its continuation.")])]); |
| 1037 | Ok(()) |
| 1038 | } |
| 1039 | |
| 1040 | /// The whitespace at either end of a block does not survive it. |
| 1041 | #[test] |
| 1042 | fn test_a_block_does_not_keep_the_space_at_its_ends_03() -> Outcome<()> { |
| 1043 | assert_eq!(said(&res!(parse("<p> padded </p>"))), vec!["padded"]); |
| 1044 | assert_eq!(said(&res!(parse("<h2>\n A Heading\n</h2>"))), vec!["A Heading"]); |
| 1045 | // A paragraph of nothing but whitespace says nothing, and is not a paragraph. |
| 1046 | assert_eq!(res!(parse("<p> \n </p>")), Vec::<Block>::new()); |
| 1047 | assert_eq!(res!(parse("<p></p>")), Vec::<Block>::new()); |
| 1048 | Ok(()) |
| 1049 | } |
| 1050 | |
| 1051 | /// The space between two inlines is a space, and the space at the start of a run is not. |
| 1052 | #[test] |
| 1053 | fn test_whitespace_around_an_inline_collapses_04() -> Outcome<()> { |
| 1054 | // A newline between two emphasised words says the space that divides them. |
| 1055 | let b = res!(parse("<p><em>a</em>\n <em>b</em></p>")); |
| 1056 | assert_eq!(b, vec![Block::Para(vec![ |
| 1057 | Inline::Emph { strong: false, content: vec![t("a")] }, |
| 1058 | t(" "), |
| 1059 | Inline::Emph { strong: false, content: vec![t("b")] }, |
| 1060 | ])]); |
| 1061 | // The whitespace an exporter put before the first inline says nothing. |
| 1062 | let b = res!(parse("<p>\n <em>a</em> b\n</p>")); |
| 1063 | assert_eq!(b, vec![Block::Para(vec![ |
| 1064 | Inline::Emph { strong: false, content: vec![t("a")] }, |
| 1065 | t(" b"), |
| 1066 | ])]); |
| 1067 | Ok(()) |
| 1068 | } |
| 1069 | |
| 1070 | /// A `pre` is the exception the rule is written around: its whitespace is what it says. |
| 1071 | #[test] |
| 1072 | fn test_a_pre_keeps_its_whitespace_exactly_05() -> Outcome<()> { |
| 1073 | let b = res!(parse("<pre>fn main() {\n\tlet x = 1;\n}\n</pre>")); |
| 1074 | assert_eq!(b, vec![Block::Code { |
| 1075 | lang: None, |
| 1076 | text: "fn main() {\n\tlet x = 1;\n}\n".to_string(), |
| 1077 | }]); |
| 1078 | Ok(()) |
| 1079 | } |
| 1080 | |
| 1081 | /// A `pre` names its language by the class every highlighter reads, on the `pre` or on the `code`. |
| 1082 | #[test] |
| 1083 | fn test_a_code_block_names_its_language_06() -> Outcome<()> { |
| 1084 | let b = res!(parse("<pre><code class=\"language-rust\">let x = 1 < 2;\n</code></pre>")); |
| 1085 | assert_eq!(b, vec![Block::Code { |
| 1086 | lang: Some("rust".to_string()), |
| 1087 | text: "let x = 1 < 2;\n".to_string(), |
| 1088 | }]); |
| 1089 | // On the `pre` itself, and beside other classes. |
| 1090 | let b = res!(parse("<pre class=\"highlight language-c\">int x;</pre>")); |
| 1091 | assert_eq!(b, vec![Block::Code { lang: Some("c".to_string()), text: "int x;".to_string() }]); |
| 1092 | // And a block that names none says none. |
| 1093 | let b = res!(parse("<pre><code>plain</code></pre>")); |
| 1094 | assert_eq!(b, vec![Block::Code { lang: None, text: "plain".to_string() }]); |
| 1095 | Ok(()) |
| 1096 | } |
| 1097 | |
| 1098 | /// A line ending directly after the opening tag is the tag's own, and does not become a blank first |
| 1099 | /// line of code. |
| 1100 | #[test] |
| 1101 | fn test_a_pre_drops_the_line_ending_that_opens_it_07() -> Outcome<()> { |
| 1102 | let b = res!(parse("<pre><code>\nfirst\nsecond\n</code></pre>")); |
| 1103 | assert_eq!(b, vec![Block::Code { lang: None, text: "first\nsecond\n".to_string() }]); |
| 1104 | Ok(()) |
| 1105 | } |
| 1106 | |
| 1107 | /// A `blockquote` holds blocks, and nests. |
| 1108 | #[test] |
| 1109 | fn test_a_quotation_holds_blocks_08() -> Outcome<()> { |
| 1110 | let b = res!(parse("<blockquote><p>One.</p><p>Two.</p></blockquote>")); |
| 1111 | assert_eq!(b, vec![Block::Quote(vec![ |
| 1112 | Block::Para(vec![t("One.")]), |
| 1113 | Block::Para(vec![t("Two.")]), |
| 1114 | ])]); |
| 1115 | let b = res!(parse("<blockquote><blockquote><p>Deep.</p></blockquote></blockquote>")); |
| 1116 | assert_eq!(b, vec![Block::Quote(vec![Block::Quote(vec![Block::Para(vec![t("Deep.")])])])]); |
| 1117 | Ok(()) |
| 1118 | } |
| 1119 | |
| 1120 | /// An `hr` is a thematic break. |
| 1121 | #[test] |
| 1122 | fn test_a_rule_is_a_rule_09() -> Outcome<()> { |
| 1123 | assert_eq!(res!(parse("<p>a</p><hr><p>b</p>")), vec![ |
| 1124 | Block::Para(vec![t("a")]), |
| 1125 | Block::Rule, |
| 1126 | Block::Para(vec![t("b")]), |
| 1127 | ]); |
| 1128 | Ok(()) |
| 1129 | } |
| 1130 | |
| 1131 | /// A `ul` is an unordered list and an `ol` an ordered one, and an item holds the blocks it holds. |
| 1132 | #[test] |
| 1133 | fn test_a_list_is_ordered_or_not_10() -> Outcome<()> { |
| 1134 | let b = res!(parse("<ul>\n <li>one</li>\n <li>two</li>\n</ul>")); |
| 1135 | assert_eq!(b, vec![Block::List { |
| 1136 | ordered: false, |
| 1137 | items: vec![ |
| 1138 | vec![Block::Para(vec![t("one")])], |
| 1139 | vec![Block::Para(vec![t("two")])], |
| 1140 | ], |
| 1141 | }]); |
| 1142 | let b = res!(parse("<ol><li><p>one</p><p>still one</p></li></ol>")); |
| 1143 | assert_eq!(b, vec![Block::List { |
| 1144 | ordered: true, |
| 1145 | items: vec![vec![Block::Para(vec![t("one")]), Block::Para(vec![t("still one")])]], |
| 1146 | }]); |
| 1147 | Ok(()) |
| 1148 | } |
| 1149 | |
| 1150 | /// A list nests within an item of a list. |
| 1151 | #[test] |
| 1152 | fn test_a_list_nests_within_a_list_11() -> Outcome<()> { |
| 1153 | let b = res!(parse("<ul><li>one<ul><li>inner</li></ul></li><li>two</li></ul>")); |
| 1154 | let want = vec![Block::List { |
| 1155 | ordered: false, |
| 1156 | items: vec![ |
| 1157 | vec![ |
| 1158 | Block::Para(vec![t("one")]), |
| 1159 | Block::List { |
| 1160 | ordered: false, |
| 1161 | items: vec![vec![Block::Para(vec![t("inner")])]], |
| 1162 | }, |
| 1163 | ], |
| 1164 | vec![Block::Para(vec![t("two")])], |
| 1165 | ], |
| 1166 | }]; |
| 1167 | assert_eq!(b, want); |
| 1168 | Ok(()) |
| 1169 | } |
| 1170 | |
| 1171 | /// A table takes its header from a `thead`, and its cells reach the grid in order. |
| 1172 | #[test] |
| 1173 | fn test_a_table_names_its_columns_in_a_thead_12() -> Outcome<()> { |
| 1174 | let src = "<table>\n<thead>\n<tr><th>Name</th><th>Age</th></tr>\n</thead>\n\ |
| 1175 | <tbody>\n<tr><td>Alice</td><td>30</td></tr>\n<tr><td>Bob</td><td>4</td></tr>\n\ |
| 1176 | </tbody>\n</table>"; |
| 1177 | let b = res!(parse(src)); |
| 1178 | assert_eq!(b.len(), 1); |
| 1179 | assert_eq!(grid(&b[0]), vec![ |
| 1180 | vec!["Name", "Age"], |
| 1181 | vec!["Alice", "30"], |
| 1182 | vec!["Bob", "4"], |
| 1183 | ]); |
| 1184 | match &b[0] { |
| 1185 | Block::Table { head, rows, cols } => { |
| 1186 | assert!(head.is_some(), "the table lost its header row"); |
| 1187 | assert_eq!(rows.len(), 2); |
| 1188 | // A column nobody aligned is aligned by nothing. |
| 1189 | assert_eq!(cols, &vec![Align::None, Align::None]); |
| 1190 | } |
| 1191 | other => panic!("expected a table, got {:?}", other), |
| 1192 | } |
| 1193 | Ok(()) |
| 1194 | } |
| 1195 | |
| 1196 | /// A first row of nothing but `th` names the columns whether or not anyone wrapped it in a `thead`, |
| 1197 | /// and a table without one has no header at all. |
| 1198 | #[test] |
| 1199 | fn test_a_row_of_th_names_the_columns_13() -> Outcome<()> { |
| 1200 | let b = res!(parse("<table><tr><th>A</th><th>B</th></tr><tr><td>1</td><td>2</td></tr></table>")); |
| 1201 | assert_eq!(grid(&b[0]), vec![vec!["A", "B"], vec!["1", "2"]]); |
| 1202 | match &b[0] { |
| 1203 | Block::Table { head, rows, .. } => { |
| 1204 | assert!(head.is_some(), "a row of th did not name the columns"); |
| 1205 | assert_eq!(rows.len(), 1); |
| 1206 | } |
| 1207 | other => panic!("expected a table, got {:?}", other), |
| 1208 | } |
| 1209 | // A grid of figures is a table whether or not anything stands at the head of it. |
| 1210 | let b = res!(parse("<table><tr><td>1</td><td>2</td></tr></table>")); |
| 1211 | match &b[0] { |
| 1212 | Block::Table { head, rows, .. } => { |
| 1213 | assert!(head.is_none(), "a table with no header grew one"); |
| 1214 | assert_eq!(rows.len(), 1); |
| 1215 | } |
| 1216 | other => panic!("expected a table, got {:?}", other), |
| 1217 | } |
| 1218 | Ok(()) |
| 1219 | } |
| 1220 | |
| 1221 | /// A cell's alignment is read from the logical keywords only. `left` and `right` are not mapped, |
| 1222 | /// because the tree does not know which side they are on. |
| 1223 | #[test] |
| 1224 | fn test_a_column_aligns_by_logical_side_only_14() -> Outcome<()> { |
| 1225 | let src = "<table><tr>\ |
| 1226 | <td style=\"text-align: start\">a</td>\ |
| 1227 | <td style=\"text-align: center\">b</td>\ |
| 1228 | <td style=\"text-align: end\">c</td>\ |
| 1229 | <td style=\"text-align: left\">d</td>\ |
| 1230 | </tr></table>"; |
| 1231 | let b = res!(parse(src)); |
| 1232 | match &b[0] { |
| 1233 | Block::Table { cols, .. } => assert_eq!( |
| 1234 | cols, |
| 1235 | &vec![Align::Start, Align::Centre, Align::End, Align::None], |
| 1236 | ), |
| 1237 | other => panic!("expected a table, got {:?}", other), |
| 1238 | } |
| 1239 | Ok(()) |
| 1240 | } |
| 1241 | |
| 1242 | /// Emphasis is `em` or `i`, strong emphasis is `strong` or `b`, and they nest. |
| 1243 | #[test] |
| 1244 | fn test_emphasis_is_ordinary_or_strong_15() -> Outcome<()> { |
| 1245 | let b = res!(parse("<p><em>a</em> <i>b</i> <strong>c</strong> <b>d</b></p>")); |
| 1246 | assert_eq!(b, vec![Block::Para(vec![ |
| 1247 | Inline::Emph { strong: false, content: vec![t("a")] }, |
| 1248 | t(" "), |
| 1249 | Inline::Emph { strong: false, content: vec![t("b")] }, |
| 1250 | t(" "), |
| 1251 | Inline::Emph { strong: true, content: vec![t("c")] }, |
| 1252 | t(" "), |
| 1253 | Inline::Emph { strong: true, content: vec![t("d")] }, |
| 1254 | ])]); |
| 1255 | let b = res!(parse("<p><strong><em>both</em></strong></p>")); |
| 1256 | assert_eq!(b, vec![Block::Para(vec![Inline::Emph { |
| 1257 | strong: true, |
| 1258 | content: vec![Inline::Emph { strong: false, content: vec![t("both")] }], |
| 1259 | }])]); |
| 1260 | Ok(()) |
| 1261 | } |
| 1262 | |
| 1263 | /// An `a` with a destination is a link; one without is an anchor, and its words stay in the line. |
| 1264 | #[test] |
| 1265 | fn test_a_link_carries_its_destination_16() -> Outcome<()> { |
| 1266 | let b = res!(parse("<p>See <a href=\"https://example.com\">here</a> now.</p>")); |
| 1267 | assert_eq!(b, vec![Block::Para(vec![ |
| 1268 | t("See "), |
| 1269 | Inline::Link { to: "https://example.com".to_string(), content: vec![t("here")] }, |
| 1270 | t(" now."), |
| 1271 | ])]); |
| 1272 | // An anchor is not a link: there is nowhere for a reader to go, so only the tag is lost. |
| 1273 | let b = res!(parse("<p>An <a name=\"x\">anchor</a> here.</p>")); |
| 1274 | assert_eq!(b, vec![Block::Para(vec![t("An anchor here.")])]); |
| 1275 | Ok(()) |
| 1276 | } |
| 1277 | |
| 1278 | /// An `img` is an image, by its source and the text that stands for it. |
| 1279 | #[test] |
| 1280 | fn test_an_image_carries_its_source_and_alt_17() -> Outcome<()> { |
| 1281 | let b = res!(parse("<p><img src=\"fig.png\" alt=\"a figure\"></p>")); |
| 1282 | assert_eq!(b, vec![Block::Para(vec![Inline::Image { |
| 1283 | src: "fig.png".to_string(), |
| 1284 | alt: "a figure".to_string(), |
| 1285 | }])]); |
| 1286 | // An image that stands for nothing says nothing, and is still an image. |
| 1287 | let b = res!(parse("<p><img src=\"fig.png\"></p>")); |
| 1288 | assert_eq!(b, vec![Block::Para(vec![Inline::Image { |
| 1289 | src: "fig.png".to_string(), |
| 1290 | alt: String::new(), |
| 1291 | }])]); |
| 1292 | Ok(()) |
| 1293 | } |
| 1294 | |
| 1295 | /// A `code` within a line is a code span, and a `br` is the one thing that makes a break. |
| 1296 | #[test] |
| 1297 | fn test_a_code_span_and_a_break_18() -> Outcome<()> { |
| 1298 | let b = res!(parse("<p>Call <code>inline()</code> now.</p>")); |
| 1299 | assert_eq!(b, vec![Block::Para(vec![ |
| 1300 | t("Call "), |
| 1301 | Inline::Code("inline()".to_string()), |
| 1302 | t(" now."), |
| 1303 | ])]); |
| 1304 | let b = res!(parse("<p>one<br>two</p>")); |
| 1305 | assert_eq!(b, vec![Block::Para(vec![t("one"), Inline::Break, t("two")])]); |
| 1306 | Ok(()) |
| 1307 | } |
| 1308 | |
| 1309 | /// A break comes from a `br` and from nothing else. A newline in the source is not one. |
| 1310 | #[test] |
| 1311 | fn test_only_a_br_makes_a_break_19() -> Outcome<()> { |
| 1312 | let b = res!(parse("<p>one\ntwo\n\nthree</p>")); |
| 1313 | assert_eq!(b, vec![Block::Para(vec![t("one two three")])]); |
| 1314 | assert!(!b.iter().any(|blk| match blk { |
| 1315 | Block::Para(c) => c.contains(&Inline::Break), |
| 1316 | _ => false, |
| 1317 | }), "a newline became a break"); |
| 1318 | // And the whitespace after a break is the line's, not a word's. |
| 1319 | let b = res!(parse("<p>one<br>\n two</p>")); |
| 1320 | assert_eq!(b, vec![Block::Para(vec![t("one"), Inline::Break, t("two")])]); |
| 1321 | Ok(()) |
| 1322 | } |
| 1323 | |
| 1324 | /// Entities are decoded, named and numeric alike, and a decoded newline collapses like any other. |
| 1325 | #[test] |
| 1326 | fn test_entities_are_decoded_20() -> Outcome<()> { |
| 1327 | assert_eq!(said(&res!(parse("<p>Tom & Jerry <3</p>"))), vec!["Tom & Jerry <3"]); |
| 1328 | assert_eq!(said(&res!(parse("<p>it’s</p>"))), vec!["it\u{2019}s"]); |
| 1329 | assert_eq!(said(&res!(parse("<p>it’s</p>"))), vec!["it\u{2019}s"]); |
| 1330 | assert_eq!(said(&res!(parse("<p>a b</p>"))), vec!["a b"]); |
| 1331 | // An entity in an attribute is decoded too, so a destination says what it names. |
| 1332 | let b = res!(parse("<p><a href=\"?a=1&b=2\">x</a></p>")); |
| 1333 | assert_eq!(b, vec![Block::Para(vec![Inline::Link { |
| 1334 | to: "?a=1&b=2".to_string(), |
| 1335 | content: vec![t("x")], |
| 1336 | }])]); |
| 1337 | // A `&` that begins nothing is an ampersand. |
| 1338 | assert_eq!(said(&res!(parse("<p>a & b</p>"))), vec!["a & b"]); |
| 1339 | Ok(()) |
| 1340 | } |
| 1341 | |
| 1342 | /// A `div` and a `span` are unwrapped: the tag goes and the content stays exactly where it stood. |
| 1343 | #[test] |
| 1344 | fn test_a_div_and_a_span_are_unwrapped_21() -> Outcome<()> { |
| 1345 | // A div holding paragraphs is those paragraphs, at no depth of their own. |
| 1346 | let b = res!(parse("<div><div><p>one</p><p>two</p></div></div>")); |
| 1347 | assert_eq!(b, vec![Block::Para(vec![t("one")]), Block::Para(vec![t("two")])]); |
| 1348 | // A div holding bare words is the paragraph those words are. |
| 1349 | let b = res!(parse("<div><em>Just words.</em></div>")); |
| 1350 | assert_eq!(b, vec![Block::Para(vec![Inline::Emph { |
| 1351 | strong: false, |
| 1352 | content: vec![t("Just words.")], |
| 1353 | }])]); |
| 1354 | // A span in the middle of a sentence leaves the sentence whole across the hole it left. |
| 1355 | let b = res!(parse("<p>a <span>b</span> c</p>")); |
| 1356 | assert_eq!(b, vec![Block::Para(vec![t("a b c")])]); |
| 1357 | // The same at block level, where there is no paragraph to sit in. |
| 1358 | let b = res!(parse("<div>a <span>b</span> c</div>")); |
| 1359 | assert_eq!(b, vec![Block::Para(vec![t("a b c")])]); |
| 1360 | Ok(()) |
| 1361 | } |
| 1362 | |
| 1363 | /// An element the reader has never heard of is unwrapped, and the prose within it is kept. |
| 1364 | #[test] |
| 1365 | fn test_an_unknown_element_is_unwrapped_22() -> Outcome<()> { |
| 1366 | let b = res!(parse("<html><body><section><p>Kept.</p></section></body></html>")); |
| 1367 | assert_eq!(b, vec![Block::Para(vec![t("Kept.")])]); |
| 1368 | let b = res!(parse("<p>a <mark>b</mark> <custom-tag attr=\"x\">c</custom-tag> d</p>")); |
| 1369 | assert_eq!(said(&b), vec!["a b c d"]); |
| 1370 | // Even one carrying blocks, and one that stands where a block would. |
| 1371 | let b = res!(parse("<figure><figcaption>A caption.</figcaption></figure>")); |
| 1372 | assert_eq!(said(&b), vec!["A caption."]); |
| 1373 | Ok(()) |
| 1374 | } |
| 1375 | |
| 1376 | /// A script, a stylesheet, a head and a comment hold no prose, and go entirely. |
| 1377 | #[test] |
| 1378 | fn test_what_holds_no_prose_is_dropped_23() -> Outcome<()> { |
| 1379 | let src = "<html><head><title>Title</title><meta charset=\"utf-8\"></head>\ |
| 1380 | <body><script>if (a < b) { drop(); }</script>\ |
| 1381 | <style>p { colour: red; }</style>\ |
| 1382 | <!-- a comment, with <p>markup</p> in it -->\ |
| 1383 | <p>Kept.</p></body></html>"; |
| 1384 | assert_eq!(res!(parse(src)), vec![Block::Para(vec![t("Kept.")])]); |
| 1385 | // A `<` within a script opens nothing, so what follows it is not swallowed. |
| 1386 | let src = "<script>for (i = 0; i < n; i++) { x(); }</script><p>After.</p>"; |
| 1387 | assert_eq!(res!(parse(src)), vec![Block::Para(vec![t("After.")])]); |
| 1388 | Ok(()) |
| 1389 | } |
| 1390 | |
| 1391 | /// A void element holds nothing, whether or not anyone closed it, and however it was written. |
| 1392 | #[test] |
| 1393 | fn test_void_elements_close_themselves_24() -> Outcome<()> { |
| 1394 | let b = res!(parse("<p>a<br>b<br/>c<br />d</p>")); |
| 1395 | assert_eq!(b, vec![Block::Para(vec![ |
| 1396 | t("a"), Inline::Break, t("b"), Inline::Break, t("c"), Inline::Break, t("d"), |
| 1397 | ])]); |
| 1398 | assert_eq!(res!(parse("<hr><hr/>")), vec![Block::Rule, Block::Rule]); |
| 1399 | // A trailing slash is the tag's and not the attribute's. |
| 1400 | let b = res!(parse("<p><img src=\"a.png\"/></p>")); |
| 1401 | assert_eq!(b, vec![Block::Para(vec![Inline::Image { |
| 1402 | src: "a.png".to_string(), |
| 1403 | alt: String::new(), |
| 1404 | }])]); |
| 1405 | Ok(()) |
| 1406 | } |
| 1407 | |
| 1408 | /// An attribute's value is read from double quotes, single quotes or none at all, and its name is |
| 1409 | /// read whatever case it was written in. |
| 1410 | #[test] |
| 1411 | fn test_attributes_take_every_quoting_25() -> Outcome<()> { |
| 1412 | let want = vec![Block::Para(vec![Inline::Link { |
| 1413 | to: "x.html".to_string(), |
| 1414 | content: vec![t("go")], |
| 1415 | }])]; |
| 1416 | assert_eq!(res!(parse("<p><a href=\"x.html\">go</a></p>"), ), want); |
| 1417 | assert_eq!(res!(parse("<p><a href='x.html'>go</a></p>")), want); |
| 1418 | assert_eq!(res!(parse("<p><a href=x.html>go</a></p>")), want); |
| 1419 | assert_eq!(res!(parse("<p><A HREF = \"x.html\" >go</A></p>")), want); |
| 1420 | // A quote mark the other kind does not close a value, and a `>` within one ends no tag. |
| 1421 | let b = res!(parse("<p><a href=\"a'b>c\" title='say \"x\"'>go</a></p>")); |
| 1422 | assert_eq!(b, vec![Block::Para(vec![Inline::Link { |
| 1423 | to: "a'b>c".to_string(), |
| 1424 | content: vec![t("go")], |
| 1425 | }])]); |
| 1426 | Ok(()) |
| 1427 | } |
| 1428 | |
| 1429 | /// A close tag that answers nothing closes nothing, and the rest of the document survives it. |
| 1430 | #[test] |
| 1431 | fn test_a_stray_close_tag_loses_nothing_26() -> Outcome<()> { |
| 1432 | let b = res!(parse("<p>one</p></div></em></p><p>two</p>")); |
| 1433 | assert_eq!(said(&b), vec!["one", "two"]); |
| 1434 | let b = res!(parse("</p><h2>A Heading</h2><p>After.</p>")); |
| 1435 | assert_eq!(said(&b), vec!["A Heading", "After."]); |
| 1436 | Ok(()) |
| 1437 | } |
| 1438 | |
| 1439 | /// Nesting past the limit is refused, which is the reader's one refusal. |
| 1440 | #[test] |
| 1441 | fn test_nesting_past_the_limit_is_refused_27() -> Outcome<()> { |
| 1442 | // A quotation for every level the limit allows is read. |
| 1443 | let ok = format!("{}deep{}", |
| 1444 | "<blockquote>".repeat(DEPTH_LIMIT - 1), |
| 1445 | "</blockquote>".repeat(DEPTH_LIMIT - 1)); |
| 1446 | assert!(parse(&ok).is_ok()); |
| 1447 | // Past it, and past it by far, is not. |
| 1448 | let deep = format!("{}deep", "<blockquote>".repeat(DEPTH_LIMIT + 8)); |
| 1449 | assert!(parse(&deep).is_err()); |
| 1450 | let very = format!("{}deep", "<blockquote>".repeat(2000)); |
| 1451 | assert!(parse(&very).is_err()); |
| 1452 | // Inlines are held to the same limit. |
| 1453 | let very = format!("<p>{}deep", "<em>".repeat(2000)); |
| 1454 | assert!(parse(&very).is_err()); |
| 1455 | Ok(()) |
| 1456 | } |
| 1457 | |
| 1458 | /// An element the tree has no node for costs no stack, so a document built to exhaust one is read |
| 1459 | /// as the flat prose it says rather than refused. This is why the limit can be as low as it is. |
| 1460 | #[test] |
| 1461 | fn test_unwrapped_elements_cost_no_depth_28() -> Outcome<()> { |
| 1462 | let deep = format!("{}<p>Kept.</p>{}", "<div>".repeat(50_000), "</div>".repeat(50_000)); |
| 1463 | assert_eq!(res!(parse(&deep)), vec![Block::Para(vec![t("Kept.")])]); |
| 1464 | Ok(()) |
| 1465 | } |
| 1466 | |
| 1467 | /// An empty document is a document with nothing in it, and not a failure. |
| 1468 | #[test] |
| 1469 | fn test_an_empty_document_holds_nothing_29() -> Outcome<()> { |
| 1470 | assert_eq!(res!(parse("")), Vec::<Block>::new()); |
| 1471 | assert_eq!(res!(parse(" \n \t ")), Vec::<Block>::new()); |
| 1472 | assert_eq!(res!(parse("<!DOCTYPE html>\n<html>\n<body>\n</body>\n</html>\n")), |
| 1473 | Vec::<Block>::new()); |
| 1474 | Ok(()) |
| 1475 | } |
| 1476 | |
| 1477 | /// A less-than that nobody escaped is a less-than, and does not open an element. |
| 1478 | #[test] |
| 1479 | fn test_a_bare_less_than_is_text_30() -> Outcome<()> { |
| 1480 | assert_eq!(said(&res!(parse("<p>a < b and c > d</p>"))), vec!["a < b and c > d"]); |
| 1481 | assert_eq!(said(&res!(parse("<p>1 <2</p><p>after</p>"))), vec!["1 <2", "after"]); |
| 1482 | Ok(()) |
| 1483 | } |
| 1484 | |
| 1485 | /// A cell is given inlines and nothing else, so a block within one is unwrapped -- and stands as |
| 1486 | /// the boundary it is, rather than running two words together. |
| 1487 | #[test] |
| 1488 | fn test_a_block_within_a_cell_is_unwrapped_31() -> Outcome<()> { |
| 1489 | let b = res!(parse("<table><tr><td><p>one</p><p>two</p></td></tr></table>")); |
| 1490 | assert_eq!(grid(&b[0]), vec![vec!["one two"]]); |
| 1491 | Ok(()) |
| 1492 | } |
| 1493 | |
| 1494 | /// A tree written out by the sibling writer and read back is the tree that went in. |
| 1495 | /// |
| 1496 | /// This is worth more than it looks. The reader and the writer were written apart and agree on |
| 1497 | /// nothing but the tree between them, so a round trip that holds is two implementations checking |
| 1498 | /// each other rather than one checking itself. It is also the claim the tree's own documentation |
| 1499 | /// makes -- that a second front-end produces the same tree -- put to a test rather than asserted. |
| 1500 | #[test] |
| 1501 | fn test_a_tree_survives_a_round_trip_through_the_writer_32() -> Outcome<()> { |
| 1502 | use crate::doc::{Doc, html::render, markdown}; |
| 1503 | |
| 1504 | let src = "\ |
| 1505 | # A Heading\n\ |
| 1506 | \n\ |
| 1507 | A paragraph with *emphasis*, **strong emphasis**, a [link](https://example.com), \ |
| 1508 | `a code span`, and an .\n\ |
| 1509 | \n\ |
| 1510 | A line that ends hard \nand carries on.\n\ |
| 1511 | \n\ |
| 1512 | > A quotation.\n\ |
| 1513 | >\n\ |
| 1514 | > - with a list\n\ |
| 1515 | > - of two items\n\ |
| 1516 | \n\ |
| 1517 | 1. An ordered item\n\ |
| 1518 | 2. Another, holding\n\ |
| 1519 | - a nested list\n\ |
| 1520 | \n\ |
| 1521 | ```rust\n\ |
| 1522 | let x = 1 < 2;\n\ |
| 1523 | ```\n\ |
| 1524 | \n\ |
| 1525 | | Name | Age |\n\ |
| 1526 | | :---- | --: |\n\ |
| 1527 | | Alice | 30 |\n\ |
| 1528 | \n\ |
| 1529 | ---\n"; |
| 1530 | let doc = res!(markdown::parse(src)); |
| 1531 | // The source is worth having only if it exercises the tree, so check that it did. |
| 1532 | assert!(doc.blocks.len() >= 8, "the round trip is not testing much: {:?}", doc); |
| 1533 | let out = render(&doc); |
| 1534 | let back = Doc { blocks: res!(parse(&out)) }; |
| 1535 | assert_eq!(back, doc, "\n--- html ---\n{}\n", out); |
| 1536 | Ok(()) |
| 1537 | } |
| 1538 | |
| 1539 | /// A `pre` that never closes runs to the end, and an element left open runs to the end of what |
| 1540 | /// encloses it. Neither is a failure, and neither loses the prose. |
| 1541 | #[test] |
| 1542 | fn test_an_unclosed_element_runs_to_the_end_33() -> Outcome<()> { |
| 1543 | let b = res!(parse("<pre>code and more")); |
| 1544 | assert_eq!(b, vec![Block::Code { lang: None, text: "code and more".to_string() }]); |
| 1545 | let b = res!(parse("<blockquote><p>inside")); |
| 1546 | assert_eq!(b, vec![Block::Quote(vec![Block::Para(vec![t("inside")])])]); |
| 1547 | Ok(()) |
| 1548 | } |
| 1549 | } |