Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/doc/html/read.rs

51.3 KiB, 1 run

created by r1870400018:14324, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Reading HTML into the document tree: the pass that turns tags into blocks and inlines, and throws
2//! the exporter's whitespace away.
3//!
4//! # The dialect
5//!
6//! The elements read are the ones the tree has a node for, and no others: the six headings,
7//! paragraphs, lists, `pre`, quotations, thematic breaks and tables, and the inline run of emphasis,
8//! links, images, code spans and breaks. Comments, `head`, `script` and `style` hold no prose and are
9//! dropped, content and all.
10//!
11//! # Everything else is unwrapped
12//!
13//! A `div`, a `span`, and every element this has never heard of, contribute nothing themselves and are
14//! erased: the tag goes and the content is read exactly where the tag stood. There is no generic
15//! container in the tree to put them in, and inventing one would make every consumer learn HTML --
16//! which is the one thing the tree exists to prevent.
17//!
18//! Erasing rather than recursing is what makes this work in both directions at once. A `div` holding
19//! paragraphs is read as those paragraphs; a `div` holding bare words is read as the paragraph those
20//! words are; a `span` in the middle of a sentence stays in the middle of that sentence, with the text
21//! either side of it joined across the hole the tag left. The rule is the same in each case, and no
22//! prose is lost in any of them.
23//!
24//! # Robustness
25//!
26//! The input this is written for is a generator's output: well formed, and wrong in none of the ways a
27//! browser must survive. So there is no error recovery here beyond the cheap kind -- a stray close tag
28//! closes nothing and the document carries on, an element left open runs to the end of what encloses
29//! it. The one refusal is [`DEPTH_LIMIT`].
30
31use crate::doc::{
32 Align,
33 Block,
34 Cell,
35 Inline,
36 Row,
37};
38use crate::html::decode_entities;
39
40use oxedyne_fe2o3_core::prelude::*;
41
42use std::mem::take;
43
44/// How deep a document may nest its elements before the reader refuses it.
45///
46/// A quotation inside a list inside a quotation is legitimate; a thousand of them is a document built
47/// to exhaust the stack of whatever reads it. The limit is generous beside anything a generator
48/// exports and far below what would trouble the machine.
49///
50/// Only the elements the tree has a node for are counted, because only those are read by a recursive
51/// walk. A `div` and an unknown element are erased where they stand and cost no stack at all, so a
52/// document of a million nested `div`s is read as the flat prose it says rather than refused.
53pub const DEPTH_LIMIT: usize = 32;
54
55/// Reads HTML into the blocks it is made of.
56pub fn parse(src: &str) -> Outcome<Vec<Block>> {
57 let mut lex = Lex { src, i: 0, raw: None };
58 blocks(&mut lex, None, 0)
59}
60
61/// What an element is, which is what decides where its content goes.
62#[derive(Clone, Copy, Debug, PartialEq)]
63enum Kind {
64 /// A heading, of the level its name gives.
65 Heading(u8),
66 /// A paragraph.
67 Para,
68 /// A list, ordered where the flag says so.
69 List(bool),
70 /// A run of code, whose whitespace is what it says.
71 Code,
72 /// A quotation.
73 Quote,
74 /// A table.
75 Table,
76 /// A thematic break.
77 Rule,
78 /// Emphasis, strong where the flag says so.
79 Emph(bool),
80 /// A link.
81 Link,
82 /// An image.
83 Image,
84 /// A code span within a line.
85 Span,
86 /// A break the author asked for.
87 Break,
88 /// Content that is not prose, and goes nowhere.
89 Drop,
90 /// Anything else: the tag is erased and its content read in its place.
91 Bare,
92}
93
94impl Kind {
95
96 /// Whether the element belongs within a line rather than standing on its own.
97 fn is_inline(&self) -> bool {
98 matches!(self, Self::Emph(_) | Self::Link | Self::Image | Self::Span | Self::Break)
99 }
100}
101
102/// What an element is, by the name it was written with.
103///
104/// The name is lowered for the match, because HTML does not care how a tag is written and neither does
105/// this. Anything not named here is [`Kind::Bare`].
106fn kind(name: &str) -> Kind {
107 match name.to_ascii_lowercase().as_str() {
108 "h1" => Kind::Heading(1),
109 "h2" => Kind::Heading(2),
110 "h3" => Kind::Heading(3),
111 "h4" => Kind::Heading(4),
112 "h5" => Kind::Heading(5),
113 "h6" => Kind::Heading(6),
114 "p" => Kind::Para,
115 "ul" => Kind::List(false),
116 "ol" => Kind::List(true),
117 "pre" => Kind::Code,
118 "blockquote" => Kind::Quote,
119 "table" => Kind::Table,
120 "hr" => Kind::Rule,
121 "em" | "i" => Kind::Emph(false),
122 "strong" | "b" => Kind::Emph(true),
123 "a" => Kind::Link,
124 "img" => Kind::Image,
125 "code" => Kind::Span,
126 "br" => Kind::Break,
127 "head" | "script" | "style" => Kind::Drop,
128 _ => Kind::Bare,
129 }
130}
131
132/// The elements that hold nothing, and so need no closing tag.
133const VOID: [&str; 14] = [
134 "area",
135 "base",
136 "br",
137 "col",
138 "embed",
139 "hr",
140 "img",
141 "input",
142 "link",
143 "meta",
144 "param",
145 "source",
146 "track",
147 "wbr",
148];
149
150/// The elements whose content is text and not markup, so that a `<` within them opens nothing.
151const RAW: [&str; 2] = [
152 "script",
153 "style",
154];
155
156/// Whether a name is the given element's, without regard to the case it was written in.
157fn is(name: &str, want: &str) -> bool {
158 name.eq_ignore_ascii_case(want)
159}
160
161// -----------------------------------------------------------------------------------------------
162// The walk.
163// -----------------------------------------------------------------------------------------------
164
165/// Reads the children of an element into the blocks they are, until the given element closes or the
166/// input ends.
167///
168/// Loose inline content -- words that reach block level with no paragraph around them, which is what a
169/// `div` full of prose leaves behind once it is erased -- is gathered into the paragraph it is.
170fn blocks(lex: &mut Lex, until: Option<&str>, depth: usize) -> Outcome<Vec<Block>> {
171 if depth > DEPTH_LIMIT {
172 return Err(err!(
173 "HTML elements nest more than {} deep, which no document written to be read \
174 does.", DEPTH_LIMIT;
175 Excessive, Input));
176 }
177 let mut out = Vec::new();
178 let mut run = Run::default(); // Loose inline content, awaiting the paragraph it makes.
179 while let Some(tok) = lex.next() {
180 match tok {
181 Tok::Text(t) => run.text(t),
182 Tok::Close(name) => {
183 // A close tag that answers nothing closes nothing, and the document carries on.
184 if let Some(until) = until {
185 if is(name, until) {
186 break;
187 }
188 }
189 }
190 Tok::Open { name, attrs, void } => {
191 let k = kind(name);
192 match k {
193 Kind::Drop => skip(lex, name, void),
194 // The tag is erased, and its content read as though it had never been there.
195 Kind::Bare => {}
196 _ if k.is_inline() => {
197 res!(inline(lex, &mut run, k, name, attrs, void, depth));
198 }
199 _ => {
200 flush(&mut out, &mut run);
201 if let Some(b) = res!(block(lex, k, name, attrs, void, depth)) {
202 out.push(b);
203 }
204 }
205 }
206 }
207 }
208 }
209 flush(&mut out, &mut run);
210 Ok(out)
211}
212
213/// Turns the loose inline content gathered so far into the paragraph it is, where it says anything.
214///
215/// A run of nothing but the whitespace that lay between two blocks says nothing, and makes no
216/// paragraph.
217fn flush(out: &mut Vec<Block>, run: &mut Run) {
218 let content = take(run).end();
219 if !content.is_empty() {
220 out.push(Block::Para(content));
221 }
222}
223
224/// Reads the block an opening tag begins, where the tree has one for it.
225fn block(lex: &mut Lex, k: Kind, name: &str, attrs: &str, void: bool, depth: usize)
226 -> Outcome<Option<Block>>
227{
228 let b = match k {
229 Kind::Rule => Block::Rule,
230 Kind::Code => code(lex, attrs, void),
231 Kind::List(ordered) => match void {
232 true => Block::List { ordered, items: Vec::new() },
233 false => res!(list(lex, ordered, name, depth)),
234 },
235 Kind::Table => match void {
236 true => Block::Table { head: None, rows: Vec::new(), cols: Vec::new() },
237 false => res!(table(lex, depth)),
238 },
239 Kind::Quote => match void {
240 true => Block::Quote(Vec::new()),
241 false => Block::Quote(res!(blocks(lex, Some(name), depth + 1))),
242 },
243 Kind::Heading(level) => Block::Heading {
244 level,
245 content: match void {
246 true => Vec::new(),
247 false => res!(inlines(lex, name, depth + 1)),
248 },
249 },
250 Kind::Para => {
251 let content = match void {
252 true => Vec::new(),
253 false => res!(inlines(lex, name, depth + 1)),
254 };
255 // A paragraph of nothing but the whitespace an exporter laid it out with says nothing,
256 // and the tree is better off without it.
257 if content.is_empty() {
258 return Ok(None);
259 }
260 Block::Para(content)
261 }
262 // Everything else here is a part of a list or a table that only its own reader sees. One that
263 // reaches this stood outside the thing it belongs to, where it says nothing: the tag is erased
264 // and the content within it is read where it stands.
265 _ => return Ok(None),
266 };
267 Ok(Some(b))
268}
269
270/// Reads the children of an element into the run of inlines they are, until the given element closes
271/// or the input ends.
272fn inlines(lex: &mut Lex, until: &str, depth: usize) -> Outcome<Vec<Inline>> {
273 if depth > DEPTH_LIMIT {
274 return Err(err!(
275 "HTML elements nest more than {} deep, which no prose written to be read \
276 does.", DEPTH_LIMIT;
277 Excessive, Input));
278 }
279 let mut run = Run::default();
280 while let Some(tok) = lex.next() {
281 match tok {
282 Tok::Text(t) => run.text(t),
283 Tok::Close(name) => {
284 if is(name, until) {
285 break;
286 }
287 }
288 Tok::Open { name, attrs, void } => {
289 let k = kind(name);
290 match k {
291 Kind::Drop => skip(lex, name, void),
292 Kind::Bare => {}
293 _ if k.is_inline() => {
294 res!(inline(lex, &mut run, k, name, attrs, void, depth));
295 }
296 // A block within a line is a thing the tree cannot hold: a cell is given inlines
297 // and nothing else, deliberately. The tag is erased like any other it has no node
298 // for, but it stands as the boundary it is, so that the words either side of it
299 // stay words apart rather than running together into one.
300 _ => run.space(),
301 }
302 }
303 }
304 }
305 Ok(run.end())
306}
307
308/// Reads the inline element an opening tag begins into the run it belongs to.
309fn inline(lex: &mut Lex, run: &mut Run, k: Kind, name: &str, attrs: &str, void: bool, depth: usize)
310 -> Outcome<()>
311{
312 match k {
313 Kind::Break => run.push(Inline::Break),
314 Kind::Image => run.push(Inline::Image {
315 src: attr(attrs, "src").unwrap_or_default(),
316 alt: attr(attrs, "alt").unwrap_or_default(),
317 }),
318 Kind::Span => {
319 // A code span is not a `pre`: HTML collapses the whitespace within one exactly as it does
320 // anywhere else, and so does this. What is not trimmed is the space at either end, because
321 // a span sits in a line and the space beside it is the line's.
322 let text = match void {
323 true => String::new(),
324 false => text_in(lex, name),
325 };
326 run.push(Inline::Code(collapse(&text)));
327 }
328 Kind::Emph(strong) => {
329 let content = match void {
330 true => Vec::new(),
331 false => res!(inlines(lex, name, depth + 1)),
332 };
333 run.push(Inline::Emph { strong, content });
334 }
335 Kind::Link => match attr(attrs, "href") {
336 Some(to) => {
337 let content = match void {
338 true => Vec::new(),
339 false => res!(inlines(lex, name, depth + 1)),
340 };
341 run.push(Inline::Link { to, content });
342 }
343 // An `a` with no destination is an anchor and not a link: there is nowhere for a reader to
344 // go. The tag is erased, and the words within it stay in the line they were in.
345 None => {}
346 },
347 // Nothing else reaches here: this is called only where the kind is inline.
348 _ => {}
349 }
350 Ok(())
351}
352
353/// Reads a `ul` or an `ol` into the list it is.
354fn list(lex: &mut Lex, ordered: bool, until: &str, depth: usize) -> Outcome<Block> {
355 let mut items: Vec<Vec<Block>> = Vec::new();
356 while let Some(tok) = lex.next() {
357 match tok {
358 Tok::Text(t) => {
359 // The whitespace an exporter lays a list out with says nothing. Words that reach a
360 // list with no item to sit in are still words, and stand as an item of their own
361 // rather than being dropped.
362 let s = collapse(&decode_entities(t));
363 let s = s.trim();
364 if !s.is_empty() {
365 items.push(vec![Block::Para(vec![Inline::Text(s.to_string())])]);
366 }
367 }
368 Tok::Close(name) => {
369 if is(name, until) {
370 break;
371 }
372 }
373 Tok::Open { name, void, .. } => {
374 if is(name, "li") {
375 items.push(match void {
376 true => Vec::new(),
377 false => res!(blocks(lex, Some(name), depth + 1)),
378 });
379 } else if kind(name) == Kind::Drop {
380 skip(lex, name, void);
381 }
382 // Anything else between the items is erased, so a list whose items are wrapped in
383 // something the tree does not know still finds them.
384 }
385 }
386 }
387 Ok(Block::List { ordered, items })
388}
389
390/// Reads a `table` into the grid it is.
391fn table(lex: &mut Lex, depth: usize) -> Outcome<Block> {
392 let mut head: Option<Row> = None;
393 let mut rows: Vec<Row> = Vec::new();
394 let mut cols: Vec<Align> = Vec::new();
395 let mut in_head = false; // Whether the rows being read are the table's header.
396 while let Some(tok) = lex.next() {
397 match tok {
398 // The whitespace a table is laid out with says nothing, and a table has nowhere to put a
399 // word that reached it without a cell to sit in.
400 Tok::Text(_) => {}
401 Tok::Close(name) => {
402 if is(name, "table") {
403 break;
404 }
405 if is(name, "thead") {
406 in_head = false;
407 }
408 }
409 Tok::Open { name, void, .. } => {
410 if is(name, "thead") {
411 in_head = true;
412 } else if is(name, "tr") && !void {
413 let (r, all_th) = res!(row(lex, &mut cols, depth + 1));
414 // A table names its columns in a `thead`, or in a first row of nothing but `th`.
415 // Both say the same thing, and an exporter picks whichever it likes.
416 if head.is_none() && rows.is_empty() && (in_head || all_th) {
417 head = Some(r);
418 } else {
419 rows.push(r);
420 }
421 } else if kind(name) == Kind::Drop {
422 skip(lex, name, void);
423 }
424 // A `tbody` or a `tfoot` groups rows and says nothing else, so it is erased and the
425 // rows within it are read where they stand.
426 }
427 }
428 }
429 Ok(Block::Table { head, rows, cols })
430}
431
432/// Reads a `tr` into the row it is, filling in the alignment its cells declare for their columns.
433///
434/// Whether every cell was a `th` comes back with the row, because a first row of nothing but `th` is a
435/// table naming its columns whether or not anyone wrapped it in a `thead`.
436fn row(lex: &mut Lex, cols: &mut Vec<Align>, depth: usize) -> Outcome<(Row, bool)> {
437 let mut cells: Vec<Cell> = Vec::new();
438 let mut all_th = true;
439 while let Some(tok) = lex.next() {
440 match tok {
441 Tok::Text(_) => {}
442 Tok::Close(name) => {
443 if is(name, "tr") {
444 break;
445 }
446 }
447 Tok::Open { name, attrs, void } => {
448 let th = is(name, "th");
449 if th || is(name, "td") {
450 if !th {
451 all_th = false;
452 }
453 let i = cells.len();
454 while cols.len() <= i {
455 cols.push(Align::None);
456 }
457 // A column takes its alignment from the first cell that declares one.
458 if cols[i] == Align::None {
459 cols[i] = align_of(attrs);
460 }
461 cells.push(Cell(match void {
462 true => Vec::new(),
463 false => res!(inlines(lex, name, depth + 1)),
464 }));
465 } else if kind(name) == Kind::Drop {
466 skip(lex, name, void);
467 }
468 }
469 }
470 }
471 // A row of no cells names no columns, whatever its cells were not.
472 let named = all_th && !cells.is_empty();
473 Ok((Row(cells), named))
474}
475
476/// Reads a `pre` into the run of code it holds.
477fn code(lex: &mut Lex, attrs: &str, void: bool) -> Block {
478 // A `pre` may name the language on itself, or on the `code` within it, which is where the
479 // convention every highlighter reads puts it.
480 let mut lang = lang_of(attrs);
481 let mut text = String::new();
482 if !void {
483 while let Some(tok) = lex.next() {
484 match tok {
485 // Not collapsed, and not trimmed: within a `pre` the whitespace is the content, which
486 // is the whole reason the element exists.
487 Tok::Text(t) => text.push_str(&decode_entities(t)),
488 Tok::Close(name) => {
489 if is(name, "pre") {
490 break;
491 }
492 }
493 Tok::Open { name, attrs, void } => {
494 if kind(name) == Kind::Drop {
495 skip(lex, name, void);
496 } else if is(name, "code") && lang.is_none() {
497 lang = lang_of(attrs);
498 }
499 // Every other tag within is decoration -- a highlighter's own markup around a
500 // keyword -- and what the block says is the text under it.
501 }
502 }
503 }
504 }
505 // A line ending directly after the opening tag is the tag's own and not the code's. HTML's own
506 // parser drops it, and a reader that kept it would grow a blank first line on every block written
507 // the way most are.
508 let text = match text.strip_prefix("\r\n") {
509 Some(rest) => rest.to_string(),
510 None => match text.strip_prefix('\n') {
511 Some(rest) => rest.to_string(),
512 None => text,
513 },
514 };
515 Block::Code { lang, text }
516}
517
518/// The text of an element and everything within it, its entities decoded and its whitespace left as it
519/// was written.
520fn text_in(lex: &mut Lex, until: &str) -> String {
521 let mut out = String::new();
522 while let Some(tok) = lex.next() {
523 match tok {
524 Tok::Text(t) => out.push_str(&decode_entities(t)),
525 Tok::Close(name) => {
526 if is(name, until) {
527 break;
528 }
529 }
530 Tok::Open { name, void, .. } => {
531 if kind(name) == Kind::Drop {
532 skip(lex, name, void);
533 }
534 }
535 }
536 }
537 out
538}
539
540/// Skips an element and everything within it, tags and all.
541///
542/// The skip counts its own element's tags rather than recursing, so a document built to nest deeply
543/// costs nothing here.
544fn skip(lex: &mut Lex, name: &str, void: bool) {
545 if void {
546 return;
547 }
548 let mut depth = 0usize;
549 while let Some(tok) = lex.next() {
550 match tok {
551 Tok::Text(_) => {}
552 Tok::Open { name: n, void: v, .. } => {
553 if !v && is(n, name) {
554 depth += 1;
555 }
556 }
557 Tok::Close(n) => {
558 if is(n, name) {
559 if depth == 0 {
560 return;
561 }
562 depth -= 1;
563 }
564 }
565 }
566 }
567}
568
569// -----------------------------------------------------------------------------------------------
570// Whitespace.
571// -----------------------------------------------------------------------------------------------
572
573/// A run of inline content, gathered with HTML's whitespace rules applied as it grows.
574///
575/// This is where the rule the module documentation states is actually kept: a run of whitespace says
576/// one space, a space at the start of a run says nothing, and the space at the end is taken off when
577/// the run closes. Only a `pre` escapes it, and a `pre` is read by [`code`] and never reaches here.
578#[derive(Default)]
579struct Run {
580 /// The inlines gathered so far.
581 out: Vec<Inline>,
582}
583
584impl Run {
585
586 /// Whether a space would say nothing here: at the start of the run, where a space has been said
587 /// already, or after a break, there is nothing for one to separate.
588 fn open(&self) -> bool {
589 match self.out.last() {
590 None => true,
591 Some(Inline::Break) => true,
592 Some(Inline::Text(t)) => t.ends_with(' '),
593 Some(_) => false,
594 }
595 }
596
597 /// Adds a run of text, its entities decoded and its whitespace collapsed.
598 fn text(&mut self, raw: &str) {
599 // Decoded first and collapsed second, which is the order that matters: a `&#10;` says a
600 // newline, and a newline is whitespace like any other. Only a no-break space survives, and it
601 // survives because it is not whitespace this collapses.
602 let s = collapse(&decode_entities(raw));
603 let s = match self.open() {
604 true => s.strip_prefix(' ').unwrap_or(&s),
605 false => s.as_str(),
606 };
607 if s.is_empty() {
608 return;
609 }
610 self.push(Inline::Text(s.to_string()));
611 }
612
613 /// Says the space an erased block stands for, where the run does not say one already.
614 fn space(&mut self) {
615 if !self.open() {
616 self.push(Inline::Text(" ".to_string()));
617 }
618 }
619
620 /// Adds a settled inline, joining it to the run before it where both are text.
621 fn push(&mut self, item: Inline) {
622 if let Inline::Text(t) = &item {
623 if let Some(Inline::Text(last)) = self.out.last_mut() {
624 last.push_str(t);
625 return;
626 }
627 }
628 self.out.push(item);
629 }
630
631 /// The run, with the trailing space that HTML does not say taken off it.
632 fn end(mut self) -> Vec<Inline> {
633 if let Some(Inline::Text(t)) = self.out.last_mut() {
634 while t.ends_with(' ') {
635 t.pop();
636 }
637 if t.is_empty() {
638 self.out.pop();
639 }
640 }
641 self.out
642 }
643}
644
645/// Whether a character is whitespace that HTML collapses.
646///
647/// A no-break space is deliberately not among them. It is not this whitespace, it does not collapse,
648/// and an author who wrote one meant it.
649fn is_ws(c: char) -> bool {
650 matches!(c, ' ' | '\t' | '\n' | '\r' | '\u{c}')
651}
652
653/// Collapses every run of whitespace in a run of text to the single space it says.
654fn collapse(s: &str) -> String {
655 let mut out = String::with_capacity(s.len());
656 let mut ws = false; // Whether whitespace has been passed over since the last character kept.
657 for c in s.chars() {
658 if is_ws(c) {
659 ws = true;
660 } else {
661 if ws {
662 out.push(' ');
663 ws = false;
664 }
665 out.push(c);
666 }
667 }
668 if ws {
669 out.push(' ');
670 }
671 out
672}
673
674// -----------------------------------------------------------------------------------------------
675// Attributes.
676// -----------------------------------------------------------------------------------------------
677
678/// The value of one attribute from a tag's unparsed run of them, where the tag carries it.
679///
680/// Names are matched without regard to case, and a value is taken from double quotes, single quotes or
681/// no quotes at all, because all three are HTML and a generator picks whichever it likes. An attribute
682/// written with no value at all is present, and says the empty string. What comes back has its entities
683/// decoded, so a destination written `a&amp;b` is read as the `a&b` it names.
684fn attr(attrs: &str, want: &str) -> Option<String> {
685 let b = attrs.as_bytes();
686 let mut i = 0;
687 while i < b.len() {
688 while i < b.len() && (b[i].is_ascii_whitespace() || b[i] == b'/') {
689 i += 1;
690 }
691 let ns = i;
692 while i < b.len() && !b[i].is_ascii_whitespace() && b[i] != b'=' && b[i] != b'/' {
693 i += 1;
694 }
695 let name = &attrs[ns..i];
696 while i < b.len() && b[i].is_ascii_whitespace() {
697 i += 1;
698 }
699 let mut val = "";
700 if i < b.len() && b[i] == b'=' {
701 i += 1;
702 while i < b.len() && b[i].is_ascii_whitespace() {
703 i += 1;
704 }
705 if i < b.len() && (b[i] == b'"' || b[i] == b'\'') {
706 let q = b[i];
707 i += 1;
708 let vs = i;
709 while i < b.len() && b[i] != q {
710 i += 1;
711 }
712 val = &attrs[vs..i];
713 if i < b.len() {
714 i += 1;
715 }
716 } else {
717 let vs = i;
718 while i < b.len() && !b[i].is_ascii_whitespace() {
719 i += 1;
720 }
721 val = &attrs[vs..i];
722 }
723 }
724 if !name.is_empty() && name.eq_ignore_ascii_case(want) {
725 return Some(decode_entities(val));
726 }
727 if name.is_empty() && val.is_empty() {
728 // Nothing was read, so nothing more will be: this is the end of the run.
729 break;
730 }
731 }
732 None
733}
734
735/// The language a `class="language-x"` names, where the class names one.
736fn lang_of(attrs: &str) -> Option<String> {
737 let class = match attr(attrs, "class") {
738 Some(c) => c,
739 None => return None,
740 };
741 for word in class.split_ascii_whitespace() {
742 if let Some(lang) = word.strip_prefix("language-") {
743 if !lang.is_empty() {
744 return Some(lang.to_string());
745 }
746 }
747 }
748 None
749}
750
751/// The alignment a cell declares, where it declares one this can honour.
752///
753/// Only the logical keywords are read: `start`, `end`, and `center`, which is unambiguous. CSS's
754/// `left` and `right` are deliberately not mapped, for the reason [`Align`] gives at length -- the
755/// tree does not know which way its text runs, so it cannot know which side `left` is on. A cell that
756/// says `left` is read as a cell that says nothing, and the consumer's own default stands, which is
757/// wrong for nobody. Mapping it would be wrong for half the world's prose, and silently.
758fn align_of(attrs: &str) -> Align {
759 let style = match attr(attrs, "style") {
760 Some(s) => s,
761 None => return Align::None,
762 };
763 // The lowered copy is only used to find the property, and an ASCII lowering does not move a byte,
764 // so the index it gives is an index into the original.
765 let at = match style.to_ascii_lowercase().find("text-align") {
766 Some(i) => i + "text-align".len(),
767 None => return Align::None,
768 };
769 let val = match style[at..].trim_start().strip_prefix(':') {
770 Some(v) => v,
771 None => return Align::None,
772 };
773 let val = val.split(';').next().unwrap_or("").trim();
774 match val.to_ascii_lowercase().as_str() {
775 "start" => Align::Start,
776 "center" => Align::Centre,
777 "end" => Align::End,
778 _ => Align::None,
779 }
780}
781
782// -----------------------------------------------------------------------------------------------
783// The lexer.
784// -----------------------------------------------------------------------------------------------
785
786/// One thing the reader takes from the source: a tag, or the text between tags.
787enum Tok<'a> {
788 /// An opening tag.
789 Open {
790 /// The element's name, in whatever case it was written.
791 name: &'a str,
792 /// The attributes, unparsed: [`attr`] reads one out where something wants it.
793 attrs: &'a str,
794 /// Whether the tag holds nothing, either by being a void element or by closing itself.
795 void: bool,
796 },
797 /// A closing tag, by its name.
798 Close(&'a str),
799 /// The text between two tags, its entities undecoded and its whitespace uncollapsed.
800 Text(&'a str),
801}
802
803/// The source, and how far into it the reader has come.
804struct Lex<'a> {
805 /// The HTML being read.
806 src: &'a str,
807 /// Where the next token begins.
808 i: usize,
809 /// The raw text element being read, where one is.
810 raw: Option<&'a str>,
811}
812
813impl<'a> Lex<'a> {
814
815 /// The next token, or nothing where the source is spent.
816 ///
817 /// Comments, doctypes and processing instructions are passed over here rather than being handed on,
818 /// because there is nothing above this that would do anything with them but drop them.
819 fn next(&mut self) -> Option<Tok<'a>> {
820 let b = self.src.as_bytes();
821 loop {
822 if self.i >= b.len() {
823 return None;
824 }
825 // Within a script or a stylesheet everything up to the closing tag is text, so a `<` in
826 // the code opens nothing. The content is dropped above, but it must be read as what it is
827 // or a comparison in a script would be read as an element.
828 if let Some(name) = self.raw.take() {
829 let end = self.raw_end(name);
830 let t = &self.src[self.i..end];
831 self.i = end;
832 return Some(Tok::Text(t));
833 }
834 if b[self.i] == b'<' && opens_tag(b, self.i) {
835 if self.src[self.i..].starts_with("<!--") {
836 // A comment says nothing to a document tree.
837 self.i = match self.src[self.i..].find("-->") {
838 Some(k) => self.i + k + 3,
839 None => b.len(),
840 };
841 continue;
842 }
843 if b[self.i + 1] == b'!' || b[self.i + 1] == b'?' {
844 // A doctype or a processing instruction says nothing either.
845 self.i = self.to_gt(self.i + 1);
846 continue;
847 }
848 if b[self.i + 1] == b'/' {
849 let ns = self.i + 2;
850 let ne = name_end(b, ns);
851 let name = &self.src[ns..ne];
852 self.i = self.to_gt(ne);
853 return Some(Tok::Close(name));
854 }
855 let ns = self.i + 1;
856 let ne = name_end(b, ns);
857 let name = &self.src[ns..ne];
858 let (attrs, end) = self.attrs_of(ne);
859 self.i = end;
860 // A trailing slash closes the tag itself; a void element closes itself whether or not
861 // anyone wrote one.
862 let slash = attrs.ends_with('/');
863 let attrs = match slash {
864 true => &attrs[..attrs.len() - 1],
865 false => attrs,
866 };
867 let void = slash || VOID.iter().any(|v| is(name, v));
868 if !void && RAW.iter().any(|r| is(name, r)) {
869 self.raw = Some(name);
870 }
871 return Some(Tok::Open { name, attrs, void });
872 }
873 // Text, up to the next tag. A `<` that opens nothing -- a less-than in prose that nobody
874 // escaped -- is text like any other, so the scan steps over it and carries on.
875 let mut j = self.i;
876 let end = loop {
877 match self.src[j..].find('<') {
878 None => break b.len(),
879 Some(k) => {
880 let at = j + k;
881 if opens_tag(b, at) {
882 break at;
883 }
884 j = at + 1;
885 }
886 }
887 };
888 let t = &self.src[self.i..end];
889 self.i = end;
890 return Some(Tok::Text(t));
891 }
892 }
893
894 /// Where the current raw text element's content ends: at its closing tag, or at the end of the
895 /// source where it has none.
896 fn raw_end(&self, name: &str) -> usize {
897 let mut j = self.i;
898 loop {
899 match self.src[j..].find('<') {
900 None => return self.src.len(),
901 Some(k) => {
902 let at = j + k;
903 let rest = &self.src[at..];
904 if rest.starts_with("</") && rest[2..].to_ascii_lowercase().starts_with(name) {
905 return at;
906 }
907 j = at + 1;
908 }
909 }
910 }
911 }
912
913 /// The run of attributes a tag carries, and where the tag ends.
914 fn attrs_of(&self, from: usize) -> (&'a str, usize) {
915 let end = self.to_gt(from);
916 // `to_gt` steps past the `>`, which is not the tag's to give away.
917 let close = match end > from && self.src.as_bytes()[end - 1] == b'>' {
918 true => end - 1,
919 false => end,
920 };
921 (self.src[from..close].trim(), end)
922 }
923
924 /// Where the tag that is being read ends: just past its `>`, or at the end of the source where it
925 /// has none. A `>` within a quoted value ends nothing.
926 fn to_gt(&self, from: usize) -> usize {
927 let b = self.src.as_bytes();
928 let mut k = from;
929 let mut q = 0u8; // The quote mark a value is sitting in, or zero for none.
930 while k < b.len() {
931 let c = b[k];
932 if q != 0 {
933 if c == q {
934 q = 0;
935 }
936 } else if c == b'"' || c == b'\'' {
937 q = c;
938 } else if c == b'>' {
939 return k + 1;
940 }
941 k += 1;
942 }
943 b.len()
944 }
945}
946
947/// Whether the `<` at the given index opens a tag, rather than being a less-than nobody escaped.
948fn opens_tag(b: &[u8], i: usize) -> bool {
949 match b.get(i + 1) {
950 Some(c) => c.is_ascii_alphabetic() || *c == b'!' || *c == b'/' || *c == b'?',
951 None => false,
952 }
953}
954
955/// Where the element name beginning at the given index ends.
956fn name_end(b: &[u8], from: usize) -> usize {
957 let mut k = from;
958 while k < b.len() && (b[k].is_ascii_alphanumeric() || b[k] == b'-' || b[k] == b':') {
959 k += 1;
960 }
961 k
962}
963
964#[cfg(test)]
965mod tests {
966 use super::*;
967
968 use crate::doc::text_of;
969
970 /// A run of literal text, for the tests that expect one.
971 fn t(s: &str) -> Inline {
972 Inline::Text(s.to_string())
973 }
974
975 /// The text of a block's inlines, for tests that care what a block says and not how.
976 fn said(blocks: &[Block]) -> Vec<String> {
977 blocks.iter().map(|b| match b {
978 Block::Para(c) => text_of(c),
979 Block::Heading { content, .. } => text_of(content),
980 Block::Code { text, .. } => text.clone(),
981 _ => String::new(),
982 }).collect()
983 }
984
985 /// What a table's rows say, cell by cell, the header first.
986 fn grid(b: &Block) -> Vec<Vec<String>> {
987 match b {
988 Block::Table { head, rows, .. } => {
989 let mut out = Vec::new();
990 if let Some(head) = head {
991 out.push(head.0.iter().map(|c| c.text_of()).collect());
992 }
993 for row in rows {
994 out.push(row.0.iter().map(|c| c.text_of()).collect());
995 }
996 out
997 }
998 other => panic!("expected a table, got {:?}", other),
999 }
1000 }
1001
1002 /// The six headings reach the six levels, whatever case they were written in.
1003 #[test]
1004 fn test_the_headings_give_their_levels_00() -> Outcome<()> {
1005 let b = res!(parse("<h1>One</h1><h3>Three</h3><H6>Six</H6>"));
1006 assert_eq!(b, vec![
1007 Block::Heading { level: 1, content: vec![t("One")] },
1008 Block::Heading { level: 3, content: vec![t("Three")] },
1009 Block::Heading { level: 6, content: vec![t("Six")] },
1010 ]);
1011 Ok(())
1012 }
1013
1014 /// A `p` is a paragraph, and the blank space an exporter laid it out with is not part of it.
1015 #[test]
1016 fn test_a_paragraph_is_a_paragraph_01() -> Outcome<()> {
1017 let b = res!(parse("<p>One.</p>\n\n<p>Two.</p>\n"));
1018 assert_eq!(b, vec![Block::Para(vec![t("One.")]), Block::Para(vec![t("Two.")])]);
1019 Ok(())
1020 }
1021
1022 /// THE RULE. A run of spaces, tabs and newlines between two words says one space, whatever the
1023 /// exporter wrote. A reader that kept them would freeze a book at the width it was exported at.
1024 #[test]
1025 fn test_a_run_of_whitespace_says_one_space_02() -> Outcome<()> {
1026 // A newline the exporter's line wrapping put there.
1027 assert_eq!(said(&res!(parse("<p>One line\nand its continuation.</p>"))),
1028 vec!["One line and its continuation."]);
1029 // Indentation, on its own line, as an exporter lays a document out.
1030 assert_eq!(said(&res!(parse("<p>\n\tOne line\n\tand its continuation.\n</p>"))),
1031 vec!["One line and its continuation."]);
1032 // Spaces, tabs and newlines together, in a run of any length.
1033 assert_eq!(said(&res!(parse("<p>a \t \n\r\n b</p>"))), vec!["a b"]);
1034 // And the break the exporter's wrapping made is never a break the author asked for.
1035 let b = res!(parse("<p>One line\nand its continuation.</p>"));
1036 assert_eq!(b, vec![Block::Para(vec![t("One line and its continuation.")])]);
1037 Ok(())
1038 }
1039
1040 /// The whitespace at either end of a block does not survive it.
1041 #[test]
1042 fn test_a_block_does_not_keep_the_space_at_its_ends_03() -> Outcome<()> {
1043 assert_eq!(said(&res!(parse("<p> padded </p>"))), vec!["padded"]);
1044 assert_eq!(said(&res!(parse("<h2>\n A Heading\n</h2>"))), vec!["A Heading"]);
1045 // A paragraph of nothing but whitespace says nothing, and is not a paragraph.
1046 assert_eq!(res!(parse("<p> \n </p>")), Vec::<Block>::new());
1047 assert_eq!(res!(parse("<p></p>")), Vec::<Block>::new());
1048 Ok(())
1049 }
1050
1051 /// The space between two inlines is a space, and the space at the start of a run is not.
1052 #[test]
1053 fn test_whitespace_around_an_inline_collapses_04() -> Outcome<()> {
1054 // A newline between two emphasised words says the space that divides them.
1055 let b = res!(parse("<p><em>a</em>\n <em>b</em></p>"));
1056 assert_eq!(b, vec![Block::Para(vec![
1057 Inline::Emph { strong: false, content: vec![t("a")] },
1058 t(" "),
1059 Inline::Emph { strong: false, content: vec![t("b")] },
1060 ])]);
1061 // The whitespace an exporter put before the first inline says nothing.
1062 let b = res!(parse("<p>\n <em>a</em> b\n</p>"));
1063 assert_eq!(b, vec![Block::Para(vec![
1064 Inline::Emph { strong: false, content: vec![t("a")] },
1065 t(" b"),
1066 ])]);
1067 Ok(())
1068 }
1069
1070 /// A `pre` is the exception the rule is written around: its whitespace is what it says.
1071 #[test]
1072 fn test_a_pre_keeps_its_whitespace_exactly_05() -> Outcome<()> {
1073 let b = res!(parse("<pre>fn main() {\n\tlet x = 1;\n}\n</pre>"));
1074 assert_eq!(b, vec![Block::Code {
1075 lang: None,
1076 text: "fn main() {\n\tlet x = 1;\n}\n".to_string(),
1077 }]);
1078 Ok(())
1079 }
1080
1081 /// A `pre` names its language by the class every highlighter reads, on the `pre` or on the `code`.
1082 #[test]
1083 fn test_a_code_block_names_its_language_06() -> Outcome<()> {
1084 let b = res!(parse("<pre><code class=\"language-rust\">let x = 1 &lt; 2;\n</code></pre>"));
1085 assert_eq!(b, vec![Block::Code {
1086 lang: Some("rust".to_string()),
1087 text: "let x = 1 < 2;\n".to_string(),
1088 }]);
1089 // On the `pre` itself, and beside other classes.
1090 let b = res!(parse("<pre class=\"highlight language-c\">int x;</pre>"));
1091 assert_eq!(b, vec![Block::Code { lang: Some("c".to_string()), text: "int x;".to_string() }]);
1092 // And a block that names none says none.
1093 let b = res!(parse("<pre><code>plain</code></pre>"));
1094 assert_eq!(b, vec![Block::Code { lang: None, text: "plain".to_string() }]);
1095 Ok(())
1096 }
1097
1098 /// A line ending directly after the opening tag is the tag's own, and does not become a blank first
1099 /// line of code.
1100 #[test]
1101 fn test_a_pre_drops_the_line_ending_that_opens_it_07() -> Outcome<()> {
1102 let b = res!(parse("<pre><code>\nfirst\nsecond\n</code></pre>"));
1103 assert_eq!(b, vec![Block::Code { lang: None, text: "first\nsecond\n".to_string() }]);
1104 Ok(())
1105 }
1106
1107 /// A `blockquote` holds blocks, and nests.
1108 #[test]
1109 fn test_a_quotation_holds_blocks_08() -> Outcome<()> {
1110 let b = res!(parse("<blockquote><p>One.</p><p>Two.</p></blockquote>"));
1111 assert_eq!(b, vec![Block::Quote(vec![
1112 Block::Para(vec![t("One.")]),
1113 Block::Para(vec![t("Two.")]),
1114 ])]);
1115 let b = res!(parse("<blockquote><blockquote><p>Deep.</p></blockquote></blockquote>"));
1116 assert_eq!(b, vec![Block::Quote(vec![Block::Quote(vec![Block::Para(vec![t("Deep.")])])])]);
1117 Ok(())
1118 }
1119
1120 /// An `hr` is a thematic break.
1121 #[test]
1122 fn test_a_rule_is_a_rule_09() -> Outcome<()> {
1123 assert_eq!(res!(parse("<p>a</p><hr><p>b</p>")), vec![
1124 Block::Para(vec![t("a")]),
1125 Block::Rule,
1126 Block::Para(vec![t("b")]),
1127 ]);
1128 Ok(())
1129 }
1130
1131 /// A `ul` is an unordered list and an `ol` an ordered one, and an item holds the blocks it holds.
1132 #[test]
1133 fn test_a_list_is_ordered_or_not_10() -> Outcome<()> {
1134 let b = res!(parse("<ul>\n <li>one</li>\n <li>two</li>\n</ul>"));
1135 assert_eq!(b, vec![Block::List {
1136 ordered: false,
1137 items: vec![
1138 vec![Block::Para(vec![t("one")])],
1139 vec![Block::Para(vec![t("two")])],
1140 ],
1141 }]);
1142 let b = res!(parse("<ol><li><p>one</p><p>still one</p></li></ol>"));
1143 assert_eq!(b, vec![Block::List {
1144 ordered: true,
1145 items: vec![vec![Block::Para(vec![t("one")]), Block::Para(vec![t("still one")])]],
1146 }]);
1147 Ok(())
1148 }
1149
1150 /// A list nests within an item of a list.
1151 #[test]
1152 fn test_a_list_nests_within_a_list_11() -> Outcome<()> {
1153 let b = res!(parse("<ul><li>one<ul><li>inner</li></ul></li><li>two</li></ul>"));
1154 let want = vec![Block::List {
1155 ordered: false,
1156 items: vec![
1157 vec![
1158 Block::Para(vec![t("one")]),
1159 Block::List {
1160 ordered: false,
1161 items: vec![vec![Block::Para(vec![t("inner")])]],
1162 },
1163 ],
1164 vec![Block::Para(vec![t("two")])],
1165 ],
1166 }];
1167 assert_eq!(b, want);
1168 Ok(())
1169 }
1170
1171 /// A table takes its header from a `thead`, and its cells reach the grid in order.
1172 #[test]
1173 fn test_a_table_names_its_columns_in_a_thead_12() -> Outcome<()> {
1174 let src = "<table>\n<thead>\n<tr><th>Name</th><th>Age</th></tr>\n</thead>\n\
1175 <tbody>\n<tr><td>Alice</td><td>30</td></tr>\n<tr><td>Bob</td><td>4</td></tr>\n\
1176 </tbody>\n</table>";
1177 let b = res!(parse(src));
1178 assert_eq!(b.len(), 1);
1179 assert_eq!(grid(&b[0]), vec![
1180 vec!["Name", "Age"],
1181 vec!["Alice", "30"],
1182 vec!["Bob", "4"],
1183 ]);
1184 match &b[0] {
1185 Block::Table { head, rows, cols } => {
1186 assert!(head.is_some(), "the table lost its header row");
1187 assert_eq!(rows.len(), 2);
1188 // A column nobody aligned is aligned by nothing.
1189 assert_eq!(cols, &vec![Align::None, Align::None]);
1190 }
1191 other => panic!("expected a table, got {:?}", other),
1192 }
1193 Ok(())
1194 }
1195
1196 /// A first row of nothing but `th` names the columns whether or not anyone wrapped it in a `thead`,
1197 /// and a table without one has no header at all.
1198 #[test]
1199 fn test_a_row_of_th_names_the_columns_13() -> Outcome<()> {
1200 let b = res!(parse("<table><tr><th>A</th><th>B</th></tr><tr><td>1</td><td>2</td></tr></table>"));
1201 assert_eq!(grid(&b[0]), vec![vec!["A", "B"], vec!["1", "2"]]);
1202 match &b[0] {
1203 Block::Table { head, rows, .. } => {
1204 assert!(head.is_some(), "a row of th did not name the columns");
1205 assert_eq!(rows.len(), 1);
1206 }
1207 other => panic!("expected a table, got {:?}", other),
1208 }
1209 // A grid of figures is a table whether or not anything stands at the head of it.
1210 let b = res!(parse("<table><tr><td>1</td><td>2</td></tr></table>"));
1211 match &b[0] {
1212 Block::Table { head, rows, .. } => {
1213 assert!(head.is_none(), "a table with no header grew one");
1214 assert_eq!(rows.len(), 1);
1215 }
1216 other => panic!("expected a table, got {:?}", other),
1217 }
1218 Ok(())
1219 }
1220
1221 /// A cell's alignment is read from the logical keywords only. `left` and `right` are not mapped,
1222 /// because the tree does not know which side they are on.
1223 #[test]
1224 fn test_a_column_aligns_by_logical_side_only_14() -> Outcome<()> {
1225 let src = "<table><tr>\
1226 <td style=\"text-align: start\">a</td>\
1227 <td style=\"text-align: center\">b</td>\
1228 <td style=\"text-align: end\">c</td>\
1229 <td style=\"text-align: left\">d</td>\
1230 </tr></table>";
1231 let b = res!(parse(src));
1232 match &b[0] {
1233 Block::Table { cols, .. } => assert_eq!(
1234 cols,
1235 &vec![Align::Start, Align::Centre, Align::End, Align::None],
1236 ),
1237 other => panic!("expected a table, got {:?}", other),
1238 }
1239 Ok(())
1240 }
1241
1242 /// Emphasis is `em` or `i`, strong emphasis is `strong` or `b`, and they nest.
1243 #[test]
1244 fn test_emphasis_is_ordinary_or_strong_15() -> Outcome<()> {
1245 let b = res!(parse("<p><em>a</em> <i>b</i> <strong>c</strong> <b>d</b></p>"));
1246 assert_eq!(b, vec![Block::Para(vec![
1247 Inline::Emph { strong: false, content: vec![t("a")] },
1248 t(" "),
1249 Inline::Emph { strong: false, content: vec![t("b")] },
1250 t(" "),
1251 Inline::Emph { strong: true, content: vec![t("c")] },
1252 t(" "),
1253 Inline::Emph { strong: true, content: vec![t("d")] },
1254 ])]);
1255 let b = res!(parse("<p><strong><em>both</em></strong></p>"));
1256 assert_eq!(b, vec![Block::Para(vec![Inline::Emph {
1257 strong: true,
1258 content: vec![Inline::Emph { strong: false, content: vec![t("both")] }],
1259 }])]);
1260 Ok(())
1261 }
1262
1263 /// An `a` with a destination is a link; one without is an anchor, and its words stay in the line.
1264 #[test]
1265 fn test_a_link_carries_its_destination_16() -> Outcome<()> {
1266 let b = res!(parse("<p>See <a href=\"https://example.com\">here</a> now.</p>"));
1267 assert_eq!(b, vec![Block::Para(vec![
1268 t("See "),
1269 Inline::Link { to: "https://example.com".to_string(), content: vec![t("here")] },
1270 t(" now."),
1271 ])]);
1272 // An anchor is not a link: there is nowhere for a reader to go, so only the tag is lost.
1273 let b = res!(parse("<p>An <a name=\"x\">anchor</a> here.</p>"));
1274 assert_eq!(b, vec![Block::Para(vec![t("An anchor here.")])]);
1275 Ok(())
1276 }
1277
1278 /// An `img` is an image, by its source and the text that stands for it.
1279 #[test]
1280 fn test_an_image_carries_its_source_and_alt_17() -> Outcome<()> {
1281 let b = res!(parse("<p><img src=\"fig.png\" alt=\"a figure\"></p>"));
1282 assert_eq!(b, vec![Block::Para(vec![Inline::Image {
1283 src: "fig.png".to_string(),
1284 alt: "a figure".to_string(),
1285 }])]);
1286 // An image that stands for nothing says nothing, and is still an image.
1287 let b = res!(parse("<p><img src=\"fig.png\"></p>"));
1288 assert_eq!(b, vec![Block::Para(vec![Inline::Image {
1289 src: "fig.png".to_string(),
1290 alt: String::new(),
1291 }])]);
1292 Ok(())
1293 }
1294
1295 /// A `code` within a line is a code span, and a `br` is the one thing that makes a break.
1296 #[test]
1297 fn test_a_code_span_and_a_break_18() -> Outcome<()> {
1298 let b = res!(parse("<p>Call <code>inline()</code> now.</p>"));
1299 assert_eq!(b, vec![Block::Para(vec![
1300 t("Call "),
1301 Inline::Code("inline()".to_string()),
1302 t(" now."),
1303 ])]);
1304 let b = res!(parse("<p>one<br>two</p>"));
1305 assert_eq!(b, vec![Block::Para(vec![t("one"), Inline::Break, t("two")])]);
1306 Ok(())
1307 }
1308
1309 /// A break comes from a `br` and from nothing else. A newline in the source is not one.
1310 #[test]
1311 fn test_only_a_br_makes_a_break_19() -> Outcome<()> {
1312 let b = res!(parse("<p>one\ntwo\n\nthree</p>"));
1313 assert_eq!(b, vec![Block::Para(vec![t("one two three")])]);
1314 assert!(!b.iter().any(|blk| match blk {
1315 Block::Para(c) => c.contains(&Inline::Break),
1316 _ => false,
1317 }), "a newline became a break");
1318 // And the whitespace after a break is the line's, not a word's.
1319 let b = res!(parse("<p>one<br>\n two</p>"));
1320 assert_eq!(b, vec![Block::Para(vec![t("one"), Inline::Break, t("two")])]);
1321 Ok(())
1322 }
1323
1324 /// Entities are decoded, named and numeric alike, and a decoded newline collapses like any other.
1325 #[test]
1326 fn test_entities_are_decoded_20() -> Outcome<()> {
1327 assert_eq!(said(&res!(parse("<p>Tom &amp; Jerry &lt;3</p>"))), vec!["Tom & Jerry <3"]);
1328 assert_eq!(said(&res!(parse("<p>it&#8217;s</p>"))), vec!["it\u{2019}s"]);
1329 assert_eq!(said(&res!(parse("<p>it&#x2019;s</p>"))), vec!["it\u{2019}s"]);
1330 assert_eq!(said(&res!(parse("<p>a&nbsp;b</p>"))), vec!["a b"]);
1331 // An entity in an attribute is decoded too, so a destination says what it names.
1332 let b = res!(parse("<p><a href=\"?a=1&amp;b=2\">x</a></p>"));
1333 assert_eq!(b, vec![Block::Para(vec![Inline::Link {
1334 to: "?a=1&b=2".to_string(),
1335 content: vec![t("x")],
1336 }])]);
1337 // A `&` that begins nothing is an ampersand.
1338 assert_eq!(said(&res!(parse("<p>a & b</p>"))), vec!["a & b"]);
1339 Ok(())
1340 }
1341
1342 /// A `div` and a `span` are unwrapped: the tag goes and the content stays exactly where it stood.
1343 #[test]
1344 fn test_a_div_and_a_span_are_unwrapped_21() -> Outcome<()> {
1345 // A div holding paragraphs is those paragraphs, at no depth of their own.
1346 let b = res!(parse("<div><div><p>one</p><p>two</p></div></div>"));
1347 assert_eq!(b, vec![Block::Para(vec![t("one")]), Block::Para(vec![t("two")])]);
1348 // A div holding bare words is the paragraph those words are.
1349 let b = res!(parse("<div><em>Just words.</em></div>"));
1350 assert_eq!(b, vec![Block::Para(vec![Inline::Emph {
1351 strong: false,
1352 content: vec![t("Just words.")],
1353 }])]);
1354 // A span in the middle of a sentence leaves the sentence whole across the hole it left.
1355 let b = res!(parse("<p>a <span>b</span> c</p>"));
1356 assert_eq!(b, vec![Block::Para(vec![t("a b c")])]);
1357 // The same at block level, where there is no paragraph to sit in.
1358 let b = res!(parse("<div>a <span>b</span> c</div>"));
1359 assert_eq!(b, vec![Block::Para(vec![t("a b c")])]);
1360 Ok(())
1361 }
1362
1363 /// An element the reader has never heard of is unwrapped, and the prose within it is kept.
1364 #[test]
1365 fn test_an_unknown_element_is_unwrapped_22() -> Outcome<()> {
1366 let b = res!(parse("<html><body><section><p>Kept.</p></section></body></html>"));
1367 assert_eq!(b, vec![Block::Para(vec![t("Kept.")])]);
1368 let b = res!(parse("<p>a <mark>b</mark> <custom-tag attr=\"x\">c</custom-tag> d</p>"));
1369 assert_eq!(said(&b), vec!["a b c d"]);
1370 // Even one carrying blocks, and one that stands where a block would.
1371 let b = res!(parse("<figure><figcaption>A caption.</figcaption></figure>"));
1372 assert_eq!(said(&b), vec!["A caption."]);
1373 Ok(())
1374 }
1375
1376 /// A script, a stylesheet, a head and a comment hold no prose, and go entirely.
1377 #[test]
1378 fn test_what_holds_no_prose_is_dropped_23() -> Outcome<()> {
1379 let src = "<html><head><title>Title</title><meta charset=\"utf-8\"></head>\
1380 <body><script>if (a < b) { drop(); }</script>\
1381 <style>p { colour: red; }</style>\
1382 <!-- a comment, with <p>markup</p> in it -->\
1383 <p>Kept.</p></body></html>";
1384 assert_eq!(res!(parse(src)), vec![Block::Para(vec![t("Kept.")])]);
1385 // A `<` within a script opens nothing, so what follows it is not swallowed.
1386 let src = "<script>for (i = 0; i < n; i++) { x(); }</script><p>After.</p>";
1387 assert_eq!(res!(parse(src)), vec![Block::Para(vec![t("After.")])]);
1388 Ok(())
1389 }
1390
1391 /// A void element holds nothing, whether or not anyone closed it, and however it was written.
1392 #[test]
1393 fn test_void_elements_close_themselves_24() -> Outcome<()> {
1394 let b = res!(parse("<p>a<br>b<br/>c<br />d</p>"));
1395 assert_eq!(b, vec![Block::Para(vec![
1396 t("a"), Inline::Break, t("b"), Inline::Break, t("c"), Inline::Break, t("d"),
1397 ])]);
1398 assert_eq!(res!(parse("<hr><hr/>")), vec![Block::Rule, Block::Rule]);
1399 // A trailing slash is the tag's and not the attribute's.
1400 let b = res!(parse("<p><img src=\"a.png\"/></p>"));
1401 assert_eq!(b, vec![Block::Para(vec![Inline::Image {
1402 src: "a.png".to_string(),
1403 alt: String::new(),
1404 }])]);
1405 Ok(())
1406 }
1407
1408 /// An attribute's value is read from double quotes, single quotes or none at all, and its name is
1409 /// read whatever case it was written in.
1410 #[test]
1411 fn test_attributes_take_every_quoting_25() -> Outcome<()> {
1412 let want = vec![Block::Para(vec![Inline::Link {
1413 to: "x.html".to_string(),
1414 content: vec![t("go")],
1415 }])];
1416 assert_eq!(res!(parse("<p><a href=\"x.html\">go</a></p>"), ), want);
1417 assert_eq!(res!(parse("<p><a href='x.html'>go</a></p>")), want);
1418 assert_eq!(res!(parse("<p><a href=x.html>go</a></p>")), want);
1419 assert_eq!(res!(parse("<p><A HREF = \"x.html\" >go</A></p>")), want);
1420 // A quote mark the other kind does not close a value, and a `>` within one ends no tag.
1421 let b = res!(parse("<p><a href=\"a'b>c\" title='say \"x\"'>go</a></p>"));
1422 assert_eq!(b, vec![Block::Para(vec![Inline::Link {
1423 to: "a'b>c".to_string(),
1424 content: vec![t("go")],
1425 }])]);
1426 Ok(())
1427 }
1428
1429 /// A close tag that answers nothing closes nothing, and the rest of the document survives it.
1430 #[test]
1431 fn test_a_stray_close_tag_loses_nothing_26() -> Outcome<()> {
1432 let b = res!(parse("<p>one</p></div></em></p><p>two</p>"));
1433 assert_eq!(said(&b), vec!["one", "two"]);
1434 let b = res!(parse("</p><h2>A Heading</h2><p>After.</p>"));
1435 assert_eq!(said(&b), vec!["A Heading", "After."]);
1436 Ok(())
1437 }
1438
1439 /// Nesting past the limit is refused, which is the reader's one refusal.
1440 #[test]
1441 fn test_nesting_past_the_limit_is_refused_27() -> Outcome<()> {
1442 // A quotation for every level the limit allows is read.
1443 let ok = format!("{}deep{}",
1444 "<blockquote>".repeat(DEPTH_LIMIT - 1),
1445 "</blockquote>".repeat(DEPTH_LIMIT - 1));
1446 assert!(parse(&ok).is_ok());
1447 // Past it, and past it by far, is not.
1448 let deep = format!("{}deep", "<blockquote>".repeat(DEPTH_LIMIT + 8));
1449 assert!(parse(&deep).is_err());
1450 let very = format!("{}deep", "<blockquote>".repeat(2000));
1451 assert!(parse(&very).is_err());
1452 // Inlines are held to the same limit.
1453 let very = format!("<p>{}deep", "<em>".repeat(2000));
1454 assert!(parse(&very).is_err());
1455 Ok(())
1456 }
1457
1458 /// An element the tree has no node for costs no stack, so a document built to exhaust one is read
1459 /// as the flat prose it says rather than refused. This is why the limit can be as low as it is.
1460 #[test]
1461 fn test_unwrapped_elements_cost_no_depth_28() -> Outcome<()> {
1462 let deep = format!("{}<p>Kept.</p>{}", "<div>".repeat(50_000), "</div>".repeat(50_000));
1463 assert_eq!(res!(parse(&deep)), vec![Block::Para(vec![t("Kept.")])]);
1464 Ok(())
1465 }
1466
1467 /// An empty document is a document with nothing in it, and not a failure.
1468 #[test]
1469 fn test_an_empty_document_holds_nothing_29() -> Outcome<()> {
1470 assert_eq!(res!(parse("")), Vec::<Block>::new());
1471 assert_eq!(res!(parse(" \n \t ")), Vec::<Block>::new());
1472 assert_eq!(res!(parse("<!DOCTYPE html>\n<html>\n<body>\n</body>\n</html>\n")),
1473 Vec::<Block>::new());
1474 Ok(())
1475 }
1476
1477 /// A less-than that nobody escaped is a less-than, and does not open an element.
1478 #[test]
1479 fn test_a_bare_less_than_is_text_30() -> Outcome<()> {
1480 assert_eq!(said(&res!(parse("<p>a < b and c > d</p>"))), vec!["a < b and c > d"]);
1481 assert_eq!(said(&res!(parse("<p>1 <2</p><p>after</p>"))), vec!["1 <2", "after"]);
1482 Ok(())
1483 }
1484
1485 /// A cell is given inlines and nothing else, so a block within one is unwrapped -- and stands as
1486 /// the boundary it is, rather than running two words together.
1487 #[test]
1488 fn test_a_block_within_a_cell_is_unwrapped_31() -> Outcome<()> {
1489 let b = res!(parse("<table><tr><td><p>one</p><p>two</p></td></tr></table>"));
1490 assert_eq!(grid(&b[0]), vec![vec!["one two"]]);
1491 Ok(())
1492 }
1493
1494 /// A tree written out by the sibling writer and read back is the tree that went in.
1495 ///
1496 /// This is worth more than it looks. The reader and the writer were written apart and agree on
1497 /// nothing but the tree between them, so a round trip that holds is two implementations checking
1498 /// each other rather than one checking itself. It is also the claim the tree's own documentation
1499 /// makes -- that a second front-end produces the same tree -- put to a test rather than asserted.
1500 #[test]
1501 fn test_a_tree_survives_a_round_trip_through_the_writer_32() -> Outcome<()> {
1502 use crate::doc::{Doc, html::render, markdown};
1503
1504 let src = "\
1505 # A Heading\n\
1506 \n\
1507 A paragraph with *emphasis*, **strong emphasis**, a [link](https://example.com), \
1508 `a code span`, and an ![image](fig.png).\n\
1509 \n\
1510 A line that ends hard \nand carries on.\n\
1511 \n\
1512 > A quotation.\n\
1513 >\n\
1514 > - with a list\n\
1515 > - of two items\n\
1516 \n\
1517 1. An ordered item\n\
1518 2. Another, holding\n\
1519 - a nested list\n\
1520 \n\
1521 ```rust\n\
1522 let x = 1 < 2;\n\
1523 ```\n\
1524 \n\
1525 | Name | Age |\n\
1526 | :---- | --: |\n\
1527 | Alice | 30 |\n\
1528 \n\
1529 ---\n";
1530 let doc = res!(markdown::parse(src));
1531 // The source is worth having only if it exercises the tree, so check that it did.
1532 assert!(doc.blocks.len() >= 8, "the round trip is not testing much: {:?}", doc);
1533 let out = render(&doc);
1534 let back = Doc { blocks: res!(parse(&out)) };
1535 assert_eq!(back, doc, "\n--- html ---\n{}\n", out);
1536 Ok(())
1537 }
1538
1539 /// A `pre` that never closes runs to the end, and an element left open runs to the end of what
1540 /// encloses it. Neither is a failure, and neither loses the prose.
1541 #[test]
1542 fn test_an_unclosed_element_runs_to_the_end_33() -> Outcome<()> {
1543 let b = res!(parse("<pre>code and more"));
1544 assert_eq!(b, vec![Block::Code { lang: None, text: "code and more".to_string() }]);
1545 let b = res!(parse("<blockquote><p>inside"));
1546 assert_eq!(b, vec![Block::Quote(vec![Block::Para(vec![t("inside")])])]);
1547 Ok(())
1548 }
1549}