oxedyne/fe2o3/fe2o3_text/src/html.rs
11.8 KiB, 1 run
created by r1870400018:13501, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Reducing an HTML document to the text a reader would take from it. |
| 2 | //! |
| 3 | //! Not a parser, and deliberately not one. Nothing here builds a DOM, resolves |
| 4 | //! a namespace or recovers from a mis-nested tag the way a browser must, |
| 5 | //! because the question being answered is narrower: what does this page *say*? |
| 6 | //! A single pass that drops the markup, drops the elements that hold no prose |
| 7 | //! (a script, a stylesheet, a navigation bar), keeps the ones that do, and |
| 8 | //! collapses what is left into lines, answers it -- and answers it on malformed |
| 9 | //! input too, which a strict parser would refuse. |
| 10 | //! |
| 11 | //! The two callers this exists for are a server fetching a page on a user's |
| 12 | //! behalf, and a mail client rendering an HTML part as text. |
| 13 | |
| 14 | use std::collections::BTreeSet; |
| 15 | |
| 16 | |
| 17 | /// Elements whose content is not prose, and is dropped along with the tags. |
| 18 | /// |
| 19 | /// A script or a stylesheet is code; a navigation bar and a footer are the |
| 20 | /// furniture around the page rather than the page. |
| 21 | const DROPPED: [&str; 9] = [ |
| 22 | "script", |
| 23 | "style", |
| 24 | "noscript", |
| 25 | "svg", |
| 26 | "canvas", |
| 27 | "template", |
| 28 | "iframe", |
| 29 | "nav", |
| 30 | "footer", |
| 31 | ]; |
| 32 | |
| 33 | /// Elements that begin a new line of text. Anything not named here is inline, |
| 34 | /// so the text of a link, an emphasis or a span joins the sentence it sits in |
| 35 | /// rather than breaking it. |
| 36 | const BLOCK: [&str; 27] = [ |
| 37 | "address", |
| 38 | "article", |
| 39 | "aside", |
| 40 | "blockquote", |
| 41 | "br", |
| 42 | "dd", |
| 43 | "div", |
| 44 | "dl", |
| 45 | "dt", |
| 46 | "figcaption", |
| 47 | "figure", |
| 48 | "h1", |
| 49 | "h2", |
| 50 | "h3", |
| 51 | "h4", |
| 52 | "h5", |
| 53 | "h6", |
| 54 | "header", |
| 55 | "hr", |
| 56 | "li", |
| 57 | "main", |
| 58 | "ol", |
| 59 | "p", |
| 60 | "pre", |
| 61 | "section", |
| 62 | "table", |
| 63 | "tr", |
| 64 | ]; |
| 65 | |
| 66 | /// The named character references worth knowing, which are the handful that |
| 67 | /// appear in prose. Anything else numeric is decoded by its code point. |
| 68 | const NAMED: [(&str, &str); 14] = [ |
| 69 | ("amp", "&"), |
| 70 | ("lt", "<"), |
| 71 | ("gt", ">"), |
| 72 | ("quot", "\""), |
| 73 | ("apos", "'"), |
| 74 | ("nbsp", " "), |
| 75 | ("ndash", "\u{2013}"), |
| 76 | ("mdash", "\u{2014}"), |
| 77 | ("hellip", "\u{2026}"), |
| 78 | ("lsquo", "\u{2018}"), |
| 79 | ("rsquo", "\u{2019}"), |
| 80 | ("ldquo", "\u{201c}"), |
| 81 | ("rdquo", "\u{201d}"), |
| 82 | ("middot", "\u{b7}"), |
| 83 | ]; |
| 84 | |
| 85 | |
| 86 | /// What a page says, once its markup is gone. |
| 87 | #[derive(Clone, Debug, Default, Eq, PartialEq)] |
| 88 | pub struct PageText { |
| 89 | /// The document title, empty when the page names none. |
| 90 | pub title: String, |
| 91 | /// The readable text, one block element to a line. |
| 92 | pub text: String, |
| 93 | } |
| 94 | |
| 95 | /// Strip an HTML document to its title and its readable text. |
| 96 | /// |
| 97 | /// Headings, paragraphs, list items and the text of links all survive; scripts, |
| 98 | /// stylesheets, navigation and footers do not; runs of whitespace collapse to |
| 99 | /// one space, and blank lines are dropped. |
| 100 | pub fn html_to_text(html: &str) -> PageText { |
| 101 | let dropped: BTreeSet<&str> = DROPPED.iter().copied().collect(); |
| 102 | let block: BTreeSet<&str> = BLOCK.iter().copied().collect(); |
| 103 | |
| 104 | let mut title_raw = String::new(); |
| 105 | let mut body_raw = String::with_capacity(html.len() / 2); |
| 106 | // The element whose content is being discarded, if any. Only its own |
| 107 | // closing tag ends the discard, so a `<script>` containing `if (a < b)` |
| 108 | // cannot end it early. |
| 109 | let mut skip: Option<String> = None; |
| 110 | let mut in_title = false; |
| 111 | let mut rest = html; |
| 112 | |
| 113 | loop { |
| 114 | let lt = match rest.find('<') { |
| 115 | Some(i) => i, |
| 116 | None => { |
| 117 | push(&mut body_raw, &mut title_raw, rest, &skip, in_title); |
| 118 | break; |
| 119 | } |
| 120 | }; |
| 121 | push(&mut body_raw, &mut title_raw, &rest[..lt], &skip, in_title); |
| 122 | let tag = &rest[lt..]; |
| 123 | |
| 124 | // A comment, or a declaration such as the doctype. Neither says |
| 125 | // anything, and a comment may hold markup that must not be read. |
| 126 | if tag.starts_with("<!--") { |
| 127 | rest = match tag.find("-->") { |
| 128 | Some(i) => &tag[i + 3..], |
| 129 | None => "", |
| 130 | }; |
| 131 | continue; |
| 132 | } |
| 133 | if tag.starts_with("<!") { |
| 134 | rest = match tag.find('>') { |
| 135 | Some(i) => &tag[i + 1..], |
| 136 | None => "", |
| 137 | }; |
| 138 | continue; |
| 139 | } |
| 140 | |
| 141 | let end = match tag_end(tag) { |
| 142 | Some(i) => i, |
| 143 | // An unterminated tag: there is no more markup, and no more text. |
| 144 | None => break, |
| 145 | }; |
| 146 | let inner = &tag[1..end]; |
| 147 | rest = &tag[end + 1..]; |
| 148 | |
| 149 | let closing = inner.starts_with('/'); |
| 150 | let name: String = inner |
| 151 | .trim_start_matches('/') |
| 152 | .chars() |
| 153 | .take_while(|c| c.is_ascii_alphanumeric()) |
| 154 | .collect::<String>() |
| 155 | .to_lowercase(); |
| 156 | if name.is_empty() { |
| 157 | continue; |
| 158 | } |
| 159 | |
| 160 | if let Some(open) = &skip { |
| 161 | if closing && &name == open { |
| 162 | skip = None; |
| 163 | } |
| 164 | continue; |
| 165 | } |
| 166 | if !closing && dropped.contains(name.as_str()) { |
| 167 | // A self-closing tag encloses nothing, so nothing is discarded. |
| 168 | if !inner.trim_end().ends_with('/') { |
| 169 | skip = Some(name); |
| 170 | } |
| 171 | continue; |
| 172 | } |
| 173 | if name == "title" { |
| 174 | in_title = !closing; |
| 175 | continue; |
| 176 | } |
| 177 | if block.contains(name.as_str()) { |
| 178 | body_raw.push('\n'); |
| 179 | } |
| 180 | } |
| 181 | |
| 182 | PageText { |
| 183 | title: squash(&decode_entities(&title_raw)), |
| 184 | text: lines(&decode_entities(&body_raw)), |
| 185 | } |
| 186 | } |
| 187 | |
| 188 | /// Add a run of text to the title or the body, unless it belongs to an element |
| 189 | /// whose content is being discarded. |
| 190 | /// |
| 191 | /// Newlines within the text are not breaks: HTML wraps its source wherever it |
| 192 | /// likes, and a sentence split across two lines of markup is still one |
| 193 | /// sentence. Only a block tag breaks a line, so every whitespace character in |
| 194 | /// the text itself becomes a space, and the layout of the file is forgotten. |
| 195 | fn push( |
| 196 | body: &mut String, |
| 197 | title: &mut String, |
| 198 | text: &str, |
| 199 | skip: &Option<String>, |
| 200 | in_title: bool, |
| 201 | ) { |
| 202 | if skip.is_some() || text.is_empty() { |
| 203 | return; |
| 204 | } |
| 205 | let flat: String = text |
| 206 | .chars() |
| 207 | .map(|c| if c.is_whitespace() { ' ' } else { c }) |
| 208 | .collect(); |
| 209 | if in_title { |
| 210 | title.push_str(&flat); |
| 211 | } else { |
| 212 | body.push_str(&flat); |
| 213 | } |
| 214 | } |
| 215 | |
| 216 | /// The index of the `>` that closes a tag, ignoring any inside a quoted |
| 217 | /// attribute value -- `<a title="a > b">` is one tag, not two. |
| 218 | fn tag_end(s: &str) -> Option<usize> { |
| 219 | let mut quote: Option<char> = None; |
| 220 | for (i, c) in s.char_indices().skip(1) { |
| 221 | match quote { |
| 222 | Some(q) => if c == q { |
| 223 | quote = None; |
| 224 | }, |
| 225 | None => match c { |
| 226 | '"' | '\'' => quote = Some(c), |
| 227 | '>' => return Some(i), |
| 228 | _ => (), |
| 229 | }, |
| 230 | } |
| 231 | } |
| 232 | None |
| 233 | } |
| 234 | |
| 235 | /// Decode the character references a page's prose actually uses: the named ones |
| 236 | /// worth knowing, and any numeric one, decimal or hexadecimal. |
| 237 | /// |
| 238 | /// An `&` that begins nothing recognisable is left exactly as it is, which is |
| 239 | /// what a browser does and what a reader expects. |
| 240 | pub fn decode_entities(s: &str) -> String { |
| 241 | let mut out = String::with_capacity(s.len()); |
| 242 | let mut rest = s; |
| 243 | loop { |
| 244 | let amp = match rest.find('&') { |
| 245 | Some(i) => i, |
| 246 | None => { |
| 247 | out.push_str(rest); |
| 248 | return out; |
| 249 | } |
| 250 | }; |
| 251 | out.push_str(&rest[..amp]); |
| 252 | let after = &rest[amp + 1..]; |
| 253 | // A reference is short; a `&` with no `;` close behind it is just an |
| 254 | // ampersand. |
| 255 | let semi = match after.char_indices().take(12).find(|(_, c)| *c == ';') { |
| 256 | Some((i, _)) => i, |
| 257 | None => { |
| 258 | out.push('&'); |
| 259 | rest = after; |
| 260 | continue; |
| 261 | } |
| 262 | }; |
| 263 | let name = &after[..semi]; |
| 264 | match entity(name) { |
| 265 | Some(c) => out.push_str(&c), |
| 266 | None => { |
| 267 | out.push('&'); |
| 268 | out.push_str(name); |
| 269 | out.push(';'); |
| 270 | } |
| 271 | } |
| 272 | rest = &after[semi + 1..]; |
| 273 | } |
| 274 | } |
| 275 | |
| 276 | /// One character reference, by name or by code point. |
| 277 | fn entity(name: &str) -> Option<String> { |
| 278 | for (n, v) in NAMED { |
| 279 | if name.eq_ignore_ascii_case(n) { |
| 280 | return Some(v.to_string()); |
| 281 | } |
| 282 | } |
| 283 | let digits = match name.strip_prefix('#') { |
| 284 | Some(d) => d, |
| 285 | None => return None, |
| 286 | }; |
| 287 | let code = match digits.strip_prefix('x').or_else(|| digits.strip_prefix('X')) { |
| 288 | Some(hex) => match u32::from_str_radix(hex, 16) { |
| 289 | Ok(n) => n, |
| 290 | Err(_) => return None, |
| 291 | }, |
| 292 | None => match digits.parse::<u32>() { |
| 293 | Ok(n) => n, |
| 294 | Err(_) => return None, |
| 295 | }, |
| 296 | }; |
| 297 | char::from_u32(code).map(|c| c.to_string()) |
| 298 | } |
| 299 | |
| 300 | /// Collapse every run of whitespace to one space and trim, leaving one line. |
| 301 | pub fn squash(s: &str) -> String { |
| 302 | s.split_whitespace().collect::<Vec<_>>().join(" ") |
| 303 | } |
| 304 | |
| 305 | /// Collapse the whitespace within each line, and drop the lines that hold |
| 306 | /// nothing. |
| 307 | fn lines(s: &str) -> String { |
| 308 | s.split('\n') |
| 309 | .map(squash) |
| 310 | .filter(|l| !l.is_empty()) |
| 311 | .collect::<Vec<_>>() |
| 312 | .join("\n") |
| 313 | } |
| 314 | |
| 315 | |
| 316 | // ┌───────────────────────────────────────────────────────────────────────────┐ |
| 317 | // │ TESTS │ |
| 318 | // └───────────────────────────────────────────────────────────────────────────┘ |
| 319 | |
| 320 | #[cfg(test)] |
| 321 | mod tests { |
| 322 | use super::*; |
| 323 | |
| 324 | #[test] |
| 325 | fn test_the_title_is_taken_and_kept_out_of_the_text() { |
| 326 | let p = html_to_text("<html><head><title> The Page </title></head>\ |
| 327 | <body><p>Hello</p></body></html>"); |
| 328 | assert_eq!(p.title, "The Page"); |
| 329 | assert_eq!(p.text, "Hello"); |
| 330 | } |
| 331 | |
| 332 | #[test] |
| 333 | fn test_scripts_styles_navigation_and_footers_are_dropped() { |
| 334 | let p = html_to_text(" |
| 335 | <nav><a href='/x'>Home</a></nav> |
| 336 | <script>var a = 1 < 2 && 3 > 2;</script> |
| 337 | <style>body { color: red; }</style> |
| 338 | <p>The only prose.</p> |
| 339 | <footer>Copyright</footer> |
| 340 | "); |
| 341 | assert_eq!(p.text, "The only prose."); |
| 342 | } |
| 343 | |
| 344 | #[test] |
| 345 | fn test_headings_paragraphs_list_items_and_link_text_survive() { |
| 346 | let p = html_to_text(" |
| 347 | <h1>Title</h1> |
| 348 | <p>A <a href='/a'>link</a> in a sentence.</p> |
| 349 | <ul><li>One</li><li>Two</li></ul> |
| 350 | "); |
| 351 | // A link is inline, so its text joins the sentence rather than |
| 352 | // breaking it; a list item is a block, so each takes a line. |
| 353 | assert_eq!(p.text, "Title\nA link in a sentence.\nOne\nTwo"); |
| 354 | } |
| 355 | |
| 356 | #[test] |
| 357 | fn test_entities_are_decoded() { |
| 358 | let p = html_to_text("<p>Tom & Jerry <3 café — \ |
| 359 | "quoted" … R&D</p>"); |
| 360 | assert_eq!(p.text, "Tom & Jerry <3 café — \"quoted\" … R&D"); |
| 361 | } |
| 362 | |
| 363 | #[test] |
| 364 | fn test_an_angle_bracket_in_an_attribute_does_not_end_the_tag() { |
| 365 | let p = html_to_text("<p title=\"a > b\">Text</p>"); |
| 366 | assert_eq!(p.text, "Text"); |
| 367 | } |
| 368 | |
| 369 | #[test] |
| 370 | fn test_comments_and_the_doctype_say_nothing() { |
| 371 | let p = html_to_text("<!DOCTYPE html><!-- <p>hidden</p> --><p>shown</p>"); |
| 372 | assert_eq!(p.text, "shown"); |
| 373 | } |
| 374 | |
| 375 | #[test] |
| 376 | fn test_broken_markup_still_yields_its_text() { |
| 377 | // No closing tags, an unterminated tag at the end: a browser reads |
| 378 | // this, and so must this. |
| 379 | let p = html_to_text("<p>One<p>Two<p>Three<p"); |
| 380 | assert_eq!(p.text, "One\nTwo\nThree"); |
| 381 | } |
| 382 | } |