Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/html.rs

11.8 KiB, 1 run

created by r1870400018:13501, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Reducing an HTML document to the text a reader would take from it.
2//!
3//! Not a parser, and deliberately not one. Nothing here builds a DOM, resolves
4//! a namespace or recovers from a mis-nested tag the way a browser must,
5//! because the question being answered is narrower: what does this page *say*?
6//! A single pass that drops the markup, drops the elements that hold no prose
7//! (a script, a stylesheet, a navigation bar), keeps the ones that do, and
8//! collapses what is left into lines, answers it -- and answers it on malformed
9//! input too, which a strict parser would refuse.
10//!
11//! The two callers this exists for are a server fetching a page on a user's
12//! behalf, and a mail client rendering an HTML part as text.
13
14use std::collections::BTreeSet;
15
16
17/// Elements whose content is not prose, and is dropped along with the tags.
18///
19/// A script or a stylesheet is code; a navigation bar and a footer are the
20/// furniture around the page rather than the page.
21const DROPPED: [&str; 9] = [
22 "script",
23 "style",
24 "noscript",
25 "svg",
26 "canvas",
27 "template",
28 "iframe",
29 "nav",
30 "footer",
31];
32
33/// Elements that begin a new line of text. Anything not named here is inline,
34/// so the text of a link, an emphasis or a span joins the sentence it sits in
35/// rather than breaking it.
36const BLOCK: [&str; 27] = [
37 "address",
38 "article",
39 "aside",
40 "blockquote",
41 "br",
42 "dd",
43 "div",
44 "dl",
45 "dt",
46 "figcaption",
47 "figure",
48 "h1",
49 "h2",
50 "h3",
51 "h4",
52 "h5",
53 "h6",
54 "header",
55 "hr",
56 "li",
57 "main",
58 "ol",
59 "p",
60 "pre",
61 "section",
62 "table",
63 "tr",
64];
65
66/// The named character references worth knowing, which are the handful that
67/// appear in prose. Anything else numeric is decoded by its code point.
68const NAMED: [(&str, &str); 14] = [
69 ("amp", "&"),
70 ("lt", "<"),
71 ("gt", ">"),
72 ("quot", "\""),
73 ("apos", "'"),
74 ("nbsp", " "),
75 ("ndash", "\u{2013}"),
76 ("mdash", "\u{2014}"),
77 ("hellip", "\u{2026}"),
78 ("lsquo", "\u{2018}"),
79 ("rsquo", "\u{2019}"),
80 ("ldquo", "\u{201c}"),
81 ("rdquo", "\u{201d}"),
82 ("middot", "\u{b7}"),
83];
84
85
86/// What a page says, once its markup is gone.
87#[derive(Clone, Debug, Default, Eq, PartialEq)]
88pub struct PageText {
89 /// The document title, empty when the page names none.
90 pub title: String,
91 /// The readable text, one block element to a line.
92 pub text: String,
93}
94
95/// Strip an HTML document to its title and its readable text.
96///
97/// Headings, paragraphs, list items and the text of links all survive; scripts,
98/// stylesheets, navigation and footers do not; runs of whitespace collapse to
99/// one space, and blank lines are dropped.
100pub fn html_to_text(html: &str) -> PageText {
101 let dropped: BTreeSet<&str> = DROPPED.iter().copied().collect();
102 let block: BTreeSet<&str> = BLOCK.iter().copied().collect();
103
104 let mut title_raw = String::new();
105 let mut body_raw = String::with_capacity(html.len() / 2);
106 // The element whose content is being discarded, if any. Only its own
107 // closing tag ends the discard, so a `<script>` containing `if (a < b)`
108 // cannot end it early.
109 let mut skip: Option<String> = None;
110 let mut in_title = false;
111 let mut rest = html;
112
113 loop {
114 let lt = match rest.find('<') {
115 Some(i) => i,
116 None => {
117 push(&mut body_raw, &mut title_raw, rest, &skip, in_title);
118 break;
119 }
120 };
121 push(&mut body_raw, &mut title_raw, &rest[..lt], &skip, in_title);
122 let tag = &rest[lt..];
123
124 // A comment, or a declaration such as the doctype. Neither says
125 // anything, and a comment may hold markup that must not be read.
126 if tag.starts_with("<!--") {
127 rest = match tag.find("-->") {
128 Some(i) => &tag[i + 3..],
129 None => "",
130 };
131 continue;
132 }
133 if tag.starts_with("<!") {
134 rest = match tag.find('>') {
135 Some(i) => &tag[i + 1..],
136 None => "",
137 };
138 continue;
139 }
140
141 let end = match tag_end(tag) {
142 Some(i) => i,
143 // An unterminated tag: there is no more markup, and no more text.
144 None => break,
145 };
146 let inner = &tag[1..end];
147 rest = &tag[end + 1..];
148
149 let closing = inner.starts_with('/');
150 let name: String = inner
151 .trim_start_matches('/')
152 .chars()
153 .take_while(|c| c.is_ascii_alphanumeric())
154 .collect::<String>()
155 .to_lowercase();
156 if name.is_empty() {
157 continue;
158 }
159
160 if let Some(open) = &skip {
161 if closing && &name == open {
162 skip = None;
163 }
164 continue;
165 }
166 if !closing && dropped.contains(name.as_str()) {
167 // A self-closing tag encloses nothing, so nothing is discarded.
168 if !inner.trim_end().ends_with('/') {
169 skip = Some(name);
170 }
171 continue;
172 }
173 if name == "title" {
174 in_title = !closing;
175 continue;
176 }
177 if block.contains(name.as_str()) {
178 body_raw.push('\n');
179 }
180 }
181
182 PageText {
183 title: squash(&decode_entities(&title_raw)),
184 text: lines(&decode_entities(&body_raw)),
185 }
186}
187
188/// Add a run of text to the title or the body, unless it belongs to an element
189/// whose content is being discarded.
190///
191/// Newlines within the text are not breaks: HTML wraps its source wherever it
192/// likes, and a sentence split across two lines of markup is still one
193/// sentence. Only a block tag breaks a line, so every whitespace character in
194/// the text itself becomes a space, and the layout of the file is forgotten.
195fn push(
196 body: &mut String,
197 title: &mut String,
198 text: &str,
199 skip: &Option<String>,
200 in_title: bool,
201) {
202 if skip.is_some() || text.is_empty() {
203 return;
204 }
205 let flat: String = text
206 .chars()
207 .map(|c| if c.is_whitespace() { ' ' } else { c })
208 .collect();
209 if in_title {
210 title.push_str(&flat);
211 } else {
212 body.push_str(&flat);
213 }
214}
215
216/// The index of the `>` that closes a tag, ignoring any inside a quoted
217/// attribute value -- `<a title="a > b">` is one tag, not two.
218fn tag_end(s: &str) -> Option<usize> {
219 let mut quote: Option<char> = None;
220 for (i, c) in s.char_indices().skip(1) {
221 match quote {
222 Some(q) => if c == q {
223 quote = None;
224 },
225 None => match c {
226 '"' | '\'' => quote = Some(c),
227 '>' => return Some(i),
228 _ => (),
229 },
230 }
231 }
232 None
233}
234
235/// Decode the character references a page's prose actually uses: the named ones
236/// worth knowing, and any numeric one, decimal or hexadecimal.
237///
238/// An `&` that begins nothing recognisable is left exactly as it is, which is
239/// what a browser does and what a reader expects.
240pub fn decode_entities(s: &str) -> String {
241 let mut out = String::with_capacity(s.len());
242 let mut rest = s;
243 loop {
244 let amp = match rest.find('&') {
245 Some(i) => i,
246 None => {
247 out.push_str(rest);
248 return out;
249 }
250 };
251 out.push_str(&rest[..amp]);
252 let after = &rest[amp + 1..];
253 // A reference is short; a `&` with no `;` close behind it is just an
254 // ampersand.
255 let semi = match after.char_indices().take(12).find(|(_, c)| *c == ';') {
256 Some((i, _)) => i,
257 None => {
258 out.push('&');
259 rest = after;
260 continue;
261 }
262 };
263 let name = &after[..semi];
264 match entity(name) {
265 Some(c) => out.push_str(&c),
266 None => {
267 out.push('&');
268 out.push_str(name);
269 out.push(';');
270 }
271 }
272 rest = &after[semi + 1..];
273 }
274}
275
276/// One character reference, by name or by code point.
277fn entity(name: &str) -> Option<String> {
278 for (n, v) in NAMED {
279 if name.eq_ignore_ascii_case(n) {
280 return Some(v.to_string());
281 }
282 }
283 let digits = match name.strip_prefix('#') {
284 Some(d) => d,
285 None => return None,
286 };
287 let code = match digits.strip_prefix('x').or_else(|| digits.strip_prefix('X')) {
288 Some(hex) => match u32::from_str_radix(hex, 16) {
289 Ok(n) => n,
290 Err(_) => return None,
291 },
292 None => match digits.parse::<u32>() {
293 Ok(n) => n,
294 Err(_) => return None,
295 },
296 };
297 char::from_u32(code).map(|c| c.to_string())
298}
299
300/// Collapse every run of whitespace to one space and trim, leaving one line.
301pub fn squash(s: &str) -> String {
302 s.split_whitespace().collect::<Vec<_>>().join(" ")
303}
304
305/// Collapse the whitespace within each line, and drop the lines that hold
306/// nothing.
307fn lines(s: &str) -> String {
308 s.split('\n')
309 .map(squash)
310 .filter(|l| !l.is_empty())
311 .collect::<Vec<_>>()
312 .join("\n")
313}
314
315
316// ┌───────────────────────────────────────────────────────────────────────────┐
317// │ TESTS │
318// └───────────────────────────────────────────────────────────────────────────┘
319
320#[cfg(test)]
321mod tests {
322 use super::*;
323
324 #[test]
325 fn test_the_title_is_taken_and_kept_out_of_the_text() {
326 let p = html_to_text("<html><head><title> The Page </title></head>\
327 <body><p>Hello</p></body></html>");
328 assert_eq!(p.title, "The Page");
329 assert_eq!(p.text, "Hello");
330 }
331
332 #[test]
333 fn test_scripts_styles_navigation_and_footers_are_dropped() {
334 let p = html_to_text("
335 <nav><a href='/x'>Home</a></nav>
336 <script>var a = 1 < 2 && 3 > 2;</script>
337 <style>body { color: red; }</style>
338 <p>The only prose.</p>
339 <footer>Copyright</footer>
340 ");
341 assert_eq!(p.text, "The only prose.");
342 }
343
344 #[test]
345 fn test_headings_paragraphs_list_items_and_link_text_survive() {
346 let p = html_to_text("
347 <h1>Title</h1>
348 <p>A <a href='/a'>link</a> in a sentence.</p>
349 <ul><li>One</li><li>Two</li></ul>
350 ");
351 // A link is inline, so its text joins the sentence rather than
352 // breaking it; a list item is a block, so each takes a line.
353 assert_eq!(p.text, "Title\nA link in a sentence.\nOne\nTwo");
354 }
355
356 #[test]
357 fn test_entities_are_decoded() {
358 let p = html_to_text("<p>Tom &amp; Jerry &lt;3 caf&#233; &#x2014; \
359 &quot;quoted&quot;&nbsp;&hellip; R&D</p>");
360 assert_eq!(p.text, "Tom & Jerry <3 café — \"quoted\" … R&D");
361 }
362
363 #[test]
364 fn test_an_angle_bracket_in_an_attribute_does_not_end_the_tag() {
365 let p = html_to_text("<p title=\"a > b\">Text</p>");
366 assert_eq!(p.text, "Text");
367 }
368
369 #[test]
370 fn test_comments_and_the_doctype_say_nothing() {
371 let p = html_to_text("<!DOCTYPE html><!-- <p>hidden</p> --><p>shown</p>");
372 assert_eq!(p.text, "shown");
373 }
374
375 #[test]
376 fn test_broken_markup_still_yields_its_text() {
377 // No closing tags, an unterminated tag at the end: a browser reads
378 // this, and so must this.
379 let p = html_to_text("<p>One<p>Two<p>Three<p");
380 assert_eq!(p.text, "One\nTwo\nThree");
381 }
382}