Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/doc/markdown/inline.rs

26.9 KiB, 6 runs

created by r1870400018:14264, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Inline structure: the pass that reads a line of prose into text, emphasis, links, images, code
2//! spans and hard breaks.
3//!
4//! This is the second of the two passes described in [`crate::doc::markdown::block`]. It is given the text
5//! of one block and returns the run of inlines it is made of.
6
7use crate::doc::{
8 Inline,
9 text_of,
10 markdown::block::DEPTH_LIMIT,
11};
12
13use oxedyne_fe2o3_core::prelude::*;
14
15/// Reads a block's text into its run of inline elements.
16pub fn parse(src: &str) -> Outcome<Vec<Inline>> {
17 run(src, 0)
18}
19
20/// One element of the scan: either an inline that is settled, or a run of emphasis characters still
21/// looking for the run that answers it.
22///
23/// Emphasis cannot be read in one pass, because whether a `*` opens anything is only known once
24/// something closes it. So the scan lays the delimiters out alongside the text it is sure of, and
25/// [`resolve`] pairs them off afterwards.
26enum Node {
27 /// An inline that needs nothing more, and how deep the tree it makes runs.
28 Done(Inline, usize),
29 /// A run of emphasis characters.
30 Delim {
31 /// The character the run is made of.
32 ch: u8,
33 /// How many characters the run has left to spend.
34 len: usize,
35 /// Whether the run may open emphasis.
36 open: bool,
37 /// Whether the run may close emphasis.
38 close: bool,
39 },
40}
41
42/// Reads a run of text into inlines, at the given nesting depth.
43fn run(src: &str, depth: usize) -> Outcome<Vec<Inline>> {
44 if depth > DEPTH_LIMIT {
45 return Err(err!(
46 "Markdown inlines nest more than {} deep, which no prose written to be read \
47 does.", DEPTH_LIMIT;
48 Excessive, Input));
49 }
50 let nodes = res!(scan(src, depth));
51 let (out, _) = res!(resolve(nodes, depth));
52 Ok(out)
53}
54
55/// Lays the text out as settled inlines and unsettled emphasis delimiters.
56fn scan(src: &str, depth: usize) -> Outcome<Vec<Node>> {
57 let mut out: Vec<Node> = Vec::new();
58 let mut buf = String::new(); // Text gathered since the last settled inline.
59 let b = src.as_bytes();
60 let mut i = 0;
61 while i < b.len() {
62 match b[i] {
63 b'\\' => {
64 if i + 1 < b.len() && b[i + 1] == b'\n' {
65 // A backslash at the end of a line is a break the author asked for.
66 flush(&mut out, &mut buf);
67 out.push(Node::Done(Inline::Break, 1));
68 i += 2;
69 } else if i + 1 < b.len() && b[i + 1].is_ascii_punctuation() {
70 // A backslash before punctuation says the punctuation is only itself.
71 buf.push(b[i + 1] as char);
72 i += 2;
73 } else {
74 buf.push('\\');
75 i += 1;
76 }
77 }
78 b'\n' => {
79 // The spaces before a line ending say what the ending meant.
80 let sp = buf.len() - buf.trim_end_matches(' ').len();
81 buf.truncate(buf.len() - sp);
82 if sp >= 2 {
83 // Two or more are a break the author asked for.
84 flush(&mut out, &mut buf);
85 out.push(Node::Done(Inline::Break, 1));
86 } else if !(out.is_empty() && buf.is_empty()) && i + 1 < b.len() {
87 // Fewer are a soft break: where the author's editor wrapped the line,
88 // not where the author meant a break. It says a space, so that prose
89 // hard wrapped to one width reflows to whatever width reads it. A soft
90 // break at either end of the run says nothing at all.
91 buf.push(' ');
92 }
93 i += 1;
94 }
95 b'`' => {
96 let n = run_len(b, i, b'`');
97 match code_span(src, i, n) {
98 Some((code, end)) => {
99 flush(&mut out, &mut buf);
100 out.push(Node::Done(Inline::Code(code), 1));
101 i = end;
102 }
103 None => {
104 // Nothing closed it, so the backticks are backticks.
105 for _ in 0..n {
106 buf.push('`');
107 }
108 i += n;
109 }
110 }
111 }
112 b'<' => {
113 match autolink(src, i) {
114 Some((to, end)) => {
115 flush(&mut out, &mut buf);
116 out.push(Node::Done(Inline::Link {
117 to: to.clone(),
118 content: vec![Inline::Text(to)],
119 }, 1));
120 i = end;
121 }
122 None => {
123 buf.push('<');
124 i += 1;
125 }
126 }
127 }
128 b'!' if i + 1 < b.len() && b[i + 1] == b'[' => {
129 match bracket(src, i + 1) {
130 Some((alt, to, end)) => {
131 flush(&mut out, &mut buf);
132 // An image stands for itself in words, so its alt is flattened.
133 let alt = text_of(&res!(run(&alt, depth + 1)));
134 out.push(Node::Done(Inline::Image { src: to, alt }, 1));
135 i = end;
136 }
137 None => {
138 buf.push('!');
139 i += 1;
140 }
141 }
142 }
143 b'[' => {
144 match bracket(src, i) {
145 Some((txt, to, end)) => {
146 flush(&mut out, &mut buf);
147 out.push(Node::Done(Inline::Link {
148 to,
149 content: res!(run(&txt, depth + 1)),
150 }, 1));
151 i = end;
152 }
153 None => {
154 buf.push('[');
155 i += 1;
156 }
157 }
158 }
159 b'*' | b'_' => {
160 let ch = b[i];
161 let len = run_len(b, i, ch);
162 let (open, close) = flank(src, i, i + len, ch);
163 flush(&mut out, &mut buf);
164 out.push(Node::Delim { ch, len, open, close });
165 i += len;
166 }
167 _ => {
168 // Anything else is itself, taken a whole character at a time.
169 let j = char_end(src, i);
170 buf.push_str(&src[i..j]);
171 i = j;
172 }
173 }
174 }
175 flush(&mut out, &mut buf);
176 Ok(out)
177}
178
179/// Adds the text gathered so far as one run, so that adjacent text is never split in two.
180fn flush(out: &mut Vec<Node>, buf: &mut String) {
181 if !buf.is_empty() {
182 out.push(Node::Done(Inline::Text(std::mem::take(buf)), 1));
183 }
184}
185
186// ── Emphasis ─────────────────────────────────────────────────────
187
188/// Pairs emphasis delimiters with the runs that answer them, and makes text of the rest.
189///
190/// Each run that could close is offered to the nearest run before it that could open. A run that
191/// finds no partner is not markup at all, and comes out as the characters it is made of -- which is
192/// why a stray asterisk is an asterisk.
193///
194/// Returns the inlines, and how deep the deepest of them runs. The depth is carried rather than
195/// measured afterwards because a run of emphasis characters nests one level per pair without the
196/// parser recursing once: ten thousand asterisks would build a tree too deep to walk, and only a
197/// count kept as it is built catches that before it exists.
198fn resolve(mut nodes: Vec<Node>, depth: usize) -> Outcome<(Vec<Inline>, usize)> {
199 let mut i = 0;
200 while i < nodes.len() {
201 let ch = match &nodes[i] {
202 Node::Delim { ch, close: true, .. } => *ch,
203 _ => {
204 i += 1;
205 continue;
206 }
207 };
208 // Look back for the nearest run of the same character that could open.
209 let mut found = None;
210 let mut k = i;
211 while k > 0 {
212 k -= 1;
213 if let Node::Delim { ch: c, open: true, .. } = &nodes[k] {
214 if *c == ch {
215 found = Some(k);
216 break;
217 }
218 }
219 }
220 let k = match found {
221 Some(k) => k,
222 None => {
223 // Nothing opened it, so it closes nothing and is only what it looks like.
224 if let Node::Delim { close, .. } = &mut nodes[i] {
225 *close = false;
226 }
227 i += 1;
228 continue;
229 }
230 };
231 // Two characters at each end make emphasis strong; one makes it ordinary.
232 let n = match (&nodes[k], &nodes[i]) {
233 (Node::Delim { len: a, .. }, Node::Delim { len: b, .. })
234 if *a >= 2 && *b >= 2 => 2,
235 _ => 1,
236 };
237 // Everything between the two runs is what they emphasise.
238 let inner: Vec<Node> = nodes.drain(k + 1..i).collect();
239 let (content, cd) = res!(resolve(inner, depth + 1));
240 let d = cd + 1;
241 if depth + d > DEPTH_LIMIT {
242 return Err(err!(
243 "Markdown emphasis nests more than {} deep, which no prose written to be \
244 read does.", DEPTH_LIMIT;
245 Excessive, Input));
246 }
247 if let Node::Delim { len, .. } = &mut nodes[k] {
248 *len -= n;
249 }
250 if let Node::Delim { len, .. } = &mut nodes[k + 1] {
251 *len -= n;
252 }
253 nodes.insert(k + 1, Node::Done(Inline::Emph { strong: n == 2, content }, d));
254 // The closing run now sits past the emphasis it made. A run with characters left over may
255 // still make more, so it is looked at again.
256 let mut ci = k + 2;
257 if matches!(&nodes[ci], Node::Delim { len: 0, .. }) {
258 nodes.remove(ci);
259 }
260 if matches!(&nodes[k], Node::Delim { len: 0, .. }) {
261 nodes.remove(k);
262 ci -= 1;
263 }
264 i = ci;
265 }
266 let mut out = Vec::new();
267 let mut md = 0; // How deep the deepest inline runs.
268 for node in nodes {
269 match node {
270 Node::Done(item, d) => {
271 if d > md {
272 md = d;
273 }
274 push(&mut out, item);
275 }
276 Node::Delim { ch, len, .. } => {
277 if len > 0 {
278 md = md.max(1);
279 push(&mut out, Inline::Text(
280 std::iter::repeat(ch as char).take(len).collect()));
281 }
282 }
283 }
284 }
285 Ok((out, md))
286}
287
288/// Adds an inline, joining it to the run before it when both are text.
289fn push(out: &mut Vec<Inline>, item: Inline) {
290 if let Inline::Text(t) = &item {
291 if let Some(Inline::Text(last)) = out.last_mut() {
292 last.push_str(t);
293 return;
294 }
295 }
296 out.push(item);
297}
298
299/// Whether a run of emphasis characters may open, and may close, by what sits either side of it.
300fn flank(src: &str, start: usize, end: usize, ch: u8) -> (bool, bool) {
301 let prev = src[..start].chars().next_back();
302 let next = src[end..].chars().next();
303 let pre_ws = match prev { Some(c) => c.is_whitespace(), None => true };
304 let post_ws = match next { Some(c) => c.is_whitespace(), None => true };
305 let pre_pn = match prev { Some(c) => is_punct(c), None => false };
306 let post_pn = match next { Some(c) => is_punct(c), None => false };
307 // A run leans left or right by which side has text against it: `*a` leans right onto the `a`,
308 // `a*` leans left onto it, and `a * b` leans nowhere and so is an asterisk.
309 let left = !post_ws && (!post_pn || pre_ws || pre_pn);
310 let right = !pre_ws && (!pre_pn || post_ws || post_pn);
311 if ch == b'_' {
312 // An underscore within a word is part of the word, so that an identifier survives.
313 (left && (!right || pre_pn), right && (!left || post_pn))
314 } else {
315 (left, right)
316 }
317}
318
319/// Whether the character counts as punctuation in judging a run of emphasis characters.
320///
321/// Beyond ASCII this is an approximation of Unicode's punctuation and symbol categories: what is
322/// neither a letter, a digit, a space nor a control is taken to be punctuation.
323fn is_punct(c: char) -> bool {
324 c.is_ascii_punctuation() || (!c.is_alphanumeric() && !c.is_whitespace() && !c.is_control())
325}
326
327// ── Code spans, links and images ─────────────────────────────────
328
329/// A code span opened by a run of `n` backticks at `i`, and the offset just past it.
330fn code_span(src: &str, i: usize, n: usize) -> Option<(String, usize)> {
331 let b = src.as_bytes();
332 let mut j = i + n;
333 while j < b.len() {
334 if b[j] == b'`' {
335 // Only a run of the same length closes: a longer or shorter one is code.
336 let m = run_len(b, j, b'`');
337 if m == n {
338 return Some((code_text(&src[i + n..j]), j + m));
339 }
340 j += m;
341 continue;
342 }
343 j += 1;
344 }
345 None
346}
347
348/// A code span's text: line endings become spaces, and a space at each end is dropped so that a span
349/// may hold a backtick of its own.
350fn code_text(raw: &str) -> String {
351 let s: String = raw.chars().map(|c| if c == '\n' { ' ' } else { c }).collect();
352 if s.len() >= 2 && s.starts_with(' ') && s.ends_with(' ') && !s.trim().is_empty() {
353 s[1..s.len() - 1].to_string()
354 } else {
355 s
356 }
357}
358
359/// An autolink at the offset: the URI it names, and the offset just past it.
360fn autolink(src: &str, i: usize) -> Option<(String, usize)> {
361 let b = src.as_bytes();
362 let mut j = i + 1;
363 while j < b.len() {
364 match b[j] {
365 b'>' => break,
366 b'<' | b' ' | b'\t' | b'\n' => return None,
367 _ => j += 1,
368 }
369 }
370 if j >= b.len() {
371 return None;
372 }
373 let inner = &src[i + 1..j];
374 if !is_uri(inner) {
375 return None;
376 }
377 Some((inner.to_string(), j + 1))
378}
379
380/// Whether the text is a URI with a scheme, which is what an autolink must be.
381fn is_uri(s: &str) -> bool {
382 let colon = match s.find(':') {
383 Some(c) => c,
384 None => return false,
385 };
386 let b = s.as_bytes();
387 if colon < 2 || colon > 32 || !b[0].is_ascii_alphabetic() {
388 return false;
389 }
390 b[1..colon].iter().all(|c| c.is_ascii_alphanumeric() || *c == b'+' || *c == b'.' || *c == b'-')
391}
392
393/// A bracketed link at the offset: its text, its destination, and the offset just past it.
394fn bracket(src: &str, i: usize) -> Option<(String, String, usize)> {
395 let b = src.as_bytes();
396 let mut j = i + 1;
397 let mut d = 1; // Bracket depth.
398 while j < b.len() {
399 match b[j] {
400 b'\\' => {
401 j = skip_esc(src, j);
402 continue;
403 }
404 b'`' => {
405 // A bracket within a code span is code, not a bracket.
406 let n = run_len(b, j, b'`');
407 j = match code_span(src, j, n) {
408 Some((_, end)) => end,
409 None => j + n,
410 };
411 continue;
412 }
413 b'[' => d += 1,
414 b']' => {
415 d -= 1;
416 if d == 0 {
417 break;
418 }
419 }
420 _ => {}
421 }
422 j += 1;
423 }
424 if j >= b.len() || d != 0 || j + 1 >= b.len() || b[j + 1] != b'(' {
425 return None; // Without a destination against it, a bracket is a bracket.
426 }
427 match dest(src, j + 1) {
428 Some((to, end)) => Some((src[i + 1..j].to_string(), to, end)),
429 None => None,
430 }
431}
432
433/// A link destination in parentheses at the offset: the destination, and the offset just past the
434/// parenthesis that closes it.
435///
436/// A title, which this tree keeps no room for, is read only so as to be stepped over.
437fn dest(src: &str, i: usize) -> Option<(String, usize)> {
438 let b = src.as_bytes();
439 let mut j = skip_ws(b, i + 1);
440 let to;
441 if j < b.len() && b[j] == b'<' {
442 // An angled destination runs to its closing angle, and may hold spaces.
443 let s = j + 1;
444 let mut k = s;
445 while k < b.len() && b[k] != b'>' && b[k] != b'\n' {
446 k = if b[k] == b'\\' { skip_esc(src, k) } else { k + 1 };
447 }
448 if k >= b.len() || b[k] != b'>' {
449 return None;
450 }
451 to = unescape(&src[s..k]);
452 j = k + 1;
453 } else {
454 let s = j;
455 let mut d = 0; // Parenthesis depth.
456 let mut k = j;
457 while k < b.len() {
458 match b[k] {
459 b'\\' => {
460 k = skip_esc(src, k);
461 continue;
462 }
463 b'(' => d += 1,
464 b')' => {
465 if d == 0 {
466 break;
467 }
468 d -= 1;
469 }
470 b' ' | b'\t' | b'\n' => break,
471 _ => {}
472 }
473 k += 1;
474 }
475 to = unescape(&src[s..k]);
476 j = k;
477 }
478 j = skip_ws(b, j);
479 if j < b.len() && (b[j] == b'"' || b[j] == b'\'' || b[j] == b'(') {
480 let shut = if b[j] == b'(' { b')' } else { b[j] };
481 let mut k = j + 1;
482 while k < b.len() && b[k] != shut {
483 k = if b[k] == b'\\' { skip_esc(src, k) } else { k + 1 };
484 }
485 if k >= b.len() {
486 return None;
487 }
488 j = skip_ws(b, k + 1);
489 }
490 if j >= b.len() || b[j] != b')' {
491 return None;
492 }
493 Some((to, j + 1))
494}
495
496/// Text with its backslash escapes of punctuation resolved.
497fn unescape(s: &str) -> String {
498 let b = s.as_bytes();
499 let mut out = String::new();
500 let mut i = 0;
501 while i < b.len() {
502 if b[i] == b'\\' && i + 1 < b.len() && b[i + 1].is_ascii_punctuation() {
503 out.push(b[i + 1] as char);
504 i += 2;
505 continue;
506 }
507 let j = char_end(s, i);
508 out.push_str(&s[i..j]);
509 i = j;
510 }
511 out
512}
513
514// ── Offsets ──────────────────────────────────────────────────────
515
516/// How many of the given character run on from the offset.
517fn run_len(b: &[u8], i: usize, ch: u8) -> usize {
518 let mut n = 0;
519 while i + n < b.len() && b[i + n] == ch {
520 n += 1;
521 }
522 n
523}
524
525/// The offset just past the character at the offset.
526fn char_end(s: &str, i: usize) -> usize {
527 let mut j = i + 1;
528 while j < s.len() && !s.is_char_boundary(j) {
529 j += 1;
530 }
531 j
532}
533
534/// The offset just past a backslash at `i` and whatever it escapes.
535fn skip_esc(s: &str, i: usize) -> usize {
536 if i + 1 < s.len() { char_end(s, i + 1) } else { i + 1 }
537}
538
539/// The offset past any spaces, tabs and line endings at `i`.
540fn skip_ws(b: &[u8], i: usize) -> usize {
541 let mut j = i;
542 while j < b.len() && (b[j] == b' ' || b[j] == b'\t' || b[j] == b'\n') {
543 j += 1;
544 }
545 j
546}
547
548#[cfg(test)]
549mod tests {
550 use super::*;
551
552 /// A run of literal text, for the tests that expect one.
553 fn t(s: &str) -> Inline {
554 Inline::Text(s.to_string())
555 }
556
557 /// Plain prose is one run of text and not a run for every character.
558 #[test]
559 fn test_plain_prose_is_one_run_of_text_00() -> Outcome<()> {
560 assert_eq!(res!(parse("Just some prose.")), vec![t("Just some prose.")]);
561 Ok(())
562 }
563
564 /// One asterisk or underscore either side makes ordinary emphasis.
565 #[test]
566 fn test_one_character_either_side_emphasises_01() -> Outcome<()> {
567 let want = vec![t("a "), Inline::Emph { strong: false, content: vec![t("b")] }, t(" c")];
568 assert_eq!(res!(parse("a *b* c")), want);
569 assert_eq!(res!(parse("a _b_ c")), want);
570 Ok(())
571 }
572
573 /// Two characters either side make strong emphasis.
574 #[test]
575 fn test_two_characters_either_side_emphasise_strongly_02() -> Outcome<()> {
576 let want = vec![t("a "), Inline::Emph { strong: true, content: vec![t("b")] }, t(" c")];
577 assert_eq!(res!(parse("a **b** c")), want);
578 assert_eq!(res!(parse("a __b__ c")), want);
579 Ok(())
580 }
581
582 /// Emphasis nests within emphasis.
583 #[test]
584 fn test_emphasis_nests_03() -> Outcome<()> {
585 assert_eq!(res!(parse("*a **b** c*")), vec![
586 Inline::Emph {
587 strong: false,
588 content: vec![
589 t("a "),
590 Inline::Emph { strong: true, content: vec![t("b")] },
591 t(" c"),
592 ],
593 },
594 ]);
595 // Three characters either side are both at once.
596 assert_eq!(res!(parse("***a***")), vec![
597 Inline::Emph {
598 strong: false,
599 content: vec![Inline::Emph { strong: true, content: vec![t("a")] }],
600 },
601 ]);
602 Ok(())
603 }
604
605 /// An asterisk with space around it is arithmetic, not emphasis.
606 #[test]
607 fn test_an_asterisk_alone_is_an_asterisk_04() -> Outcome<()> {
608 assert_eq!(res!(parse("2 * 3 * 4")), vec![t("2 * 3 * 4")]);
609 assert_eq!(res!(parse("a * b")), vec![t("a * b")]);
610 Ok(())
611 }
612
613 /// An asterisk that opens nothing that closes is an asterisk.
614 #[test]
615 fn test_an_unmatched_delimiter_is_text_05() -> Outcome<()> {
616 assert_eq!(res!(parse("*not emphasis")), vec![t("*not emphasis")]);
617 assert_eq!(res!(parse("closing only*")), vec![t("closing only*")]);
618 // The nearest opener wins, and the one left over is text.
619 assert_eq!(res!(parse("*a *b*")), vec![
620 t("*a "),
621 Inline::Emph { strong: false, content: vec![t("b")] },
622 ]);
623 Ok(())
624 }
625
626 /// An underscore within a word is part of the word, so an identifier survives.
627 #[test]
628 fn test_an_underscore_within_a_word_is_a_word_06() -> Outcome<()> {
629 assert_eq!(res!(parse("snake_case_name")), vec![t("snake_case_name")]);
630 // An asterisk within a word still emphasises, as Markdown has it.
631 assert_eq!(res!(parse("a*b*c")), vec![
632 t("a"),
633 Inline::Emph { strong: false, content: vec![t("b")] },
634 t("c"),
635 ]);
636 Ok(())
637 }
638
639 /// Backticks make a code span, and what is in it is exactly what was written.
640 #[test]
641 fn test_backticks_make_a_code_span_07() -> Outcome<()> {
642 assert_eq!(res!(parse("a `let x = *y*;` b")), vec![
643 t("a "),
644 Inline::Code("let x = *y*;".to_string()),
645 t(" b"),
646 ]);
647 Ok(())
648 }
649
650 /// A longer run of backticks lets a span hold a backtick of its own.
651 #[test]
652 fn test_a_longer_run_of_backticks_holds_one_08() -> Outcome<()> {
653 assert_eq!(res!(parse("`` a ` b ``")), vec![Inline::Code("a ` b".to_string())]);
654 // A run nothing closes is only backticks.
655 assert_eq!(res!(parse("`` unclosed")), vec![t("`` unclosed")]);
656 Ok(())
657 }
658
659 /// A bracket against a destination is a link.
660 #[test]
661 fn test_a_bracket_against_a_destination_is_a_link_09() -> Outcome<()> {
662 assert_eq!(res!(parse("[text](https://a.b)")), vec![
663 Inline::Link {
664 to: "https://a.b".to_string(),
665 content: vec![t("text")],
666 },
667 ]);
668 // A link's text is prose in its own right.
669 assert_eq!(res!(parse("[a *b*](c)")), vec![
670 Inline::Link {
671 to: "c".to_string(),
672 content: vec![t("a "), Inline::Emph { strong: false, content: vec![t("b")] }],
673 },
674 ]);
675 Ok(())
676 }
677
678 /// A destination may be angled, may be nothing, and may be followed by a title nobody keeps.
679 #[test]
680 fn test_a_destination_takes_several_forms_10() -> Outcome<()> {
681 assert_eq!(res!(parse("[a](<b c>)")), vec![
682 Inline::Link { to: "b c".to_string(), content: vec![t("a")] },
683 ]);
684 assert_eq!(res!(parse("[a]()")), vec![
685 Inline::Link { to: String::new(), content: vec![t("a")] },
686 ]);
687 assert_eq!(res!(parse("[a](b \"a title\")")), vec![
688 Inline::Link { to: "b".to_string(), content: vec![t("a")] },
689 ]);
690 Ok(())
691 }
692
693 /// A bracket with no destination against it is a bracket.
694 #[test]
695 fn test_a_bracket_with_no_destination_is_text_11() -> Outcome<()> {
696 assert_eq!(res!(parse("[just brackets]")), vec![t("[just brackets]")]);
697 assert_eq!(res!(parse("an [unclosed bracket")), vec![t("an [unclosed bracket")]);
698 assert_eq!(res!(parse("a ] alone")), vec![t("a ] alone")]);
699 Ok(())
700 }
701
702 /// A bang before a link makes an image, whose alt is its text in words.
703 #[test]
704 fn test_a_bang_before_a_link_makes_an_image_12() -> Outcome<()> {
705 assert_eq!(res!(parse("![a picture](p.png)")), vec![
706 Inline::Image { src: "p.png".to_string(), alt: "a picture".to_string() },
707 ]);
708 // The alt is flattened, so emphasis within it does not lose its words.
709 assert_eq!(res!(parse("![a *loud* picture](p.png)")), vec![
710 Inline::Image { src: "p.png".to_string(), alt: "a loud picture".to_string() },
711 ]);
712 // A bang before anything else is a bang.
713 assert_eq!(res!(parse("Look! [here](x)")), vec![
714 t("Look! "),
715 Inline::Link { to: "x".to_string(), content: vec![t("here")] },
716 ]);
717 Ok(())
718 }
719
720 /// A URI in angle brackets is a link to itself.
721 #[test]
722 fn test_a_uri_in_angle_brackets_is_a_link_13() -> Outcome<()> {
723 assert_eq!(res!(parse("See <https://a.b/c>.")), vec![
724 t("See "),
725 Inline::Link {
726 to: "https://a.b/c".to_string(),
727 content: vec![t("https://a.b/c")],
728 },
729 t("."),
730 ]);
731 // Without a scheme, angle brackets are angle brackets.
732 assert_eq!(res!(parse("a <b> c")), vec![t("a <b> c")]);
733 assert_eq!(res!(parse("1 < 2")), vec![t("1 < 2")]);
734 Ok(())
735 }
736
737 /// Two spaces before a line ending are a break the author asked for.
738 #[test]
739 fn test_trailing_spaces_make_a_break_14() -> Outcome<()> {
740 assert_eq!(res!(parse("one \ntwo")), vec![t("one"), Inline::Break, t("two")]);
741 // Three or more do as well.
742 assert_eq!(res!(parse("one \ntwo")), vec![t("one"), Inline::Break, t("two")]);
743 Ok(())
744 }
745
746 /// A line ending the author did not ask for is a space, so that prose reflows.
747 ///
748 /// Prose is hard wrapped to whatever width its author was writing at. That width is an artefact
749 /// of their editor and means nothing, so a soft break says a space and the reader's window
750 /// decides where the lines fall.
751 #[test]
752 fn test_a_soft_line_ending_is_a_space_23() -> Outcome<()> {
753 assert_eq!(res!(parse("one\ntwo")), vec![t("one two")]);
754 // A single trailing space is not two, so it is soft, and says one space and not two.
755 assert_eq!(res!(parse("one \ntwo")), vec![t("one two")]);
756 // A paragraph wrapped across three lines is one run that reflows.
757 assert_eq!(res!(parse("A paragraph that the author\nhard wrapped at a narrow width\nacross three lines.")),
758 vec![t("A paragraph that the author hard wrapped at a narrow width across three lines.")]);
759 Ok(())
760 }
761
762 /// A soft break at either end of a run says nothing, and leaves no space behind.
763 #[test]
764 fn test_a_soft_line_ending_at_an_end_says_nothing_24() -> Outcome<()> {
765 assert_eq!(res!(parse("one\n")), vec![t("one")]);
766 assert_eq!(res!(parse("\none")), vec![t("one")]);
767 assert_eq!(res!(parse("one \n")), vec![t("one")]);
768 assert_eq!(res!(parse("\n")), Vec::<Inline>::new());
769 Ok(())
770 }
771
772 /// A soft break beside an inline is still a space, and joins the text around it.
773 #[test]
774 fn test_a_soft_line_ending_beside_an_inline_25() -> Outcome<()> {
775 assert_eq!(res!(parse("*a*\nb")), vec![
776 Inline::Emph { strong: false, content: vec![t("a")] },
777 t(" b"),
778 ]);
779 assert_eq!(res!(parse("a\n`b`")), vec![t("a "), Inline::Code("b".to_string())]);
780 Ok(())
781 }
782
783 /// A backslash before a line ending is a break the author asked for.
784 #[test]
785 fn test_a_backslash_before_a_line_ending_breaks_15() -> Outcome<()> {
786 assert_eq!(res!(parse("one\\\ntwo")), vec![t("one"), Inline::Break, t("two")]);
787 Ok(())
788 }
789
790 /// A backslash before punctuation says the punctuation is only itself.
791 #[test]
792 fn test_a_backslash_escapes_punctuation_16() -> Outcome<()> {
793 assert_eq!(res!(parse("\\*not emphasis\\*")), vec![t("*not emphasis*")]);
794 assert_eq!(res!(parse("\\[not a link\\]")), vec![t("[not a link]")]);
795 assert_eq!(res!(parse("a \\\\ backslash")), vec![t("a \\ backslash")]);
796 // Before anything else, a backslash is a backslash.
797 assert_eq!(res!(parse("\\a")), vec![t("\\a")]);
798 Ok(())
799 }
800
801 /// Text either side of an inline joins into one run, and is never split per character.
802 #[test]
803 fn test_text_around_an_inline_coalesces_17() -> Outcome<()> {
804 // The delimiters that came to nothing rejoin the text around them.
805 match &res!(parse("a * b _ c"))[..] {
806 [Inline::Text(s)] => assert_eq!(s, "a * b _ c"),
807 other => panic!("expected one run of text, got {:?}", other),
808 }
809 // Text on both sides of emphasis is one run each side.
810 let got = res!(parse("one *two* three four"));
811 assert_eq!(got.len(), 3);
812 assert_eq!(got[2], t(" three four"));
813 Ok(())
814 }
815
816 /// Punctuation and Unicode either side of a delimiter do not stop emphasis.
817 #[test]
818 fn test_emphasis_survives_punctuation_around_it_18() -> Outcome<()> {
819 assert_eq!(res!(parse("(*a*)")), vec![
820 t("("),
821 Inline::Emph { strong: false, content: vec![t("a")] },
822 t(")"),
823 ]);
824 assert_eq!(res!(parse("*naïve*")), vec![
825 Inline::Emph { strong: false, content: vec![t("naïve")] },
826 ]);
827 Ok(())
828 }
829
830 /// A code span outranks the emphasis and brackets that appear to be within it.
831 #[test]
832 fn test_a_code_span_outranks_what_is_in_it_19() -> Outcome<()> {
833 assert_eq!(res!(parse("`[a](b)`")), vec![Inline::Code("[a](b)".to_string())]);
834 // And a code span within a link's text is code.
835 assert_eq!(res!(parse("[`a]`](b)")), vec![
836 Inline::Link { to: "b".to_string(), content: vec![Inline::Code("a]".to_string())] },
837 ]);
838 Ok(())
839 }
840
841 /// Nothing at all is a run of nothing.
842 #[test]
843 fn test_nothing_is_a_run_of_nothing_20() -> Outcome<()> {
844 assert_eq!(res!(parse("")), Vec::<Inline>::new());
845 Ok(())
846 }
847
848 /// Emphasis nested past the limit is refused, as deep block nesting is.
849 ///
850 /// A long run of emphasis characters nests a level for every pair it spends, and spends two a
851 /// level when it makes strong emphasis. So the limit is reached at twice its own count.
852 #[test]
853 fn test_nesting_past_the_limit_is_refused_21() -> Outcome<()> {
854 // Nesting the limit allows is read.
855 let ok = format!("{}a{}", "*".repeat(2 * (DEPTH_LIMIT - 1)), "*".repeat(2 * (DEPTH_LIMIT - 1)));
856 assert!(parse(&ok).is_ok());
857 // Past it is not, and a run built to exhaust the stack of whatever reads the tree is
858 // refused rather than built.
859 for n in [2 * DEPTH_LIMIT, 2 * DEPTH_LIMIT + 8, 100_000] {
860 let src = format!("{}a{}", "*".repeat(n), "*".repeat(n));
861 assert!(parse(&src).is_err(), "for a run of {}", n);
862 }
863 // Links nested past the limit are refused likewise.
864 let mut deep = "x".to_string();
865 for _ in 0..DEPTH_LIMIT + 4 {
866 deep = format!("[{}](y)", deep);
867 }
868 assert!(parse(&deep).is_err());
869 Ok(())
870 }
871
872 /// A run of characters that emphasises across a line ending still does.
873 #[test]
874 fn test_emphasis_crosses_a_line_ending_22() -> Outcome<()> {
875 assert_eq!(res!(parse("*a\nb*")), vec![
876 Inline::Emph { strong: false, content: vec![t("a b")] },
877 ]);
878 Ok(())
879 }
880}