Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/doc/djot/inline.rs

31.6 KiB, 1 run

created by r1870400018:14700, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Inline structure: the pass that reads a line of prose into text, emphasis, links, images, code
2//! spans, attributed spans and hard breaks.
3//!
4//! This is the second of the two passes described in [`crate::doc::djot::block`]. It is given the text
5//! of one block, along with the document's reference definitions, and returns the run of inlines it is
6//! made of.
7//!
8//! # Where Djot parts from Markdown
9//!
10//! The markers are swapped and single. A single `_` is ordinary emphasis and a single `*` is strong,
11//! so Djot needs no doubling: `_it_` is italic and `*it*` is bold, the reverse of a Markdown reader's
12//! reflex. A run of a marker either side opens and closes by whitespace alone -- a marker may sit
13//! against a word on the inside and not on the outside -- which is what lets emphasis fall inside a
14//! word without the intraword rule Markdown carries for its underscore.
15//!
16//! A bracket that is not a link may be a span. `[text]{.c}` is an [`Inline::Span`] carrying the
17//! attributes in the braces, the inline counterpart of a `:::` division; `[text](url)` is a link as
18//! ever; and `[text]` on its own is a reference link where the document defined the reference, and
19//! literal brackets where it did not.
20
21use crate::doc::{
22 Attrs,
23 Inline,
24 text_of,
25 djot::block::DEPTH_LIMIT,
26};
27
28use std::collections::HashMap;
29
30use oxedyne_fe2o3_core::prelude::*;
31
32/// Reads a block's text into its run of inline elements.
33///
34/// The reference definitions are the map [`crate::doc::djot::block`] gathered in its first pass, by
35/// which a `[text][ref]` or a bare `[text]` resolves to the destination the document named elsewhere.
36pub fn parse(src: &str, refs: &HashMap<String, String>) -> Outcome<Vec<Inline>> {
37 run(src, 0, refs)
38}
39
40/// One element of the scan: either an inline that is settled, or a run of emphasis characters still
41/// looking for the run that answers it.
42///
43/// Emphasis cannot be read in one pass, because whether a `*` opens anything is only known once
44/// something closes it. So the scan lays the delimiters out alongside the text it is sure of, and
45/// [`resolve`] pairs them off afterwards.
46enum Node {
47 /// An inline that needs nothing more, and how deep the tree it makes runs.
48 Done(Inline, usize),
49 /// A run of emphasis characters.
50 Delim {
51 /// The character the run is made of.
52 ch: u8,
53 /// How many characters the run has left to spend.
54 len: usize,
55 /// Whether the run may open emphasis.
56 open: bool,
57 /// Whether the run may close emphasis.
58 close: bool,
59 },
60}
61
62/// Reads a run of text into inlines, at the given nesting depth.
63fn run(src: &str, depth: usize, refs: &HashMap<String, String>) -> Outcome<Vec<Inline>> {
64 if depth > DEPTH_LIMIT {
65 return Err(err!(
66 "Djot inlines nest more than {} deep, which no prose written to be read \
67 does.", DEPTH_LIMIT;
68 Excessive, Input));
69 }
70 let nodes = res!(scan(src, depth, refs));
71 let (out, _) = res!(resolve(nodes, depth));
72 Ok(out)
73}
74
75/// Lays the text out as settled inlines and unsettled emphasis delimiters.
76fn scan(src: &str, depth: usize, refs: &HashMap<String, String>) -> Outcome<Vec<Node>> {
77 let mut out: Vec<Node> = Vec::new();
78 let mut buf = String::new(); // Text gathered since the last settled inline.
79 let b = src.as_bytes();
80 let mut i = 0;
81 while i < b.len() {
82 match b[i] {
83 b'\\' => {
84 if i + 1 < b.len() && b[i + 1] == b'\n' {
85 // A backslash at the end of a line is a break the author asked for.
86 flush(&mut out, &mut buf);
87 out.push(Node::Done(Inline::Break, 1));
88 i += 2;
89 } else if i + 1 < b.len() && b[i + 1].is_ascii_punctuation() {
90 // A backslash before punctuation says the punctuation is only itself.
91 buf.push(b[i + 1] as char);
92 i += 2;
93 } else {
94 buf.push('\\');
95 i += 1;
96 }
97 }
98 b'\n' => {
99 // Djot has no two-space hard break, so a bare line ending is always soft. The
100 // spaces before it mean nothing and are dropped.
101 let sp = buf.len() - buf.trim_end_matches(' ').len();
102 buf.truncate(buf.len() - sp);
103 if !(out.is_empty() && buf.is_empty()) && i + 1 < b.len() {
104 // A soft break: where the author's editor wrapped the line, not where the
105 // author meant a break. It says a space, so that prose hard wrapped to one
106 // width reflows to whatever width reads it. A soft break at either end of the
107 // run says nothing at all.
108 buf.push(' ');
109 }
110 i += 1;
111 }
112 b'`' => {
113 let n = run_len(b, i, b'`');
114 match code_span(src, i, n) {
115 Some((code, end)) => {
116 flush(&mut out, &mut buf);
117 out.push(Node::Done(Inline::Code(code), 1));
118 i = end;
119 }
120 None => {
121 // Nothing closed it, so the backticks are backticks.
122 for _ in 0..n {
123 buf.push('`');
124 }
125 i += n;
126 }
127 }
128 }
129 b'<' => {
130 match autolink(src, i) {
131 Some((to, end)) => {
132 flush(&mut out, &mut buf);
133 out.push(Node::Done(Inline::Link {
134 to: to.clone(),
135 content: vec![Inline::Text(to)],
136 }, 1));
137 i = end;
138 }
139 None => {
140 buf.push('<');
141 i += 1;
142 }
143 }
144 }
145 b'!' if i + 1 < b.len() && b[i + 1] == b'[' => {
146 match image(src, i, depth, refs) {
147 Some((res, end)) => {
148 flush(&mut out, &mut buf);
149 out.push(Node::Done(res!(res), 1));
150 i = end;
151 }
152 None => {
153 buf.push('!');
154 i += 1;
155 }
156 }
157 }
158 b'[' => {
159 match bracket(src, i, depth, refs) {
160 Some((res, end)) => {
161 flush(&mut out, &mut buf);
162 out.push(Node::Done(res!(res), 1));
163 i = end;
164 }
165 None => {
166 buf.push('[');
167 i += 1;
168 }
169 }
170 }
171 b'*' | b'_' => {
172 let ch = b[i];
173 let len = run_len(b, i, ch);
174 let (open, close) = flank(src, i, i + len);
175 flush(&mut out, &mut buf);
176 out.push(Node::Delim { ch, len, open, close });
177 i += len;
178 }
179 _ => {
180 // Anything else is itself, taken a whole character at a time.
181 let j = char_end(src, i);
182 buf.push_str(&src[i..j]);
183 i = j;
184 }
185 }
186 }
187 flush(&mut out, &mut buf);
188 Ok(out)
189}
190
191/// Adds the text gathered so far as one run, so that adjacent text is never split in two.
192fn flush(out: &mut Vec<Node>, buf: &mut String) {
193 if !buf.is_empty() {
194 out.push(Node::Done(Inline::Text(std::mem::take(buf)), 1));
195 }
196}
197
198// ── Emphasis ─────────────────────────────────────────────────────
199
200/// Pairs emphasis delimiters with the runs that answer them, and makes text of the rest.
201///
202/// Each run that could close is offered to the nearest run before it that could open. A run that
203/// finds no partner is not markup at all, and comes out as the characters it is made of -- which is
204/// why a stray asterisk is an asterisk. A pairing spends one character of each run, and the character
205/// decides the kind: a `*` makes strong emphasis and a `_` ordinary emphasis, the Djot reading rather
206/// than the Markdown one.
207///
208/// Returns the inlines, and how deep the deepest of them runs. The depth is carried rather than
209/// measured afterwards because a run of emphasis characters nests one level per pair without the
210/// parser recursing once: ten thousand asterisks would build a tree too deep to walk, and only a
211/// count kept as it is built catches that before it exists.
212fn resolve(mut nodes: Vec<Node>, depth: usize) -> Outcome<(Vec<Inline>, usize)> {
213 let mut i = 0;
214 while i < nodes.len() {
215 let ch = match &nodes[i] {
216 Node::Delim { ch, close: true, .. } => *ch,
217 _ => {
218 i += 1;
219 continue;
220 }
221 };
222 // Look back for the nearest run of the same character that could open.
223 let mut found = None;
224 let mut k = i;
225 while k > 0 {
226 k -= 1;
227 if let Node::Delim { ch: c, open: true, .. } = &nodes[k] {
228 if *c == ch {
229 found = Some(k);
230 break;
231 }
232 }
233 }
234 let k = match found {
235 Some(k) => k,
236 None => {
237 // Nothing opened it, so it closes nothing and is only what it looks like.
238 if let Node::Delim { close, .. } = &mut nodes[i] {
239 *close = false;
240 }
241 i += 1;
242 continue;
243 }
244 };
245 // Everything between the two runs is what they emphasise.
246 let inner: Vec<Node> = nodes.drain(k + 1..i).collect();
247 let (content, cd) = res!(resolve(inner, depth + 1));
248 let d = cd + 1;
249 if depth + d > DEPTH_LIMIT {
250 return Err(err!(
251 "Djot emphasis nests more than {} deep, which no prose written to be \
252 read does.", DEPTH_LIMIT;
253 Excessive, Input));
254 }
255 // A pairing spends one character of each run, whichever length the runs are.
256 if let Node::Delim { len, .. } = &mut nodes[k] {
257 *len -= 1;
258 }
259 if let Node::Delim { len, .. } = &mut nodes[k + 1] {
260 *len -= 1;
261 }
262 nodes.insert(k + 1, Node::Done(Inline::Emph { strong: ch == b'*', content }, d));
263 // The closing run now sits past the emphasis it made. A run with characters left over may
264 // still make more, so it is looked at again.
265 let mut ci = k + 2;
266 if matches!(&nodes[ci], Node::Delim { len: 0, .. }) {
267 nodes.remove(ci);
268 }
269 if matches!(&nodes[k], Node::Delim { len: 0, .. }) {
270 nodes.remove(k);
271 ci -= 1;
272 }
273 i = ci;
274 }
275 let mut out = Vec::new();
276 let mut md = 0; // How deep the deepest inline runs.
277 for node in nodes {
278 match node {
279 Node::Done(item, d) => {
280 if d > md {
281 md = d;
282 }
283 push(&mut out, item);
284 }
285 Node::Delim { ch, len, .. } => {
286 if len > 0 {
287 md = md.max(1);
288 push(&mut out, Inline::Text(
289 std::iter::repeat(ch as char).take(len).collect()));
290 }
291 }
292 }
293 }
294 Ok((out, md))
295}
296
297/// Adds an inline, joining it to the run before it when both are text.
298fn push(out: &mut Vec<Inline>, item: Inline) {
299 if let Inline::Text(t) = &item {
300 if let Some(Inline::Text(last)) = out.last_mut() {
301 last.push_str(t);
302 return;
303 }
304 }
305 out.push(item);
306}
307
308/// Whether a run of emphasis characters may open, and may close, by the whitespace either side of it.
309///
310/// Djot's rule is the plain one: a run may open where a non-space follows it, and close where a
311/// non-space precedes it. There is no intraword exception, because there is nothing to except: an
312/// underscore against letters on both sides both opens and closes, and so emphasis falls inside a word
313/// where an author writes it there.
314fn flank(src: &str, start: usize, end: usize) -> (bool, bool) {
315 let prev = src[..start].chars().next_back();
316 let next = src[end..].chars().next();
317 let pre_ws = match prev { Some(c) => c.is_whitespace(), None => true };
318 let post_ws = match next { Some(c) => c.is_whitespace(), None => true };
319 // Opens where a non-space follows; closes where a non-space precedes.
320 (!post_ws, !pre_ws)
321}
322
323// ── Code spans, links, images and spans ──────────────────────────
324
325/// A code span opened by a run of `n` backticks at `i`, and the offset just past it.
326fn code_span(src: &str, i: usize, n: usize) -> Option<(String, usize)> {
327 let b = src.as_bytes();
328 let mut j = i + n;
329 while j < b.len() {
330 if b[j] == b'`' {
331 // Only a run of the same length closes: a longer or shorter one is code.
332 let m = run_len(b, j, b'`');
333 if m == n {
334 return Some((code_text(&src[i + n..j]), j + m));
335 }
336 j += m;
337 continue;
338 }
339 j += 1;
340 }
341 None
342}
343
344/// A code span's text: line endings become spaces, and a space at each end is dropped so that a span
345/// may hold a backtick of its own.
346fn code_text(raw: &str) -> String {
347 let s: String = raw.chars().map(|c| if c == '\n' { ' ' } else { c }).collect();
348 if s.len() >= 2 && s.starts_with(' ') && s.ends_with(' ') && !s.trim().is_empty() {
349 s[1..s.len() - 1].to_string()
350 } else {
351 s
352 }
353}
354
355/// An autolink at the offset: the URI it names, and the offset just past it.
356fn autolink(src: &str, i: usize) -> Option<(String, usize)> {
357 let b = src.as_bytes();
358 let mut j = i + 1;
359 while j < b.len() {
360 match b[j] {
361 b'>' => break,
362 b'<' | b' ' | b'\t' | b'\n' => return None,
363 _ => j += 1,
364 }
365 }
366 if j >= b.len() {
367 return None;
368 }
369 let inner = &src[i + 1..j];
370 if !is_uri(inner) {
371 return None;
372 }
373 Some((inner.to_string(), j + 1))
374}
375
376/// Whether the text is a URI with a scheme, which is what an autolink must be.
377fn is_uri(s: &str) -> bool {
378 let colon = match s.find(':') {
379 Some(c) => c,
380 None => return false,
381 };
382 let b = s.as_bytes();
383 if colon < 2 || colon > 32 || !b[0].is_ascii_alphabetic() {
384 return false;
385 }
386 b[1..colon].iter().all(|c| c.is_ascii_alphanumeric() || *c == b'+' || *c == b'.' || *c == b'-')
387}
388
389/// What a `[` at the offset settles to, and the offset just past it, or nothing where the bracket is
390/// only a bracket.
391///
392/// The character against the closing bracket decides. A `(` makes an inline link, a `{` an attributed
393/// span, and a `[` a reference link; a bare bracket is a shortcut reference where the document defined
394/// its text, and otherwise nothing, so the caller writes the `[` as the character it is.
395fn bracket(src: &str, i: usize, depth: usize, refs: &HashMap<String, String>)
396 -> Option<(Outcome<Inline>, usize)>
397{
398 let b = src.as_bytes();
399 let close = match bracket_end(src, i) {
400 Some(c) => c,
401 None => return None,
402 };
403 let inner = &src[i + 1..close];
404 let after = close + 1;
405 // A destination against the bracket makes a link.
406 if after < b.len() && b[after] == b'(' {
407 if let Some((to, end)) = dest(src, after) {
408 return Some((link(to, inner, depth, refs), end));
409 }
410 }
411 // An attribute block against the bracket makes a span.
412 if after < b.len() && b[after] == b'{' {
413 if let Some((attrs, used)) = attrs_of(&src[after..]) {
414 return Some((span(attrs, inner, depth, refs), after + used));
415 }
416 }
417 // A second bracket names a reference the document may have defined.
418 if after < b.len() && b[after] == b'[' {
419 if let Some((label, end)) = ref_label(src, after) {
420 let key = if label.trim().is_empty() { normalise(inner) } else { normalise(&label) };
421 if let Some(to) = refs.get(&key) {
422 return Some((link(to.clone(), inner, depth, refs), end));
423 }
424 }
425 }
426 // A bracket on its own is a shortcut reference where its own text was defined.
427 if let Some(to) = refs.get(&normalise(inner)) {
428 return Some((link(to.clone(), inner, depth, refs), after));
429 }
430 None
431}
432
433/// What a `![` at the offset settles to, and the offset just past it, or nothing where it is not an
434/// image at all.
435fn image(src: &str, i: usize, depth: usize, refs: &HashMap<String, String>)
436 -> Option<(Outcome<Inline>, usize)>
437{
438 let b = src.as_bytes();
439 let close = match bracket_end(src, i + 1) {
440 Some(c) => c,
441 None => return None,
442 };
443 let inner = &src[i + 2..close];
444 let after = close + 1;
445 if after < b.len() && b[after] == b'(' {
446 if let Some((to, end)) = dest(src, after) {
447 return Some((img(to, inner, depth, refs), end));
448 }
449 }
450 if after < b.len() && b[after] == b'[' {
451 if let Some((label, end)) = ref_label(src, after) {
452 let key = if label.trim().is_empty() { normalise(inner) } else { normalise(&label) };
453 if let Some(to) = refs.get(&key) {
454 return Some((img(to.clone(), inner, depth, refs), end));
455 }
456 }
457 }
458 if let Some(to) = refs.get(&normalise(inner)) {
459 return Some((img(to.clone(), inner, depth, refs), after));
460 }
461 None
462}
463
464/// A link to `to`, its text read as prose in its own right.
465fn link(to: String, inner: &str, depth: usize, refs: &HashMap<String, String>) -> Outcome<Inline> {
466 let content = res!(run(inner, depth + 1, refs));
467 Ok(Inline::Link { to, content })
468}
469
470/// An image at `src`, its alt text the flattened words of what stood for it.
471fn img(to: String, inner: &str, depth: usize, refs: &HashMap<String, String>) -> Outcome<Inline> {
472 // An image stands for itself in words, so its alt is flattened.
473 let alt = text_of(&res!(run(inner, depth + 1, refs)));
474 Ok(Inline::Image { src: to, alt })
475}
476
477/// A span carrying the attributes in the braces, its content read as prose.
478fn span(attrs: Attrs, inner: &str, depth: usize, refs: &HashMap<String, String>) -> Outcome<Inline> {
479 let content = res!(run(inner, depth + 1, refs));
480 Ok(Inline::Span { attrs, content })
481}
482
483/// The offset of the `]` that closes the `[` at `i`, if one does.
484///
485/// Brackets nest, and a bracket within a code span is code and not a bracket, so both are stepped
486/// over the way the inline pass steps over them everywhere else.
487fn bracket_end(src: &str, i: usize) -> Option<usize> {
488 let b = src.as_bytes();
489 let mut j = i + 1;
490 let mut d = 1; // Bracket depth.
491 while j < b.len() {
492 match b[j] {
493 b'\\' => {
494 j = skip_esc(src, j);
495 continue;
496 }
497 b'`' => {
498 let n = run_len(b, j, b'`');
499 j = match code_span(src, j, n) {
500 Some((_, end)) => end,
501 None => j + n,
502 };
503 continue;
504 }
505 b'[' => d += 1,
506 b']' => {
507 d -= 1;
508 if d == 0 {
509 return Some(j);
510 }
511 }
512 _ => {}
513 }
514 j += 1;
515 }
516 None
517}
518
519/// The label of a `[ref]` at `i`, and the offset just past its closing bracket.
520///
521/// A collapsed reference `[]` gives an empty label, which the caller reads as a call to use the link's
522/// own text as the label instead.
523fn ref_label(src: &str, i: usize) -> Option<(String, usize)> {
524 let b = src.as_bytes();
525 let mut j = i + 1;
526 while j < b.len() {
527 match b[j] {
528 b'\\' => {
529 j = skip_esc(src, j);
530 continue;
531 }
532 b']' => return Some((src[i + 1..j].to_string(), j + 1)),
533 _ => j += 1,
534 }
535 }
536 None
537}
538
539/// A link destination in parentheses at the offset: the destination, and the offset just past the
540/// parenthesis that closes it.
541///
542/// A title, which this tree keeps no room for, is read only so as to be stepped over.
543fn dest(src: &str, i: usize) -> Option<(String, usize)> {
544 let b = src.as_bytes();
545 let mut j = skip_ws(b, i + 1);
546 let to;
547 if j < b.len() && b[j] == b'<' {
548 // An angled destination runs to its closing angle, and may hold spaces.
549 let s = j + 1;
550 let mut k = s;
551 while k < b.len() && b[k] != b'>' && b[k] != b'\n' {
552 k = if b[k] == b'\\' { skip_esc(src, k) } else { k + 1 };
553 }
554 if k >= b.len() || b[k] != b'>' {
555 return None;
556 }
557 to = unescape(&src[s..k]);
558 j = k + 1;
559 } else {
560 let s = j;
561 let mut d = 0; // Parenthesis depth.
562 let mut k = j;
563 while k < b.len() {
564 match b[k] {
565 b'\\' => {
566 k = skip_esc(src, k);
567 continue;
568 }
569 b'(' => d += 1,
570 b')' => {
571 if d == 0 {
572 break;
573 }
574 d -= 1;
575 }
576 b' ' | b'\t' | b'\n' => break,
577 _ => {}
578 }
579 k += 1;
580 }
581 to = unescape(&src[s..k]);
582 j = k;
583 }
584 j = skip_ws(b, j);
585 if j < b.len() && (b[j] == b'"' || b[j] == b'\'' || b[j] == b'(') {
586 let shut = if b[j] == b'(' { b')' } else { b[j] };
587 let mut k = j + 1;
588 while k < b.len() && b[k] != shut {
589 k = if b[k] == b'\\' { skip_esc(src, k) } else { k + 1 };
590 }
591 if k >= b.len() {
592 return None;
593 }
594 j = skip_ws(b, k + 1);
595 }
596 if j >= b.len() || b[j] != b')' {
597 return None;
598 }
599 Some((to, j + 1))
600}
601
602/// Text with its backslash escapes of punctuation resolved.
603fn unescape(s: &str) -> String {
604 let b = s.as_bytes();
605 let mut out = String::new();
606 let mut i = 0;
607 while i < b.len() {
608 if b[i] == b'\\' && i + 1 < b.len() && b[i + 1].is_ascii_punctuation() {
609 out.push(b[i + 1] as char);
610 i += 2;
611 continue;
612 }
613 let j = char_end(s, i);
614 out.push_str(&s[i..j]);
615 i = j;
616 }
617 out
618}
619
620// ── Attributes ───────────────────────────────────────────────────
621
622/// The attributes a `{...}` at the start of `src` names, and how many bytes it runs to, or nothing
623/// where the braces do not close.
624///
625/// Shared by the inline span, the `:::` division and the standalone attributes line, so that all three
626/// read a brace group the one way. A quoted value may hold a `}` of its own, so the scan for the
627/// closing brace steps over what a quote encloses.
628pub fn attrs_of(src: &str) -> Option<(Attrs, usize)> {
629 let b = src.as_bytes();
630 if b.is_empty() || b[0] != b'{' {
631 return None;
632 }
633 let mut j = 1;
634 let mut quoted = false;
635 while j < b.len() {
636 match b[j] {
637 b'"' => quoted = !quoted,
638 b'}' if !quoted => return Some((parse_attrs(&src[1..j]), j + 1)),
639 _ => {}
640 }
641 j += 1;
642 }
643 None
644}
645
646/// Parses the interior of a brace group into its id, classes and pairs.
647///
648/// The items are space-separated: `.name` adds a class, `#name` sets the id, and `key=value` or
649/// `key="quoted value"` adds a pair. The reading is lenient, since an attribute block is the author's
650/// note to a stylesheet and not a program: what does not parse is passed over rather than refused.
651pub fn parse_attrs(inner: &str) -> Attrs {
652 let mut a = Attrs::default();
653 let b = inner.as_bytes();
654 let mut i = 0;
655 while i < b.len() {
656 match b[i] {
657 b' ' | b'\t' | b'\n' | b'\r' | b',' => i += 1,
658 b'.' => {
659 let s = i + 1;
660 let mut j = s;
661 while j < b.len() && !is_attr_sep(b[j]) {
662 j += 1;
663 }
664 if j > s {
665 a.classes.push(inner[s..j].to_string());
666 }
667 i = j;
668 }
669 b'#' => {
670 let s = i + 1;
671 let mut j = s;
672 while j < b.len() && !is_attr_sep(b[j]) {
673 j += 1;
674 }
675 if j > s {
676 // The last id written wins, as Djot has it.
677 a.id = Some(inner[s..j].to_string());
678 }
679 i = j;
680 }
681 _ => {
682 let s = i;
683 let mut j = i;
684 while j < b.len() && b[j] != b'=' && !is_attr_sep(b[j]) {
685 j += 1;
686 }
687 let key = &inner[s..j];
688 if j < b.len() && b[j] == b'=' {
689 let k = j + 1;
690 if k < b.len() && b[k] == b'"' {
691 let vs = k + 1;
692 let mut m = vs;
693 while m < b.len() && b[m] != b'"' {
694 m += 1;
695 }
696 if !key.is_empty() {
697 a.pairs.push((key.to_string(), inner[vs..m].to_string()));
698 }
699 i = if m < b.len() { m + 1 } else { m };
700 } else {
701 let vs = k;
702 let mut m = vs;
703 while m < b.len() && !is_attr_sep(b[m]) {
704 m += 1;
705 }
706 if !key.is_empty() {
707 a.pairs.push((key.to_string(), inner[vs..m].to_string()));
708 }
709 i = m;
710 }
711 } else {
712 // A bare word with no value names nothing this tree carries, so it is passed
713 // over. The offset still advances, so a stray character cannot loop.
714 i = if j > s { j } else { i + 1 };
715 }
716 }
717 }
718 }
719 a
720}
721
722/// Whether the byte ends an attribute item.
723fn is_attr_sep(c: u8) -> bool {
724 c == b' ' || c == b'\t' || c == b'\n' || c == b'\r' || c == b','
725}
726
727/// A reference label folded to the form two spellings of it share: trimmed, its inner whitespace
728/// collapsed, and its case set aside.
729///
730/// A reference is matched by what it names and not by how it was typed, so `[My Ref]` and `[my ref]`
731/// reach the one definition. The block pass folds a definition's label the same way, so the two meet.
732pub fn normalise(s: &str) -> String {
733 s.split_whitespace().collect::<Vec<_>>().join(" ").to_lowercase()
734}
735
736// ── Offsets ──────────────────────────────────────────────────────
737
738/// How many of the given character run on from the offset.
739fn run_len(b: &[u8], i: usize, ch: u8) -> usize {
740 let mut n = 0;
741 while i + n < b.len() && b[i + n] == ch {
742 n += 1;
743 }
744 n
745}
746
747/// The offset just past the character at the offset.
748fn char_end(s: &str, i: usize) -> usize {
749 let mut j = i + 1;
750 while j < s.len() && !s.is_char_boundary(j) {
751 j += 1;
752 }
753 j
754}
755
756/// The offset just past a backslash at `i` and whatever it escapes.
757fn skip_esc(s: &str, i: usize) -> usize {
758 if i + 1 < s.len() { char_end(s, i + 1) } else { i + 1 }
759}
760
761/// The offset past any spaces, tabs and line endings at `i`.
762fn skip_ws(b: &[u8], i: usize) -> usize {
763 let mut j = i;
764 while j < b.len() && (b[j] == b' ' || b[j] == b'\t' || b[j] == b'\n') {
765 j += 1;
766 }
767 j
768}
769
770#[cfg(test)]
771mod tests {
772 use super::*;
773
774 /// A run of literal text, for the tests that expect one.
775 fn t(s: &str) -> Inline {
776 Inline::Text(s.to_string())
777 }
778
779 /// Reads inline text with no reference definitions, which is what most tests want.
780 fn parse0(src: &str) -> Outcome<Vec<Inline>> {
781 parse(src, &HashMap::new())
782 }
783
784 /// Plain prose is one run of text and not a run for every character.
785 #[test]
786 fn test_plain_prose_is_one_run_of_text_00() -> Outcome<()> {
787 assert_eq!(res!(parse0("Just some prose.")), vec![t("Just some prose.")]);
788 Ok(())
789 }
790
791 /// An underscore either side is ordinary emphasis, and an asterisk either side is strong: the
792 /// Djot reading, and the reverse of Markdown's.
793 #[test]
794 fn test_the_markers_are_the_djot_way_round_01() -> Outcome<()> {
795 assert_eq!(res!(parse0("an _italic_ word")), vec![
796 t("an "),
797 Inline::Emph { strong: false, content: vec![t("italic")] },
798 t(" word"),
799 ]);
800 assert_eq!(res!(parse0("a *bold* word")), vec![
801 t("a "),
802 Inline::Emph { strong: true, content: vec![t("bold")] },
803 t(" word"),
804 ]);
805 Ok(())
806 }
807
808 /// The markers are not swapped: `_` is never strong and `*` is never ordinary.
809 #[test]
810 fn test_the_markers_are_not_swapped_02() -> Outcome<()> {
811 match &res!(parse0("_x_"))[0] {
812 Inline::Emph { strong, .. } => assert!(!strong, "an underscore is ordinary emphasis"),
813 other => panic!("expected emphasis, got {:?}", other),
814 }
815 match &res!(parse0("*x*"))[0] {
816 Inline::Emph { strong, .. } => assert!(*strong, "an asterisk is strong emphasis"),
817 other => panic!("expected emphasis, got {:?}", other),
818 }
819 Ok(())
820 }
821
822 /// Emphasis nests within emphasis, of either kind.
823 #[test]
824 fn test_emphasis_nests_03() -> Outcome<()> {
825 assert_eq!(res!(parse0("_a *b* c_")), vec![
826 Inline::Emph {
827 strong: false,
828 content: vec![
829 t("a "),
830 Inline::Emph { strong: true, content: vec![t("b")] },
831 t(" c"),
832 ],
833 },
834 ]);
835 Ok(())
836 }
837
838 /// Emphasis falls inside a word where an author writes it there, since Djot keeps no intraword
839 /// exception.
840 #[test]
841 fn test_emphasis_falls_inside_a_word_04() -> Outcome<()> {
842 assert_eq!(res!(parse0("a_b_c")), vec![
843 t("a"),
844 Inline::Emph { strong: false, content: vec![t("b")] },
845 t("c"),
846 ]);
847 Ok(())
848 }
849
850 /// A marker with space against it emphasises nothing, and stays the character it is.
851 #[test]
852 fn test_a_marker_with_space_is_a_character_05() -> Outcome<()> {
853 assert_eq!(res!(parse0("2 * 3 * 4")), vec![t("2 * 3 * 4")]);
854 assert_eq!(res!(parse0("a _ b")), vec![t("a _ b")]);
855 // A marker that opens nothing that closes is only itself.
856 assert_eq!(res!(parse0("*not strong")), vec![t("*not strong")]);
857 Ok(())
858 }
859
860 /// Backticks make a verbatim span, and what is in it is exactly what was written.
861 #[test]
862 fn test_backticks_make_a_verbatim_span_06() -> Outcome<()> {
863 assert_eq!(res!(parse0("a `let x = *y*;` b")), vec![
864 t("a "),
865 Inline::Code("let x = *y*;".to_string()),
866 t(" b"),
867 ]);
868 // A longer run lets a span hold a backtick of its own.
869 assert_eq!(res!(parse0("`` a ` b ``")), vec![Inline::Code("a ` b".to_string())]);
870 Ok(())
871 }
872
873 /// A bracket against a destination is a link, its text prose in its own right.
874 #[test]
875 fn test_a_bracket_against_a_destination_is_a_link_07() -> Outcome<()> {
876 assert_eq!(res!(parse0("[text](https://a.b)")), vec![
877 Inline::Link {
878 to: "https://a.b".to_string(),
879 content: vec![t("text")],
880 },
881 ]);
882 assert_eq!(res!(parse0("[a _b_](c)")), vec![
883 Inline::Link {
884 to: "c".to_string(),
885 content: vec![t("a "), Inline::Emph { strong: false, content: vec![t("b")] }],
886 },
887 ]);
888 Ok(())
889 }
890
891 /// A bang before a link makes an image, whose alt is its text in words.
892 #[test]
893 fn test_a_bang_before_a_link_makes_an_image_08() -> Outcome<()> {
894 assert_eq!(res!(parse0("![a picture](p.png)")), vec![
895 Inline::Image { src: "p.png".to_string(), alt: "a picture".to_string() },
896 ]);
897 // The alt is flattened, so emphasis within it does not lose its words.
898 assert_eq!(res!(parse0("![an _italic_ picture](p.png)")), vec![
899 Inline::Image { src: "p.png".to_string(), alt: "an italic picture".to_string() },
900 ]);
901 Ok(())
902 }
903
904 /// A bracket followed by an attribute block is a span, carrying the attributes the braces named.
905 #[test]
906 fn test_a_bracket_with_attributes_is_a_span_09() -> Outcome<()> {
907 assert_eq!(res!(parse0("[text]{.cls #id key=val}")), vec![
908 Inline::Span {
909 attrs: Attrs {
910 id: Some("id".to_string()),
911 classes: vec!["cls".to_string()],
912 pairs: vec![("key".to_string(), "val".to_string())],
913 },
914 content: vec![t("text")],
915 },
916 ]);
917 // A quoted value holds its spaces.
918 match &res!(parse0("[t]{key=\"a value\"}"))[0] {
919 Inline::Span { attrs, .. } => assert_eq!(
920 attrs.pairs,
921 vec![("key".to_string(), "a value".to_string())],
922 ),
923 other => panic!("expected a span, got {:?}", other),
924 }
925 Ok(())
926 }
927
928 /// A bracket against neither a destination nor an attribute block is only brackets.
929 #[test]
930 fn test_a_bare_bracket_is_text_10() -> Outcome<()> {
931 assert_eq!(res!(parse0("[just brackets]")), vec![t("[just brackets]")]);
932 assert_eq!(res!(parse0("an [unclosed bracket")), vec![t("an [unclosed bracket")]);
933 Ok(())
934 }
935
936 /// A reference link resolves to the destination the document defined, whether named or collapsed
937 /// or a bare shortcut.
938 #[test]
939 fn test_a_reference_link_resolves_11() -> Outcome<()> {
940 let mut refs = HashMap::new();
941 refs.insert("ref".to_string(), "https://x".to_string());
942 // A named reference.
943 assert_eq!(res!(parse("[text][ref]", &refs)), vec![
944 Inline::Link { to: "https://x".to_string(), content: vec![t("text")] },
945 ]);
946 // A collapsed reference reads the label from the text.
947 assert_eq!(res!(parse("[ref][]", &refs)), vec![
948 Inline::Link { to: "https://x".to_string(), content: vec![t("ref")] },
949 ]);
950 // A shortcut reference, the same way.
951 assert_eq!(res!(parse("[ref]", &refs)), vec![
952 Inline::Link { to: "https://x".to_string(), content: vec![t("ref")] },
953 ]);
954 // A reference nobody defined is only brackets.
955 assert_eq!(res!(parse("[undefined]", &refs)), vec![t("[undefined]")]);
956 Ok(())
957 }
958
959 /// A soft line ending says a space, so that prose reflows to whatever reads it.
960 #[test]
961 fn test_a_soft_line_ending_is_a_space_12() -> Outcome<()> {
962 assert_eq!(res!(parse0("one\ntwo")), vec![t("one two")]);
963 // A soft break at either end says nothing.
964 assert_eq!(res!(parse0("one\n")), vec![t("one")]);
965 assert_eq!(res!(parse0("\none")), vec![t("one")]);
966 Ok(())
967 }
968
969 /// A backslash before a line ending is a break the author asked for, and two trailing spaces are
970 /// not, since Djot has no such break.
971 #[test]
972 fn test_a_backslash_before_a_line_ending_breaks_13() -> Outcome<()> {
973 assert_eq!(res!(parse0("one\\\ntwo")), vec![t("one"), Inline::Break, t("two")]);
974 // Two trailing spaces are only a soft break, and say a space.
975 assert_eq!(res!(parse0("one \ntwo")), vec![t("one two")]);
976 Ok(())
977 }
978
979 /// A backslash before punctuation says the punctuation is only itself.
980 #[test]
981 fn test_a_backslash_escapes_punctuation_14() -> Outcome<()> {
982 assert_eq!(res!(parse0("\\*not strong\\*")), vec![t("*not strong*")]);
983 assert_eq!(res!(parse0("\\[not a link\\]")), vec![t("[not a link]")]);
984 Ok(())
985 }
986
987 /// Syntax the reader does not yet read survives as the text it is, and does not crash it.
988 #[test]
989 fn test_deferred_syntax_survives_as_text_15() -> Outcome<()> {
990 // Inline maths, a symbol, a footnote reference and a superscript are all not yet read.
991 assert_eq!(res!(parse0("an equation $x$ here")), vec![t("an equation $x$ here")]);
992 assert_eq!(res!(parse0("a smile :smile: here")), vec![t("a smile :smile: here")]);
993 assert_eq!(res!(parse0("a note[^1] here")), vec![t("a note[^1] here")]);
994 assert_eq!(res!(parse0("H~2~O and x^2^")), vec![t("H~2~O and x^2^")]);
995 Ok(())
996 }
997
998 /// Emphasis nested past the limit is refused, as it is in the block pass.
999 #[test]
1000 fn test_nesting_past_the_limit_is_refused_16() -> Outcome<()> {
1001 // A run the limit allows is read.
1002 let ok = format!("{}a{}", "*".repeat(DEPTH_LIMIT - 1), "*".repeat(DEPTH_LIMIT - 1));
1003 assert!(parse0(&ok).is_ok());
1004 // A run built to exhaust the stack of whatever reads the tree is refused rather than built.
1005 for n in [DEPTH_LIMIT + 2, 100_000] {
1006 let src = format!("{}a{}", "*".repeat(n), "*".repeat(n));
1007 assert!(parse0(&src).is_err(), "for a run of {}", n);
1008 }
1009 Ok(())
1010 }
1011
1012 /// A verbatim span outranks the emphasis and brackets that appear to be within it.
1013 #[test]
1014 fn test_a_verbatim_span_outranks_what_is_in_it_17() -> Outcome<()> {
1015 assert_eq!(res!(parse0("`[a](b)`")), vec![Inline::Code("[a](b)".to_string())]);
1016 Ok(())
1017 }
1018}