oxedyne/fe2o3/fe2o3_text/tests/xml.rs
11.3 KiB, 3 runs
created by r1870400018:22566, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | use oxedyne_fe2o3_text::xml::{ |
| 2 | Node, |
| 3 | Xml, |
| 4 | write::{ |
| 5 | Out, |
| 6 | decode, |
| 7 | escape, |
| 8 | escape_attr, |
| 9 | }, |
| 10 | }; |
| 11 | |
| 12 | use oxedyne_fe2o3_core::{ |
| 13 | prelude::*, |
| 14 | test::test_it, |
| 15 | }; |
| 16 | |
| 17 | /// The main part of a `.docx` LibreOffice wrote: fourteen namespace declarations, an `mc:Ignorable`, |
| 18 | /// and the paragraph-run-text nesting every Word document is made of. Somebody else's output, which |
| 19 | /// is the only kind worth reading a foreign format against. |
| 20 | const DOCUMENT: &str = include_str!("data/document.xml"); |
| 21 | |
| 22 | /// A synthetic document holding one of everything that is not an element, so the tiling check has |
| 23 | /// something to lose. |
| 24 | const MIXED: &str = concat!( |
| 25 | "<?xml version=\"1.0\"?>\n", |
| 26 | "<!DOCTYPE root>\n", |
| 27 | "<!-- a comment -->\n", |
| 28 | "<root xmlns=\"urn:d\" xmlns:a=\"urn:a\" xml:lang=\"en\">\n", |
| 29 | " text before <a:one k=\"v\" a:k=\"w\"/> text after\n", |
| 30 | " <two>a & b A </two>\n", |
| 31 | " <three><![CDATA[ <not> markup ]]></three>\n", |
| 32 | " <four attr=\"a > b\">held</four>\n", |
| 33 | "</root>\n", |
| 34 | ); |
| 35 | |
| 36 | /// Every node's span, in document order, concatenated. |
| 37 | /// |
| 38 | /// The property this exists to check is that the result is the source: nothing lost between two |
| 39 | /// nodes, and nothing counted twice. A span-preserving tree that did not tile would write back a |
| 40 | /// document with a hole in it, and no other check would see it. |
| 41 | fn tiled(xml: &Xml) -> String { |
| 42 | let mut out = String::new(); |
| 43 | for node in &xml.nodes { |
| 44 | walk(xml, node, &mut out); |
| 45 | } |
| 46 | out |
| 47 | } |
| 48 | |
| 49 | /// Adds a node's own bytes to the tiling, descending through an element rather than taking its whole |
| 50 | /// span, so a gap inside one is caught rather than papered over by its parent. |
| 51 | fn walk(xml: &Xml, node: &Node, out: &mut String) { |
| 52 | match node { |
| 53 | Node::Elem(e) => { |
| 54 | out.push_str(xml.raw(&e.open)); |
| 55 | for kid in &e.kids { |
| 56 | walk(xml, kid, out); |
| 57 | } |
| 58 | if let Some(inner) = &e.inner { |
| 59 | // Whatever is left of the element after its open tag and its content is its close |
| 60 | // tag, and that is the last thing it is made of. |
| 61 | out.push_str(&xml.source()[inner.end..e.span.end]); |
| 62 | } |
| 63 | } |
| 64 | other => out.push_str(xml.raw(&other.span())), |
| 65 | } |
| 66 | } |
| 67 | |
| 68 | pub fn test_xml(filter: &'static str) -> Outcome<()> { |
| 69 | |
| 70 | res!(test_it(filter, &["The spans tile the source exactly 000", "all", "xml"], || { |
| 71 | for (what, src) in [("the synthetic document", MIXED), ("a real word/document.xml", DOCUMENT)] { |
| 72 | let xml = res!(Xml::parse(src)); |
| 73 | let got = tiled(&xml); |
| 74 | assert_eq!(got.len(), src.len(), "{} did not tile: length differs", what); |
| 75 | match got == src { |
| 76 | true => {} |
| 77 | false => { |
| 78 | let at = got.char_indices().zip(src.chars()).position(|((_, a), b)| a != b); |
| 79 | panic!("{} did not tile: first difference at {:?}", what, at); |
| 80 | } |
| 81 | } |
| 82 | } |
| 83 | Ok(()) |
| 84 | })); |
| 85 | |
| 86 | // THIS CHECK IS VACUOUS BY CONSTRUCTION AND IS NOT DRESSED UP AS EVIDENCE. |
| 87 | // |
| 88 | // `render` copies the source and drops the splices into it. With no splices there is nothing to |
| 89 | // drop in, so it returns the source, and no implementation of `render` that anyone would write |
| 90 | // could fail this. It is here to say what the editing model IS, not to prove that it holds. |
| 91 | // |
| 92 | // The property with teeth is the one above it: the spans TILE the source. A reader whose spans |
| 93 | // had a hole in it would pass this and fail that, and a future reader who takes a green result |
| 94 | // here as proof that editing preserves a document has been told something this cannot say. |
| 95 | res!(test_it(filter, &["A document nobody edited renders as its source 001", "all", "xml"], || { |
| 96 | for src in [MIXED, DOCUMENT] { |
| 97 | let xml = res!(Xml::parse(src)); |
| 98 | assert!(xml.is_pristine()); |
| 99 | assert_eq!(xml.render(), src); |
| 100 | } |
| 101 | Ok(()) |
| 102 | })); |
| 103 | |
| 104 | res!(test_it(filter, &["A splice replaces its bytes and copies the rest 002", "all", "xml"], || { |
| 105 | // The whole editing model in one test. The theme, the settings and the tab stops of a real |
| 106 | // document are the elements this knows nothing about, and they come through because nothing |
| 107 | // re-serialises them. |
| 108 | let mut xml = res!(Xml::parse(DOCUMENT)); |
| 109 | let texts = xml.all("w:t"); |
| 110 | assert_eq!(texts.len(), 3, "the document holds three runs of text"); |
| 111 | let target = res!(texts.iter().find(|e| xml.text_of(e).starts_with("A paragraph")) |
| 112 | .ok_or_else(|| err!("the paragraph went missing"; Missing))); |
| 113 | let span = target.span.clone(); |
| 114 | let was = xml.raw(&span).to_string(); |
| 115 | res!(xml.splice(span, "<w:t>Replaced.</w:t>".to_string())); |
| 116 | let out = xml.render(); |
| 117 | assert!(!out.contains("A paragraph of ordinary prose"), "the old text went"); |
| 118 | assert!(out.contains("<w:t>Replaced.</w:t>"), "the new text arrived"); |
| 119 | // And everything else is exactly what it was: the rendered document differs from the source by |
| 120 | // that one substitution and by nothing else. |
| 121 | let mut expect = DOCUMENT.to_string(); |
| 122 | let at = res!(expect.find(&was).ok_or_else(|| err!("the run was not found"; Missing))); |
| 123 | expect.replace_range(at..at + was.len(), "<w:t>Replaced.</w:t>"); |
| 124 | assert_eq!(out, expect); |
| 125 | // It still parses, which is what says the splice did not break the document. |
| 126 | let again = res!(Xml::parse(&out)); |
| 127 | assert_eq!(again.all("w:p").len(), 5); |
| 128 | Ok(()) |
| 129 | })); |
| 130 | |
| 131 | res!(test_it(filter, &["Names resolve against the declarations in scope 003", "all", "xml"], || { |
| 132 | let xml = res!(Xml::parse(MIXED)); |
| 133 | let root = res!(xml.root()); |
| 134 | assert_eq!(root.name.qname, "root"); |
| 135 | assert_eq!(root.name.local(), "root"); |
| 136 | assert_eq!(root.name.prefix(), ""); |
| 137 | // An element with no prefix takes the default namespace. |
| 138 | let urn_d = res!(xml.uri_index("urn:d").ok_or_else(|| err!("urn:d not declared"; Missing))); |
| 139 | assert_eq!(root.name.ns, Some(urn_d)); |
| 140 | let one = res!(root.child("a:one").ok_or_else(|| err!("a:one is missing"; Missing))); |
| 141 | let urn_a = res!(xml.uri_index("urn:a").ok_or_else(|| err!("urn:a not declared"; Missing))); |
| 142 | assert_eq!(one.name.ns, Some(urn_a), "a prefixed element takes its prefix's namespace"); |
| 143 | assert_eq!(one.name.local(), "one"); |
| 144 | // An attribute with no prefix is in NO namespace, which is not the same rule as an element's. |
| 145 | // A reader that treated them alike would put every unprefixed attribute in the default. |
| 146 | let bare = res!(one.attrs.iter().find(|a| a.name.qname == "k") |
| 147 | .ok_or_else(|| err!("attribute k is missing"; Missing))); |
| 148 | assert_eq!(bare.name.ns, None, "an unprefixed attribute is in no namespace"); |
| 149 | let pref = res!(one.attrs.iter().find(|a| a.name.qname == "a:k") |
| 150 | .ok_or_else(|| err!("attribute a:k is missing"; Missing))); |
| 151 | assert_eq!(pref.name.ns, Some(urn_a)); |
| 152 | assert_eq!(one.attr("k"), Some("v")); |
| 153 | assert_eq!(one.attr("a:k"), Some("w")); |
| 154 | // `xml:` is bound everywhere and declared nowhere. |
| 155 | let lang = res!(root.attrs.iter().find(|a| a.name.qname == "xml:lang") |
| 156 | .ok_or_else(|| err!("xml:lang is missing"; Missing))); |
| 157 | assert!(lang.name.ns.is_some(), "the xml prefix is bound without a declaration"); |
| 158 | // And a real document's prefixes resolve too. |
| 159 | let doc = res!(Xml::parse(DOCUMENT)); |
| 160 | let w = res!(doc.uri_index("http://schemas.openxmlformats.org/wordprocessingml/2006/main") |
| 161 | .ok_or_else(|| err!("the wordprocessingml namespace is missing"; Missing))); |
| 162 | assert_eq!(res!(doc.root()).name.ns, Some(w)); |
| 163 | Ok(()) |
| 164 | })); |
| 165 | |
| 166 | res!(test_it(filter, &["The entities are the five XML has 004", "all", "xml"], || { |
| 167 | assert_eq!(decode("a & b < c > d " e '"), "a & b < c > d \" e '"); |
| 168 | assert_eq!(decode("AB☺"), "AB\u{263A}"); |
| 169 | // ` ` is HTML's, not XML's. Inventing a character here would make a malformed document |
| 170 | // look well formed, and the bug would surface somewhere with less context. |
| 171 | assert_eq!(decode("a b"), "a b"); |
| 172 | assert_eq!(decode("50% off & more"), "50% off & more", "a bare ampersand is left alone"); |
| 173 | assert_eq!(escape("a & b < c > d"), "a & b < c > d"); |
| 174 | assert_eq!(escape_attr("a \"b\" & c"), "a "b" & c"); |
| 175 | // The text of an element resolves them, and a CDATA section does not. |
| 176 | let xml = res!(Xml::parse(MIXED)); |
| 177 | let root = res!(xml.root()); |
| 178 | let two = res!(root.child("two").ok_or_else(|| err!("two is missing"; Missing))); |
| 179 | assert_eq!(xml.text_of(two), "a & b A "); |
| 180 | let three = res!(root.child("three").ok_or_else(|| err!("three is missing"; Missing))); |
| 181 | assert_eq!(xml.text_of(three), " <not> markup ", "a CDATA section says what it says"); |
| 182 | Ok(()) |
| 183 | })); |
| 184 | |
| 185 | res!(test_it(filter, &["An angle bracket inside a value ends nothing 005", "all", "xml"], || { |
| 186 | // The bug a reader that searched for `>` would have. `>` is legal in an attribute value and |
| 187 | // appears in real documents. |
| 188 | let xml = res!(Xml::parse("<a b=\"x > y\" c='p > q'>held</a>")); |
| 189 | let root = res!(xml.root()); |
| 190 | assert_eq!(root.attr("b"), Some("x > y")); |
| 191 | assert_eq!(root.attr("c"), Some("p > q"), "single quotes hold a value too"); |
| 192 | assert_eq!(xml.text_of(root), "held"); |
| 193 | Ok(()) |
| 194 | })); |
| 195 | |
| 196 | res!(test_it(filter, &["Malformed markup is refused by name 006", "all", "xml"], || { |
| 197 | // XML is not HTML: what a browser recovers from, this refuses, because the documents it reads |
| 198 | // are generator output and a generator emitting these has a bug worth hearing about. |
| 199 | for (what, src) in [ |
| 200 | ("a mismatched close tag", "<a><b></a></b>"), |
| 201 | ("an element left open", "<a><b>text</b>"), |
| 202 | ("a close tag with nothing open", "<a></a></b>"), |
| 203 | ("an unquoted attribute value", "<a b=c/>"), |
| 204 | ("a bare attribute", "<a b/>"), |
| 205 | ("an unclosed value", "<a b=\"c/>"), |
| 206 | ("an unclosed comment", "<a><!-- forever</a>"), |
| 207 | ("an unbound prefix", "<z:a/>"), |
| 208 | ] { |
| 209 | assert!(Xml::parse(src).is_err(), "{} was accepted: {:?}", what, src); |
| 210 | } |
| 211 | // And a document with no root element is not a document. |
| 212 | let xml = res!(Xml::parse("<?xml version=\"1.0\"?>\n<!-- nothing here -->")); |
| 213 | assert!(xml.root().is_err()); |
| 214 | Ok(()) |
| 215 | })); |
| 216 | |
| 217 | res!(test_it(filter, &["Edits that overlap are refused rather than resolved 007", "all", "xml"], || { |
| 218 | let mut xml = res!(Xml::parse("<a><b>one</b><c>two</c></a>")); |
| 219 | res!(xml.splice(3..13, "<b>ONE</b>".to_string())); |
| 220 | assert!(!xml.is_pristine()); |
| 221 | // Overlapping the first. |
| 222 | assert!(xml.splice(6..16, "x".to_string()).is_err()); |
| 223 | // Past the end. |
| 224 | assert!(xml.splice(100..200, "x".to_string()).is_err()); |
| 225 | // Backwards. |
| 226 | assert!(xml.splice(10..3, "x".to_string()).is_err()); |
| 227 | // A second, disjoint edit is fine, and the two render in the right order however they arrive. |
| 228 | res!(xml.splice(13..23, "<c>TWO</c>".to_string())); |
| 229 | assert_eq!(xml.render(), "<a><b>ONE</b><c>TWO</c></a>"); |
| 230 | xml.revert(); |
| 231 | assert!(xml.is_pristine()); |
| 232 | assert_eq!(xml.render(), "<a><b>one</b><c>two</c></a>"); |
| 233 | Ok(()) |
| 234 | })); |
| 235 | |
| 236 | res!(test_it(filter, &["The emitter refuses a document it would break 008", "all", "xml"], || { |
| 237 | let mut out = Out::declared(); |
| 238 | // The declaration goes on the root, as it does in a real part: a prefix nothing bound is a |
| 239 | // document the reader below refuses, which is how a missing `xmlns` is found here rather than |
| 240 | // by Word. |
| 241 | out.open("w:p", &[("xmlns:w", "urn:w")]); |
| 242 | out.leaf("w:t", &[("xml:space", "preserve")], "a < b & c"); |
| 243 | assert!(out.close("w:r").is_err(), "closing what is not open is refused"); |
| 244 | res!(out.close("w:p")); |
| 245 | let s = res!(out.finish()); |
| 246 | assert!(s.starts_with("<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>")); |
| 247 | assert!(s.contains("<w:t xml:space=\"preserve\">a < b & c</w:t>")); |
| 248 | // What it emits reads back as what was put in. |
| 249 | let xml = res!(Xml::parse(&s)); |
| 250 | let root = res!(xml.root()); |
| 251 | assert_eq!(xml.text_of(root), "a < b & c"); |
| 252 | // And an unfinished document is refused. |
| 253 | let mut bad = Out::new(); |
| 254 | bad.open("a", &[]); |
| 255 | assert!(bad.finish().is_err()); |
| 256 | Ok(()) |
| 257 | })); |
| 258 | |
| 259 | Ok(()) |
| 260 | } |