oxedyne/fe2o3/fe2o3_text/tests/unicode.rs
22.7 KiB, 1 run
created by r1870400018:13922, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Conformance tests for `fe2o3_text::unicode`, run against the Unicode Consortium's own test |
| 2 | //! files. |
| 3 | //! |
| 4 | //! The files live in `tests/unicode_data/`, and come from the same UCD release as the tables, so |
| 5 | //! the data can never drift from what it is testing. |
| 6 | //! |
| 7 | //! Four of the six are kept in the repository. The two bidi suites are 15 MB between them, which is |
| 8 | //! not worth carrying in a repository's history forever, so they are fetched on demand at the |
| 9 | //! pinned UCD version the first time a test needs them. A test that has no data fetches it; a test |
| 10 | //! that cannot fetch it fails. Nothing is ever skipped. |
| 11 | |
| 12 | use oxedyne_fe2o3_text::unicode::{ |
| 13 | bidi::{ |
| 14 | self, |
| 15 | BidiInfo, |
| 16 | Direction, |
| 17 | }, |
| 18 | linebreak, |
| 19 | lookup::Partitioned, |
| 20 | norm::{ |
| 21 | self, |
| 22 | Form, |
| 23 | }, |
| 24 | prop::BidiClass as B, |
| 25 | segment, |
| 26 | UCD_VERSION, |
| 27 | }; |
| 28 | |
| 29 | use oxedyne_fe2o3_core::prelude::*; |
| 30 | |
| 31 | use std::{ |
| 32 | fs, |
| 33 | path::{ |
| 34 | Path, |
| 35 | PathBuf, |
| 36 | }, |
| 37 | process::Command, |
| 38 | }; |
| 39 | |
| 40 | /// How many failing cases a report lists before it stops. |
| 41 | const SHOWN: usize = 8; |
| 42 | |
| 43 | /// The tally of a conformance run. |
| 44 | struct Tally { |
| 45 | /// The name of the test file. |
| 46 | name: String, |
| 47 | /// The number of cases run. |
| 48 | total: usize, |
| 49 | /// The cases that failed, as a line number and a description. |
| 50 | failed: Vec<(usize, String)>, |
| 51 | } |
| 52 | |
| 53 | impl Tally { |
| 54 | |
| 55 | /// Begins a tally. |
| 56 | fn new(name: &str) -> Self { |
| 57 | Self { |
| 58 | name: name.to_string(), |
| 59 | total: 0, |
| 60 | failed: Vec::new(), |
| 61 | } |
| 62 | } |
| 63 | |
| 64 | /// Records a case. |
| 65 | fn case(&mut self, line: usize, ok: bool, why: String) { |
| 66 | self.total += 1; |
| 67 | if !ok { |
| 68 | self.failed.push((line, why)); |
| 69 | } |
| 70 | } |
| 71 | |
| 72 | /// Reports the pass rate, and fails the test if any case failed. |
| 73 | fn finish(self) -> Outcome<()> { |
| 74 | let passed = self.total - self.failed.len(); |
| 75 | let rate = if self.total == 0 { |
| 76 | 0.0 |
| 77 | } else { |
| 78 | 100.0 * (passed as f64) / (self.total as f64) |
| 79 | }; |
| 80 | msg!("{}: {}/{} passed ({:.4}%), UCD {}.", self.name, passed, self.total, rate, |
| 81 | UCD_VERSION); |
| 82 | if self.failed.is_empty() { |
| 83 | return Ok(()); |
| 84 | } |
| 85 | for (line, why) in self.failed.iter().take(SHOWN) { |
| 86 | msg!(" line {}: {}", line, why); |
| 87 | } |
| 88 | if self.failed.len() > SHOWN { |
| 89 | msg!(" ... and {} more.", self.failed.len() - SHOWN); |
| 90 | } |
| 91 | Err(err!( |
| 92 | "{}: {} of {} conformance cases failed.", self.name, self.failed.len(), self.total; |
| 93 | Test, Mismatch)) |
| 94 | } |
| 95 | } |
| 96 | |
| 97 | /// Where the Unicode Consortium publishes its data. |
| 98 | const UCD_BASE: &str = "https://www.unicode.org/Public"; |
| 99 | |
| 100 | /// Each conformance file, and its path below the UCD version directory. |
| 101 | /// |
| 102 | /// The four small suites are kept in the repository. The two bidi suites are 15 MB between them, |
| 103 | /// which is not worth carrying in a repository's history forever, so they are absent and fetched on |
| 104 | /// demand by [`fetch`]. |
| 105 | const TEST_FILES: [(&str, &str); 6] = [ |
| 106 | ("NormalizationTest.txt", "ucd/NormalizationTest.txt"), |
| 107 | ("BidiTest.txt", "ucd/BidiTest.txt"), |
| 108 | ("BidiCharacterTest.txt", "ucd/BidiCharacterTest.txt"), |
| 109 | ("GraphemeBreakTest.txt", "ucd/auxiliary/GraphemeBreakTest.txt"), |
| 110 | ("WordBreakTest.txt", "ucd/auxiliary/WordBreakTest.txt"), |
| 111 | ("LineBreakTest.txt", "ucd/auxiliary/LineBreakTest.txt"), |
| 112 | ]; |
| 113 | |
| 114 | /// Reads a conformance file, fetching it first if the repository does not carry it. |
| 115 | /// |
| 116 | /// A test is never skipped for want of its data: either the file is found, or it is fetched, or the |
| 117 | /// test fails saying why. |
| 118 | fn data(name: &str) -> Outcome<String> { |
| 119 | let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")) |
| 120 | .join("tests") |
| 121 | .join("unicode_data") |
| 122 | .join(name); |
| 123 | if !path.exists() { |
| 124 | res!(fetch(name, &path)); |
| 125 | } |
| 126 | match fs::read_to_string(&path) { |
| 127 | Ok(text) => Ok(text), |
| 128 | Err(e) => Err(err!(e, |
| 129 | "The conformance file {:?} could not be read. Run \ |
| 130 | `cargo run -p oxedyne_fe2o3_text --bin gen_unicode` to rebuild the tables and \ |
| 131 | refetch it.", path; |
| 132 | IO, File, Missing)), |
| 133 | } |
| 134 | } |
| 135 | |
| 136 | /// Fetches a conformance file the repository does not carry, at the pinned UCD version. |
| 137 | /// |
| 138 | /// The version fetched is [`UCD_VERSION`], the one the committed tables were generated from, so the |
| 139 | /// data can never drift from the tables it is testing. A newer Unicode release does not silently |
| 140 | /// become the thing under test. |
| 141 | fn fetch(name: &str, dest: &Path) -> Outcome<()> { |
| 142 | let rel = match TEST_FILES.iter().find(|(n, _)| *n == name) { |
| 143 | Some((_, p)) => *p, |
| 144 | None => return Err(err!( |
| 145 | "There is no Unicode conformance file named {}.", name; Invalid, Input)), |
| 146 | }; |
| 147 | if let Some(dir) = dest.parent() { |
| 148 | if let Err(e) = fs::create_dir_all(dir) { |
| 149 | return Err(err!(e, "While creating {:?}.", dir; IO, Path)); |
| 150 | } |
| 151 | } |
| 152 | let url = fmt!("{}/{}/{}", UCD_BASE, UCD_VERSION, rel); |
| 153 | msg!("{} is not in the repository. Fetching {}.", name, url); |
| 154 | let out = match Command::new("curl") |
| 155 | .arg("-sS") |
| 156 | .arg("--fail") |
| 157 | .arg(&url) |
| 158 | .output() |
| 159 | { |
| 160 | Ok(out) => out, |
| 161 | Err(e) => return Err(err!(e, |
| 162 | "While running curl to fetch {}. The conformance suites this file feeds cannot run \ |
| 163 | without it, and fetching needs curl on the path and a network.", url; |
| 164 | IO, Network)), |
| 165 | }; |
| 166 | if !out.status.success() { |
| 167 | return Err(err!( |
| 168 | "curl could not fetch {}: {}", url, String::from_utf8_lossy(&out.stderr); |
| 169 | IO, Network)); |
| 170 | } |
| 171 | let text = String::from_utf8_lossy(&out.stdout); |
| 172 | let body = fmt!( |
| 173 | "# {}\n\ |
| 174 | # Fetched from {}/{}/{} by fe2o3_text/tests/unicode.rs.\n\ |
| 175 | # Comment and blank lines have been stripped; the data lines are unchanged.\n\ |
| 176 | {}", |
| 177 | name, UCD_BASE, UCD_VERSION, rel, strip_comments(&text), |
| 178 | ); |
| 179 | if let Err(e) = fs::write(dest, &body) { |
| 180 | return Err(err!(e, "While writing {:?}.", dest; IO, File)); |
| 181 | } |
| 182 | Ok(()) |
| 183 | } |
| 184 | |
| 185 | /// Drops comment and blank lines, leaving the data lines untouched, as the generator does. |
| 186 | fn strip_comments(text: &str) -> String { |
| 187 | let mut out = String::new(); |
| 188 | for line in text.lines() { |
| 189 | if line.starts_with('#') { |
| 190 | continue; |
| 191 | } |
| 192 | let data = match line.find('#') { |
| 193 | Some(i) => &line[..i], |
| 194 | None => line, |
| 195 | }; |
| 196 | let data = data.trim_end(); |
| 197 | if data.trim().is_empty() { |
| 198 | continue; |
| 199 | } |
| 200 | out.push_str(data); |
| 201 | out.push('\n'); |
| 202 | } |
| 203 | out |
| 204 | } |
| 205 | |
| 206 | /// Parses a hexadecimal code point. |
| 207 | fn hex(s: &str) -> Outcome<char> { |
| 208 | let cp = match u32::from_str_radix(s.trim(), 16) { |
| 209 | Ok(cp) => cp, |
| 210 | Err(e) => return Err(err!(e, "{:?} is not a code point.", s; Invalid, Input)), |
| 211 | }; |
| 212 | match char::from_u32(cp) { |
| 213 | Some(c) => Ok(c), |
| 214 | None => Err(err!("U+{:04X} is not a character.", cp; Invalid, Input)), |
| 215 | } |
| 216 | } |
| 217 | |
| 218 | /// Parses a space separated run of hexadecimal code points. |
| 219 | fn hexes(s: &str) -> Outcome<String> { |
| 220 | let mut out = String::new(); |
| 221 | for part in s.split_whitespace() { |
| 222 | out.push(res!(hex(part))); |
| 223 | } |
| 224 | Ok(out) |
| 225 | } |
| 226 | |
| 227 | // ┌───────────────────────────────────────────────────────────────────────────────────────────┐ |
| 228 | // │ Tables │ |
| 229 | // └───────────────────────────────────────────────────────────────────────────────────────────┘ |
| 230 | |
| 231 | #[test] |
| 232 | fn tables_are_consistent() -> Outcome<()> { |
| 233 | res!(norm::check_tables()); |
| 234 | Ok(()) |
| 235 | } |
| 236 | |
| 237 | // ┌───────────────────────────────────────────────────────────────────────────────────────────┐ |
| 238 | // │ UAX #15, normalisation │ |
| 239 | // └───────────────────────────────────────────────────────────────────────────────────────────┘ |
| 240 | |
| 241 | #[test] |
| 242 | fn normalisation_conformance() -> Outcome<()> { |
| 243 | |
| 244 | let text = res!(data("NormalizationTest.txt")); |
| 245 | let mut tally = Tally::new("NormalizationTest"); |
| 246 | |
| 247 | for (n, line) in text.lines().enumerate() { |
| 248 | if line.starts_with('#') || line.starts_with('@') || line.trim().is_empty() { |
| 249 | continue; |
| 250 | } |
| 251 | let f: Vec<&str> = line.split(';').collect(); |
| 252 | if f.len() < 5 { |
| 253 | return Err(err!( |
| 254 | "Line {} of NormalizationTest.txt has {} fields, expected 5.", n + 1, f.len(); |
| 255 | Invalid, Input)); |
| 256 | } |
| 257 | let c1 = res!(hexes(f[0])); |
| 258 | let c2 = res!(hexes(f[1])); |
| 259 | let c3 = res!(hexes(f[2])); |
| 260 | let c4 = res!(hexes(f[3])); |
| 261 | let c5 = res!(hexes(f[4])); |
| 262 | |
| 263 | // The invariants the file itself states. |
| 264 | let mut bad = Vec::new(); |
| 265 | for (src, want, form, name) in [ |
| 266 | (&c1, &c2, Form::Nfc, "NFC(c1)"), (&c2, &c2, Form::Nfc, "NFC(c2)"), |
| 267 | (&c3, &c2, Form::Nfc, "NFC(c3)"), (&c4, &c4, Form::Nfc, "NFC(c4)"), |
| 268 | (&c5, &c4, Form::Nfc, "NFC(c5)"), |
| 269 | (&c1, &c3, Form::Nfd, "NFD(c1)"), (&c2, &c3, Form::Nfd, "NFD(c2)"), |
| 270 | (&c3, &c3, Form::Nfd, "NFD(c3)"), (&c4, &c5, Form::Nfd, "NFD(c4)"), |
| 271 | (&c5, &c5, Form::Nfd, "NFD(c5)"), |
| 272 | (&c1, &c4, Form::Nfkc, "NFKC(c1)"), (&c2, &c4, Form::Nfkc, "NFKC(c2)"), |
| 273 | (&c3, &c4, Form::Nfkc, "NFKC(c3)"), (&c4, &c4, Form::Nfkc, "NFKC(c4)"), |
| 274 | (&c5, &c4, Form::Nfkc, "NFKC(c5)"), |
| 275 | (&c1, &c5, Form::Nfkd, "NFKD(c1)"), (&c2, &c5, Form::Nfkd, "NFKD(c2)"), |
| 276 | (&c3, &c5, Form::Nfkd, "NFKD(c3)"), (&c4, &c5, Form::Nfkd, "NFKD(c4)"), |
| 277 | (&c5, &c5, Form::Nfkd, "NFKD(c5)"), |
| 278 | ] { |
| 279 | let got = norm::normalise(src, form); |
| 280 | if got != *want { |
| 281 | bad.push(fmt!("{} gave {:?}, expected {:?}", name, got, want)); |
| 282 | } |
| 283 | } |
| 284 | |
| 285 | tally.case(n + 1, bad.is_empty(), bad.join("; ")); |
| 286 | } |
| 287 | |
| 288 | tally.finish() |
| 289 | } |
| 290 | |
| 291 | #[test] |
| 292 | fn normalisation_leaves_the_rest_alone() -> Outcome<()> { |
| 293 | |
| 294 | // Every character that Part 1 of the test file does not list is its own normalisation, in |
| 295 | // every form. |
| 296 | let text = res!(data("NormalizationTest.txt")); |
| 297 | let mut listed = vec![false; 0x110000]; |
| 298 | let mut part1 = false; |
| 299 | for line in text.lines() { |
| 300 | if line.starts_with("@Part1") { |
| 301 | part1 = true; |
| 302 | continue; |
| 303 | } |
| 304 | if line.starts_with('@') { |
| 305 | part1 = false; |
| 306 | continue; |
| 307 | } |
| 308 | if !part1 || line.starts_with('#') || line.trim().is_empty() { |
| 309 | continue; |
| 310 | } |
| 311 | if let Some(first) = line.split(';').next() { |
| 312 | let c = res!(hex(first)); |
| 313 | listed[c as usize] = true; |
| 314 | } |
| 315 | } |
| 316 | |
| 317 | let mut tally = Tally::new("NormalizationTest, Part 1 invariant"); |
| 318 | for cp in 0..=0x10FFFFu32 { |
| 319 | let c = match char::from_u32(cp) { |
| 320 | Some(c) => c, |
| 321 | None => continue, |
| 322 | }; |
| 323 | if listed[cp as usize] { |
| 324 | continue; |
| 325 | } |
| 326 | let s = c.to_string(); |
| 327 | let ok = norm::nfc(&s) == s |
| 328 | && norm::nfd(&s) == s |
| 329 | && norm::nfkc(&s) == s |
| 330 | && norm::nfkd(&s) == s; |
| 331 | if !ok { |
| 332 | tally.case(cp as usize, false, fmt!("U+{:04X} is not its own normalisation.", cp)); |
| 333 | } else { |
| 334 | tally.total += 1; |
| 335 | } |
| 336 | } |
| 337 | |
| 338 | tally.finish() |
| 339 | } |
| 340 | |
| 341 | // ┌───────────────────────────────────────────────────────────────────────────────────────────┐ |
| 342 | // │ UAX #29, segmentation │ |
| 343 | // └───────────────────────────────────────────────────────────────────────────────────────────┘ |
| 344 | |
| 345 | /// Parses a line of a break test into the string and the byte offsets of the breaks in it, |
| 346 | /// including the offset zero if the line begins with a break. |
| 347 | fn break_case(line: &str) -> Outcome<(String, Vec<usize>)> { |
| 348 | let mut s = String::new(); |
| 349 | let mut breaks = Vec::new(); |
| 350 | for tok in line.split_whitespace() { |
| 351 | match tok { |
| 352 | "\u{00F7}" => breaks.push(s.len()), |
| 353 | "\u{00D7}" => (), |
| 354 | hx => s.push(res!(hex(hx))), |
| 355 | } |
| 356 | } |
| 357 | Ok((s, breaks)) |
| 358 | } |
| 359 | |
| 360 | #[test] |
| 361 | fn grapheme_conformance() -> Outcome<()> { |
| 362 | |
| 363 | let text = res!(data("GraphemeBreakTest.txt")); |
| 364 | let mut tally = Tally::new("GraphemeBreakTest"); |
| 365 | |
| 366 | for (n, line) in text.lines().enumerate() { |
| 367 | if line.starts_with('#') || line.trim().is_empty() { |
| 368 | continue; |
| 369 | } |
| 370 | let (s, want) = res!(break_case(line)); |
| 371 | let got = segment::grapheme_boundaries(&s); |
| 372 | tally.case(n + 1, got == want, fmt!("{:?}: got {:?}, expected {:?}", s, got, want)); |
| 373 | } |
| 374 | |
| 375 | tally.finish() |
| 376 | } |
| 377 | |
| 378 | #[test] |
| 379 | fn word_conformance() -> Outcome<()> { |
| 380 | |
| 381 | let text = res!(data("WordBreakTest.txt")); |
| 382 | let mut tally = Tally::new("WordBreakTest"); |
| 383 | |
| 384 | for (n, line) in text.lines().enumerate() { |
| 385 | if line.starts_with('#') || line.trim().is_empty() { |
| 386 | continue; |
| 387 | } |
| 388 | let (s, want) = res!(break_case(line)); |
| 389 | let got = segment::word_boundaries(&s); |
| 390 | tally.case(n + 1, got == want, fmt!("{:?}: got {:?}, expected {:?}", s, got, want)); |
| 391 | } |
| 392 | |
| 393 | tally.finish() |
| 394 | } |
| 395 | |
| 396 | // ┌───────────────────────────────────────────────────────────────────────────────────────────┐ |
| 397 | // │ UAX #14, line breaking │ |
| 398 | // └───────────────────────────────────────────────────────────────────────────────────────────┘ |
| 399 | |
| 400 | #[test] |
| 401 | fn linebreak_conformance() -> Outcome<()> { |
| 402 | |
| 403 | let text = res!(data("LineBreakTest.txt")); |
| 404 | let mut tally = Tally::new("LineBreakTest"); |
| 405 | |
| 406 | for (n, line) in text.lines().enumerate() { |
| 407 | if line.starts_with('#') || line.trim().is_empty() { |
| 408 | continue; |
| 409 | } |
| 410 | let (s, want) = res!(break_case(line)); |
| 411 | let got = linebreak::break_offsets(&s); |
| 412 | tally.case(n + 1, got == want, fmt!("{:?}: got {:?}, expected {:?}", s, got, want)); |
| 413 | } |
| 414 | |
| 415 | tally.finish() |
| 416 | } |
| 417 | |
| 418 | // ┌───────────────────────────────────────────────────────────────────────────────────────────┐ |
| 419 | // │ UAX #9, the bidirectional algorithm │ |
| 420 | // └───────────────────────────────────────────────────────────────────────────────────────────┘ |
| 421 | |
| 422 | /// A character of each Bidi_Class, for the test file that gives classes rather than characters. |
| 423 | /// None of them is a paired bracket, which is what `BidiTest.txt` assumes. |
| 424 | const REPS: &[(&str, char)] = &[ |
| 425 | ("L", 'A'), |
| 426 | ("R", '\u{05D0}'), |
| 427 | ("AL", '\u{0627}'), |
| 428 | ("EN", '0'), |
| 429 | ("ES", '+'), |
| 430 | ("ET", '#'), |
| 431 | ("AN", '\u{0660}'), |
| 432 | ("CS", ','), |
| 433 | ("NSM", '\u{0300}'), |
| 434 | ("BN", '\u{00AD}'), |
| 435 | ("B", '\u{2029}'), |
| 436 | ("S", '\t'), |
| 437 | ("WS", ' '), |
| 438 | ("ON", '!'), |
| 439 | ("LRE", '\u{202A}'), |
| 440 | ("RLE", '\u{202B}'), |
| 441 | ("PDF", '\u{202C}'), |
| 442 | ("LRO", '\u{202D}'), |
| 443 | ("RLO", '\u{202E}'), |
| 444 | ("LRI", '\u{2066}'), |
| 445 | ("RLI", '\u{2067}'), |
| 446 | ("FSI", '\u{2068}'), |
| 447 | ("PDI", '\u{2069}'), |
| 448 | ]; |
| 449 | |
| 450 | /// Returns the representative character of a Bidi_Class name. |
| 451 | fn rep(name: &str) -> Outcome<char> { |
| 452 | for (n, c) in REPS { |
| 453 | if *n == name { |
| 454 | return Ok(*c); |
| 455 | } |
| 456 | } |
| 457 | Err(err!("The Bidi_Class {:?} has no representative character.", name; Invalid, Input)) |
| 458 | } |
| 459 | |
| 460 | /// Checks the resolved levels and the visual order against what a test file expects, where an |
| 461 | /// expected level of `x` means the character is one the algorithm removes. |
| 462 | fn bidi_case( |
| 463 | info: &BidiInfo, |
| 464 | levels: &str, |
| 465 | order: &str, |
| 466 | ) |
| 467 | -> Outcome<Option<String>> |
| 468 | { |
| 469 | let want: Vec<&str> = levels.split_whitespace().collect(); |
| 470 | if want.len() != info.levels.len() { |
| 471 | return Ok(Some(fmt!( |
| 472 | "expected {} levels, the text has {} characters", want.len(), info.levels.len()))); |
| 473 | } |
| 474 | for (i, w) in want.iter().enumerate() { |
| 475 | if *w == "x" { |
| 476 | continue; |
| 477 | } |
| 478 | let lv = match w.parse::<u8>() { |
| 479 | Ok(lv) => lv, |
| 480 | Err(e) => return Err(err!(e, "{:?} is not a level.", w; Invalid, Input)), |
| 481 | }; |
| 482 | if info.levels[i] != lv { |
| 483 | return Ok(Some(fmt!( |
| 484 | "level {} is {}, expected {}; all levels {:?}", i, info.levels[i], lv, |
| 485 | info.levels))); |
| 486 | } |
| 487 | } |
| 488 | |
| 489 | let mut want_order = Vec::new(); |
| 490 | for part in order.split_whitespace() { |
| 491 | match part.parse::<usize>() { |
| 492 | Ok(i) => want_order.push(i), |
| 493 | Err(e) => return Err(err!(e, "{:?} is not an index.", part; Invalid, Input)), |
| 494 | } |
| 495 | } |
| 496 | let got = info.visual_order(); |
| 497 | if got != want_order { |
| 498 | return Ok(Some(fmt!("visual order {:?}, expected {:?}", got, want_order))); |
| 499 | } |
| 500 | |
| 501 | Ok(None) |
| 502 | } |
| 503 | |
| 504 | #[test] |
| 505 | fn bidi_conformance() -> Outcome<()> { |
| 506 | |
| 507 | let text = res!(data("BidiTest.txt")); |
| 508 | let mut tally = Tally::new("BidiTest"); |
| 509 | |
| 510 | // The representative characters must really carry the class they stand for. |
| 511 | for (name, c) in REPS { |
| 512 | let got = fmt!("{:?}", B::of(*c)); |
| 513 | if got != *name { |
| 514 | return Err(err!( |
| 515 | "The representative U+{:04X} of Bidi_Class {} is {}.", *c as u32, name, got; |
| 516 | Bug, Invalid)); |
| 517 | } |
| 518 | } |
| 519 | |
| 520 | let mut levels = String::new(); |
| 521 | let mut order = String::new(); |
| 522 | |
| 523 | for (n, line) in text.lines().enumerate() { |
| 524 | let line = line.trim(); |
| 525 | if line.is_empty() || line.starts_with('#') { |
| 526 | continue; |
| 527 | } |
| 528 | if let Some(rest) = line.strip_prefix("@Levels:") { |
| 529 | levels = rest.trim().to_string(); |
| 530 | continue; |
| 531 | } |
| 532 | if let Some(rest) = line.strip_prefix("@Reorder:") { |
| 533 | order = rest.trim().to_string(); |
| 534 | continue; |
| 535 | } |
| 536 | if line.starts_with('@') { |
| 537 | continue; |
| 538 | } |
| 539 | |
| 540 | let (input, bits) = match line.split_once(';') { |
| 541 | Some((a, b)) => (a, b), |
| 542 | None => return Err(err!( |
| 543 | "Line {} of BidiTest.txt has no semicolon.", n + 1; Invalid, Input)), |
| 544 | }; |
| 545 | let bits = match u8::from_str_radix(bits.trim(), 16) { |
| 546 | Ok(bits) => bits, |
| 547 | Err(e) => return Err(err!(e, |
| 548 | "{:?} on line {} is not a bitset.", bits, n + 1; Invalid, Input)), |
| 549 | }; |
| 550 | |
| 551 | let mut chars = Vec::new(); |
| 552 | let mut classes = Vec::new(); |
| 553 | for name in input.split_whitespace() { |
| 554 | let c = res!(rep(name)); |
| 555 | chars.push(c); |
| 556 | classes.push(B::of(c)); |
| 557 | } |
| 558 | |
| 559 | for (bit, dir) in [(1u8, Direction::Auto), (2, Direction::Ltr), (4, Direction::Rtl)] { |
| 560 | if bits & bit == 0 { |
| 561 | continue; |
| 562 | } |
| 563 | let info = bidi::resolve_classes(&chars, &classes, dir); |
| 564 | match res!(bidi_case(&info, &levels, &order)) { |
| 565 | Some(why) => tally.case(n + 1, false, |
| 566 | fmt!("{} with {:?}: {}", input.trim(), dir, why)), |
| 567 | None => tally.case(n + 1, true, String::new()), |
| 568 | } |
| 569 | } |
| 570 | } |
| 571 | |
| 572 | tally.finish() |
| 573 | } |
| 574 | |
| 575 | #[test] |
| 576 | fn bidi_character_conformance() -> Outcome<()> { |
| 577 | |
| 578 | let text = res!(data("BidiCharacterTest.txt")); |
| 579 | let mut tally = Tally::new("BidiCharacterTest"); |
| 580 | |
| 581 | for (n, line) in text.lines().enumerate() { |
| 582 | if line.starts_with('#') || line.trim().is_empty() { |
| 583 | continue; |
| 584 | } |
| 585 | let f: Vec<&str> = line.split(';').collect(); |
| 586 | if f.len() < 5 { |
| 587 | return Err(err!( |
| 588 | "Line {} of BidiCharacterTest.txt has {} fields, expected 5.", n + 1, f.len(); |
| 589 | Invalid, Input)); |
| 590 | } |
| 591 | |
| 592 | let mut chars = Vec::new(); |
| 593 | for part in f[0].split_whitespace() { |
| 594 | chars.push(res!(hex(part))); |
| 595 | } |
| 596 | let dir = match f[1].trim() { |
| 597 | "0" => Direction::Ltr, |
| 598 | "1" => Direction::Rtl, |
| 599 | "2" => Direction::Auto, |
| 600 | other => return Err(err!( |
| 601 | "{:?} on line {} is not a paragraph direction.", other, n + 1; Invalid, Input)), |
| 602 | }; |
| 603 | let want_para = match f[2].trim().parse::<u8>() { |
| 604 | Ok(lv) => lv, |
| 605 | Err(e) => return Err(err!(e, |
| 606 | "{:?} on line {} is not a level.", f[2], n + 1; Invalid, Input)), |
| 607 | }; |
| 608 | |
| 609 | let classes: Vec<B> = chars.iter().map(|c| B::of(*c)).collect(); |
| 610 | let info = bidi::resolve_classes(&chars, &classes, dir); |
| 611 | |
| 612 | if info.para_level != want_para { |
| 613 | tally.case(n + 1, false, fmt!( |
| 614 | "paragraph level {}, expected {}", info.para_level, want_para)); |
| 615 | continue; |
| 616 | } |
| 617 | match res!(bidi_case(&info, f[3], f[4])) { |
| 618 | Some(why) => tally.case(n + 1, false, why), |
| 619 | None => tally.case(n + 1, true, String::new()), |
| 620 | } |
| 621 | } |
| 622 | |
| 623 | tally.finish() |
| 624 | } |
| 625 | |
| 626 | // ┌───────────────────────────────────────────────────────────────────────────────────────────┐ |
| 627 | // │ The shape of the API │ |
| 628 | // └───────────────────────────────────────────────────────────────────────────────────────────┘ |
| 629 | |
| 630 | #[test] |
| 631 | fn normalisation_basics() -> Outcome<()> { |
| 632 | |
| 633 | req!(norm::nfc("e\u{0301}"), "\u{00E9}"); |
| 634 | req!(norm::nfd("\u{00E9}"), "e\u{0301}"); |
| 635 | req!(norm::nfkc("\u{FB01}"), "fi"); |
| 636 | req!(norm::nfkd("\u{2460}"), "1"); |
| 637 | |
| 638 | // A Hangul syllable composes and decomposes arithmetically. |
| 639 | req!(norm::nfd("\u{AC01}"), "\u{1100}\u{1161}\u{11A8}"); |
| 640 | req!(norm::nfc("\u{1100}\u{1161}\u{11A8}"), "\u{AC01}"); |
| 641 | |
| 642 | // The marks come out in canonical order whichever order they went in. |
| 643 | req!(norm::nfd("q\u{0307}\u{0323}"), norm::nfd("q\u{0323}\u{0307}")); |
| 644 | req!(norm::eq_canonical("\u{00C5}", "A\u{030A}"), true); |
| 645 | |
| 646 | // A singleton does not compose back. |
| 647 | req!(norm::nfc("\u{2126}"), "\u{03A9}"); |
| 648 | |
| 649 | req!(norm::is_normalised("abc", Form::Nfc), true); |
| 650 | req!(norm::is_normalised("e\u{0301}", Form::Nfc), false); |
| 651 | |
| 652 | Ok(()) |
| 653 | } |
| 654 | |
| 655 | #[test] |
| 656 | fn segmentation_basics() -> Outcome<()> { |
| 657 | |
| 658 | req!(segment::graphemes("hi"), vec!["h", "i"]); |
| 659 | req!(segment::graphemes("e\u{0301}"), vec!["e\u{0301}"]); |
| 660 | req!(segment::graphemes("\u{1F1E6}\u{1F1FA}"), vec!["\u{1F1E6}\u{1F1FA}"]); |
| 661 | req!(segment::graphemes("\r\n"), vec!["\r\n"]); |
| 662 | |
| 663 | // A cursor steps over a whole cluster. |
| 664 | let s = "e\u{0301}x"; |
| 665 | req!(segment::next_grapheme(s, 0), 3); |
| 666 | req!(segment::prev_grapheme(s, 4), 3); |
| 667 | |
| 668 | req!(segment::words("one two"), vec!["one", " ", "two"]); |
| 669 | |
| 670 | Ok(()) |
| 671 | } |
| 672 | |
| 673 | #[test] |
| 674 | fn linebreak_basics() -> Outcome<()> { |
| 675 | |
| 676 | let opps = linebreak::line_breaks("one two"); |
| 677 | req!(opps.len(), 2); |
| 678 | req!(opps[0].offset, 4); |
| 679 | req!(opps[0].kind, linebreak::Break::Optional); |
| 680 | req!(opps[1].offset, 7); |
| 681 | req!(opps[1].kind, linebreak::Break::Mandatory); |
| 682 | |
| 683 | // A hard break is mandatory, and no break falls inside a non-breaking space. |
| 684 | let opps = linebreak::line_breaks("a\nb"); |
| 685 | req!(opps[0].offset, 2); |
| 686 | req!(opps[0].kind, linebreak::Break::Mandatory); |
| 687 | |
| 688 | req!(linebreak::break_offsets("a\u{00A0}b"), vec![4]); |
| 689 | |
| 690 | Ok(()) |
| 691 | } |
| 692 | |
| 693 | #[test] |
| 694 | fn bidi_basics() -> Outcome<()> { |
| 695 | |
| 696 | // A left to right paragraph with a right to left word in it. |
| 697 | let info = bidi::resolve("he said \u{05D0}\u{05D1}\u{05D2} to me", Direction::Auto); |
| 698 | req!(info.para_level, 0); |
| 699 | req!(info.has_rtl(), true); |
| 700 | req!(info.levels[8], 1); |
| 701 | |
| 702 | // A right to left paragraph, taken from its first strong character. |
| 703 | let info = bidi::resolve("\u{05D0} a", Direction::Auto); |
| 704 | req!(info.para_level, 1); |
| 705 | req!(info.visual_order(), vec![2, 1, 0]); |
| 706 | |
| 707 | // Plain text needs nothing done to it. |
| 708 | let info = bidi::resolve("plain", Direction::Auto); |
| 709 | req!(info.has_rtl(), false); |
| 710 | req!(info.visual_order(), vec![0, 1, 2, 3, 4]); |
| 711 | |
| 712 | Ok(()) |
| 713 | } |