Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/bin/gen_unicode.rs

58.0 KiB, 15 runs

created by r1870400018:13895, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Generator for the committed Unicode tables in `fe2o3_text::unicode`.
2//!
3//! This binary downloads the Unicode Character Database files for the pinned UCD version, parses
4//! them, and writes Rust source into `src/unicode/tables/`, plus the Unicode Consortium
5//! conformance test files into `tests/unicode_data/`. Both sets of outputs are committed, so the
6//! library builds with no build script, no runtime download and no third party dependency.
7//!
8//! Downloads go through `curl`, which keeps the crate free of an HTTP dependency; the files are
9//! cached under the system temporary directory so that repeated runs do not hit unicode.org.
10//!
11//! Run it from the crate root:
12//!
13//! ```text
14//! cargo run -p oxedyne_fe2o3_text --bin gen_unicode
15//! ```
16//!
17#![forbid(unsafe_code)]
18
19use oxedyne_fe2o3_core::prelude::*;
20
21use std::{
22 collections::{
23 BTreeMap,
24 BTreeSet,
25 },
26 fmt::Write as _,
27 fs,
28 path::{
29 Path,
30 PathBuf,
31 },
32 process::Command,
33};
34
35/// The pinned Unicode Character Database version. Every generated table and vendored test file
36/// comes from this release.
37const UCD_VERSION: &str = "17.0.0";
38
39/// Root of the Unicode public file server.
40const UCD_BASE: &str = "https://www.unicode.org/Public";
41
42/// The highest Unicode code point.
43const MAX_CP: u32 = 0x10FFFF;
44
45/// The number of code points, including the surrogate gap.
46const NUM_CP: usize = (MAX_CP as usize) + 1;
47
48/// Data files parsed into tables, given as paths below the version directory.
49const DATA_FILES: &[&str] = &[
50 "ucd/UnicodeData.txt",
51 "ucd/DerivedNormalizationProps.txt",
52 "ucd/DerivedCoreProperties.txt",
53 "ucd/extracted/DerivedBidiClass.txt",
54 "ucd/BidiBrackets.txt",
55 "ucd/LineBreak.txt",
56 "ucd/EastAsianWidth.txt",
57 "ucd/auxiliary/GraphemeBreakProperty.txt",
58 "ucd/auxiliary/WordBreakProperty.txt",
59 "ucd/emoji/emoji-data.txt",
60 "ucd/Scripts.txt",
61 "ucd/ScriptExtensions.txt",
62 "ucd/PropList.txt",
63 "ucd/PropertyAliases.txt",
64 "ucd/PropertyValueAliases.txt",
65 "ucd/auxiliary/SentenceBreakProperty.txt",
66];
67
68/// Conformance test files vendored into `tests/unicode_data/`.
69const TEST_FILES: &[&str] = &[
70 "ucd/NormalizationTest.txt",
71 "ucd/BidiTest.txt",
72 "ucd/BidiCharacterTest.txt",
73 "ucd/auxiliary/GraphemeBreakTest.txt",
74 "ucd/auxiliary/WordBreakTest.txt",
75 "ucd/auxiliary/LineBreakTest.txt",
76];
77
78// The property value lists below fix the enum variant order. A value in a UCD file that is absent
79// from these lists is an error, so a new Unicode release cannot silently drop characters into a
80// wrong class.
81
82/// Line_Break property values, in the order UAX #14 lists them.
83const LB_CLASSES: &[(&str, &str, &str)] = &[
84 ("XX", "XX", "Unknown, treated as AL."),
85 ("AI", "AI", "Ambiguous, treated as AL by default."),
86 ("AK", "AK", "Aksara."),
87 ("AL", "AL", "Alphabetic."),
88 ("AP", "AP", "Aksara pre-base."),
89 ("AS", "AS", "Aksara start."),
90 ("B2", "B2", "Break opportunity before and after."),
91 ("BA", "BA", "Break after."),
92 ("BB", "BB", "Break before."),
93 ("BK", "BK", "Mandatory break."),
94 ("CB", "CB", "Contingent break opportunity."),
95 ("CJ", "CJ", "Conditional Japanese starter, treated as NS by default."),
96 ("CL", "CL", "Close punctuation."),
97 ("CM", "CM", "Combining mark."),
98 ("CP", "CP", "Close parenthesis."),
99 ("CR", "CR", "Carriage return."),
100 ("EB", "EB", "Emoji base."),
101 ("EM", "EM", "Emoji modifier."),
102 ("EX", "EX", "Exclamation or interrogation."),
103 ("GL", "GL", "Non-breaking glue."),
104 ("H2", "H2", "Hangul LV syllable."),
105 ("H3", "H3", "Hangul LVT syllable."),
106 ("HH", "HH", "Unambiguous hyphen."),
107 ("HL", "HL", "Hebrew letter."),
108 ("HY", "HY", "Hyphen."),
109 ("ID", "ID", "Ideographic."),
110 ("IN", "IN", "Inseparable."),
111 ("IS", "IS", "Infix numeric separator."),
112 ("JL", "JL", "Hangul leading jamo."),
113 ("JT", "JT", "Hangul trailing jamo."),
114 ("JV", "JV", "Hangul vowel jamo."),
115 ("LF", "LF", "Line feed."),
116 ("NL", "NL", "Next line."),
117 ("NS", "NS", "Nonstarter."),
118 ("NU", "NU", "Numeric."),
119 ("OP", "OP", "Open punctuation."),
120 ("PO", "PO", "Postfix numeric."),
121 ("PR", "PR", "Prefix numeric."),
122 ("QU", "QU", "Quotation."),
123 ("RI", "RI", "Regional indicator."),
124 ("SA", "SA", "Complex context, South East Asian."),
125 ("SG", "SG", "Surrogate, treated as AL."),
126 ("SP", "SP", "Space."),
127 ("SY", "SY", "Symbols allowing break after."),
128 ("VF", "VF", "Virama final."),
129 ("VI", "VI", "Virama."),
130 ("WJ", "WJ", "Word joiner."),
131 ("ZW", "ZW", "Zero width space."),
132 ("ZWJ", "ZWJ", "Zero width joiner."),
133];
134
135/// Grapheme_Cluster_Break property values, as the UCD name, the Rust variant, and its doc.
136const GCB_CLASSES: &[(&str, &str, &str)] = &[
137 ("Other", "Other", "Any character not in another class."),
138 ("CR", "CR", "Carriage return."),
139 ("LF", "LF", "Line feed."),
140 ("Control", "Control", "A control, format or line separator character."),
141 ("Extend", "Extend", "A character that extends the preceding one."),
142 ("ZWJ", "ZWJ", "Zero width joiner."),
143 ("Regional_Indicator", "RegionalIndicator", "One half of a flag sequence."),
144 ("Prepend", "Prepend", "A character that prefixes the following one."),
145 ("SpacingMark", "SpacingMark", "A spacing combining mark."),
146 ("L", "L", "Hangul leading jamo."),
147 ("V", "V", "Hangul vowel jamo."),
148 ("T", "T", "Hangul trailing jamo."),
149 ("LV", "LV", "Hangul LV syllable."),
150 ("LVT", "LVT", "Hangul LVT syllable."),
151];
152
153/// Word_Break property values, as the UCD name, the Rust variant, and its doc.
154const WB_CLASSES: &[(&str, &str, &str)] = &[
155 ("Other", "Other", "Any character not in another class."),
156 ("CR", "CR", "Carriage return."),
157 ("LF", "LF", "Line feed."),
158 ("Newline", "Newline", "A newline character other than CR or LF."),
159 ("Extend", "Extend", "A character that extends the preceding one."),
160 ("Format", "Format", "A format character."),
161 ("Katakana", "Katakana", "A katakana character."),
162 ("ALetter", "ALetter", "A letter that takes part in words."),
163 ("MidLetter", "MidLetter", "A character found inside a word, such as a colon."),
164 ("MidNum", "MidNum", "A character found inside a number, such as a comma."),
165 ("MidNumLet", "MidNumLet", "A character found inside either a word or a number."),
166 ("Numeric", "Numeric", "A decimal digit."),
167 ("ExtendNumLet", "ExtendNumLet", "A character that joins words and numbers, such as an underscore."),
168 ("Regional_Indicator", "RegionalIndicator", "One half of a flag sequence."),
169 ("Hebrew_Letter", "HebrewLetter", "A Hebrew letter."),
170 ("Double_Quote", "DoubleQuote", "A double quotation mark."),
171 ("Single_Quote", "SingleQuote", "A single quotation mark."),
172 ("ZWJ", "ZWJ", "Zero width joiner."),
173 ("WSegSpace", "WSegSpace", "A space that separates words."),
174];
175
176/// Bidi_Class property values, by their UCD long names.
177const BIDI_CLASSES: &[(&str, &str, &str)] = &[
178 ("Left_To_Right", "L", "Strong left to right."),
179 ("Right_To_Left", "R", "Strong right to left."),
180 ("Arabic_Letter", "AL", "Strong right to left Arabic."),
181 ("European_Number", "EN", "European number."),
182 ("European_Separator", "ES", "European number separator."),
183 ("European_Terminator", "ET", "European number terminator."),
184 ("Arabic_Number", "AN", "Arabic number."),
185 ("Common_Separator", "CS", "Common number separator."),
186 ("Nonspacing_Mark", "NSM", "Nonspacing mark."),
187 ("Boundary_Neutral", "BN", "Boundary neutral."),
188 ("Paragraph_Separator", "B", "Paragraph separator."),
189 ("Segment_Separator", "S", "Segment separator."),
190 ("White_Space", "WS", "Whitespace."),
191 ("Other_Neutral", "ON", "Other neutral."),
192 ("Left_To_Right_Embedding", "LRE", "Left to right embedding."),
193 ("Left_To_Right_Override", "LRO", "Left to right override."),
194 ("Right_To_Left_Embedding", "RLE", "Right to left embedding."),
195 ("Right_To_Left_Override", "RLO", "Right to left override."),
196 ("Pop_Directional_Format", "PDF", "Pop directional format."),
197 ("Left_To_Right_Isolate", "LRI", "Left to right isolate."),
198 ("Right_To_Left_Isolate", "RLI", "Right to left isolate."),
199 ("First_Strong_Isolate", "FSI", "First strong isolate."),
200 ("Pop_Directional_Isolate", "PDI", "Pop directional isolate."),
201];
202
203/// Indic_Conjunct_Break property values.
204const INCB_CLASSES: &[(&str, &str, &str)] = &[
205 ("None", "None", "Not part of a conjunct."),
206 ("Consonant", "Consonant", "A consonant that can begin or end a conjunct."),
207 ("Extend", "Extend", "A mark that does not interrupt a conjunct."),
208 ("Linker", "Linker", "A virama that joins two consonants."),
209];
210
211/// Bit in the line breaking flags table marking a character of East Asian width F, W or H.
212const LB_FLAG_EAST_ASIAN: u8 = 1 << 0;
213/// Bit marking an initial quotation mark, general category Pi.
214const LB_FLAG_PI: u8 = 1 << 1;
215/// Bit marking a final quotation mark, general category Pf.
216const LB_FLAG_PF: u8 = 1 << 2;
217/// Bit marking an unassigned code point with Extended_Pictographic.
218const LB_FLAG_EXT_PICT_UNASSIGNED: u8 = 1 << 3;
219/// Bit marking general category Mn or Mc, which decides how class SA resolves.
220const LB_FLAG_MARK: u8 = 1 << 4;
221
222/// Bit in the segmentation flags table marking Extended_Pictographic.
223const SEG_FLAG_EXT_PICT: u8 = 1 << 0;
224/// Shift of the two bit Indic_Conjunct_Break field in the segmentation flags table.
225const SEG_INCB_SHIFT: u8 = 1;
226
227fn main() {
228 match run() {
229 Ok(()) => println!("Done."),
230 Err(e) => {
231 eprintln!("gen_unicode failed: {}", e);
232 std::process::exit(1);
233 },
234 }
235}
236
237/// Downloads, parses and emits everything.
238fn run() -> Outcome<()> {
239
240 let root = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
241 let cache = std::env::temp_dir().join("fe2o3_ucd").join(UCD_VERSION);
242 res!(mkdir(&cache));
243
244 let mut src = BTreeMap::new();
245 for path in DATA_FILES {
246 let text = res!(fetch(&cache, path));
247 src.insert(*path, text);
248 }
249
250 let ucd = res!(Ucd::parse(&src));
251
252 let tables = root.join("src").join("unicode").join("tables");
253 res!(mkdir(&tables));
254
255 res!(write_file(&tables.join("mod.rs"), &emit_mod()));
256 let cats = res!(Cats::parse(&src, &ucd));
257
258 res!(write_file(&tables.join("prop.rs"), &emit_prop(&cats)));
259 res!(write_file(&tables.join("cat.rs"), &res!(emit_cat(&cats))));
260 res!(write_file(&tables.join("norm.rs"), &res!(emit_norm(&ucd))));
261 res!(write_file(&tables.join("lb.rs"), &emit_lb(&ucd)));
262 res!(write_file(&tables.join("seg.rs"), &emit_seg(&ucd)));
263 res!(write_file(&tables.join("bidi.rs"), &emit_bidi(&ucd)));
264
265 // Vendor the conformance data.
266 let data = root.join("tests").join("unicode_data");
267 res!(mkdir(&data));
268 for path in TEST_FILES {
269 let text = res!(fetch(&cache, path));
270 let name = res!(base_name(path));
271 let stripped = strip_comments(&text);
272 let out = fmt!(
273 "# {}\n\
274 # Vendored from {}/{}/{} by fe2o3_text/src/bin/gen_unicode.rs.\n\
275 # Comment and blank lines have been stripped; the data lines are unchanged.\n\
276 {}",
277 name, UCD_BASE, UCD_VERSION, path, stripped,
278 );
279 res!(write_file(&data.join(&name), &out));
280 }
281
282 Ok(())
283}
284
285// ┌───────────────────────────────────────────────────────────────────────────────────────────┐
286// │ Fetching │
287// └───────────────────────────────────────────────────────────────────────────────────────────┘
288
289/// Creates a directory and any missing parents.
290fn mkdir(path: &Path) -> Outcome<()> {
291 match fs::create_dir_all(path) {
292 Ok(()) => Ok(()),
293 Err(e) => Err(err!(e, "While creating {:?}.", path; IO, Path)),
294 }
295}
296
297/// Returns the file name at the end of a UCD path.
298fn base_name(path: &str) -> Outcome<String> {
299 match path.rsplit('/').next() {
300 Some(name) => Ok(name.to_string()),
301 None => Err(err!("The path {} has no file name.", path; Invalid, Input)),
302 }
303}
304
305/// Reads a UCD file from the cache, downloading it with `curl` if it is not there.
306fn fetch(cache: &Path, path: &str) -> Outcome<String> {
307
308 let name = res!(base_name(path));
309 let dest = cache.join(&name);
310
311 if !dest.exists() {
312 let url = fmt!("{}/{}/{}", UCD_BASE, UCD_VERSION, path);
313 println!("Downloading {}", url);
314 let out = match Command::new("curl")
315 .arg("-sS")
316 .arg("--fail")
317 .arg("-o")
318 .arg(&dest)
319 .arg(&url)
320 .output()
321 {
322 Ok(out) => out,
323 Err(e) => return Err(err!(e,
324 "While running curl for {}. The generator needs curl on the path.", url;
325 IO, Network)),
326 };
327 if !out.status.success() {
328 return Err(err!(
329 "curl could not fetch {}: {}", url, String::from_utf8_lossy(&out.stderr);
330 IO, Network));
331 }
332 }
333
334 match fs::read_to_string(&dest) {
335 Ok(text) => Ok(text),
336 Err(e) => Err(err!(e, "While reading {:?}.", dest; IO, File)),
337 }
338}
339
340/// Writes a generated file.
341fn write_file(path: &Path, text: &str) -> Outcome<()> {
342 match fs::write(path, text) {
343 Ok(()) => {
344 println!("Wrote {:?} ({} bytes)", path, text.len());
345 Ok(())
346 },
347 Err(e) => Err(err!(e, "While writing {:?}.", path; IO, File)),
348 }
349}
350
351/// Removes comment and blank lines, and any trailing comment on a data line.
352fn strip_comments(text: &str) -> String {
353 let mut out = String::new();
354 for line in text.lines() {
355 if line.starts_with('#') {
356 continue;
357 }
358 let data = match line.find('#') {
359 Some(i) => &line[..i],
360 None => line,
361 };
362 let data = data.trim_end();
363 if data.trim().is_empty() {
364 continue;
365 }
366 out.push_str(data);
367 out.push('\n');
368 }
369 out
370}
371
372// ┌───────────────────────────────────────────────────────────────────────────────────────────┐
373// │ Parsing │
374// └───────────────────────────────────────────────────────────────────────────────────────────┘
375
376/// Everything parsed out of the UCD, indexed by code point where it is dense.
377struct Ucd {
378 /// General category, as the two letter abbreviation. Unassigned code points are `Cn`.
379 gc: Vec<[u8; 2]>,
380 /// Canonical combining class.
381 ccc: Vec<u8>,
382 /// Canonical decomposition mappings.
383 canon: BTreeMap<u32, Vec<u32>>,
384 /// Compatibility decomposition mappings, excluding the canonical ones.
385 compat: BTreeMap<u32, Vec<u32>>,
386 /// Code points with Full_Composition_Exclusion.
387 excl: BTreeSet<u32>,
388 /// Line_Break class index into [`LB_CLASSES`].
389 lb: Vec<u8>,
390 /// Line breaking flags, a bitset of the `LB_FLAG_*` constants.
391 lb_flags: Vec<u8>,
392 /// Grapheme_Cluster_Break class index into [`GCB_CLASSES`].
393 gcb: Vec<u8>,
394 /// Word_Break class index into [`WB_CLASSES`].
395 wb: Vec<u8>,
396 /// Segmentation flags: Extended_Pictographic and Indic_Conjunct_Break.
397 seg_flags: Vec<u8>,
398 /// Bidi_Class index into [`BIDI_CLASSES`].
399 bidi: Vec<u8>,
400 /// Bidi paired brackets, mapping a bracket to its pair and its kind, 0 open and 1 close.
401 brackets: BTreeMap<u32, (u32, u8)>,
402}
403
404/// Parses a hexadecimal code point.
405fn hex(s: &str) -> Outcome<u32> {
406 match u32::from_str_radix(s.trim(), 16) {
407 Ok(v) if v <= MAX_CP => Ok(v),
408 Ok(v) => Err(err!("The code point {:X} is out of range.", v; Invalid, Input, Range)),
409 Err(e) => Err(err!(e, "The string {:?} is not a code point.", s; Invalid, Input)),
410 }
411}
412
413/// Parses a `start..end` or `cp` range from the first field of a property file line.
414fn range(s: &str) -> Outcome<(u32, u32)> {
415 let s = s.trim();
416 match s.split_once("..") {
417 Some((a, b)) => Ok((res!(hex(a)), res!(hex(b)))),
418 None => {
419 let cp = res!(hex(s));
420 Ok((cp, cp))
421 },
422 }
423}
424
425/// Splits a property file line into its fields, dropping any comment.
426fn fields(line: &str) -> Option<Vec<&str>> {
427 let data = match line.find('#') {
428 Some(i) => &line[..i],
429 None => line,
430 };
431 if data.trim().is_empty() {
432 return None;
433 }
434 Some(data.split(';').map(|f| f.trim()).collect())
435}
436
437/// Returns the index of `name` in a property value list, or an error naming the file.
438fn class_index(list: &[(&str, &str, &str)], name: &str, file: &str) -> Outcome<u8> {
439 for (i, (ucd, rust, _)) in list.iter().enumerate() {
440 if *ucd == name || *rust == name {
441 return Ok(i as u8);
442 }
443 }
444 Err(err!(
445 "The property value {:?} in {} is not known to the generator. \
446 A new Unicode release may have added it.", name, file;
447 Invalid, Input, Mismatch))
448}
449
450impl Ucd {
451
452 /// Parses every downloaded data file.
453 fn parse(src: &BTreeMap<&str, String>) -> Outcome<Self> {
454
455 let get = |path: &str| -> Outcome<&String> {
456 match src.get(path) {
457 Some(text) => Ok(text),
458 None => Err(err!("The file {} was not downloaded.", path; Missing, Input)),
459 }
460 };
461
462 let mut ucd = Self {
463 gc: vec![*b"Cn"; NUM_CP],
464 ccc: vec![0; NUM_CP],
465 canon: BTreeMap::new(),
466 compat: BTreeMap::new(),
467 excl: BTreeSet::new(),
468 lb: vec![0; NUM_CP], // XX
469 lb_flags: vec![0; NUM_CP],
470 gcb: vec![0; NUM_CP], // Other
471 wb: vec![0; NUM_CP], // Other
472 seg_flags: vec![0; NUM_CP],
473 bidi: vec![0; NUM_CP], // L
474 brackets: BTreeMap::new(),
475 };
476
477 res!(ucd.parse_unicode_data(res!(get("ucd/UnicodeData.txt"))));
478 res!(ucd.parse_norm_props(res!(get("ucd/DerivedNormalizationProps.txt"))));
479 res!(ucd.parse_bidi_class(res!(get("ucd/extracted/DerivedBidiClass.txt"))));
480 res!(ucd.parse_brackets(res!(get("ucd/BidiBrackets.txt"))));
481 res!(ucd.parse_line_break(res!(get("ucd/LineBreak.txt"))));
482
483 let eaw = res!(ucd.parse_east_asian(res!(get("ucd/EastAsianWidth.txt"))));
484 let ext = res!(parse_ext_pict(res!(get("ucd/emoji/emoji-data.txt"))));
485 res!(ucd.parse_gcb(res!(get("ucd/auxiliary/GraphemeBreakProperty.txt"))));
486 res!(ucd.parse_wb(res!(get("ucd/auxiliary/WordBreakProperty.txt"))));
487 let incb = res!(parse_incb(res!(get("ucd/DerivedCoreProperties.txt"))));
488
489 res!(ucd.derive_flags(&eaw, &ext, &incb));
490
491 Ok(ucd)
492 }
493
494 /// Parses UnicodeData.txt for general category, combining class and decompositions.
495 fn parse_unicode_data(&mut self, text: &str) -> Outcome<()> {
496
497 let mut first: Option<u32> = None;
498
499 for line in text.lines() {
500 if line.trim().is_empty() {
501 continue;
502 }
503 let f: Vec<&str> = line.split(';').collect();
504 if f.len() < 6 {
505 return Err(err!(
506 "UnicodeData.txt line {:?} has {} fields, expected at least 6.",
507 line, f.len(); Invalid, Input));
508 }
509 let cp = res!(hex(f[0]));
510 let name = f[1];
511 let gc = f[2].as_bytes();
512 if gc.len() != 2 {
513 return Err(err!(
514 "The general category {:?} at U+{:04X} is not two letters.", f[2], cp;
515 Invalid, Input));
516 }
517 let ccc = match f[3].trim().parse::<u16>() {
518 Ok(v) if v <= 255 => v as u8,
519 _ => return Err(err!(
520 "The combining class {:?} at U+{:04X} is not a byte.", f[3], cp;
521 Invalid, Input)),
522 };
523
524 // A range is written as a pair of lines ending in <..., First> and <..., Last>.
525 let (lo, hi) = if name.ends_with(", First>") {
526 first = Some(cp);
527 continue;
528 } else if name.ends_with(", Last>") {
529 let lo = match first.take() {
530 Some(lo) => lo,
531 None => return Err(err!(
532 "The range ending at U+{:04X} has no First line.", cp; Invalid, Input)),
533 };
534 (lo, cp)
535 } else {
536 (cp, cp)
537 };
538
539 for c in lo..=hi {
540 let i = c as usize;
541 self.gc[i] = [gc[0], gc[1]];
542 self.ccc[i] = ccc;
543 }
544
545 // Field 5 is the decomposition mapping, prefixed by a tag if it is a compatibility
546 // mapping.
547 let dec = f[5].trim();
548 if !dec.is_empty() && lo == hi {
549 let (compat, body) = if dec.starts_with('<') {
550 match dec.split_once('>') {
551 Some((_, rest)) => (true, rest.trim()),
552 None => return Err(err!(
553 "The decomposition {:?} at U+{:04X} has an unclosed tag.", dec, cp;
554 Invalid, Input)),
555 }
556 } else {
557 (false, dec)
558 };
559 let mut seq = Vec::new();
560 for part in body.split_whitespace() {
561 seq.push(res!(hex(part)));
562 }
563 if seq.is_empty() {
564 return Err(err!(
565 "The decomposition {:?} at U+{:04X} is empty.", dec, cp; Invalid, Input));
566 }
567 if compat {
568 self.compat.insert(cp, seq);
569 } else {
570 self.canon.insert(cp, seq);
571 }
572 }
573 }
574
575 Ok(())
576 }
577
578 /// Parses the Full_Composition_Exclusion property.
579 fn parse_norm_props(&mut self, text: &str) -> Outcome<()> {
580 for line in text.lines() {
581 let f = match fields(line) {
582 Some(f) => f,
583 None => continue,
584 };
585 if f.len() < 2 || f[1] != "Full_Composition_Exclusion" {
586 continue;
587 }
588 let (lo, hi) = res!(range(f[0]));
589 for c in lo..=hi {
590 self.excl.insert(c);
591 }
592 }
593 Ok(())
594 }
595
596 /// Parses the Bidi_Class property, honouring the `@missing` defaults for unassigned code
597 /// points.
598 fn parse_bidi_class(&mut self, text: &str) -> Outcome<()> {
599
600 let index = |name: &str| -> Outcome<u8> {
601 class_index(BIDI_CLASSES, name, "DerivedBidiClass.txt")
602 };
603
604 // The @missing lines give the class of the unassigned code points in a range, so they must
605 // be applied before the explicit assignments.
606 for line in text.lines() {
607 let body = match line.strip_prefix("# @missing:") {
608 Some(body) => body,
609 None => continue,
610 };
611 let f: Vec<&str> = body.split(';').map(|s| s.trim()).collect();
612 if f.len() < 2 {
613 return Err(err!(
614 "The @missing line {:?} has too few fields.", line; Invalid, Input));
615 }
616 let (lo, hi) = res!(range(f[0]));
617 let cls = res!(index(f[1]));
618 for c in lo..=hi {
619 self.bidi[c as usize] = cls;
620 }
621 }
622
623 for line in text.lines() {
624 let f = match fields(line) {
625 Some(f) => f,
626 None => continue,
627 };
628 if f.len() < 2 {
629 continue;
630 }
631 let (lo, hi) = res!(range(f[0]));
632 let cls = res!(index(f[1]));
633 for c in lo..=hi {
634 self.bidi[c as usize] = cls;
635 }
636 }
637
638 Ok(())
639 }
640
641 /// Parses BidiBrackets.txt.
642 fn parse_brackets(&mut self, text: &str) -> Outcome<()> {
643 for line in text.lines() {
644 let f = match fields(line) {
645 Some(f) => f,
646 None => continue,
647 };
648 if f.len() < 3 {
649 continue;
650 }
651 let cp = res!(hex(f[0]));
652 let pair = res!(hex(f[1]));
653 let kind = match f[2] {
654 "o" => 0u8,
655 "c" => 1u8,
656 other => return Err(err!(
657 "The bracket kind {:?} at U+{:04X} is neither o nor c.", other, cp;
658 Invalid, Input)),
659 };
660 self.brackets.insert(cp, (pair, kind));
661 }
662 Ok(())
663 }
664
665 /// Parses LineBreak.txt.
666 ///
667 /// The file lists the reserved code points that take a value other than the default, so the
668 /// defaults its header describes in prose need no code here: a code point the file does not
669 /// list takes XX, which the algorithm resolves to AL. The Extended_Pictographic code points
670 /// that are still unassigned are the ones this matters for, and they are deliberately left at
671 /// XX.
672 fn parse_line_break(&mut self, text: &str) -> Outcome<()> {
673
674 for line in text.lines() {
675 let f = match fields(line) {
676 Some(f) => f,
677 None => continue,
678 };
679 if f.len() < 2 {
680 continue;
681 }
682 let (lo, hi) = res!(range(f[0]));
683 let cls = res!(class_index(LB_CLASSES, f[1], "LineBreak.txt"));
684 for c in lo..=hi {
685 self.lb[c as usize] = cls;
686 }
687 }
688
689 Ok(())
690 }
691
692 /// Parses EastAsianWidth.txt into a per code point width letter. As with LineBreak.txt, the
693 /// file lists the reserved ranges that take a value other than the default of N.
694 fn parse_east_asian(&self, text: &str) -> Outcome<Vec<u8>> {
695
696 let mut eaw = vec![b'N'; NUM_CP];
697
698 for line in text.lines() {
699 let f = match fields(line) {
700 Some(f) => f,
701 None => continue,
702 };
703 if f.len() < 2 {
704 continue;
705 }
706 let (lo, hi) = res!(range(f[0]));
707 let w = match f[1] {
708 "A" => b'A',
709 "F" => b'F',
710 "H" => b'H',
711 "N" => b'N',
712 "Na" => b'n',
713 "W" => b'W',
714 other => return Err(err!(
715 "The East_Asian_Width value {:?} is not known.", other; Invalid, Input)),
716 };
717 for c in lo..=hi {
718 eaw[c as usize] = w;
719 }
720 }
721
722 Ok(eaw)
723 }
724
725 /// Parses GraphemeBreakProperty.txt.
726 fn parse_gcb(&mut self, text: &str) -> Outcome<()> {
727 for line in text.lines() {
728 let f = match fields(line) {
729 Some(f) => f,
730 None => continue,
731 };
732 if f.len() < 2 {
733 continue;
734 }
735 let (lo, hi) = res!(range(f[0]));
736 let cls = res!(class_index(GCB_CLASSES, f[1], "GraphemeBreakProperty.txt"));
737 for c in lo..=hi {
738 self.gcb[c as usize] = cls;
739 }
740 }
741 Ok(())
742 }
743
744 /// Parses WordBreakProperty.txt.
745 fn parse_wb(&mut self, text: &str) -> Outcome<()> {
746 for line in text.lines() {
747 let f = match fields(line) {
748 Some(f) => f,
749 None => continue,
750 };
751 if f.len() < 2 {
752 continue;
753 }
754 let (lo, hi) = res!(range(f[0]));
755 let cls = res!(class_index(WB_CLASSES, f[1], "WordBreakProperty.txt"));
756 for c in lo..=hi {
757 self.wb[c as usize] = cls;
758 }
759 }
760 Ok(())
761 }
762
763 /// Combines the auxiliary properties into the two flag tables.
764 fn derive_flags(
765 &mut self,
766 eaw: &[u8],
767 ext: &[bool],
768 incb: &[u8],
769 )
770 -> Outcome<()>
771 {
772 for c in 0..NUM_CP {
773 let gc = self.gc[c];
774
775 let mut lb = 0u8;
776 if matches!(eaw[c], b'F' | b'W' | b'H') {
777 lb |= LB_FLAG_EAST_ASIAN;
778 }
779 if gc == *b"Pi" {
780 lb |= LB_FLAG_PI;
781 }
782 if gc == *b"Pf" {
783 lb |= LB_FLAG_PF;
784 }
785 if ext[c] && gc == *b"Cn" {
786 lb |= LB_FLAG_EXT_PICT_UNASSIGNED;
787 }
788 if gc == *b"Mn" || gc == *b"Mc" {
789 lb |= LB_FLAG_MARK;
790 }
791 self.lb_flags[c] = lb;
792
793 let mut seg = 0u8;
794 if ext[c] {
795 seg |= SEG_FLAG_EXT_PICT;
796 }
797 seg |= incb[c] << SEG_INCB_SHIFT;
798 self.seg_flags[c] = seg;
799 }
800 Ok(())
801 }
802}
803
804/// Parses the Extended_Pictographic property from emoji-data.txt.
805fn parse_ext_pict(text: &str) -> Outcome<Vec<bool>> {
806 let mut ext = vec![false; NUM_CP];
807 for line in text.lines() {
808 let f = match fields(line) {
809 Some(f) => f,
810 None => continue,
811 };
812 if f.len() < 2 || f[1] != "Extended_Pictographic" {
813 continue;
814 }
815 let (lo, hi) = res!(range(f[0]));
816 for c in lo..=hi {
817 ext[c as usize] = true;
818 }
819 }
820 Ok(ext)
821}
822
823/// Parses the Indic_Conjunct_Break property from DerivedCoreProperties.txt.
824fn parse_incb(text: &str) -> Outcome<Vec<u8>> {
825 let mut incb = vec![0u8; NUM_CP];
826 for line in text.lines() {
827 let f = match fields(line) {
828 Some(f) => f,
829 None => continue,
830 };
831 if f.len() < 3 || f[1] != "InCB" {
832 continue;
833 }
834 let (lo, hi) = res!(range(f[0]));
835 let cls = res!(class_index(INCB_CLASSES, f[2], "DerivedCoreProperties.txt"));
836 for c in lo..=hi {
837 incb[c as usize] = cls;
838 }
839 }
840 Ok(incb)
841}
842
843// ┌───────────────────────────────────────────────────────────────────────────────────────────┐
844// │ Emission │
845// └───────────────────────────────────────────────────────────────────────────────────────────┘
846
847/// Returns the header that every generated file carries.
848fn header(what: &str) -> String {
849 fmt!(
850 "// GENERATED FILE. DO NOT EDIT.\n\
851 //\n\
852 // {}\n\
853 //\n\
854 // Generated by fe2o3_text/src/bin/gen_unicode.rs from the Unicode Character Database,\n\
855 // version {}, at {}/{}/.\n\
856 // Run `cargo run -p oxedyne_fe2o3_text --bin gen_unicode` to rebuild.\n\n",
857 what, UCD_VERSION, UCD_BASE, UCD_VERSION,
858 )
859}
860
861/// Compresses a dense per code point value array into a partition: the code point at which each
862/// run starts, and the value of that run. A lookup binary searches the starts.
863fn partition(vals: &[u8]) -> (Vec<u32>, Vec<u8>) {
864 let mut starts = Vec::new();
865 let mut runs = Vec::new();
866 let mut prev = None;
867 for (c, v) in vals.iter().enumerate() {
868 if prev != Some(*v) {
869 starts.push(c as u32);
870 runs.push(*v);
871 prev = Some(*v);
872 }
873 }
874 (starts, runs)
875}
876
877/// Emits a `u32` array, eight values to the line.
878fn emit_u32(out: &mut String, name: &str, doc: &str, vals: &[u32]) {
879 let _ = write!(out, "/// {}\npub static {}: [u32; {}] = [\n", doc, name, vals.len());
880 for chunk in vals.chunks(8) {
881 let mut line = String::from("\t");
882 for v in chunk {
883 let _ = write!(line, "0x{:X}, ", v);
884 }
885 let _ = write!(out, "{}\n", line.trim_end());
886 }
887 out.push_str("];\n\n");
888}
889
890/// Emits a `u8` array, sixteen values to the line.
891fn emit_u8(out: &mut String, name: &str, doc: &str, vals: &[u8]) {
892 let _ = write!(out, "/// {}\npub static {}: [u8; {}] = [\n", doc, name, vals.len());
893 for chunk in vals.chunks(16) {
894 let mut line = String::from("\t");
895 for v in chunk {
896 let _ = write!(line, "{}, ", v);
897 }
898 let _ = write!(out, "{}\n", line.trim_end());
899 }
900 out.push_str("];\n\n");
901}
902
903/// Emits an array of enum variants, eight values to the line.
904fn emit_enum_vals(
905 out: &mut String,
906 name: &str,
907 doc: &str,
908 typ: &str,
909 list: &[(&str, &str, &str)],
910 vals: &[u8],
911)
912 -> Outcome<()>
913{
914 let _ = write!(out, "/// {}\npub static {}: [{}; {}] = [\n", doc, name, typ, vals.len());
915 for chunk in vals.chunks(8) {
916 let mut line = String::from("\t");
917 for v in chunk {
918 let variant = match list.get(*v as usize) {
919 Some((_, rust, _)) => *rust,
920 None => return Err(err!(
921 "The class index {} is out of range for {}.", v, typ; Bug, Index)),
922 };
923 let _ = write!(line, "{}::{}, ", typ, variant);
924 }
925 let _ = write!(out, "{}\n", line.trim_end());
926 }
927 out.push_str("];\n\n");
928 Ok(())
929}
930
931/// Emits an array of `char` literals, eight values to the line.
932fn emit_chars(out: &mut String, name: &str, doc: &str, vals: &[u32]) {
933 let _ = write!(out, "/// {}\npub static {}: [char; {}] = [\n", doc, name, vals.len());
934 for chunk in vals.chunks(8) {
935 let mut line = String::from("\t");
936 for v in chunk {
937 let _ = write!(line, "'\\u{{{:X}}}', ", v);
938 }
939 let _ = write!(out, "{}\n", line.trim_end());
940 }
941 out.push_str("];\n\n");
942}
943
944/// Emits the table module root.
945fn emit_mod() -> String {
946 let mut out = header("The generated Unicode tables.");
947 out.push_str(
948"//! The tables are partitions of the code point space: a sorted array of run start code points,
949//! and a parallel array of the value each run takes. A lookup is a binary search of the starts,
950//! about eleven steps for the largest table. This representation was chosen over a staged trie
951//! because the tables are committed source: a partition of the Line_Break property is a few
952//! thousand entries where a two level trie is tens of thousands, and the difference in lookup cost
953//! is invisible next to the rest of a segmentation pass.
954
955pub mod bidi;
956pub mod cat;
957pub mod lb;
958pub mod norm;
959pub mod prop;
960pub mod seg;
961
962/// The Unicode Character Database version every table in this module was generated from.
963pub const UCD_VERSION: &str = \"");
964 out.push_str(UCD_VERSION);
965 out.push_str("\";\n");
966 out
967}
968
969/// Emits one property enum and its `Partitioned` implementation.
970fn emit_enum(
971 out: &mut String,
972 doc: &str,
973 typ: &str,
974 list: &[(&str, &str, &str)],
975 dflt: &str,
976 table: Option<(&str, &str)>,
977) {
978 let _ = write!(out, "/// {}\n", doc);
979 out.push_str("#[repr(u8)]\n#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]\n");
980 let _ = write!(out, "pub enum {} {{\n", typ);
981 for (ucd, rust, vdoc) in list {
982 if ucd == rust {
983 let _ = write!(out, "\t/// {}\n\t{},\n", vdoc, rust);
984 } else {
985 let _ = write!(out, "\t/// {} The UCD name is `{}`.\n\t{},\n", vdoc, ucd, rust);
986 }
987 }
988 out.push_str("}\n\n");
989
990 if let Some((module, prefix)) = table {
991 let _ = write!(out,
992"impl Partitioned for {typ} {{
993
994 const DEFAULT: Self = Self::{dflt};
995
996 fn table() -> (&'static [u32], &'static [Self]) {{
997 (&super::{module}::{prefix}_STARTS, &super::{module}::{prefix}_VALS)
998 }}
999}}
1000
1001");
1002 }
1003}
1004
1005/// Emits the property enums.
1006fn emit_prop(cats: &Cats) -> String {
1007
1008 let mut out = header("The Unicode character property enums.");
1009 out.push_str("\nuse crate::unicode::lookup::Partitioned;\n\n");
1010
1011 emit_enum(&mut out,
1012 "The Line_Break property of UAX #14, as the UCD gives it, before any tailoring.",
1013 "LineBreakClass", LB_CLASSES, "XX", Some(("lb", "LB")));
1014
1015 emit_enum(&mut out,
1016 "The Grapheme_Cluster_Break property of UAX #29.",
1017 "GraphemeClass", GCB_CLASSES, "Other", Some(("seg", "GCB")));
1018
1019 emit_enum(&mut out,
1020 "The Word_Break property of UAX #29.",
1021 "WordClass", WB_CLASSES, "Other", Some(("seg", "WB")));
1022
1023 emit_enum(&mut out,
1024 "The Bidi_Class property of UAX #9.",
1025 "BidiClass", BIDI_CLASSES, "L", Some(("bidi", "BIDI")));
1026
1027 emit_enum(&mut out,
1028 "The Indic_Conjunct_Break property of UAX #44, which grapheme rule GB9c consults.",
1029 "ConjunctBreak", INCB_CLASSES, "None", None);
1030
1031 out.push_str(
1032"/// The Bidi_Paired_Bracket_Type property of UAX #9.
1033#[repr(u8)]
1034#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
1035pub enum BracketKind {
1036 /// Not a paired bracket.
1037 None,
1038 /// An opening bracket.
1039 Open,
1040 /// A closing bracket.
1041 Close,
1042}
1043
1044");
1045 emit_cat_enums(&mut out, cats);
1046 out
1047}
1048
1049/// Emits the normalisation tables.
1050fn emit_norm(ucd: &Ucd) -> Outcome<String> {
1051
1052 let mut out = header("Tables for the normalisation of UAX #15.");
1053
1054 let (starts, vals) = partition(&ucd.ccc);
1055 emit_u32(&mut out, "CCC_STARTS", "Start code points of the canonical combining class runs.", &starts);
1056 emit_u8(&mut out, "CCC_VALS", "The canonical combining class of each run.", &vals);
1057
1058 // Canonical decompositions, flattened into a pool of characters.
1059 let mut keys = Vec::new();
1060 let mut offs = Vec::new();
1061 let mut pool = Vec::new();
1062 for (cp, seq) in &ucd.canon {
1063 keys.push(*cp);
1064 offs.push(pool.len() as u16);
1065 pool.extend_from_slice(seq);
1066 }
1067 offs.push(pool.len() as u16);
1068 emit_u32(&mut out, "CANON_KEYS",
1069 "Code points with a canonical decomposition, sorted.", &keys);
1070 let offs32: Vec<u32> = offs.iter().map(|o| *o as u32).collect();
1071 emit_u32(&mut out, "CANON_OFFS",
1072 "Offsets into `CANON_POOL`, one per key plus a final bound.", &offs32);
1073 emit_chars(&mut out, "CANON_POOL",
1074 "The canonical decompositions, concatenated.", &pool);
1075
1076 // Compatibility decompositions. Only the code points whose mapping is tagged appear here; a
1077 // caller wanting the full compatibility decomposition falls back on the canonical table.
1078 let mut keys = Vec::new();
1079 let mut offs = Vec::new();
1080 let mut pool = Vec::new();
1081 for (cp, seq) in &ucd.compat {
1082 keys.push(*cp);
1083 offs.push(pool.len() as u16);
1084 pool.extend_from_slice(seq);
1085 }
1086 offs.push(pool.len() as u16);
1087 emit_u32(&mut out, "COMPAT_KEYS",
1088 "Code points with a compatibility decomposition, sorted.", &keys);
1089 let offs32: Vec<u32> = offs.iter().map(|o| *o as u32).collect();
1090 emit_u32(&mut out, "COMPAT_OFFS",
1091 "Offsets into `COMPAT_POOL`, one per key plus a final bound.", &offs32);
1092 emit_chars(&mut out, "COMPAT_POOL",
1093 "The compatibility decompositions, concatenated.", &pool);
1094
1095 // Primary composites: a canonical decomposition of exactly two characters whose code point is
1096 // not excluded from composition.
1097 let mut comp: Vec<(u64, u32)> = Vec::new();
1098 for (cp, seq) in &ucd.canon {
1099 if seq.len() != 2 || ucd.excl.contains(cp) {
1100 continue;
1101 }
1102 let key = ((seq[0] as u64) << 32) | (seq[1] as u64);
1103 comp.push((key, *cp));
1104 }
1105 comp.sort();
1106 let mut hi = Vec::new();
1107 let mut lo = Vec::new();
1108 let mut val = Vec::new();
1109 for (key, cp) in &comp {
1110 hi.push((*key >> 32) as u32);
1111 lo.push((*key & 0xFFFF_FFFF) as u32);
1112 val.push(*cp);
1113 }
1114 emit_u32(&mut out, "COMPOSE_FIRST",
1115 "The first character of each primary composite pair, sorted with `COMPOSE_SECOND`.", &hi);
1116 emit_u32(&mut out, "COMPOSE_SECOND",
1117 "The second character of each primary composite pair.", &lo);
1118 emit_chars(&mut out, "COMPOSE_VALS",
1119 "The composite each pair yields.", &val);
1120
1121 Ok(out)
1122}
1123
1124/// Emits the line breaking tables.
1125fn emit_lb(ucd: &Ucd) -> String {
1126
1127 let mut out = header("Tables for the line breaking of UAX #14.");
1128 out.push_str("\nuse crate::unicode::prop::LineBreakClass;\n\n");
1129
1130 let (starts, vals) = partition(&ucd.lb);
1131 emit_u32(&mut out, "LB_STARTS", "Start code points of the Line_Break class runs.", &starts);
1132 let _ = emit_enum_vals(&mut out, "LB_VALS", "The Line_Break class of each run.",
1133 "LineBreakClass", LB_CLASSES, &vals);
1134
1135 let (starts, vals) = partition(&ucd.lb_flags);
1136 emit_u32(&mut out, "LB_FLAG_STARTS", "Start code points of the line breaking flag runs.", &starts);
1137 emit_u8(&mut out, "LB_FLAG_VALS",
1138 "The line breaking flags of each run. See the `flag` constants in `unicode::linebreak`.",
1139 &vals);
1140
1141 out
1142}
1143
1144/// Emits the segmentation tables.
1145fn emit_seg(ucd: &Ucd) -> String {
1146
1147 let mut out = header("Tables for the segmentation of UAX #29.");
1148 out.push_str("\nuse crate::unicode::prop::{\n\tGraphemeClass,\n\tWordClass,\n};\n\n");
1149
1150 let (starts, vals) = partition(&ucd.gcb);
1151 emit_u32(&mut out, "GCB_STARTS",
1152 "Start code points of the Grapheme_Cluster_Break class runs.", &starts);
1153 let _ = emit_enum_vals(&mut out, "GCB_VALS", "The Grapheme_Cluster_Break class of each run.",
1154 "GraphemeClass", GCB_CLASSES, &vals);
1155
1156 let (starts, vals) = partition(&ucd.wb);
1157 emit_u32(&mut out, "WB_STARTS", "Start code points of the Word_Break class runs.", &starts);
1158 let _ = emit_enum_vals(&mut out, "WB_VALS", "The Word_Break class of each run.",
1159 "WordClass", WB_CLASSES, &vals);
1160
1161 let (starts, vals) = partition(&ucd.seg_flags);
1162 emit_u32(&mut out, "SEG_FLAG_STARTS",
1163 "Start code points of the segmentation flag runs.", &starts);
1164 emit_u8(&mut out, "SEG_FLAG_VALS",
1165 "The segmentation flags of each run: bit 0 is Extended_Pictographic, bits 1 and 2 are the \
1166 Indic_Conjunct_Break value.",
1167 &vals);
1168
1169 out
1170}
1171
1172/// Emits the bidirectional tables.
1173fn emit_bidi(ucd: &Ucd) -> String {
1174
1175 let mut out = header("Tables for the bidirectional algorithm of UAX #9.");
1176 out.push_str("\nuse crate::unicode::prop::BidiClass;\n\n");
1177
1178 let (starts, vals) = partition(&ucd.bidi);
1179 emit_u32(&mut out, "BIDI_STARTS", "Start code points of the Bidi_Class runs.", &starts);
1180 let _ = emit_enum_vals(&mut out, "BIDI_VALS", "The Bidi_Class of each run.",
1181 "BidiClass", BIDI_CLASSES, &vals);
1182
1183 let keys: Vec<u32> = ucd.brackets.keys().copied().collect();
1184 let pairs: Vec<u32> = ucd.brackets.values().map(|(p, _)| *p).collect();
1185 let kinds: Vec<u8> = ucd.brackets.values().map(|(_, k)| *k).collect();
1186 emit_u32(&mut out, "BRACKET_KEYS", "The paired bracket code points, sorted.", &keys);
1187 emit_chars(&mut out, "BRACKET_PAIRS", "The bracket each key pairs with.", &pairs);
1188 emit_u8(&mut out, "BRACKET_KINDS", "The kind of each bracket: 0 opening, 1 closing.", &kinds);
1189
1190 out
1191}
1192
1193// ┌───────────────────────────────────────────────────────────────────────────────────────────┐
1194// │ Character classes: General_Category, Script, Script_Extensions, binary properties │
1195// └───────────────────────────────────────────────────────────────────────────────────────────┘
1196
1197/// The General_Category values in the order the generated enum takes them. The groups (`L`, `LC`,
1198/// ...) are not variants; they become bit masks over this order.
1199const GC_CODES: &[&str] = &[
1200 "Lu", "Ll", "Lt", "Lm", "Lo",
1201 "Mn", "Mc", "Me",
1202 "Nd", "Nl", "No",
1203 "Pc", "Pd", "Ps", "Pe", "Pi", "Pf", "Po",
1204 "Sm", "Sc", "Sk", "So",
1205 "Zs", "Zl", "Zp",
1206 "Cc", "Cf", "Cs", "Co", "Cn",
1207];
1208
1209/// Normalises a property name or value alias by UAX44-LM3, less the `is` prefix, which the
1210/// run time lookup tries both with and without: ASCII lower case, and no spaces, underscores
1211/// or hyphens.
1212fn loose(name: &str) -> String {
1213 name.chars()
1214 .filter(|c| *c != ' ' && *c != '_' && *c != '-')
1215 .map(|c| c.to_ascii_lowercase())
1216 .collect()
1217}
1218
1219/// Sorts and merges a list of inclusive code point ranges, joining those that touch.
1220fn merge(mut ranges: Vec<(u32, u32)>) -> Vec<(u32, u32)> {
1221 ranges.sort();
1222 let mut out: Vec<(u32, u32)> = Vec::with_capacity(ranges.len());
1223 for (lo, hi) in ranges {
1224 match out.last_mut() {
1225 Some(last) if lo <= last.1.saturating_add(1) => {
1226 if hi > last.1 {
1227 last.1 = hi;
1228 }
1229 },
1230 _ => out.push((lo, hi)),
1231 }
1232 }
1233 out
1234}
1235
1236/// Inserts an alias, refusing one that already names something else.
1237fn alias<T: Copy + PartialEq + std::fmt::Debug>(
1238 map: &mut BTreeMap<String, T>,
1239 name: &str,
1240 val: T,
1241 what: &str,
1242)
1243 -> Outcome<()>
1244{
1245 let key = loose(name);
1246 match map.get(&key) {
1247 Some(old) if *old != val => Err(err!(
1248 "The {} alias {:?} names both {:?} and {:?}.", what, name, old, val;
1249 Invalid, Input, Mismatch)),
1250 _ => {
1251 map.insert(key, val);
1252 Ok(())
1253 },
1254 }
1255}
1256
1257/// The character class properties, parsed.
1258struct Cats {
1259 gc: Vec<u8>, // GC_CODES index per code point
1260 gc_long: Vec<String>, // long name per GC_CODES entry
1261 gc_alias: BTreeMap<String, u32>, // loose alias to category mask
1262 scripts: Vec<(String, String)>, // (short, long), enum order
1263 sc_alias: BTreeMap<String, u8>, // loose alias to script index
1264 sc: Vec<u8>, // script index per code point
1265 scx: Vec<u16>, // 0, or 1 + index into scx_sets
1266 scx_sets: Vec<Vec<u8>>,
1267 bins: BTreeMap<String, Vec<(u32, u32)>>, // long name to merged ranges
1268 bin_alias: BTreeMap<String, u8>, // loose alias to index in `bins`
1269 gcb_alias: BTreeMap<String, u8>, // loose alias to GCB_CLASSES index
1270 wb_alias: BTreeMap<String, u8>, // loose alias to WB_CLASSES index
1271 sbs: Vec<(String, String)>, // Sentence_Break (short, long), enum order
1272 sb_alias: BTreeMap<String, u8>,
1273 sb: Vec<u8>, // Sentence_Break index per code point
1274}
1275
1276impl Cats {
1277
1278 fn parse(src: &BTreeMap<&str, String>, ucd: &Ucd) -> Outcome<Self> {
1279
1280 let get = |path: &str| -> Outcome<&String> {
1281 match src.get(path) {
1282 Some(text) => Ok(text),
1283 None => Err(err!("The file {} was not downloaded.", path; Missing, Input)),
1284 }
1285 };
1286 let pva = res!(get("ucd/PropertyValueAliases.txt"));
1287
1288 // General_Category names and groups.
1289 let mut gc_long = vec![String::new(); GC_CODES.len()];
1290 let mut gc_alias = BTreeMap::new();
1291 for line in pva.lines() {
1292 let (data, note) = match line.split_once('#') {
1293 Some((d, n)) => (d, n),
1294 None => (line, ""),
1295 };
1296 let f: Vec<&str> = data.split(';').map(|x| x.trim()).collect();
1297 if f.len() < 3 || f[0] != "gc" {
1298 continue;
1299 }
1300 let mask = if note.contains('|') {
1301 let mut m = 0u32;
1302 for part in note.split('|') {
1303 let code = part.trim();
1304 match GC_CODES.iter().position(|c| *c == code) {
1305 Some(i) => m |= 1 << i,
1306 None => return Err(err!(
1307 "The category group {} names an unknown member {:?}.", f[1], code;
1308 Invalid, Input)),
1309 }
1310 }
1311 m
1312 } else {
1313 match GC_CODES.iter().position(|c| *c == f[1]) {
1314 Some(i) => {
1315 gc_long[i] = f[2].to_string();
1316 1 << i
1317 },
1318 None => return Err(err!(
1319 "The general category {:?} is not known to the generator.", f[1];
1320 Invalid, Input, Mismatch)),
1321 }
1322 };
1323 for name in &f[1..] {
1324 res!(alias(&mut gc_alias, name, mask, "General_Category"));
1325 }
1326 }
1327 for (i, l) in gc_long.iter().enumerate() {
1328 if l.is_empty() {
1329 return Err(err!(
1330 "PropertyValueAliases.txt gives no long name for {}.", GC_CODES[i];
1331 Missing, Input));
1332 }
1333 }
1334
1335 let mut gc = vec![0u8; NUM_CP];
1336 for (c, code) in ucd.gc.iter().enumerate() {
1337 let code = String::from_utf8_lossy(code);
1338 match GC_CODES.iter().position(|x| *x == code) {
1339 Some(i) => gc[c] = i as u8,
1340 None => return Err(err!(
1341 "The general category {} at U+{:04X} is not known.", code, c; Invalid, Input)),
1342 }
1343 }
1344
1345 // Script names.
1346 let mut scripts = Vec::new();
1347 let mut sc_alias = BTreeMap::new();
1348 for line in pva.lines() {
1349 let f = match fields(line) {
1350 Some(f) => f,
1351 None => continue,
1352 };
1353 if f.len() < 3 || f[0] != "sc" {
1354 continue;
1355 }
1356 let i = scripts.len();
1357 if i > 255 {
1358 return Err(err!("There are more scripts than a byte can index."; Excessive));
1359 }
1360 scripts.push((f[1].to_string(), f[2].to_string()));
1361 for name in &f[1..] {
1362 res!(alias(&mut sc_alias, name, i as u8, "Script"));
1363 }
1364 }
1365 let unknown = match scripts.iter().position(|(s, _)| s == "Zzzz") {
1366 Some(i) => i as u8,
1367 None => return Err(err!("No script Zzzz (Unknown) was listed."; Missing, Input)),
1368 };
1369 let by_long = |name: &str| -> Outcome<u8> {
1370 match scripts.iter().position(|(_, l)| l == name) {
1371 Some(i) => Ok(i as u8),
1372 None => Err(err!("The script {:?} is not in PropertyValueAliases.txt.", name;
1373 Invalid, Input, Mismatch)),
1374 }
1375 };
1376 let by_short = |name: &str| -> Outcome<u8> {
1377 match scripts.iter().position(|(s, _)| s == name) {
1378 Some(i) => Ok(i as u8),
1379 None => Err(err!("The script {:?} is not in PropertyValueAliases.txt.", name;
1380 Invalid, Input, Mismatch)),
1381 }
1382 };
1383
1384 let mut sc = vec![unknown; NUM_CP];
1385 for line in res!(get("ucd/Scripts.txt")).lines() {
1386 let f = match fields(line) {
1387 Some(f) => f,
1388 None => continue,
1389 };
1390 if f.len() < 2 {
1391 continue;
1392 }
1393 let (lo, hi) = res!(range(f[0]));
1394 let v = res!(by_long(f[1]));
1395 for c in lo..=hi {
1396 sc[c as usize] = v;
1397 }
1398 }
1399
1400 let mut scx = vec![0u16; NUM_CP];
1401 let mut scx_sets: Vec<Vec<u8>> = Vec::new();
1402 for line in res!(get("ucd/ScriptExtensions.txt")).lines() {
1403 let f = match fields(line) {
1404 Some(f) => f,
1405 None => continue,
1406 };
1407 if f.len() < 2 {
1408 continue;
1409 }
1410 let (lo, hi) = res!(range(f[0]));
1411 let mut set = Vec::new();
1412 for s in f[1].split_whitespace() {
1413 set.push(res!(by_short(s)));
1414 }
1415 set.sort();
1416 set.dedup();
1417 let id = match scx_sets.iter().position(|x| *x == set) {
1418 Some(i) => i,
1419 None => {
1420 scx_sets.push(set);
1421 scx_sets.len() - 1
1422 },
1423 };
1424 for c in lo..=hi {
1425 scx[c as usize] = (id + 1) as u16;
1426 }
1427 }
1428
1429 // Binary properties. The contributory `Other_*` properties are left out: UAX #44 says
1430 // they exist to derive others and are not for use in their own right.
1431 let mut raw: BTreeMap<String, Vec<(u32, u32)>> = BTreeMap::new();
1432 for path in [
1433 "ucd/PropList.txt",
1434 "ucd/DerivedCoreProperties.txt",
1435 "ucd/emoji/emoji-data.txt",
1436 ] {
1437 for line in res!(get(path)).lines() {
1438 let f = match fields(line) {
1439 Some(f) => f,
1440 None => continue,
1441 };
1442 if f.len() != 2 || f[1].starts_with("Other_") {
1443 continue;
1444 }
1445 let r = res!(range(f[0]));
1446 raw.entry(f[1].to_string()).or_default().push(r);
1447 }
1448 }
1449 let bins: BTreeMap<String, Vec<(u32, u32)>> =
1450 raw.into_iter().map(|(k, v)| (k, merge(v))).collect();
1451
1452 let mut bin_alias = BTreeMap::new();
1453 for (i, name) in bins.keys().enumerate() {
1454 res!(alias(&mut bin_alias, name, i as u8, "binary property"));
1455 }
1456 let mut in_binary = false;
1457 for line in res!(get("ucd/PropertyAliases.txt")).lines() {
1458 if line.starts_with('#') {
1459 if line.contains("Binary Properties") {
1460 in_binary = true;
1461 } else if line.contains("Properties") {
1462 in_binary = false;
1463 }
1464 continue;
1465 }
1466 if !in_binary {
1467 continue;
1468 }
1469 let f = match fields(line) {
1470 Some(f) => f,
1471 None => continue,
1472 };
1473 if f.len() < 2 {
1474 continue;
1475 }
1476 if let Some(i) = bins.keys().position(|k| k == f[1]) {
1477 for name in &f {
1478 res!(alias(&mut bin_alias, name, i as u8, "binary property"));
1479 }
1480 }
1481 }
1482
1483 // The segmentation break properties, by the aliases a `\p{gcb=..}` may use. A value
1484 // no character takes (the retired emoji Word_Break values) is left out.
1485 let mut gcb_alias = BTreeMap::new();
1486 let mut wb_alias = BTreeMap::new();
1487 let mut sbs = Vec::new();
1488 let mut sb_alias = BTreeMap::new();
1489 for line in pva.lines() {
1490 let f = match fields(line) {
1491 Some(f) => f,
1492 None => continue,
1493 };
1494 if f.len() < 3 {
1495 continue;
1496 }
1497 let (list, map) = match f[0] {
1498 "GCB" => (GCB_CLASSES, &mut gcb_alias),
1499 "WB" => (WB_CLASSES, &mut wb_alias),
1500 "SB" => {
1501 let i = sbs.len() as u8;
1502 sbs.push((f[1].to_string(), f[2].to_string()));
1503 for name in &f[1..] {
1504 res!(alias(&mut sb_alias, name, i, "Sentence_Break"));
1505 }
1506 continue;
1507 },
1508 _ => continue,
1509 };
1510 if let Some(i) = list.iter().position(|(u, _, _)| *u == f[2] || *u == f[1]) {
1511 for name in &f[1..] {
1512 res!(alias(map, name, i as u8, f[0]));
1513 }
1514 }
1515 }
1516 let other = match sbs.iter().position(|(_, l)| l == "Other") {
1517 Some(i) => i as u8,
1518 None => return Err(err!("No Sentence_Break value Other was listed."; Missing, Input)),
1519 };
1520 let mut sb = vec![other; NUM_CP];
1521 for line in res!(get("ucd/auxiliary/SentenceBreakProperty.txt")).lines() {
1522 let f = match fields(line) {
1523 Some(f) => f,
1524 None => continue,
1525 };
1526 if f.len() < 2 {
1527 continue;
1528 }
1529 let (lo, hi) = res!(range(f[0]));
1530 let v = match sbs.iter().position(|(_, l)| l == f[1]) {
1531 Some(i) => i as u8,
1532 None => return Err(err!("The Sentence_Break value {:?} is not known.", f[1];
1533 Invalid, Input, Mismatch)),
1534 };
1535 for c in lo..=hi {
1536 sb[c as usize] = v;
1537 }
1538 }
1539
1540 Ok(Self {
1541 gc, gc_long, gc_alias, scripts, sc_alias, sc, scx, scx_sets, bins, bin_alias,
1542 gcb_alias, wb_alias, sbs, sb_alias, sb,
1543 })
1544 }
1545
1546 /// The Rust variant name of a script: its long name without underscores.
1547 fn variant(&self, i: usize) -> String {
1548 match self.scripts.get(i) {
1549 Some((_, long)) => long.replace('_', ""),
1550 None => String::from("Unknown"),
1551 }
1552 }
1553}
1554
1555/// Emits a `u16` array, twelve values to the line.
1556fn emit_u16(out: &mut String, name: &str, doc: &str, vals: &[u16]) {
1557 let _ = write!(out, "/// {}\npub static {}: [u16; {}] = [\n", doc, name, vals.len());
1558 for chunk in vals.chunks(12) {
1559 let mut line = String::from("\t");
1560 for v in chunk {
1561 let _ = write!(line, "{}, ", v);
1562 }
1563 let _ = write!(out, "{}\n", line.trim_end());
1564 }
1565 out.push_str("];\n\n");
1566}
1567
1568/// Emits an array of variants of an enum whose variant names are given, eight to the line.
1569fn emit_variants(
1570 out: &mut String,
1571 name: &str,
1572 doc: &str,
1573 typ: &str,
1574 short: &str,
1575 names: &[String],
1576 vals: &[u8],
1577)
1578 -> Outcome<()>
1579{
1580 let _ = write!(out, "/// {}\npub static {}: [{}; {}] = [\n", doc, name, typ, vals.len());
1581 for chunk in vals.chunks(8) {
1582 let mut line = String::from("\t");
1583 for v in chunk {
1584 match names.get(*v as usize) {
1585 Some(n) => { let _ = write!(line, "{}::{}, ", short, n); },
1586 None => return Err(err!(
1587 "The index {} is out of range for {}.", v, typ; Bug, Index)),
1588 }
1589 }
1590 let _ = write!(out, "{}\n", line.trim_end());
1591 }
1592 out.push_str("];\n\n");
1593 Ok(())
1594}
1595
1596/// Emits the General_Category and Script enums, into the property enum file.
1597fn emit_cat_enums(out: &mut String, cats: &Cats) {
1598
1599 out.push_str(
1600"/// The General_Category property of UAX #44, by its two letter abbreviation.
1601#[repr(u8)]
1602#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
1603pub enum GeneralCategory {
1604");
1605 for code in GC_CODES {
1606 let _ = write!(out, "\t{},\n", code);
1607 }
1608 out.push_str("}\n\n");
1609 out.push_str(
1610"impl Partitioned for GeneralCategory {
1611
1612 const DEFAULT: Self = Self::Cn;
1613
1614 fn table() -> (&'static [u32], &'static [Self]) {
1615 (&super::cat::GC_STARTS, &super::cat::GC_VALS)
1616 }
1617}
1618
1619impl GeneralCategory {
1620
1621 /// The long UCD name, `Uppercase_Letter` for `Lu`.
1622 pub fn name(self) -> &'static str {
1623 match self {
1624");
1625 for (i, code) in GC_CODES.iter().enumerate() {
1626 let long = match cats.gc_long.get(i) {
1627 Some(l) => l.as_str(),
1628 None => "",
1629 };
1630 let _ = write!(out, "\t\t\tSelf::{}\t=> \"{}\",\n", code, long);
1631 }
1632 out.push_str("\t\t}\n\t}\n\n\t/// The two letter abbreviation.\n\tpub fn abbr(self) -> &'static str {\n\t\tmatch self {\n");
1633 for code in GC_CODES {
1634 let _ = write!(out, "\t\t\tSelf::{}\t=> \"{}\",\n", code, code);
1635 }
1636 out.push_str("\t\t}\n\t}\n}\n\n");
1637
1638 out.push_str(
1639"/// The Script property of UAX #24. A character's Script_Extensions, the wider set of scripts it
1640/// is used with, are in `unicode::property`.
1641#[repr(u8)]
1642#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
1643pub enum Script {
1644");
1645 for i in 0..cats.scripts.len() {
1646 let _ = write!(out, "\t{},\n", cats.variant(i));
1647 }
1648 out.push_str("}\n\n");
1649 out.push_str(
1650"impl Partitioned for Script {
1651
1652 const DEFAULT: Self = Self::Unknown;
1653
1654 fn table() -> (&'static [u32], &'static [Self]) {
1655 (&super::cat::SC_STARTS, &super::cat::SC_VALS)
1656 }
1657}
1658
1659impl Script {
1660
1661 /// The long UCD name, `Old_Italic` for instance.
1662 pub fn name(self) -> &'static str {
1663 match self {
1664");
1665 for (i, (_, long)) in cats.scripts.iter().enumerate() {
1666 let _ = write!(out, "\t\t\tSelf::{}\t=> \"{}\",\n", cats.variant(i), long);
1667 }
1668 out.push_str("\t\t}\n\t}\n\n\t/// The four letter ISO 15924 code, `Ital` for Old_Italic.\n\tpub fn code(self) -> &'static str {\n\t\tmatch self {\n");
1669 for (i, (short, _)) in cats.scripts.iter().enumerate() {
1670 let _ = write!(out, "\t\t\tSelf::{}\t=> \"{}\",\n", cats.variant(i), short);
1671 }
1672 out.push_str("\t\t}\n\t}\n}\n\n");
1673
1674 out.push_str(
1675"/// The Sentence_Break property of UAX #29.
1676#[repr(u8)]
1677#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
1678pub enum SentenceClass {
1679");
1680 for (_, long) in &cats.sbs {
1681 let _ = write!(out, "\t{},\n", long.replace('_', ""));
1682 }
1683 out.push_str(
1684"}
1685
1686impl Partitioned for SentenceClass {
1687
1688 const DEFAULT: Self = Self::Other;
1689
1690 fn table() -> (&'static [u32], &'static [Self]) {
1691 (&super::cat::SB_STARTS, &super::cat::SB_VALS)
1692 }
1693}
1694");
1695}
1696
1697/// Emits the character class tables.
1698fn emit_cat(cats: &Cats) -> Outcome<String> {
1699
1700 let mut out = header(
1701 "Tables for the character classes of UAX #44 and UAX #24: General_Category, Script, \
1702 Script_Extensions and the binary properties.");
1703 // Short aliases keep the long value arrays readable and the file a third smaller.
1704 out.push_str("\nuse crate::unicode::prop::{\n\tGeneralCategory,\n\tGeneralCategory as G,\n\tGraphemeClass,\n\tScript,\n\tScript as S,\n\tSentenceClass,\n\tWordClass,\n};\n\n");
1705
1706 let gc_names: Vec<String> = GC_CODES.iter().map(|c| c.to_string()).collect();
1707 let (starts, vals) = partition(&cats.gc);
1708 emit_u32(&mut out, "GC_STARTS", "Start code points of the General_Category runs.", &starts);
1709 res!(emit_variants(&mut out, "GC_VALS", "The General_Category of each run.",
1710 "GeneralCategory", "G", &gc_names, &vals));
1711
1712 let _ = write!(out,
1713 "/// Every General_Category alias, loosely normalised and sorted, with the categories it \
1714 names as a mask over the `GeneralCategory` variants.\npub static GC_NAMES: [(&str, u32); {}] = [\n",
1715 cats.gc_alias.len());
1716 for (k, v) in &cats.gc_alias {
1717 let _ = write!(out, "\t(\"{}\", 0x{:08X}),\n", k, v);
1718 }
1719 out.push_str("];\n\n");
1720
1721 let sc_names: Vec<String> = (0..cats.scripts.len()).map(|i| cats.variant(i)).collect();
1722 let (starts, vals) = partition(&cats.sc);
1723 emit_u32(&mut out, "SC_STARTS", "Start code points of the Script runs.", &starts);
1724 res!(emit_variants(&mut out, "SC_VALS", "The Script of each run.", "Script", "S", &sc_names, &vals));
1725
1726 let _ = write!(out,
1727 "/// Every Script alias, loosely normalised and sorted.\npub static SCRIPT_NAMES: [(&str, Script); {}] = [\n",
1728 cats.sc_alias.len());
1729 for (k, v) in &cats.sc_alias {
1730 let _ = write!(out, "\t(\"{}\", S::{}),\n", k, cats.variant(*v as usize));
1731 }
1732 out.push_str("];\n\n");
1733
1734 // Script_Extensions as a partition of set numbers, zero where a character's extensions are
1735 // just its script.
1736 let mut starts = Vec::new();
1737 let mut runs = Vec::new();
1738 let mut prev = None;
1739 for (c, v) in cats.scx.iter().enumerate() {
1740 if prev != Some(*v) {
1741 starts.push(c as u32);
1742 runs.push(*v);
1743 prev = Some(*v);
1744 }
1745 }
1746 emit_u32(&mut out, "SCX_STARTS", "Start code points of the Script_Extensions runs.", &starts);
1747 emit_u16(&mut out, "SCX_VALS",
1748 "The Script_Extensions set of each run: zero for the character's own script alone, \
1749 otherwise one more than the set's index in `SCX_OFFS`.", &runs);
1750 let mut offs: Vec<u16> = vec![0];
1751 let mut pool: Vec<u8> = Vec::new();
1752 for set in &cats.scx_sets {
1753 pool.extend_from_slice(set);
1754 offs.push(pool.len() as u16);
1755 }
1756 emit_u16(&mut out, "SCX_OFFS",
1757 "Where each Script_Extensions set begins in `SCX_POOL`, with the end as the last entry.",
1758 &offs);
1759 res!(emit_variants(&mut out, "SCX_POOL", "The scripts of every Script_Extensions set, in turn.",
1760 "Script", "S", &sc_names, &pool));
1761
1762 // Binary properties as inclusive range pairs.
1763 let mut flat: Vec<u32> = Vec::new();
1764 let mut boffs: Vec<u32> = vec![0];
1765 let mut longs = String::new();
1766 for (name, ranges) in &cats.bins {
1767 for (lo, hi) in ranges {
1768 flat.push(*lo);
1769 flat.push(*hi);
1770 }
1771 boffs.push(flat.len() as u32);
1772 let _ = write!(longs, "\t\"{}\",\n", name);
1773 }
1774 emit_u32(&mut out, "BIN_RANGES",
1775 "The binary properties as inclusive code point ranges, low then high, one property after \
1776 another.", &flat);
1777 emit_u32(&mut out, "BIN_OFFS",
1778 "Where each binary property's ranges begin in `BIN_RANGES`, with the end as the last entry.",
1779 &boffs);
1780 let _ = write!(out, "/// The long name of each binary property.\npub static BIN_LONG: [&str; {}] = [\n{}];\n\n",
1781 cats.bins.len(), longs);
1782 for (i, name) in cats.bins.keys().enumerate() {
1783 let _ = write!(out, "pub const BIN_{}: u8 = {};\n", name.to_ascii_uppercase(), i);
1784 }
1785 out.push('\n');
1786 let _ = write!(out,
1787 "/// Every binary property alias, loosely normalised and sorted, with the property's index.\npub static BIN_NAMES: [(&str, u8); {}] = [\n",
1788 cats.bin_alias.len());
1789 for (k, v) in &cats.bin_alias {
1790 let _ = write!(out, "\t(\"{}\", {}),\n", k, v);
1791 }
1792 out.push_str("];\n\n");
1793
1794 // The segmentation break properties' names; their values are in `seg`.
1795 for (name, typ, list, map) in [
1796 ("GCB_NAMES", "GraphemeClass", GCB_CLASSES, &cats.gcb_alias),
1797 ("WB_NAMES", "WordClass", WB_CLASSES, &cats.wb_alias),
1798 ] {
1799 let _ = write!(out,
1800 "/// Every {} alias, loosely normalised and sorted.\npub static {}: [(&str, {}); {}] = [\n",
1801 typ, name, typ, map.len());
1802 for (k, v) in map {
1803 let variant = match list.get(*v as usize) {
1804 Some((_, rust, _)) => *rust,
1805 None => return Err(err!("Index {} is out of range for {}.", v, typ;
1806 Bug, Index)),
1807 };
1808 let _ = write!(out, "\t(\"{}\", {}::{}),\n", k, typ, variant);
1809 }
1810 out.push_str("];\n\n");
1811 }
1812
1813 let sb_names: Vec<String> = cats.sbs.iter().map(|(_, l)| l.replace('_', "")).collect();
1814 let (starts, vals) = partition(&cats.sb);
1815 emit_u32(&mut out, "SB_STARTS", "Start code points of the Sentence_Break runs.", &starts);
1816 res!(emit_variants(&mut out, "SB_VALS", "The Sentence_Break of each run.", "SentenceClass",
1817 "SentenceClass", &sb_names, &vals));
1818 let _ = write!(out,
1819 "/// Every Sentence_Break alias, loosely normalised and sorted.\npub static SB_NAMES: [(&str, SentenceClass); {}] = [\n",
1820 cats.sb_alias.len());
1821 for (k, v) in &cats.sb_alias {
1822 let n = sb_names.get(*v as usize).cloned().unwrap_or_default();
1823 let _ = write!(out, "\t(\"{}\", SentenceClass::{}),\n", k, n);
1824 }
1825 out.push_str("];\n");
1826
1827 Ok(out)
1828}