oxedyne/fe2o3/fe2o3_text/src/unicode/property.rs
8.9 KiB, 1 run
created by r1870400018:60328, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Character classes by Unicode property: General_Category, Script, Script_Extensions and the |
| 2 | //! binary properties of UAX #44, as named in a regular expression's `\p{...}`. |
| 3 | //! |
| 4 | //! Name resolution follows UTS #18 and the Rust `regex` crate, which Typst's `regex(...)` uses, so |
| 5 | //! a pattern means the same here as there. A bare name is tried first as a binary property, then |
| 6 | //! as a General_Category, then as a Script -- the Script proper, not Script_Extensions, which is |
| 7 | //! asked for as `scx=`. Names match loosely (UAX44-LM3): case, spaces, underscores, hyphens and a |
| 8 | //! leading `is` are ignored, so `\p{Greek}`, `\p{isGreek}` and `\p{sc=grek}` agree. |
| 9 | //! |
| 10 | //! ``` |
| 11 | //! use oxedyne_fe2o3_text::unicode::property::CharClass; |
| 12 | //! |
| 13 | //! let greek = CharClass::parse("Greek").expect("a known script"); |
| 14 | //! assert!(greek.contains('λ')); |
| 15 | //! assert!(!greek.contains('l')); |
| 16 | //! ``` |
| 17 | |
| 18 | use crate::unicode::{ |
| 19 | lookup::Partitioned, |
| 20 | prop::{ |
| 21 | GeneralCategory, |
| 22 | GraphemeClass, |
| 23 | Script, |
| 24 | SentenceClass, |
| 25 | WordClass, |
| 26 | }, |
| 27 | tables::cat::{ |
| 28 | BIN_ALPHABETIC, |
| 29 | BIN_JOIN_CONTROL, |
| 30 | BIN_LONG, |
| 31 | BIN_NAMES, |
| 32 | BIN_OFFS, |
| 33 | BIN_RANGES, |
| 34 | BIN_WHITE_SPACE, |
| 35 | GCB_NAMES, |
| 36 | GC_NAMES, |
| 37 | SB_NAMES, |
| 38 | SCRIPT_NAMES, |
| 39 | SCX_OFFS, |
| 40 | SCX_POOL, |
| 41 | SCX_STARTS, |
| 42 | SCX_VALS, |
| 43 | WB_NAMES, |
| 44 | }, |
| 45 | lookup, |
| 46 | }; |
| 47 | |
| 48 | use oxedyne_fe2o3_core::prelude::*; |
| 49 | |
| 50 | |
| 51 | /// The Script_Extensions of one character: the scripts it is used with. |
| 52 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 53 | pub enum Extensions { |
| 54 | One(Script), // no explicit extensions: the character's own script |
| 55 | Many(&'static [Script]), |
| 56 | } |
| 57 | |
| 58 | impl Extensions { |
| 59 | |
| 60 | pub fn of(c: char) -> Self { |
| 61 | let set = lookup::get(&SCX_VALS, lookup::run(&SCX_STARTS, c), 0) as usize; |
| 62 | if set == 0 { |
| 63 | return Self::One(Script::of(c)); |
| 64 | } |
| 65 | let a = lookup::get(&SCX_OFFS, set - 1, 0) as usize; |
| 66 | let b = lookup::get(&SCX_OFFS, set, 0) as usize; |
| 67 | match SCX_POOL.get(a..b) { |
| 68 | Some(s) => Self::Many(s), |
| 69 | None => Self::One(Script::of(c)), |
| 70 | } |
| 71 | } |
| 72 | |
| 73 | pub fn as_slice(&self) -> &[Script] { |
| 74 | match self { |
| 75 | Self::One(s) => std::slice::from_ref(s), |
| 76 | Self::Many(v) => v, |
| 77 | } |
| 78 | } |
| 79 | |
| 80 | pub fn contains(&self, s: Script) -> bool { |
| 81 | self.as_slice().contains(&s) |
| 82 | } |
| 83 | } |
| 84 | |
| 85 | /// A binary property of UAX #44, such as Alphabetic or White_Space. |
| 86 | #[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)] |
| 87 | pub struct Binary(u8); |
| 88 | |
| 89 | impl Binary { |
| 90 | |
| 91 | pub const ALPHABETIC: Self = Self(BIN_ALPHABETIC); |
| 92 | pub const JOIN_CONTROL: Self = Self(BIN_JOIN_CONTROL); |
| 93 | pub const WHITE_SPACE: Self = Self(BIN_WHITE_SPACE); |
| 94 | |
| 95 | /// Every binary property the tables carry. |
| 96 | pub fn all() -> impl Iterator<Item = Self> { |
| 97 | (0..BIN_LONG.len()).map(|i| Self(i as u8)) |
| 98 | } |
| 99 | |
| 100 | /// Finds a binary property by any of its aliases, loosely matched. |
| 101 | pub fn find(name: &str) -> Option<Self> { |
| 102 | match find_alias(&BIN_NAMES, name) { |
| 103 | Some(i) => Some(Self(i)), |
| 104 | None => None, |
| 105 | } |
| 106 | } |
| 107 | |
| 108 | /// The long UCD name, `White_Space` for instance. |
| 109 | pub fn name(self) -> &'static str { |
| 110 | lookup::get(&BIN_LONG, self.0 as usize, "") |
| 111 | } |
| 112 | |
| 113 | /// Does `c` have the property? |
| 114 | pub fn contains(self, c: char) -> bool { |
| 115 | let a = lookup::get(&BIN_OFFS, self.0 as usize, 0) as usize; |
| 116 | let b = lookup::get(&BIN_OFFS, self.0 as usize + 1, 0) as usize; |
| 117 | let pairs = match BIN_RANGES.get(a..b) { |
| 118 | Some(p) => p, |
| 119 | None => return false, |
| 120 | }; |
| 121 | // Pairs of (low, high): find the last pair whose low is at or below `c`. |
| 122 | let n = pairs.len() / 2; |
| 123 | let cp = c as u32; |
| 124 | let i = partition_pairs(pairs, n, cp); |
| 125 | if i == 0 { |
| 126 | return false; |
| 127 | } |
| 128 | match (pairs.get(2 * (i - 1)), pairs.get(2 * (i - 1) + 1)) { |
| 129 | (Some(lo), Some(hi)) => *lo <= cp && cp <= *hi, |
| 130 | _ => false, |
| 131 | } |
| 132 | } |
| 133 | } |
| 134 | |
| 135 | /// The number of pairs whose low end is at or below `cp`. |
| 136 | fn partition_pairs(pairs: &[u32], n: usize, cp: u32) -> usize { |
| 137 | let (mut lo, mut hi) = (0usize, n); |
| 138 | while lo < hi { |
| 139 | let mid = (lo + hi) / 2; |
| 140 | if lookup::get(pairs, 2 * mid, u32::MAX) <= cp { |
| 141 | lo = mid + 1; |
| 142 | } else { |
| 143 | hi = mid; |
| 144 | } |
| 145 | } |
| 146 | lo |
| 147 | } |
| 148 | |
| 149 | /// A set of characters named by Unicode property, what `\p{...}` stands for. |
| 150 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 151 | pub enum CharClass { |
| 152 | Any, |
| 153 | Ascii, |
| 154 | Assigned, |
| 155 | Category(u32), // mask over the `GeneralCategory` variants |
| 156 | Script(Script), |
| 157 | Extension(Script), // Script_Extensions contains this script |
| 158 | Binary(Binary), |
| 159 | Grapheme(GraphemeClass), // Grapheme_Cluster_Break |
| 160 | WordBreak(WordClass), |
| 161 | Sentence(SentenceClass), // Sentence_Break |
| 162 | } |
| 163 | |
| 164 | impl CharClass { |
| 165 | |
| 166 | /// Resolves the text between the braces of `\p{...}`: a bare name such as `L`, `Greek` or |
| 167 | /// `Alphabetic`, or `property=value` (also `property:value`) where the property is |
| 168 | /// General_Category (`gc`), Script (`sc`), Script_Extensions (`scx`), |
| 169 | /// Grapheme_Cluster_Break (`gcb`), Word_Break (`wb`) or Sentence_Break (`sb`). Negation, |
| 170 | /// `!=`, is the caller's business. |
| 171 | pub fn parse(query: &str) -> Outcome<Self> { |
| 172 | let split = match query.find(|c| c == '=' || c == ':') { |
| 173 | Some(i) => Some((&query[..i], &query[i + 1..])), |
| 174 | None => None, |
| 175 | }; |
| 176 | let found = match split { |
| 177 | Some((prop, val)) => { |
| 178 | let p = loose(prop); |
| 179 | match strip_is(&p).as_str() { |
| 180 | "gc" | "generalcategory" => Self::category(val), |
| 181 | "sc" | "script" => Self::script(val).map(Self::Script), |
| 182 | "scx" | "scriptextensions" => Self::script(val).map(Self::Extension), |
| 183 | "gcb" | "graphemeclusterbreak" => |
| 184 | find_alias(&GCB_NAMES, val).map(Self::Grapheme), |
| 185 | "wb" | "wordbreak" => find_alias(&WB_NAMES, val).map(Self::WordBreak), |
| 186 | "sb" | "sentencebreak" => find_alias(&SB_NAMES, val).map(Self::Sentence), |
| 187 | _ => return Err(err!( |
| 188 | "Unicode property '{}' in '\\p{{{}}}' is not supported: the properties \ |
| 189 | that take a value are gc, sc, scx, gcb, wb and sb.", prop, query; |
| 190 | Unimplemented, Input)), |
| 191 | } |
| 192 | }, |
| 193 | None => { |
| 194 | let n = loose(query); |
| 195 | let binary = if n != "cf" && n != "sc" && n != "lc" { |
| 196 | Binary::find(query).map(Self::Binary) |
| 197 | } else { |
| 198 | None |
| 199 | }; |
| 200 | binary |
| 201 | .or_else(|| Self::category(query)) |
| 202 | .or_else(|| Self::script(query).map(Self::Script)) |
| 203 | }, |
| 204 | }; |
| 205 | match found { |
| 206 | Some(c) => Ok(c), |
| 207 | None => Err(err!( |
| 208 | "Unicode property '\\p{{{}}}' is not known: it is neither a binary property, a \ |
| 209 | General_Category nor a Script.", query; Invalid, Input, Missing)), |
| 210 | } |
| 211 | } |
| 212 | |
| 213 | fn category(val: &str) -> Option<Self> { |
| 214 | match loose(val).as_str() { |
| 215 | "any" => return Some(Self::Any), |
| 216 | "ascii" => return Some(Self::Ascii), |
| 217 | "assigned" => return Some(Self::Assigned), |
| 218 | _ => {}, |
| 219 | } |
| 220 | find_alias(&GC_NAMES, val).map(Self::Category) |
| 221 | } |
| 222 | |
| 223 | fn script(val: &str) -> Option<Script> { |
| 224 | find_alias(&SCRIPT_NAMES, val) |
| 225 | } |
| 226 | |
| 227 | pub fn contains(&self, c: char) -> bool { |
| 228 | match self { |
| 229 | Self::Any => true, |
| 230 | Self::Ascii => c.is_ascii(), |
| 231 | Self::Assigned => GeneralCategory::of(c) != GeneralCategory::Cn, |
| 232 | Self::Category(m) => m & (1u32 << (GeneralCategory::of(c) as u32)) != 0, |
| 233 | Self::Script(s) => Script::of(c) == *s, |
| 234 | Self::Extension(s) => Extensions::of(c).contains(*s), |
| 235 | Self::Binary(b) => b.contains(c), |
| 236 | Self::Grapheme(g) => GraphemeClass::of(c) == *g, |
| 237 | Self::WordBreak(w) => WordClass::of(c) == *w, |
| 238 | Self::Sentence(v) => SentenceClass::of(c) == *v, |
| 239 | } |
| 240 | } |
| 241 | } |
| 242 | |
| 243 | /// Is `c` a word character in the Unicode sense of UTS #18 and `\w`: Alphabetic, a mark, a |
| 244 | /// decimal digit, connector punctuation or a join control? |
| 245 | pub fn is_word(c: char) -> bool { |
| 246 | if c.is_ascii() { |
| 247 | return c.is_ascii_alphanumeric() || c == '_'; |
| 248 | } |
| 249 | match GeneralCategory::of(c) { |
| 250 | GeneralCategory::Mn |
| 251 | | GeneralCategory::Mc |
| 252 | | GeneralCategory::Me |
| 253 | | GeneralCategory::Nd |
| 254 | | GeneralCategory::Pc => true, |
| 255 | _ => Binary::ALPHABETIC.contains(c) || Binary::JOIN_CONTROL.contains(c), |
| 256 | } |
| 257 | } |
| 258 | |
| 259 | /// Is `c` white space, the White_Space property that `\s` stands for? |
| 260 | pub fn is_space(c: char) -> bool { |
| 261 | if c.is_ascii() { |
| 262 | return matches!(c, ' ' | '\t' | '\n' | '\x0B' | '\x0C' | '\r'); |
| 263 | } |
| 264 | Binary::WHITE_SPACE.contains(c) |
| 265 | } |
| 266 | |
| 267 | /// Is `c` a decimal digit, General_Category Nd? |
| 268 | pub fn is_digit(c: char) -> bool { |
| 269 | if c.is_ascii() { |
| 270 | return c.is_ascii_digit(); |
| 271 | } |
| 272 | GeneralCategory::of(c) == GeneralCategory::Nd |
| 273 | } |
| 274 | |
| 275 | /// UAX44-LM3 without the `is` prefix rule: ASCII lower case, no spaces, underscores or hyphens, |
| 276 | /// and nothing that is not ASCII. |
| 277 | fn loose(name: &str) -> String { |
| 278 | name.chars() |
| 279 | .filter(|c| c.is_ascii() && *c != ' ' && *c != '_' && *c != '-') |
| 280 | .map(|c| c.to_ascii_lowercase()) |
| 281 | .collect() |
| 282 | } |
| 283 | |
| 284 | /// The name with a leading `is` removed, which UAX44-LM3 ignores, except that `isc` is the |
| 285 | /// ISO_Comment alias rather than `is` and `c`. |
| 286 | fn strip_is(n: &str) -> String { |
| 287 | match n.strip_prefix("is") { |
| 288 | Some(rest) if n != "isc" && !rest.is_empty() => rest.to_string(), |
| 289 | _ => n.to_string(), |
| 290 | } |
| 291 | } |
| 292 | |
| 293 | /// Looks a name up in a sorted alias table, as written and then without an `is` prefix. |
| 294 | fn find_alias<T: Copy>(table: &[(&str, T)], name: &str) -> Option<T> { |
| 295 | let n = loose(name); |
| 296 | if let Ok(i) = table.binary_search_by(|(k, _)| (*k).cmp(n.as_str())) { |
| 297 | return table.get(i).map(|(_, v)| *v); |
| 298 | } |
| 299 | let s = strip_is(&n); |
| 300 | if s != n { |
| 301 | if let Ok(i) = table.binary_search_by(|(k, _)| (*k).cmp(s.as_str())) { |
| 302 | return table.get(i).map(|(_, v)| *v); |
| 303 | } |
| 304 | } |
| 305 | None |
| 306 | } |