oxedyne/fe2o3/fe2o3_text/src/split.rs
8.2 KiB, 3 runs
created by r1870400018:1109, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | use crate::{ |
| 2 | phrase::{ |
| 3 | Phrase, |
| 4 | PhraseMeta, |
| 5 | }, |
| 6 | string::Quote, |
| 7 | }; |
| 8 | |
| 9 | use oxedyne_fe2o3_core::prelude::*; |
| 10 | |
| 11 | use std::{ |
| 12 | str, |
| 13 | }; |
| 14 | |
| 15 | use std::collections::HashSet; |
| 16 | |
| 17 | |
| 18 | #[derive(Clone, Debug)] |
| 19 | pub struct StringSplitter { |
| 20 | separators: HashSet<char>, |
| 21 | quote_protection: bool, |
| 22 | keep_protected_quotes: bool, |
| 23 | dash_as_hyphen: bool, |
| 24 | classify: bool, |
| 25 | } |
| 26 | |
| 27 | /// The default is to, at a minimum, split at spaces. |
| 28 | impl Default for StringSplitter { |
| 29 | |
| 30 | fn default() -> Self { |
| 31 | let mut seps = HashSet::new(); |
| 32 | seps.insert(' '); |
| 33 | Self { |
| 34 | separators: seps, |
| 35 | quote_protection: true, |
| 36 | keep_protected_quotes: true, |
| 37 | dash_as_hyphen: false, |
| 38 | classify: false, |
| 39 | } |
| 40 | } |
| 41 | |
| 42 | } |
| 43 | |
| 44 | impl StringSplitter { |
| 45 | |
| 46 | /// This constructor does not assume that space splitting is required. |
| 47 | pub fn new() -> Self { |
| 48 | Self { |
| 49 | separators: HashSet::new(), |
| 50 | quote_protection: true, |
| 51 | keep_protected_quotes: true, |
| 52 | dash_as_hyphen: false, |
| 53 | classify: false, |
| 54 | } |
| 55 | } |
| 56 | |
| 57 | pub fn protect_quotes(mut self) -> Self { |
| 58 | self.quote_protection = true; |
| 59 | self |
| 60 | } |
| 61 | |
| 62 | pub fn keep_protected_quotes(mut self) -> Self { |
| 63 | self.keep_protected_quotes = true; |
| 64 | self |
| 65 | } |
| 66 | |
| 67 | pub fn dashes_are_hypens(mut self) -> Self { |
| 68 | self.dash_as_hyphen = true; |
| 69 | self |
| 70 | } |
| 71 | |
| 72 | pub fn classify(mut self) -> Self { |
| 73 | self.classify = true; |
| 74 | self |
| 75 | } |
| 76 | |
| 77 | pub fn add_separators(mut self, seps: Box<[char]>) -> Self { |
| 78 | for i in 0..seps.len() { |
| 79 | self.separators.insert(seps[i]); |
| 80 | } |
| 81 | self |
| 82 | } |
| 83 | |
| 84 | pub fn clear_separators(mut self) -> Self { |
| 85 | self.separators = HashSet::new(); |
| 86 | self |
| 87 | } |
| 88 | |
| 89 | /// Split unicode string using `StringSplitter` settings. There are two cases. |
| 90 | /// Quote protection ON: |
| 91 | /// While a quote is active, words can only be terminated by the closing quote. |
| 92 | /// e.g. |
| 93 | /// "this is a test" -> word = "this is a test" |
| 94 | /// ^ ^ |
| 95 | /// +- quote active +- quote inactive |
| 96 | /// While quotes are inactive, the rules for no quote protection apply... |
| 97 | /// |
| 98 | /// Quote protection OFF: |
| 99 | /// Quotes are treated as normal characters. That is, words are terminated by the following |
| 100 | /// characters: |
| 101 | /// ' ' - space, discarded |
| 102 | /// '-' - hyphen, with the word tagged as the hyphenation of the previous word |
| 103 | /// '.', ';', ',' |
| 104 | /// |
| 105 | // quote_protection ON |
| 106 | // th" is " is -> 'th', ' is ', 'is' |
| 107 | // quote_protection OFF |
| 108 | // th" is " is -> 'th"', 'is', '"', 'is' |
| 109 | pub fn split(&self, input: &str) -> Vec<Phrase> { |
| 110 | let mut parts = Vec::new(); |
| 111 | let mut is_part = false; |
| 112 | let mut is_hyphened = false; |
| 113 | let mut part = if self.classify { |
| 114 | Phrase::Classified(String::new(), PhraseMeta::new()) |
| 115 | } else { |
| 116 | Phrase::Plain(String::new()) |
| 117 | }; |
| 118 | let mut i_start = 0; |
| 119 | let mut i = 0; |
| 120 | let mut quote: Quote = Quote::None; |
| 121 | for c in input.chars() { |
| 122 | i += 1; |
| 123 | if self.quote_protection { |
| 124 | match c { |
| 125 | '"' => { |
| 126 | if quote != Quote::Single { |
| 127 | if quote == Quote::Double { |
| 128 | quote = Quote::None; |
| 129 | } else { |
| 130 | quote = Quote::Double; |
| 131 | } |
| 132 | if !self.keep_protected_quotes { |
| 133 | continue; |
| 134 | } |
| 135 | } |
| 136 | }, |
| 137 | '\'' => { |
| 138 | if quote != Quote::Double { |
| 139 | if quote == Quote::Single { |
| 140 | quote = Quote::None; |
| 141 | } else { |
| 142 | quote = Quote::Single; |
| 143 | } |
| 144 | if !self.keep_protected_quotes { |
| 145 | continue; |
| 146 | } |
| 147 | } |
| 148 | }, |
| 149 | _ => {}, |
| 150 | } |
| 151 | } |
| 152 | if self.separators.contains(&c) { |
| 153 | if quote == Quote::None { |
| 154 | if is_part { // end of part, start new part |
| 155 | is_part = false; |
| 156 | if self.classify { |
| 157 | match &mut part { |
| 158 | Phrase::Classified(_, meta) => { |
| 159 | if meta.typ == None { |
| 160 | meta.set_type(c); |
| 161 | } |
| 162 | meta.len = i - i_start - 1; |
| 163 | }, |
| 164 | _ => {}, |
| 165 | } |
| 166 | } |
| 167 | if (self.dash_as_hyphen && c == '-') || |
| 168 | c == '.' |
| 169 | { |
| 170 | part.push(c); |
| 171 | part.inc_len(); |
| 172 | if c == '-' { |
| 173 | is_hyphened = true; |
| 174 | } |
| 175 | } |
| 176 | parts.push(part); |
| 177 | part = if self.classify { |
| 178 | Phrase::Classified(String::new(), PhraseMeta::new()) |
| 179 | } else { |
| 180 | Phrase::Plain(String::new()) |
| 181 | }; |
| 182 | i_start = i; |
| 183 | } else { // re-start of part |
| 184 | i_start += 1; |
| 185 | } |
| 186 | } else { |
| 187 | part.push(c); |
| 188 | } |
| 189 | } else { |
| 190 | if !is_part { |
| 191 | is_part = true; |
| 192 | } |
| 193 | if is_hyphened { |
| 194 | part.set_hyphen_left(); |
| 195 | is_hyphened = false; |
| 196 | } |
| 197 | part.push(c); |
| 198 | } |
| 199 | } |
| 200 | if is_part { |
| 201 | part.set_len(i - i_start - 1); |
| 202 | parts.push(part); |
| 203 | } |
| 204 | parts |
| 205 | } |
| 206 | |
| 207 | } |
| 208 | |
| 209 | //#[derive(Debug, Clone, Copy)] |
| 210 | //pub enum StringFormat { |
| 211 | // Json, |
| 212 | //} |
| 213 | // |
| 214 | //#[derive(Debug, Clone, Copy)] |
| 215 | //pub struct StrFmtCfg<'a> { |
| 216 | // pub fmt: StringFormat, |
| 217 | // pub multiline: bool, |
| 218 | // pub prepend_first_line: bool, |
| 219 | // pub show_kind: bool, |
| 220 | // pub indent: &'a str, |
| 221 | // pub next: &'a str, |
| 222 | // pub end: &'a str, |
| 223 | // pub dat_open_bracket: &'a str, |
| 224 | // pub dat_close_bracket: &'a str, |
| 225 | // pub kind_separator: &'a str, |
| 226 | // pub list_open_bracket: &'a str, |
| 227 | // pub list_close_bracket: &'a str, |
| 228 | // pub map_open_bracket: &'a str, |
| 229 | // pub map_close_bracket: &'a str, |
| 230 | // pub map_separator: &'a str, |
| 231 | //} |
| 232 | // |
| 233 | //impl<'a> StrFmtCfg<'a> { |
| 234 | // pub fn flat_json() -> StrFmtCfg<'a> { |
| 235 | // StrFmtCfg { |
| 236 | // fmt: StringFormat::Json, |
| 237 | // multiline: false, |
| 238 | // prepend_first_line: false, |
| 239 | // show_kind: true, |
| 240 | // //indent: "\t", |
| 241 | // indent: " ", |
| 242 | // next: "", |
| 243 | // end: ", ", |
| 244 | // dat_open_bracket: "(", |
| 245 | // dat_close_bracket: ")", |
| 246 | // kind_separator: "|", |
| 247 | // list_open_bracket: "[", |
| 248 | // list_close_bracket: "]", |
| 249 | // map_open_bracket: "{", |
| 250 | // map_close_bracket: "}", |
| 251 | // map_separator: ": ", |
| 252 | // } |
| 253 | // } |
| 254 | // pub fn multiline_json() -> StrFmtCfg<'a> { |
| 255 | // let mut config = StrFmtCfg::flat_json(); |
| 256 | // config.multiline = true; |
| 257 | // config.prepend_first_line = true; |
| 258 | // config.next = "\n"; |
| 259 | // config.end = ",\n"; |
| 260 | // config |
| 261 | // } |
| 262 | // pub fn show_kind(&mut self, b: bool) { |
| 263 | // self.show_kind = b; |
| 264 | // } |
| 265 | // pub fn prepend_first_line(&mut self, b: bool) { |
| 266 | // self.prepend_first_line = b; |
| 267 | // } |
| 268 | //} |