oxedyne/fe2o3/fe2o3_jdat/src/string/dec.rs
100 KiB, 50 runs
created by r1870400018:485, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! 12024-12-23 |
| 2 | //! Added implicit tuple string decoding for round brackets, e.g. explicit (tup2|[1,2]), implicit |
| 3 | //! (1,2). This has no effect on tuple string encoding, which continues to use square brackets |
| 4 | //! either with a kindicle or without, e.g. explicit (tup2[1,2]) implicit [1,2]. The latter will |
| 5 | //! continue to be decoded as a list. |
| 6 | //! |
| 7 | use crate::{ |
| 8 | prelude::*, |
| 9 | bdat::limits::DecodeLimits, |
| 10 | note::NoteConfig, |
| 11 | int::{ |
| 12 | DatInt, |
| 13 | DatIntKind, |
| 14 | }, |
| 15 | kind::{ |
| 16 | KindCase, |
| 17 | KindClass, |
| 18 | }, |
| 19 | usr::{ |
| 20 | UsrKind, |
| 21 | UsrKinds, |
| 22 | UsrKindId, |
| 23 | UsrKindCode, |
| 24 | }, |
| 25 | }; |
| 26 | |
| 27 | use oxedyne_fe2o3_core::{ |
| 28 | prelude::*, |
| 29 | map::MapMut, |
| 30 | mem::Extract, |
| 31 | }; |
| 32 | use oxedyne_fe2o3_num::{ |
| 33 | float::{ |
| 34 | Float32, |
| 35 | Float64, |
| 36 | }, |
| 37 | string::NumberString, |
| 38 | }; |
| 39 | use oxedyne_fe2o3_text::{ |
| 40 | base2x, |
| 41 | string::Quote, |
| 42 | }; |
| 43 | |
| 44 | use std::{ |
| 45 | cell::RefCell, |
| 46 | collections::BTreeMap, |
| 47 | convert::TryInto, |
| 48 | fmt, |
| 49 | str::FromStr, |
| 50 | }; |
| 51 | |
| 52 | |
| 53 | impl FromStr for Dat { |
| 54 | type Err = Error<ErrTag>; |
| 55 | |
| 56 | fn from_str(s: &str) -> std::result::Result<Self, Self::Err> { |
| 57 | Dat::decode_string(s) |
| 58 | } |
| 59 | } |
| 60 | |
| 61 | #[derive(Clone, Debug)] |
| 62 | pub struct DecoderConfig< |
| 63 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 64 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 65 | >{ |
| 66 | pub quote_protection: bool, |
| 67 | pub comment_allowed: bool, |
| 68 | pub comment_capture: bool, |
| 69 | pub comment1_start_char: char, |
| 70 | pub comment1_end_char: char, |
| 71 | pub comment2_start_char: char, |
| 72 | pub comment2_end_char: char, |
| 73 | pub default_key: DatIntKind, |
| 74 | pub trailing_comma_allowed: bool, |
| 75 | pub use_ordmaps: bool, |
| 76 | pub ukinds_opt: Option<UsrKinds<M1, M2>>, |
| 77 | pub omap_start: u64, |
| 78 | pub omap_delta: u64, |
| 79 | pub limits: DecodeLimits, |
| 80 | } |
| 81 | |
| 82 | impl< |
| 83 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 84 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 85 | > |
| 86 | Default for DecoderConfig<M1, M2> |
| 87 | { |
| 88 | fn default() -> Self { |
| 89 | Self { |
| 90 | quote_protection: true, |
| 91 | comment_allowed: true, |
| 92 | comment_capture: true, |
| 93 | comment1_start_char: '!', |
| 94 | comment1_end_char: '!', |
| 95 | comment2_start_char: '#', |
| 96 | comment2_end_char: '#', |
| 97 | default_key: DatIntKind::U32, |
| 98 | trailing_comma_allowed: true, |
| 99 | use_ordmaps: false, |
| 100 | ukinds_opt: None, |
| 101 | omap_start: Dat::OMAP_ORDER_START_DEFAULT, |
| 102 | omap_delta: Dat::OMAP_ORDER_DELTA_DEFAULT, |
| 103 | limits: DecodeLimits::text(), |
| 104 | } |
| 105 | } |
| 106 | } |
| 107 | |
| 108 | impl< |
| 109 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 110 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 111 | > |
| 112 | DecoderConfig<M1, M2> |
| 113 | { |
| 114 | pub fn jdat(ukinds_opt: Option<UsrKinds<M1, M2>>) -> Self { |
| 115 | Self { |
| 116 | ukinds_opt, |
| 117 | ..Default::default() |
| 118 | } |
| 119 | } |
| 120 | |
| 121 | pub fn json(ukinds_opt: Option<UsrKinds<M1, M2>>) -> Self { |
| 122 | Self { |
| 123 | comment_allowed: false, |
| 124 | trailing_comma_allowed: false, |
| 125 | ukinds_opt, |
| 126 | ..Default::default() |
| 127 | } |
| 128 | } |
| 129 | |
| 130 | pub fn with_limits(mut self, limits: DecodeLimits) -> Self { |
| 131 | self.limits = limits; |
| 132 | self |
| 133 | } |
| 134 | } |
| 135 | |
| 136 | #[derive(Clone, Debug)] |
| 137 | pub struct Cursor { |
| 138 | abs: usize, // Stream position. |
| 139 | curr: char, // Current character. |
| 140 | prev: char, // Previous character. |
| 141 | line: usize, // Line position. |
| 142 | x: usize, // Character position on line. |
| 143 | } |
| 144 | |
| 145 | impl Default for Cursor { |
| 146 | fn default() -> Self { |
| 147 | Self { |
| 148 | abs: 0, |
| 149 | curr: char::default(), |
| 150 | prev: char::default(), |
| 151 | line: 1, |
| 152 | x: 0, |
| 153 | } |
| 154 | } |
| 155 | } |
| 156 | |
| 157 | impl fmt::Display for Cursor { |
| 158 | fn fmt (&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
| 159 | write!(f, "char '{}' line {} pos {}", self.curr, self.line, self.x) |
| 160 | } |
| 161 | } |
| 162 | |
| 163 | impl Cursor { |
| 164 | pub fn advance( |
| 165 | &mut self, |
| 166 | c: char, |
| 167 | quote_protection: bool, |
| 168 | ) { |
| 169 | if self.abs > 0 { |
| 170 | let p = self.curr; |
| 171 | self.curr = c; |
| 172 | self.prev = p; |
| 173 | } else { |
| 174 | self.curr = c |
| 175 | } |
| 176 | self.abs += 1; |
| 177 | if c == '\n' || (quote_protection && c == '\\') { |
| 178 | self.line += 1; |
| 179 | self.x = 0; |
| 180 | } else { |
| 181 | if !c.is_control() { |
| 182 | self.x += 1; |
| 183 | } |
| 184 | } |
| 185 | } |
| 186 | } |
| 187 | |
| 188 | #[derive(Clone, Debug, PartialEq)] |
| 189 | pub enum CommentCapture { |
| 190 | Type1, |
| 191 | Type2, |
| 192 | } |
| 193 | |
| 194 | #[derive(Clone, Debug, Default, PartialEq)] |
| 195 | pub enum StringEscape { |
| 196 | #[default] |
| 197 | None, |
| 198 | Backslash, |
| 199 | Unicode { digits: u8, acc: u32 }, |
| 200 | SurrogateBackslash { high: u16 }, |
| 201 | SurrogateU { high: u16 }, |
| 202 | LowSurrogate { digits: u8, acc: u32, high: u16 }, |
| 203 | } |
| 204 | |
| 205 | #[derive(Clone, Debug)] |
| 206 | pub struct DecoderState { |
| 207 | // Data |
| 208 | //pub cursor: Cursor, |
| 209 | pub kind_outer: Kind, |
| 210 | pub depth: usize, |
| 211 | // Switches |
| 212 | pub explicit_kind: bool, // The kind was defined explicitly. |
| 213 | pub quote_protection: Quote, // "a(' &{}" etc |
| 214 | pub comment_capture: Option<CommentCapture>, |
| 215 | pub number_capture: bool, // 23_786.345 contiguous |
| 216 | pub kind_capture: bool, // (k| |
| 217 | pub atomic_capture: bool, |
| 218 | pub molecular_capture: Option<MolecularCapture>, |
| 219 | pub string_escape: StringEscape, // Inside quoted strings only. |
| 220 | } |
| 221 | |
| 222 | impl Default for DecoderState { |
| 223 | fn default() -> Self { |
| 224 | Self { |
| 225 | kind_outer: Kind::default(), |
| 226 | depth: 1, // The state a decode starts from is the root value. |
| 227 | explicit_kind: false, |
| 228 | quote_protection: Quote::default(), |
| 229 | comment_capture: None, |
| 230 | number_capture: false, |
| 231 | kind_capture: false, |
| 232 | atomic_capture: false, |
| 233 | molecular_capture: None, |
| 234 | string_escape: StringEscape::default(), |
| 235 | } |
| 236 | } |
| 237 | } |
| 238 | |
| 239 | impl fmt::Display for DecoderState { |
| 240 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
| 241 | write!(f, "<quo:{:?}|com:{:?}|num:{}|knd:{}|atm:{}|mol:{:?}>", |
| 242 | self.quote_protection, |
| 243 | self.comment_capture, |
| 244 | self.number_capture as u8, |
| 245 | self.kind_capture as u8, |
| 246 | self.atomic_capture as u8, |
| 247 | self.molecular_capture, |
| 248 | ) |
| 249 | } |
| 250 | } |
| 251 | |
| 252 | impl DecoderState { |
| 253 | pub fn recurse(&self) -> Self { |
| 254 | Self { |
| 255 | kind_outer: self.kind_outer.clone(), |
| 256 | depth: self.depth + 1, |
| 257 | ..Default::default() |
| 258 | } |
| 259 | } |
| 260 | } |
| 261 | |
| 262 | #[derive(Clone, Debug, Default)] |
| 263 | pub struct DecoderStore { |
| 264 | pub val_opt: Option<Dat>, |
| 265 | pub key_opt: Option<Dat>, |
| 266 | pub list: Vec<Dat>, |
| 267 | pub comment: String, |
| 268 | pub note_config: NoteConfig, |
| 269 | pub byts: Vec<u8>, |
| 270 | pub map: DaticleMap, |
| 271 | pub ordmap: OrdDaticleMap, |
| 272 | pub slurp: Slurp, |
| 273 | pub map_count: MapCounter, |
| 274 | } |
| 275 | |
| 276 | impl DecoderStore { |
| 277 | pub fn new< |
| 278 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 279 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 280 | >( |
| 281 | cfg: &DecoderConfig<M1, M2>, |
| 282 | ) |
| 283 | -> Self |
| 284 | { |
| 285 | Self { |
| 286 | val_opt: None, |
| 287 | key_opt: None, |
| 288 | list: Vec::new(), |
| 289 | comment: String::new(), |
| 290 | note_config: NoteConfig::default(), |
| 291 | byts: Vec::new(), |
| 292 | map: DaticleMap::new(), |
| 293 | ordmap: OrdDaticleMap::new(), |
| 294 | slurp: Slurp::new(), |
| 295 | map_count: MapCounter { |
| 296 | count: 0, |
| 297 | order: cfg.omap_start, |
| 298 | delta: cfg.omap_delta, |
| 299 | }, |
| 300 | } |
| 301 | } |
| 302 | } |
| 303 | |
| 304 | #[derive(Clone, Debug, PartialEq)] |
| 305 | pub enum MolecularCapture { |
| 306 | Bytes, |
| 307 | ListMixed, |
| 308 | ListSame, |
| 309 | Map, |
| 310 | } |
| 311 | |
| 312 | #[derive(Clone, Copy, Debug, PartialEq)] |
| 313 | enum Descent { |
| 314 | Kindicle, |
| 315 | Molecule, |
| 316 | } |
| 317 | |
| 318 | #[derive(Debug)] |
| 319 | enum Step { |
| 320 | Continue, |
| 321 | Done(Dat), |
| 322 | Descend(DecoderState, Descent), |
| 323 | } |
| 324 | |
| 325 | impl MolecularCapture { |
| 326 | #[inline(never)] |
| 327 | pub fn from_kind(kind: &Kind) -> Option<Self> { |
| 328 | match kind { |
| 329 | Kind::List | |
| 330 | Kind::Tup2 | |
| 331 | Kind::Tup3 | |
| 332 | Kind::Tup4 | |
| 333 | Kind::Tup5 | |
| 334 | Kind::Tup6 | |
| 335 | Kind::Tup7 | |
| 336 | Kind::Tup8 | |
| 337 | Kind::Tup9 | |
| 338 | Kind::Tup10 => Some(MolecularCapture::ListMixed), |
| 339 | Kind::OrdMap | |
| 340 | Kind::Map => Some(MolecularCapture::Map), |
| 341 | Kind::BU8 | |
| 342 | Kind::BU16 | |
| 343 | Kind::BU32 | |
| 344 | Kind::BU64 | |
| 345 | Kind::BC64 | |
| 346 | Kind::B2 | |
| 347 | Kind::B3 | |
| 348 | Kind::B4 | |
| 349 | Kind::B5 | |
| 350 | Kind::B6 | |
| 351 | Kind::B7 | |
| 352 | Kind::B8 | |
| 353 | Kind::B9 | |
| 354 | Kind::B10 | |
| 355 | Kind::B16 | |
| 356 | Kind::B32 => Some(MolecularCapture::Bytes), |
| 357 | Kind::Vek | |
| 358 | Kind::Tup2u8 | |
| 359 | Kind::Tup3u8 | |
| 360 | Kind::Tup4u8 | |
| 361 | Kind::Tup5u8 | |
| 362 | Kind::Tup6u8 | |
| 363 | Kind::Tup7u8 | |
| 364 | Kind::Tup8u8 | |
| 365 | Kind::Tup9u8 | |
| 366 | Kind::Tup10u8 | |
| 367 | Kind::Tup2u16 | |
| 368 | Kind::Tup3u16 | |
| 369 | Kind::Tup4u16 | |
| 370 | Kind::Tup5u16 | |
| 371 | Kind::Tup6u16 | |
| 372 | Kind::Tup7u16 | |
| 373 | Kind::Tup8u16 | |
| 374 | Kind::Tup9u16 | |
| 375 | Kind::Tup10u16 | |
| 376 | Kind::Tup2u32 | |
| 377 | Kind::Tup3u32 | |
| 378 | Kind::Tup4u32 | |
| 379 | Kind::Tup5u32 | |
| 380 | Kind::Tup6u32 | |
| 381 | Kind::Tup7u32 | |
| 382 | Kind::Tup8u32 | |
| 383 | Kind::Tup9u32 | |
| 384 | Kind::Tup10u32 | |
| 385 | Kind::Tup2u64 | |
| 386 | Kind::Tup3u64 | |
| 387 | Kind::Tup4u64 | |
| 388 | Kind::Tup5u64 | |
| 389 | Kind::Tup6u64 | |
| 390 | Kind::Tup7u64 | |
| 391 | Kind::Tup8u64 | |
| 392 | Kind::Tup9u64 | |
| 393 | Kind::Tup10u64 | |
| 394 | Kind::Tup2i8 | |
| 395 | Kind::Tup3i8 | |
| 396 | Kind::Tup4i8 | |
| 397 | Kind::Tup5i8 | |
| 398 | Kind::Tup6i8 | |
| 399 | Kind::Tup7i8 | |
| 400 | Kind::Tup8i8 | |
| 401 | Kind::Tup9i8 | |
| 402 | Kind::Tup10i8 | |
| 403 | Kind::Tup2i16 | |
| 404 | Kind::Tup3i16 | |
| 405 | Kind::Tup4i16 | |
| 406 | Kind::Tup5i16 | |
| 407 | Kind::Tup6i16 | |
| 408 | Kind::Tup7i16 | |
| 409 | Kind::Tup8i16 | |
| 410 | Kind::Tup9i16 | |
| 411 | Kind::Tup10i16 | |
| 412 | Kind::Tup2i32 | |
| 413 | Kind::Tup3i32 | |
| 414 | Kind::Tup4i32 | |
| 415 | Kind::Tup5i32 | |
| 416 | Kind::Tup6i32 | |
| 417 | Kind::Tup7i32 | |
| 418 | Kind::Tup8i32 | |
| 419 | Kind::Tup9i32 | |
| 420 | Kind::Tup10i32 | |
| 421 | Kind::Tup2i64 | |
| 422 | Kind::Tup3i64 | |
| 423 | Kind::Tup4i64 | |
| 424 | Kind::Tup5i64 | |
| 425 | Kind::Tup6i64 | |
| 426 | Kind::Tup7i64 | |
| 427 | Kind::Tup8i64 | |
| 428 | Kind::Tup9i64 | |
| 429 | Kind::Tup10i64 => Some(MolecularCapture::ListSame), |
| 430 | _ => None, |
| 431 | } |
| 432 | } |
| 433 | |
| 434 | #[inline(never)] |
| 435 | pub fn same_kind(kind: &Kind) -> Kind { |
| 436 | match kind { |
| 437 | Kind::Tup2u8 | |
| 438 | Kind::Tup3u8 | |
| 439 | Kind::Tup4u8 | |
| 440 | Kind::Tup5u8 | |
| 441 | Kind::Tup6u8 | |
| 442 | Kind::Tup7u8 | |
| 443 | Kind::Tup8u8 | |
| 444 | Kind::Tup9u8 | |
| 445 | Kind::Tup10u8 => Kind::U8, |
| 446 | Kind::Tup2u16 | |
| 447 | Kind::Tup3u16 | |
| 448 | Kind::Tup4u16 | |
| 449 | Kind::Tup5u16 | |
| 450 | Kind::Tup6u16 | |
| 451 | Kind::Tup7u16 | |
| 452 | Kind::Tup8u16 | |
| 453 | Kind::Tup9u16 | |
| 454 | Kind::Tup10u16 => Kind::U16, |
| 455 | Kind::Tup2u32 | |
| 456 | Kind::Tup3u32 | |
| 457 | Kind::Tup4u32 | |
| 458 | Kind::Tup5u32 | |
| 459 | Kind::Tup6u32 | |
| 460 | Kind::Tup7u32 | |
| 461 | Kind::Tup8u32 | |
| 462 | Kind::Tup9u32 | |
| 463 | Kind::Tup10u32 => Kind::U32, |
| 464 | Kind::Tup2u64 | |
| 465 | Kind::Tup3u64 | |
| 466 | Kind::Tup4u64 | |
| 467 | Kind::Tup5u64 | |
| 468 | Kind::Tup6u64 | |
| 469 | Kind::Tup7u64 | |
| 470 | Kind::Tup8u64 | |
| 471 | Kind::Tup9u64 | |
| 472 | Kind::Tup10u64 => Kind::U64, |
| 473 | Kind::Tup2i8 | |
| 474 | Kind::Tup3i8 | |
| 475 | Kind::Tup4i8 | |
| 476 | Kind::Tup5i8 | |
| 477 | Kind::Tup6i8 | |
| 478 | Kind::Tup7i8 | |
| 479 | Kind::Tup8i8 | |
| 480 | Kind::Tup9i8 | |
| 481 | Kind::Tup10i8 => Kind::I8, |
| 482 | Kind::Tup2i16 | |
| 483 | Kind::Tup3i16 | |
| 484 | Kind::Tup4i16 | |
| 485 | Kind::Tup5i16 | |
| 486 | Kind::Tup6i16 | |
| 487 | Kind::Tup7i16 | |
| 488 | Kind::Tup8i16 | |
| 489 | Kind::Tup9i16 | |
| 490 | Kind::Tup10i16 => Kind::I16, |
| 491 | Kind::Tup2i32 | |
| 492 | Kind::Tup3i32 | |
| 493 | Kind::Tup4i32 | |
| 494 | Kind::Tup5i32 | |
| 495 | Kind::Tup6i32 | |
| 496 | Kind::Tup7i32 | |
| 497 | Kind::Tup8i32 | |
| 498 | Kind::Tup9i32 | |
| 499 | Kind::Tup10i32 => Kind::I32, |
| 500 | Kind::Tup2i64 | |
| 501 | Kind::Tup3i64 | |
| 502 | Kind::Tup4i64 | |
| 503 | Kind::Tup5i64 | |
| 504 | Kind::Tup6i64 | |
| 505 | Kind::Tup7i64 | |
| 506 | Kind::Tup8i64 | |
| 507 | Kind::Tup9i64 | |
| 508 | Kind::Tup10i64 => Kind::I64, |
| 509 | _ => Kind::Unknown, |
| 510 | } |
| 511 | } |
| 512 | } |
| 513 | |
| 514 | #[derive(Clone, Debug, Default)] |
| 515 | pub struct Slurp { |
| 516 | s: String, |
| 517 | is_string: bool, |
| 518 | } |
| 519 | |
| 520 | impl Slurp { |
| 521 | fn new() -> Self { |
| 522 | Slurp { |
| 523 | s: String::new(), |
| 524 | is_string: false, |
| 525 | } |
| 526 | } |
| 527 | |
| 528 | fn reset(&mut self) { |
| 529 | self.s = String::new(); |
| 530 | self.is_string = false; |
| 531 | } |
| 532 | |
| 533 | fn push(&mut self, c: char) { |
| 534 | self.s.push(c); |
| 535 | } |
| 536 | |
| 537 | fn flag_as_string(&mut self) { |
| 538 | self.is_string = true; |
| 539 | } |
| 540 | |
| 541 | fn is_string(&self) -> bool { |
| 542 | self.is_string |
| 543 | } |
| 544 | |
| 545 | fn get_str(&self) -> &str { |
| 546 | &self.s |
| 547 | } |
| 548 | |
| 549 | fn clone_string(&self) -> String { |
| 550 | self.s.clone() |
| 551 | } |
| 552 | |
| 553 | fn take_string(&mut self) -> String { |
| 554 | std::mem::replace(&mut self.s, String::new()) |
| 555 | } |
| 556 | |
| 557 | fn len(&self) -> usize { |
| 558 | self.s.len() |
| 559 | } |
| 560 | |
| 561 | fn has_content(&self) -> bool { |
| 562 | self.is_string || !self.s.is_empty() |
| 563 | } |
| 564 | |
| 565 | #[inline(never)] |
| 566 | fn char_slurped( |
| 567 | &mut self, |
| 568 | (i, c): (usize, char), |
| 569 | state: &mut DecoderState, |
| 570 | ) |
| 571 | -> Outcome<bool> |
| 572 | { |
| 573 | if state.quote_protection != Quote::None { |
| 574 | self.push(c); |
| 575 | return Ok(true) |
| 576 | } |
| 577 | if state.kind_capture { |
| 578 | match c { |
| 579 | 'A' ..= 'Z' | |
| 580 | 'a' ..= 'z' | |
| 581 | '0' ..= '9' | |
| 582 | '/' | |
| 583 | '.' | |
| 584 | '_' => |
| 585 | { |
| 586 | self.push(c); |
| 587 | return Ok(true); |
| 588 | } |
| 589 | ' ' | |
| 590 | '\n' => {} |
| 591 | ',' => { |
| 592 | // Allow for the possibility that we have a tuple without a kindicle. |
| 593 | // e.g. (1,2,..) |
| 594 | state.kind_capture = false; |
| 595 | state.molecular_capture = Some(MolecularCapture::ListMixed); |
| 596 | return Ok(false); |
| 597 | } |
| 598 | '(' => { |
| 599 | // Allow for the possibility that we have a tuple without a kindicle. |
| 600 | // e.g. ((u8|1), ..) |
| 601 | state.molecular_capture = Some(MolecularCapture::ListMixed); |
| 602 | return Ok(false); |
| 603 | } |
| 604 | ')' => return Ok(false), |
| 605 | _ => { |
| 606 | return Err(err!( |
| 607 | "Found unrecognised '{}' at position {} while capturing daticle kind.", c, i + 1; |
| 608 | String, Input, Decode, Invalid)); |
| 609 | } |
| 610 | } |
| 611 | } |
| 612 | Ok(false) |
| 613 | } |
| 614 | } |
| 615 | |
| 616 | impl fmt::Display for Slurp { |
| 617 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
| 618 | f.write_str(&self.s[..]) |
| 619 | } |
| 620 | } |
| 621 | |
| 622 | #[derive(Clone, Debug, Default)] |
| 623 | pub struct MapCounter { |
| 624 | count: u64, |
| 625 | order: u64, |
| 626 | delta: u64, |
| 627 | } |
| 628 | |
| 629 | impl MapCounter { |
| 630 | fn order(&self) -> u64 { self.order } |
| 631 | fn inc(&mut self) -> Outcome<()> { |
| 632 | self.count = try_add!(self.count, 1); |
| 633 | Ok(()) |
| 634 | } |
| 635 | fn inc_all(&mut self) -> Outcome<()> { |
| 636 | self.count = try_add!(self.count, 1); |
| 637 | self.order = try_add!(self.order, self.delta); |
| 638 | Ok(()) |
| 639 | } |
| 640 | } |
| 641 | |
| 642 | impl Kind { |
| 643 | |
| 644 | pub const USR_LABEL_PREFIX: &'static str = "usr_"; |
| 645 | |
| 646 | /// Reverse lookup from name to enum value that first tries the standard names then the given |
| 647 | /// user names. |
| 648 | #[inline(never)] |
| 649 | pub fn from_label< |
| 650 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 651 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 652 | >( |
| 653 | s: &str, |
| 654 | ukinds_opt: Option<&UsrKinds<M1, M2>>, |
| 655 | ) |
| 656 | -> Outcome<Self> |
| 657 | { |
| 658 | match Self::from_str(s) { |
| 659 | Ok(k) => Ok(k), |
| 660 | Err(e1) => match ukinds_opt { |
| 661 | Some(ukids) => match ukids.get_label(s) { |
| 662 | Some(ukid) => Ok(Self::Usr(ukid.clone())), |
| 663 | None => Err(err!( |
| 664 | "Daticle kind label not recognised as standard or custom: '{}'", s; |
| 665 | Input, String, Unknown)), |
| 666 | } |
| 667 | None => Err(err!(e1, "No custom user kinds supplied"; Input, String, Unknown)), |
| 668 | } |
| 669 | } |
| 670 | } |
| 671 | |
| 672 | #[inline(never)] |
| 673 | fn decode_number(self, ns: NumberString) -> Outcome<Dat> { |
| 674 | match self { |
| 675 | Kind::U8 => { |
| 676 | if ns.is_negative() { |
| 677 | return Err(err!( |
| 678 | "U8 '{}' cannot be negative", |
| 679 | &ns.int_string(); |
| 680 | String, Input, Decode, Invalid)); |
| 681 | } |
| 682 | let n = res!(<u8>::from_str_radix(ns.abs_integer_str(), ns.radix())); |
| 683 | return Ok(Dat::U8(n)); |
| 684 | } |
| 685 | Kind::U16 => { |
| 686 | if ns.is_negative() { |
| 687 | return Err(err!( |
| 688 | "U16 '{}' cannot be negative", |
| 689 | &ns.int_string(); |
| 690 | String, Input, Decode, Invalid)); |
| 691 | } |
| 692 | let n = res!(<u16>::from_str_radix(ns.abs_integer_str(), ns.radix())); |
| 693 | return Ok(Dat::U16(n)); |
| 694 | } |
| 695 | Kind::U32 => { |
| 696 | if ns.is_negative() { |
| 697 | return Err(err!( |
| 698 | "U32 '{}' cannot be negative", |
| 699 | &ns.int_string(); |
| 700 | String, Input, Decode, Invalid)); |
| 701 | } |
| 702 | let n = res!(<u32>::from_str_radix(ns.abs_integer_str(), ns.radix())); |
| 703 | return Ok(Dat::U32(n)); |
| 704 | } |
| 705 | Kind::U64 | Kind::C64 => { |
| 706 | if ns.is_negative() { |
| 707 | return Err(err!( |
| 708 | "U64 '{}' cannot be negative", |
| 709 | &ns.int_string(); |
| 710 | String, Input, Decode, Invalid)); |
| 711 | } |
| 712 | let n = res!(<u64>::from_str_radix(ns.abs_integer_str(), ns.radix())); |
| 713 | if self == Kind::U64 { |
| 714 | return Ok(Dat::U64(n)); |
| 715 | } else { |
| 716 | return Ok(Dat::C64(n)); |
| 717 | } |
| 718 | } |
| 719 | Kind::U128 => { |
| 720 | if ns.is_negative() { |
| 721 | return Err(err!( |
| 722 | "U128 '{}' cannot be negative", |
| 723 | &ns.int_string(); |
| 724 | String, Input, Decode, Invalid)); |
| 725 | } |
| 726 | let n = res!(<u128>::from_str_radix(ns.abs_integer_str(), ns.radix())); |
| 727 | return Ok(Dat::U128(n)); |
| 728 | } |
| 729 | Kind::I8 => { |
| 730 | let n = res!(<i8>::from_str_radix(&ns.int_string(), ns.radix())); |
| 731 | return Ok(Dat::I8(n)); |
| 732 | } |
| 733 | Kind::I16 => { |
| 734 | let n = res!(<i16>::from_str_radix(&ns.int_string(), ns.radix())); |
| 735 | return Ok(Dat::I16(n)); |
| 736 | } |
| 737 | Kind::I32 => { |
| 738 | let n = res!(<i32>::from_str_radix(&ns.int_string(), ns.radix())); |
| 739 | return Ok(Dat::I32(n)); |
| 740 | } |
| 741 | Kind::I64 => { |
| 742 | let n = res!(<i64>::from_str_radix(&ns.int_string(), ns.radix())); |
| 743 | return Ok(Dat::I64(n)); |
| 744 | } |
| 745 | Kind::I128 => { |
| 746 | let n = res!(<i128>::from_str_radix(&ns.int_string(), ns.radix())); |
| 747 | return Ok(Dat::I128(n)); |
| 748 | } |
| 749 | Kind::F32 => { |
| 750 | return Ok(Dat::F32(res!(Float32::from_str(ns.source())))); |
| 751 | } |
| 752 | Kind::F64 => { |
| 753 | return Ok(Dat::F64(res!(Float64::from_str(ns.source())))); |
| 754 | } |
| 755 | Kind::Aint => { |
| 756 | if ns.has_point() || ns.has_exp() { |
| 757 | return Err(err!( |
| 758 | "Decimal or scientific notation not accepted for AINT, use ADEC"; |
| 759 | String, Input, Decode, Invalid)); |
| 760 | } |
| 761 | match ns.as_bigint() { |
| 762 | Ok(n) => return Ok(Dat::Aint(n)), |
| 763 | Err(e) => return Err(err!(e, "While decoding an integer."; Decode, Integer)), |
| 764 | } |
| 765 | } |
| 766 | Kind::Adec => { |
| 767 | return Ok(Dat::Adec(res!(ns.as_bigdecimal()))); |
| 768 | } |
| 769 | _ => return Err(err!( |
| 770 | "NumberString {:?} is of invalid kind {:?}", |
| 771 | ns, self; |
| 772 | String, Input, Decode, Invalid)), |
| 773 | } |
| 774 | } |
| 775 | |
| 776 | /// Classify which `Kind`s are numbers for the purposes of string decoding. |
| 777 | pub fn is_number(&self) -> bool { |
| 778 | match *self { |
| 779 | Self::U8 | |
| 780 | Self::U16 | |
| 781 | Self::U32 | |
| 782 | Self::U64 | |
| 783 | Self::C64 | |
| 784 | Self::U128 | |
| 785 | Self::I8 | |
| 786 | Self::I16 | |
| 787 | Self::I32 | |
| 788 | Self::I64 | |
| 789 | Self::I128 | |
| 790 | Self::F32 | |
| 791 | Self::F64 | |
| 792 | Self::Aint | |
| 793 | Self::Adec => true, |
| 794 | _ => false, |
| 795 | } |
| 796 | } |
| 797 | |
| 798 | pub fn uses_list_brackets(&self) -> bool { |
| 799 | match *self { |
| 800 | // Heterogenous |
| 801 | Self::List | |
| 802 | Self::Tup2 | |
| 803 | Self::Tup3 | |
| 804 | Self::Tup4 | |
| 805 | Self::Tup5 | |
| 806 | Self::Tup6 | |
| 807 | Self::Tup7 | |
| 808 | Self::Tup8 | |
| 809 | Self::Tup9 | |
| 810 | Self::Tup10 | |
| 811 | // Homogenous |
| 812 | Self::Vek | |
| 813 | // Variable length bytes |
| 814 | Self::BU8 | |
| 815 | Self::BU16 | |
| 816 | Self::BU32 | |
| 817 | Self::BU64 | |
| 818 | Self::BC64 | |
| 819 | // Fixed length bytes |
| 820 | Self::B2 | |
| 821 | Self::B3 | |
| 822 | Self::B4 | |
| 823 | Self::B5 | |
| 824 | Self::B6 | |
| 825 | Self::B7 | |
| 826 | Self::B8 | |
| 827 | Self::B9 | |
| 828 | Self::B10 | |
| 829 | Self::B16 | |
| 830 | Self::B32 | |
| 831 | // Fixed length numbers |
| 832 | Self::Tup2u8 | |
| 833 | Self::Tup3u8 | |
| 834 | Self::Tup4u8 | |
| 835 | Self::Tup5u8 | |
| 836 | Self::Tup6u8 | |
| 837 | Self::Tup7u8 | |
| 838 | Self::Tup8u8 | |
| 839 | Self::Tup9u8 | |
| 840 | Self::Tup10u8 | |
| 841 | Self::Tup2u16 | |
| 842 | Self::Tup3u16 | |
| 843 | Self::Tup4u16 | |
| 844 | Self::Tup5u16 | |
| 845 | Self::Tup6u16 | |
| 846 | Self::Tup7u16 | |
| 847 | Self::Tup8u16 | |
| 848 | Self::Tup9u16 | |
| 849 | Self::Tup10u16 | |
| 850 | Self::Tup2u32 | |
| 851 | Self::Tup3u32 | |
| 852 | Self::Tup4u32 | |
| 853 | Self::Tup5u32 | |
| 854 | Self::Tup6u32 | |
| 855 | Self::Tup7u32 | |
| 856 | Self::Tup8u32 | |
| 857 | Self::Tup9u32 | |
| 858 | Self::Tup10u32 | |
| 859 | Self::Tup2u64 | |
| 860 | Self::Tup3u64 | |
| 861 | Self::Tup4u64 | |
| 862 | Self::Tup5u64 | |
| 863 | Self::Tup6u64 | |
| 864 | Self::Tup7u64 | |
| 865 | Self::Tup8u64 | |
| 866 | Self::Tup9u64 | |
| 867 | Self::Tup10u64 | |
| 868 | Self::Tup2i8 | |
| 869 | Self::Tup3i8 | |
| 870 | Self::Tup4i8 | |
| 871 | Self::Tup5i8 | |
| 872 | Self::Tup6i8 | |
| 873 | Self::Tup7i8 | |
| 874 | Self::Tup8i8 | |
| 875 | Self::Tup9i8 | |
| 876 | Self::Tup10i8 | |
| 877 | Self::Tup2i16 | |
| 878 | Self::Tup3i16 | |
| 879 | Self::Tup4i16 | |
| 880 | Self::Tup5i16 | |
| 881 | Self::Tup6i16 | |
| 882 | Self::Tup7i16 | |
| 883 | Self::Tup8i16 | |
| 884 | Self::Tup9i16 | |
| 885 | Self::Tup10i16 | |
| 886 | Self::Tup2i32 | |
| 887 | Self::Tup3i32 | |
| 888 | Self::Tup4i32 | |
| 889 | Self::Tup5i32 | |
| 890 | Self::Tup6i32 | |
| 891 | Self::Tup7i32 | |
| 892 | Self::Tup8i32 | |
| 893 | Self::Tup9i32 | |
| 894 | Self::Tup10i32 | |
| 895 | Self::Tup2i64 | |
| 896 | Self::Tup3i64 | |
| 897 | Self::Tup4i64 | |
| 898 | Self::Tup5i64 | |
| 899 | Self::Tup6i64 | |
| 900 | Self::Tup7i64 | |
| 901 | Self::Tup8i64 | |
| 902 | Self::Tup9i64 | |
| 903 | Self::Tup10i64 => true, |
| 904 | _ => false, |
| 905 | } |
| 906 | } |
| 907 | } |
| 908 | |
| 909 | impl Dat { |
| 910 | |
| 911 | pub fn decode_string< |
| 912 | S: Into<String>, |
| 913 | >( |
| 914 | s: S, |
| 915 | ) |
| 916 | -> Outcome<Self> |
| 917 | { |
| 918 | let dec_cfg = DecoderConfig::< |
| 919 | BTreeMap<UsrKindCode, UsrKind>, |
| 920 | BTreeMap<String, UsrKindId>, |
| 921 | >::default(); |
| 922 | |
| 923 | Self::decode_string_with_config(s, &dec_cfg) |
| 924 | } |
| 925 | |
| 926 | pub fn decode_string_limited< |
| 927 | S: Into<String>, |
| 928 | >( |
| 929 | s: S, |
| 930 | lims: &DecodeLimits, |
| 931 | ) |
| 932 | -> Outcome<Self> |
| 933 | { |
| 934 | let dec_cfg = DecoderConfig::< |
| 935 | BTreeMap<UsrKindCode, UsrKind>, |
| 936 | BTreeMap<String, UsrKindId>, |
| 937 | >::default().with_limits(*lims); |
| 938 | |
| 939 | Self::decode_string_with_config(s, &dec_cfg) |
| 940 | } |
| 941 | |
| 942 | pub fn decode_string_with_config< |
| 943 | S: Into<String>, |
| 944 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 945 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 946 | >( |
| 947 | s: S, |
| 948 | cfg: &DecoderConfig<M1, M2>, |
| 949 | ) |
| 950 | -> Outcome<Self> |
| 951 | { |
| 952 | // We want to take ownership of the String. |
| 953 | let s = s.into(); |
| 954 | res!(cfg.limits.check_len(s.len())); |
| 955 | let mut iter = s.chars().collect::<Vec<_>>().into_iter().enumerate(); |
| 956 | Self::recursive_decode( |
| 957 | &mut iter, |
| 958 | cfg, |
| 959 | DecoderState::default(), |
| 960 | &RefCell::new(Cursor::default()), |
| 961 | ) |
| 962 | } |
| 963 | |
| 964 | /// Decodes a value at one level of nesting, recursing at each `(k|`, `[` and `{`. |
| 965 | /// |
| 966 | /// The recursion is bounded by `cfg.limits.max_depth`, and the frame is kept small by moving |
| 967 | /// each of the fatter arms of the match below into its own `#[inline(never)]` function, so that |
| 968 | /// the cost of a level of nesting is the cost of the loop and its store, not the sum of every |
| 969 | /// arm's locals. A hostile text nesting a list a million deep must return an error, not exhaust |
| 970 | /// the stack, since a stack overflow aborts the process and cannot be caught. |
| 971 | /// |
| 972 | /// * `kind_outer` - if not None, indicates that the upper recursion level is a |
| 973 | /// `List` or `Map`/`OrdMap` |
| 974 | /// |
| 975 | /// BNF (does not include comments) |
| 976 | /// <atom> ::= "" | <string> | <number> | "(" <kind> ")" | "(" <kind> "|" ")" |
| 977 | /// <daticle> ::= <atom> | <seq> | "(" <kind> "|" <daticle> ")" |
| 978 | /// <kind> ::= EMPTY | TRUE | FALSE | U8 ... |
| 979 | /// <seq> ::= <lbrac> <seq> "," <daticle> <rbrac> | <lbrac> <seq> "," <pair> <rbrac> |
| 980 | /// <pair> ::= <daticle> ":" <daticle> |
| 981 | /// <lbrac> ::= "[" | "{" |
| 982 | /// <rbrac> ::= "]" | "}" |
| 983 | pub fn recursive_decode< |
| 984 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 985 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 986 | >( |
| 987 | mut iter: &mut std::iter::Enumerate<std::vec::IntoIter<char>>, |
| 988 | cfg: &DecoderConfig<M1, M2>, |
| 989 | mut state: DecoderState, |
| 990 | cursor: &RefCell<Cursor>, |
| 991 | ) |
| 992 | -> Outcome<Self> |
| 993 | { |
| 994 | // Refuse the level before descending into it, naming the character we are looking at. |
| 995 | let pos = cursor.borrow().abs; |
| 996 | res!(cfg.limits.check_char_depth(state.depth, pos)); |
| 997 | |
| 998 | // Switches. |
| 999 | let mut comment_required = false; |
| 1000 | // Store. |
| 1001 | let mut store = DecoderStore::new(cfg); |
| 1002 | |
| 1003 | state.molecular_capture = MolecularCapture::from_kind(&state.kind_outer); |
| 1004 | |
| 1005 | while let Some((i, c)) = iter.next() { |
| 1006 | match res!(Self::step( |
| 1007 | (i, c), |
| 1008 | cfg, |
| 1009 | &mut state, |
| 1010 | &mut store, |
| 1011 | cursor, |
| 1012 | &mut comment_required, |
| 1013 | )) { |
| 1014 | Step::Continue => (), |
| 1015 | Step::Done(dat) => return Ok(dat), |
| 1016 | Step::Descend(new_state, descent) => { |
| 1017 | // The one place the decoder descends a level. Every other part of reading a |
| 1018 | // character has already returned by now, so the frame that carries the |
| 1019 | // recursion holds only the store, the state and this daticle. |
| 1020 | let dat = res!(Self::recursive_decode( |
| 1021 | &mut iter, |
| 1022 | cfg, |
| 1023 | new_state, |
| 1024 | cursor, |
| 1025 | )); |
| 1026 | store.slurp = Slurp::new(); |
| 1027 | match descent { |
| 1028 | Descent::Kindicle if Self::kindicle_completes(&state) => return Ok(dat), |
| 1029 | _ => store.val_opt = Some(dat), |
| 1030 | } |
| 1031 | } |
| 1032 | } |
| 1033 | } // while loop |
| 1034 | Self::finish(&state, &mut store) |
| 1035 | } |
| 1036 | |
| 1037 | #[inline(never)] |
| 1038 | fn kindicle_completes(state: &DecoderState) -> bool { |
| 1039 | if state.kind_outer == Kind::Unknown { |
| 1040 | state.molecular_capture == None |
| 1041 | } else { |
| 1042 | state.kind_outer.case().class() == KindClass::Atomic |
| 1043 | } |
| 1044 | } |
| 1045 | |
| 1046 | #[inline(never)] |
| 1047 | fn step< |
| 1048 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 1049 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 1050 | >( |
| 1051 | (i, c): (usize, char), |
| 1052 | cfg: &DecoderConfig<M1, M2>, |
| 1053 | state: &mut DecoderState, |
| 1054 | store: &mut DecoderStore, |
| 1055 | cursor: &RefCell<Cursor>, |
| 1056 | comment_required: &mut bool, |
| 1057 | ) |
| 1058 | -> Outcome<Step> |
| 1059 | { |
| 1060 | { |
| 1061 | cursor.borrow_mut().advance(c, state.quote_protection != Quote::None); |
| 1062 | } |
| 1063 | // String escape handling (RFC 8259 §7). Only active |
| 1064 | // inside a quoted string. Runs ahead of the outer |
| 1065 | // `"` / `'` matching so that `\"` inside a string is |
| 1066 | // interpreted as an escape, not a string terminator. |
| 1067 | if state.quote_protection != Quote::None { |
| 1068 | if res!(Self::handle_string_escape(c, state, &mut store.slurp)) { |
| 1069 | return Ok(Step::Continue); |
| 1070 | } |
| 1071 | } |
| 1072 | if cfg.quote_protection { |
| 1073 | match c { |
| 1074 | '"' => { |
| 1075 | if state.quote_protection != Quote::Single { |
| 1076 | if state.quote_protection == Quote::Double { |
| 1077 | state.quote_protection = Quote::None; |
| 1078 | // Whenever quote protection is applied to some or all of a string, |
| 1079 | // the entire string is flagged as a STR kind. |
| 1080 | store.slurp.flag_as_string(); |
| 1081 | } else { |
| 1082 | state.quote_protection = Quote::Double; |
| 1083 | } |
| 1084 | return Ok(Step::Continue); |
| 1085 | } |
| 1086 | } |
| 1087 | '\'' => { |
| 1088 | if state.quote_protection != Quote::Double { |
| 1089 | if state.quote_protection == Quote::Single { |
| 1090 | state.quote_protection = Quote::None; |
| 1091 | // Whenever quote protection is applied to some or all of a string, |
| 1092 | // the entire string is flagged as a STR kind |
| 1093 | store.slurp.flag_as_string(); |
| 1094 | } else { |
| 1095 | state.quote_protection = Quote::Single; |
| 1096 | } |
| 1097 | return Ok(Step::Continue); |
| 1098 | } |
| 1099 | } |
| 1100 | _ => {} |
| 1101 | } |
| 1102 | } |
| 1103 | if state.comment_capture.is_some() { |
| 1104 | res!(Self::capture_comment(c, cfg, state, store, cursor)); |
| 1105 | return Ok(Step::Continue); |
| 1106 | } |
| 1107 | if cfg.comment_allowed && state.quote_protection == Quote::None { |
| 1108 | // Start comment capturing? |
| 1109 | if c == cfg.comment1_start_char { |
| 1110 | state.comment_capture = Some(CommentCapture::Type1); |
| 1111 | store.note_config = store.note_config.extract().set_type1(true); |
| 1112 | return Ok(Step::Continue); |
| 1113 | } |
| 1114 | if c == cfg.comment2_start_char { |
| 1115 | state.comment_capture = Some(CommentCapture::Type2); |
| 1116 | store.note_config = store.note_config.extract().set_type1(false); |
| 1117 | return Ok(Step::Continue); |
| 1118 | } |
| 1119 | } |
| 1120 | |
| 1121 | // Every character is first offered to the slurp, which takes it while a number, a quoted |
| 1122 | // string or a kind label is being gathered. A single call site here, rather than the same |
| 1123 | // call at the head of a dozen match arms, keeps the frame small. The separator '|' is the |
| 1124 | // exception, since a kind capture must see it and the slurp would refuse it. |
| 1125 | let whitespace = matches!(c, ' ' | '\t' | '\n' | '\r'); |
| 1126 | let slurped = if c == '|' { |
| 1127 | if state.quote_protection != Quote::None { |
| 1128 | store.slurp.push(c); |
| 1129 | true |
| 1130 | } else { |
| 1131 | false |
| 1132 | } |
| 1133 | } else { |
| 1134 | res!(store.slurp.char_slurped((i, c), state)) |
| 1135 | }; |
| 1136 | // Whitespace is offered to the slurp but never ends the character's handling, since a |
| 1137 | // space may also close an atom. |
| 1138 | if slurped && !whitespace { |
| 1139 | return Ok(Step::Continue); |
| 1140 | } |
| 1141 | |
| 1142 | match c { |
| 1143 | // JSON RFC 8259 §2 allows U+0009 tab as whitespace, so |
| 1144 | // JDAT accepts it alongside space / LF / CR. This lets |
| 1145 | // JSON files using tab indentation round-trip through |
| 1146 | // `Dat::decode_string` unchanged. |
| 1147 | ' ' | '\t' | '\n' | '\r' => (), // The slurp has taken it, if it wanted it. |
| 1148 | '(' => { |
| 1149 | if state.kind_outer == Kind::Unknown || |
| 1150 | state.kind_outer.case() == KindCase::MoleculeUnitary || |
| 1151 | state.molecular_capture != None |
| 1152 | { |
| 1153 | // Begin capturing the daticle kind (or "kindicle"). |
| 1154 | state.kind_capture = true; |
| 1155 | store.slurp = Slurp::new(); |
| 1156 | } else { |
| 1157 | // "(k1|(k2|v))" |
| 1158 | // ^ already expect k1 kind, invalid unless preceded by [ or { |
| 1159 | return Err(Self::open_paren_err(state, cursor)); |
| 1160 | } |
| 1161 | } |
| 1162 | '|' => { |
| 1163 | if state.kind_capture { |
| 1164 | // We have captured a kind, triggering a new level of recursion. |
| 1165 | let kind_inner = res!(Kind::from_label( |
| 1166 | &store.slurp.clone_string().to_lowercase(), |
| 1167 | cfg.ukinds_opt.as_ref(), |
| 1168 | )); |
| 1169 | if kind_inner.case() == KindCase::AtomLogic { |
| 1170 | // An unused '|' separator is not valid, e.g. (true|), (EMPTY|) |
| 1171 | return Err(Self::superfluous_bar_err(&kind_inner, cursor)); |
| 1172 | } |
| 1173 | state.kind_capture = false; |
| 1174 | let mut new_state = state.recurse(); |
| 1175 | new_state.kind_outer = kind_inner; |
| 1176 | new_state.explicit_kind = true; |
| 1177 | return Ok(Step::Descend(new_state, Descent::Kindicle)); |
| 1178 | } |
| 1179 | } |
| 1180 | ')' => { |
| 1181 | // Deal with atoms (e.g. (FALSE)), which should only |
| 1182 | // ever be processed within a recursion level. |
| 1183 | match res!(Self::close_paren(cfg, state, store, cursor, comment_required)) { |
| 1184 | Some(dat) => return Ok(Step::Done(dat)), |
| 1185 | None => (), // The daticle may yet be one item of a molecule. |
| 1186 | } |
| 1187 | } |
| 1188 | '[' => { |
| 1189 | if state.molecular_capture == None { |
| 1190 | // A `[` that is the whole molecular payload of a dataless user kind -- |
| 1191 | // `(node|[...])` -- resolves in this frame to a bare list, the list analogue |
| 1192 | // of `{` resolving `(node|{...})` to a bare map. Capture the list here and |
| 1193 | // reset the frame's kind to List so `close_bracket` builds it and terminates |
| 1194 | // at the matching `]`, leaving the enclosing `)` and any separators to the |
| 1195 | // parent frame. Descending as the ordinary branch below does would not |
| 1196 | // terminate the payload frame at `]`, so it would swallow the parent's |
| 1197 | // separators. Without this a `[` under a user kind raised "Found a '[' |
| 1198 | // which is incompatible". |
| 1199 | let dataless_usr = matches!( |
| 1200 | &state.kind_outer, Kind::Usr(ukid) if ukid.kind().is_none()); |
| 1201 | if state.kind_outer == Kind::Unknown { |
| 1202 | state.molecular_capture = Some(MolecularCapture::ListMixed); |
| 1203 | } else if dataless_usr { |
| 1204 | state.molecular_capture = Some(MolecularCapture::ListMixed); |
| 1205 | state.kind_outer = Kind::List; |
| 1206 | return Ok(Step::Continue); |
| 1207 | } else { |
| 1208 | match MolecularCapture::from_kind(&state.kind_outer) { |
| 1209 | Some(MolecularCapture::Map) | |
| 1210 | None => return Err(Self::open_bracket_err(state, cursor)), |
| 1211 | _ => (), |
| 1212 | } |
| 1213 | } |
| 1214 | } |
| 1215 | let mut new_state = state.recurse(); |
| 1216 | if !state.kind_outer.uses_list_brackets() { |
| 1217 | new_state.kind_outer = Kind::List; |
| 1218 | } |
| 1219 | return Ok(Step::Descend(new_state, Descent::Molecule)); |
| 1220 | } |
| 1221 | ',' => { |
| 1222 | res!(Self::comma_handler(cfg, state, cursor, store)); |
| 1223 | } |
| 1224 | ']' => { |
| 1225 | // A terminal branch, so the state and the store are given away. |
| 1226 | return Ok(Step::Done(res!(Self::close_bracket( |
| 1227 | state.extract(), |
| 1228 | store.extract(), |
| 1229 | cursor, |
| 1230 | )))); |
| 1231 | } |
| 1232 | '{' => { |
| 1233 | if state.molecular_capture == None { |
| 1234 | state.molecular_capture = Some(MolecularCapture::Map); |
| 1235 | // Also honour cfg.use_ordmaps at the top |
| 1236 | // level. Without this, a root `{...}` |
| 1237 | // decoded with `use_ordmaps = true` still |
| 1238 | // lands in a `Dat::Map` because kind_outer |
| 1239 | // was never set from Kind::Unknown -- only |
| 1240 | // the nested recurse branch below did it. |
| 1241 | if !state.explicit_kind && cfg.use_ordmaps { |
| 1242 | state.kind_outer = Kind::OrdMap; |
| 1243 | } |
| 1244 | } else { |
| 1245 | let mut new_state = state.recurse(); |
| 1246 | // A nested `{` value is a fresh map unless we are continuing an explicitly |
| 1247 | // declared map kind -- the `{` right after a `(MAP|` or `(OMAP|` kindicle. |
| 1248 | // When the parent carries a non-map explicit kind (a user kind acting as a |
| 1249 | // map, say), that kind belongs to the parent, not to this value, so the value |
| 1250 | // must be defined as a map here. Mirrors the `[` branch, which already resets |
| 1251 | // to `Kind::List`. Without this a nested map value inherits the parent's |
| 1252 | // non-map kind, leaving its own frame with no map capture. |
| 1253 | if !(state.explicit_kind && state.kind_outer.is_map()) { |
| 1254 | // We are free to define the kind of this store.map. |
| 1255 | match cfg.use_ordmaps { |
| 1256 | true => new_state.kind_outer = Kind::OrdMap, |
| 1257 | false => new_state.kind_outer = Kind::Map, |
| 1258 | } |
| 1259 | } |
| 1260 | return Ok(Step::Descend(new_state, Descent::Molecule)); |
| 1261 | } |
| 1262 | } |
| 1263 | ':' => { |
| 1264 | res!(Self::colon(state, store, cursor)); |
| 1265 | } |
| 1266 | '}' => { |
| 1267 | // A terminal branch, so the store is given away. |
| 1268 | return Ok(Step::Done(res!(Self::close_brace( |
| 1269 | cfg, |
| 1270 | state, |
| 1271 | store.extract(), |
| 1272 | cursor, |
| 1273 | )))); |
| 1274 | } |
| 1275 | _ => { |
| 1276 | store.slurp.push(c); |
| 1277 | } |
| 1278 | } // match |
| 1279 | Ok(Step::Continue) |
| 1280 | } |
| 1281 | |
| 1282 | #[inline(never)] |
| 1283 | fn finish( |
| 1284 | state: &DecoderState, |
| 1285 | store: &mut DecoderStore, |
| 1286 | ) |
| 1287 | -> Outcome<Self> |
| 1288 | { |
| 1289 | if let Some(dat) = store.val_opt.take() { |
| 1290 | return Ok(dat); |
| 1291 | } |
| 1292 | match state.molecular_capture { |
| 1293 | None => Self::process_atom(&mut store.slurp, &state.kind_outer), |
| 1294 | Some(MolecularCapture::ListMixed) | |
| 1295 | Some(MolecularCapture::ListSame) | |
| 1296 | Some(MolecularCapture::Bytes) => |
| 1297 | Err(err!("Expected closure of a store.list with ']'"; |
| 1298 | String, Input, Decode, Missing)), |
| 1299 | Some(MolecularCapture::Map) => |
| 1300 | Err(err!("Expected closure of a store.map with '}}'"; |
| 1301 | String, Input, Decode, Missing)), |
| 1302 | } |
| 1303 | } |
| 1304 | |
| 1305 | #[inline(never)] |
| 1306 | fn open_paren_err( |
| 1307 | state: &DecoderState, |
| 1308 | cursor: &RefCell<Cursor>, |
| 1309 | ) |
| 1310 | -> Error<ErrTag> |
| 1311 | { |
| 1312 | match state.kind_outer.case() { |
| 1313 | KindCase::MoleculeSame => err!( |
| 1314 | "Elements of a vector of kind {:?} are not daticles, \ |
| 1315 | so no kind should be specified ({})", |
| 1316 | state.kind_outer, cursor.borrow(); |
| 1317 | String, Input, Decode, Invalid), |
| 1318 | _ => err!( |
| 1319 | "The kind for the daticle has already been specified \ |
| 1320 | as {:?} ({})", state.kind_outer, cursor.borrow(); |
| 1321 | String, Input, Decode, Invalid), |
| 1322 | } |
| 1323 | } |
| 1324 | |
| 1325 | #[inline(never)] |
| 1326 | fn superfluous_bar_err( |
| 1327 | kind_inner: &Kind, |
| 1328 | cursor: &RefCell<Cursor>, |
| 1329 | ) |
| 1330 | -> Error<ErrTag> |
| 1331 | { |
| 1332 | err!( |
| 1333 | "The separation character \"|\" for {:?} is superfluous {}.", |
| 1334 | kind_inner, cursor.borrow(); |
| 1335 | String, Input, Decode, Invalid) |
| 1336 | } |
| 1337 | |
| 1338 | #[inline(never)] |
| 1339 | fn open_bracket_err( |
| 1340 | state: &DecoderState, |
| 1341 | cursor: &RefCell<Cursor>, |
| 1342 | ) |
| 1343 | -> Error<ErrTag> |
| 1344 | { |
| 1345 | match MolecularCapture::from_kind(&state.kind_outer) { |
| 1346 | Some(MolecularCapture::Map) => err!( |
| 1347 | "Expecting a store.map bracket '{{' but found a '[' ({})", cursor.borrow(); |
| 1348 | String, Input, Decode, Invalid), |
| 1349 | _ => err!( |
| 1350 | "Found a '[' which is incompatible with a {:?} ({})", |
| 1351 | state.kind_outer, cursor.borrow(); |
| 1352 | String, Input, Decode, Invalid), |
| 1353 | } |
| 1354 | } |
| 1355 | |
| 1356 | #[inline(never)] |
| 1357 | fn capture_comment< |
| 1358 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 1359 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 1360 | >( |
| 1361 | c: char, |
| 1362 | cfg: &DecoderConfig<M1, M2>, |
| 1363 | state: &mut DecoderState, |
| 1364 | store: &mut DecoderStore, |
| 1365 | cursor: &RefCell<Cursor>, |
| 1366 | ) |
| 1367 | -> Outcome<()> |
| 1368 | { |
| 1369 | let capturing = match &state.comment_capture { |
| 1370 | Some(capturing) => capturing.clone(), |
| 1371 | None => return Ok(()), |
| 1372 | }; |
| 1373 | // Finish comment capturing? |
| 1374 | if ( |
| 1375 | capturing == CommentCapture::Type1 |
| 1376 | && (c == cfg.comment1_end_char || c == '\n') |
| 1377 | ) || ( |
| 1378 | capturing == CommentCapture::Type2 |
| 1379 | && (c == cfg.comment2_end_char || c == '\n') |
| 1380 | ) { |
| 1381 | state.comment_capture = None; |
| 1382 | store.comment = store.comment.trim_start().to_string(); |
| 1383 | if c == '\n' { |
| 1384 | if let Some(molecular_capture) = &state.molecular_capture { |
| 1385 | let mut dat = match store.val_opt.take() { |
| 1386 | Some(dat) => dat, // store.val_opt is now None. |
| 1387 | None => { |
| 1388 | let dat = if store.slurp.has_content() { |
| 1389 | res!(Self::process_atom(&mut store.slurp, &Kind::Unknown)) |
| 1390 | } else { |
| 1391 | Dat::Empty |
| 1392 | }; |
| 1393 | dat |
| 1394 | } |
| 1395 | }; |
| 1396 | dat = Dat::ABox( |
| 1397 | store.note_config.extract(), |
| 1398 | Box::new(dat), |
| 1399 | store.comment.extract(), |
| 1400 | ); |
| 1401 | store.slurp = Slurp::new(); |
| 1402 | match molecular_capture { |
| 1403 | MolecularCapture::Map => { |
| 1404 | match store.key_opt.take() { |
| 1405 | Some(key) => { |
| 1406 | res!(Self::map_insert( |
| 1407 | state.kind_outer == Kind::OrdMap, |
| 1408 | store, |
| 1409 | (key, dat), |
| 1410 | )); |
| 1411 | } |
| 1412 | None => { |
| 1413 | res!(Self::map_insert( |
| 1414 | state.kind_outer == Kind::OrdMap, |
| 1415 | store, |
| 1416 | (dat, Dat::Empty), |
| 1417 | )); |
| 1418 | } |
| 1419 | } |
| 1420 | } |
| 1421 | _ => { |
| 1422 | store.list.push(dat); |
| 1423 | } |
| 1424 | } |
| 1425 | } else { |
| 1426 | return Err(err!( |
| 1427 | "Line comments currently only allowed in molecules \ |
| 1428 | (lists and maps). ({})", cursor.borrow(); |
| 1429 | String, Input, Decode, Invalid)); |
| 1430 | } |
| 1431 | } |
| 1432 | return Ok(()); |
| 1433 | } |
| 1434 | if cfg.comment_capture { |
| 1435 | store.comment.push(c); |
| 1436 | } |
| 1437 | Ok(()) |
| 1438 | } |
| 1439 | |
| 1440 | #[inline(never)] |
| 1441 | fn close_paren< |
| 1442 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 1443 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 1444 | >( |
| 1445 | cfg: &DecoderConfig<M1, M2>, |
| 1446 | state: &mut DecoderState, |
| 1447 | store: &mut DecoderStore, |
| 1448 | cursor: &RefCell<Cursor>, |
| 1449 | comment_required: &mut bool, |
| 1450 | ) |
| 1451 | -> Outcome<Option<Self>> |
| 1452 | { |
| 1453 | // A `)` closing a `(node|{...})` or `(node|[...])` value that sits inside a PLAIN map -- |
| 1454 | // the wrapper's molecular payload decoded a level down and was left in `val_opt` by the |
| 1455 | // descend, and the payload's own `}` / `]` terminated the child frame, leaving this `)` |
| 1456 | // for the map frame. A dataless user-kind value frame does not descend its own `{` (its |
| 1457 | // `MolecularCapture::from_kind` is None), so it ends at the payload's `}` and hands the |
| 1458 | // `)` up; a map-kind value frame -- `(omap|{...})` -- does descend and consumes its own |
| 1459 | // `)`, so it never reaches here. The payload resolves to a bare map or list (the wrapper |
| 1460 | // is dropped, exactly as a top-level `(node|{...})` does), so keep it pending: the |
| 1461 | // enclosing map commits it on the next `,` or `}`. Without this the `kind == Unknown` |
| 1462 | // path below rejects the `)` for lack of a ListMixed capture ("should have triggered ... |
| 1463 | // Some(Map)"). The guard is confined to a genuine plain-map accumulator (`!explicit_kind`) |
| 1464 | // so it never fires in an explicit map-kind or user-kind value frame, whose own `)` the |
| 1465 | // existing logic below still terminates on. |
| 1466 | if !state.kind_capture |
| 1467 | && !state.explicit_kind |
| 1468 | && state.molecular_capture == Some(MolecularCapture::Map) |
| 1469 | && store.val_opt.is_some() |
| 1470 | { |
| 1471 | return Ok(None); |
| 1472 | } |
| 1473 | let kind_inner = if state.kind_capture { |
| 1474 | // We were capturing a kind, but the daticle finished before |
| 1475 | // | and no data was found. This is valid for AtomLogic |
| 1476 | // cases and usr kinds. Set the inner kind. |
| 1477 | let slurped = store.slurp.clone_string().to_lowercase(); |
| 1478 | let kind_inner = if slurped.len() == 0 { |
| 1479 | // First take care of case "()". |
| 1480 | Kind::Empty |
| 1481 | } else { |
| 1482 | // (EMPTY), (TRUE), (FALSE), (NONE), (my_kind), etc. are valid. |
| 1483 | let k = res!(Kind::from_label(&slurped, cfg.ukinds_opt.as_ref())); |
| 1484 | if !k.is_dataless() { |
| 1485 | return Err(err!( |
| 1486 | "Closing ')' without a kindicle should have triggered \ |
| 1487 | set MolecularCapture::ListMixed, but instead the state is \ |
| 1488 | {:?}. ({})", state.molecular_capture, cursor.borrow(); |
| 1489 | String, Input, Decode, Invalid)); |
| 1490 | } |
| 1491 | k |
| 1492 | }; |
| 1493 | state.kind_capture = false; |
| 1494 | store.slurp.reset(); |
| 1495 | kind_inner |
| 1496 | } else { |
| 1497 | Kind::Unknown |
| 1498 | }; |
| 1499 | let mut kind = kind_inner.clone(); |
| 1500 | if kind == Kind::Unknown { |
| 1501 | kind = state.kind_outer.clone(); |
| 1502 | } |
| 1503 | |
| 1504 | if kind == Kind::Unknown { |
| 1505 | // For unknown kinds, first attempt to interpret as a tuple. |
| 1506 | // When no kindicle is specified, items accumulate directly in store.list |
| 1507 | match state.molecular_capture { |
| 1508 | Some(MolecularCapture::ListMixed) => {} |
| 1509 | _ => { |
| 1510 | return Err(err!( |
| 1511 | "Closing ')' without a kindicle should have triggered \ |
| 1512 | set MolecularCapture::ListMixed, but instead the state is \ |
| 1513 | {:?}. ({})", state.molecular_capture, cursor.borrow(); |
| 1514 | String, Input, Decode, Invalid)); |
| 1515 | } |
| 1516 | } |
| 1517 | // Complete the capture of the store.list by converting it to the correct |
| 1518 | // type. |
| 1519 | if store.slurp.has_content() { |
| 1520 | store.val_opt = Some(res!(Self::process_atom(&mut store.slurp, &kind))); |
| 1521 | } |
| 1522 | match store.val_opt.take() { |
| 1523 | Some(dat) => { |
| 1524 | if store.comment.len() > 0 { |
| 1525 | store.list.push(Dat::ABox( |
| 1526 | store.note_config.extract(), |
| 1527 | Box::new(dat), |
| 1528 | store.comment.extract(), |
| 1529 | )); |
| 1530 | } else { |
| 1531 | store.list.push(dat); |
| 1532 | } |
| 1533 | } |
| 1534 | None => { |
| 1535 | if store.comment.len() > 0 { |
| 1536 | store.list.push(Dat::ABox( |
| 1537 | store.note_config.extract(), |
| 1538 | Box::new(Dat::Empty), |
| 1539 | store.comment.extract(), |
| 1540 | )); |
| 1541 | } |
| 1542 | } |
| 1543 | } |
| 1544 | return Ok(Some(res!(Self::implicit_tuple(store.list.extract(), cursor)))); |
| 1545 | } |
| 1546 | |
| 1547 | match &kind { |
| 1548 | // No data is needed for these kinds, so create them here. |
| 1549 | Kind::Empty => store.val_opt = Some(Dat::Empty), |
| 1550 | Kind::True => store.val_opt = Some(Dat::Bool(true)), |
| 1551 | Kind::False => store.val_opt = Some(Dat::Bool(false)), |
| 1552 | Kind::None => store.val_opt = Some(Dat::Opt(Box::new(None))), |
| 1553 | // A dataless user kind with no payload is `(node)`. But if a molecular payload |
| 1554 | // has already been captured into `val_opt` by a nested descend -- a `(node|{...})` |
| 1555 | // value, whose `{...}` decoded a level down and landed here -- keep it: overwriting |
| 1556 | // with `None` silently drops that payload. The kept payload is wrapped (mol == None) |
| 1557 | // or committed to the enclosing molecule (mol == Some) by the logic below. |
| 1558 | Kind::Usr(ukid) if ukid.kind().is_none() && store.val_opt.is_none() => |
| 1559 | store.val_opt = Some(Dat::Usr(ukid.clone(), None)), |
| 1560 | _ => (), |
| 1561 | // We don't necessarily return out of the method here because the |
| 1562 | // daticle might be part of a molecule, with the exception of the outer |
| 1563 | // kind being a Dat::ABox. |
| 1564 | } |
| 1565 | if let Kind::ABox(_) = state.kind_outer { |
| 1566 | match &store.val_opt { |
| 1567 | Some(dat) => { |
| 1568 | if store.comment.len() > 0 { |
| 1569 | return Ok(Some(Dat::ABox( |
| 1570 | store.note_config.extract(), |
| 1571 | Box::new(dat.clone()), |
| 1572 | store.comment.extract(), |
| 1573 | ))); |
| 1574 | } else { |
| 1575 | *comment_required = true; |
| 1576 | } |
| 1577 | } |
| 1578 | None => { |
| 1579 | return Err(err!( |
| 1580 | "Daticle missing in Dat::ABox ({})", cursor.borrow(); |
| 1581 | String, Input, Decode, Invalid, Missing)); |
| 1582 | } |
| 1583 | } |
| 1584 | } |
| 1585 | |
| 1586 | if !*comment_required && ( |
| 1587 | !kind.is_dataless() || state.molecular_capture == None |
| 1588 | ) { |
| 1589 | // Atom point cases were just dealt with, while molecular capture is |
| 1590 | // ongoing until a `]` or `}` is encountered, with the exception of a |
| 1591 | // base2x string for bytes. |
| 1592 | if state.molecular_capture == None || ( |
| 1593 | state.molecular_capture == Some(MolecularCapture::Bytes) |
| 1594 | && store.slurp.is_string() |
| 1595 | ) { |
| 1596 | let d = match store.val_opt.take() { |
| 1597 | Some(d) => d, |
| 1598 | None => res!(Self::process_atom(&mut store.slurp, &kind)), |
| 1599 | }; |
| 1600 | return Ok(Some(match state.kind_outer.clone() { |
| 1601 | // Unitary molecules |
| 1602 | Kind::Usr(ukid) => Dat::Usr(ukid, Some(Box::new(d))), |
| 1603 | Kind::Box(_) => Dat::Box(Box::new(d)), |
| 1604 | Kind::Some(_) => Dat::Opt(Box::new(Some(d))), |
| 1605 | _ => d, |
| 1606 | })); |
| 1607 | } else { |
| 1608 | match store.val_opt.take() { |
| 1609 | Some(d) => return Ok(Some(d)), |
| 1610 | None => { |
| 1611 | return Err(err!( |
| 1612 | "Failed to capture the daticle of kind {:?} ({})", |
| 1613 | kind, cursor.borrow(); |
| 1614 | String, Input, Decode, Invalid)); |
| 1615 | } |
| 1616 | } |
| 1617 | } |
| 1618 | } |
| 1619 | Ok(None) |
| 1620 | } |
| 1621 | |
| 1622 | #[inline(never)] |
| 1623 | fn implicit_tuple( |
| 1624 | mut list: Vec<Dat>, |
| 1625 | cursor: &RefCell<Cursor>, |
| 1626 | ) |
| 1627 | -> Outcome<Self> |
| 1628 | { |
| 1629 | match list.len() { |
| 1630 | 0 => Ok(Dat::Empty), |
| 1631 | 1 => Ok(list.remove(0)), |
| 1632 | 2 => Ok(Dat::Tup2(Box::new( |
| 1633 | res!(list.try_into().map_err(|_| err!( |
| 1634 | "While decoding a 2-item tuple ({})", cursor.borrow(); |
| 1635 | String, Input, Decode, Invalid))) |
| 1636 | ))), |
| 1637 | 3 => Ok(Dat::Tup3(Box::new( |
| 1638 | res!(list.try_into().map_err(|_| err!( |
| 1639 | "While decoding a 3-item tuple ({})", cursor.borrow(); |
| 1640 | String, Input, Decode, Invalid))) |
| 1641 | ))), |
| 1642 | 4 => Ok(Dat::Tup4(Box::new( |
| 1643 | res!(list.try_into().map_err(|_| err!( |
| 1644 | "While decoding a 4-item tuple ({})", cursor.borrow(); |
| 1645 | String, Input, Decode, Invalid))) |
| 1646 | ))), |
| 1647 | 5 => Ok(Dat::Tup5(Box::new( |
| 1648 | res!(list.try_into().map_err(|_| err!( |
| 1649 | "While decoding a 5-item tuple ({})", cursor.borrow(); |
| 1650 | String, Input, Decode, Invalid))) |
| 1651 | ))), |
| 1652 | 6 => Ok(Dat::Tup6(Box::new( |
| 1653 | res!(list.try_into().map_err(|_| err!( |
| 1654 | "While decoding a 6-item tuple ({})", cursor.borrow(); |
| 1655 | String, Input, Decode, Invalid))) |
| 1656 | ))), |
| 1657 | 7 => Ok(Dat::Tup7(Box::new( |
| 1658 | res!(list.try_into().map_err(|_| err!( |
| 1659 | "While decoding a 7-item tuple ({})", cursor.borrow(); |
| 1660 | String, Input, Decode, Invalid))) |
| 1661 | ))), |
| 1662 | 8 => Ok(Dat::Tup8(Box::new( |
| 1663 | res!(list.try_into().map_err(|_| err!( |
| 1664 | "While decoding a 8-item tuple ({})", cursor.borrow(); |
| 1665 | String, Input, Decode, Invalid))) |
| 1666 | ))), |
| 1667 | 9 => Ok(Dat::Tup9(Box::new( |
| 1668 | res!(list.try_into().map_err(|_| err!( |
| 1669 | "While decoding a 9-item tuple ({})", cursor.borrow(); |
| 1670 | String, Input, Decode, Invalid))) |
| 1671 | ))), |
| 1672 | 10 => Ok(Dat::Tup10(Box::new( |
| 1673 | res!(list.try_into().map_err(|_| err!( |
| 1674 | "While decoding a 10-item tuple ({})", cursor.borrow(); |
| 1675 | String, Input, Decode, Invalid))) |
| 1676 | ))), |
| 1677 | n => Err(err!( |
| 1678 | "Tuples are limited to 10 items, {} found ({}).", |
| 1679 | n, cursor.borrow(); |
| 1680 | String, Input, Decode, Invalid)), |
| 1681 | } |
| 1682 | } |
| 1683 | |
| 1684 | #[inline(never)] |
| 1685 | fn close_bracket( |
| 1686 | mut state: DecoderState, |
| 1687 | mut store: DecoderStore, |
| 1688 | cursor: &RefCell<Cursor>, |
| 1689 | ) |
| 1690 | -> Outcome<Self> |
| 1691 | { |
| 1692 | match state.molecular_capture { |
| 1693 | Some(MolecularCapture::Bytes) => { |
| 1694 | if store.slurp.has_content() { |
| 1695 | let n = try_extract_dat!( |
| 1696 | res!(Self::process_atom(&mut store.slurp, &Kind::U8)), U8, |
| 1697 | ); |
| 1698 | store.byts.push(n); |
| 1699 | } |
| 1700 | return Self::decode_bytes(store.byts, &state.kind_outer); |
| 1701 | } |
| 1702 | Some(MolecularCapture::ListMixed) | |
| 1703 | Some(MolecularCapture::ListSame) => { |
| 1704 | // Complete the capture of the store.list by converting it to the correct |
| 1705 | // type. |
| 1706 | if store.slurp.has_content() { |
| 1707 | let kind = match state.molecular_capture { |
| 1708 | Some(MolecularCapture::ListSame) => |
| 1709 | MolecularCapture::same_kind(&state.kind_outer), |
| 1710 | _ => state.kind_outer.clone(), |
| 1711 | }; |
| 1712 | store.val_opt = Some(res!(Self::process_atom(&mut store.slurp, &kind))); |
| 1713 | } |
| 1714 | match store.val_opt { |
| 1715 | Some(dat) => { |
| 1716 | if store.comment.len() > 0 { |
| 1717 | store.list.push(Dat::ABox( |
| 1718 | store.note_config.clone(), |
| 1719 | Box::new(dat), |
| 1720 | store.comment, |
| 1721 | )); |
| 1722 | } else { |
| 1723 | store.list.push(dat); |
| 1724 | } |
| 1725 | } |
| 1726 | None => { |
| 1727 | if store.comment.len() > 0 { |
| 1728 | store.list.push(Dat::ABox( |
| 1729 | store.note_config.clone(), |
| 1730 | Box::new(Dat::Empty), |
| 1731 | store.comment, |
| 1732 | )); |
| 1733 | } |
| 1734 | } |
| 1735 | } |
| 1736 | state.molecular_capture = None; |
| 1737 | match state.kind_outer { |
| 1738 | Kind::List | Kind::Unknown => return Ok(Dat::List(store.list)), |
| 1739 | |
| 1740 | Kind::Tup2 => string_decode_heterogenous_tuple! { Tup2, 2, store.list, state }, |
| 1741 | Kind::Tup3 => string_decode_heterogenous_tuple! { Tup3, 3, store.list, state }, |
| 1742 | Kind::Tup4 => string_decode_heterogenous_tuple! { Tup4, 4, store.list, state }, |
| 1743 | Kind::Tup5 => string_decode_heterogenous_tuple! { Tup5, 5, store.list, state }, |
| 1744 | Kind::Tup6 => string_decode_heterogenous_tuple! { Tup6, 6, store.list, state }, |
| 1745 | Kind::Tup7 => string_decode_heterogenous_tuple! { Tup7, 7, store.list, state }, |
| 1746 | Kind::Tup8 => string_decode_heterogenous_tuple! { Tup8, 8, store.list, state }, |
| 1747 | Kind::Tup9 => string_decode_heterogenous_tuple! { Tup9, 9, store.list, state }, |
| 1748 | Kind::Tup10 => string_decode_heterogenous_tuple! { Tup10, 10, store.list, state }, |
| 1749 | |
| 1750 | Kind::Vek => return Ok(Dat::Vek(res!(Vek::try_from(store.list)))), |
| 1751 | Kind::Tup2u8 => string_decode_int_tuple! { Tup2u8, U8, u8, 2, store.list, state }, |
| 1752 | Kind::Tup3u8 => string_decode_int_tuple! { Tup3u8, U8, u8, 3, store.list, state }, |
| 1753 | Kind::Tup4u8 => string_decode_int_tuple! { Tup4u8, U8, u8, 4, store.list, state }, |
| 1754 | Kind::Tup5u8 => string_decode_int_tuple! { Tup5u8, U8, u8, 5, store.list, state }, |
| 1755 | Kind::Tup6u8 => string_decode_int_tuple! { Tup6u8, U8, u8, 6, store.list, state }, |
| 1756 | Kind::Tup7u8 => string_decode_int_tuple! { Tup7u8, U8, u8, 7, store.list, state }, |
| 1757 | Kind::Tup8u8 => string_decode_int_tuple! { Tup8u8, U8, u8, 8, store.list, state }, |
| 1758 | Kind::Tup9u8 => string_decode_int_tuple! { Tup9u8, U8, u8, 9, store.list, state }, |
| 1759 | Kind::Tup10u8 => string_decode_int_tuple! { Tup10u8, U8, u8, 10, store.list, state }, |
| 1760 | |
| 1761 | Kind::Tup2u16 => string_decode_int_tuple! { Tup2u16, U16, u16, 2, store.list, state }, |
| 1762 | Kind::Tup3u16 => string_decode_int_tuple! { Tup3u16, U16, u16, 3, store.list, state }, |
| 1763 | Kind::Tup4u16 => string_decode_int_tuple! { Tup4u16, U16, u16, 4, store.list, state }, |
| 1764 | Kind::Tup5u16 => string_decode_int_tuple! { Tup5u16, U16, u16, 5, store.list, state }, |
| 1765 | Kind::Tup6u16 => string_decode_int_tuple! { Tup6u16, U16, u16, 6, store.list, state }, |
| 1766 | Kind::Tup7u16 => string_decode_int_tuple! { Tup7u16, U16, u16, 7, store.list, state }, |
| 1767 | Kind::Tup8u16 => string_decode_int_tuple! { Tup8u16, U16, u16, 8, store.list, state }, |
| 1768 | Kind::Tup9u16 => string_decode_int_tuple! { Tup9u16, U16, u16, 9, store.list, state }, |
| 1769 | Kind::Tup10u16 => string_decode_int_tuple! { Tup10u16, U16, u16, 10, store.list, state }, |
| 1770 | |
| 1771 | Kind::Tup2u32 => string_decode_int_tuple! { Tup2u32, U32, u32, 2, store.list, state }, |
| 1772 | Kind::Tup3u32 => string_decode_int_tuple! { Tup3u32, U32, u32, 3, store.list, state }, |
| 1773 | Kind::Tup4u32 => string_decode_int_tuple! { Tup4u32, U32, u32, 4, store.list, state }, |
| 1774 | Kind::Tup5u32 => string_decode_int_tuple! { Tup5u32, U32, u32, 5, store.list, state }, |
| 1775 | Kind::Tup6u32 => string_decode_int_tuple! { Tup6u32, U32, u32, 6, store.list, state }, |
| 1776 | Kind::Tup7u32 => string_decode_int_tuple! { Tup7u32, U32, u32, 7, store.list, state }, |
| 1777 | Kind::Tup8u32 => string_decode_int_tuple! { Tup8u32, U32, u32, 8, store.list, state }, |
| 1778 | Kind::Tup9u32 => string_decode_int_tuple! { Tup9u32, U32, u32, 9, store.list, state }, |
| 1779 | Kind::Tup10u32 => string_decode_int_tuple! { Tup10u32, U32, u32, 10, store.list, state }, |
| 1780 | |
| 1781 | Kind::Tup2u64 => string_decode_int_tuple! { Tup2u64, U64, u64, 2, store.list, state }, |
| 1782 | Kind::Tup3u64 => string_decode_int_tuple! { Tup3u64, U64, u64, 3, store.list, state }, |
| 1783 | Kind::Tup4u64 => string_decode_int_tuple! { Tup4u64, U64, u64, 4, store.list, state }, |
| 1784 | Kind::Tup5u64 => string_decode_int_tuple! { Tup5u64, U64, u64, 5, store.list, state }, |
| 1785 | Kind::Tup6u64 => string_decode_int_tuple! { Tup6u64, U64, u64, 6, store.list, state }, |
| 1786 | Kind::Tup7u64 => string_decode_int_tuple! { Tup7u64, U64, u64, 7, store.list, state }, |
| 1787 | Kind::Tup8u64 => string_decode_int_tuple! { Tup8u64, U64, u64, 8, store.list, state }, |
| 1788 | Kind::Tup9u64 => string_decode_int_tuple! { Tup9u64, U64, u64, 9, store.list, state }, |
| 1789 | Kind::Tup10u64 => string_decode_int_tuple! { Tup10u64, U64, u64, 10, store.list, state }, |
| 1790 | Kind::Tup2i8 => string_decode_int_tuple! { Tup2i8, I8, i8, 2, store.list, state }, |
| 1791 | Kind::Tup3i8 => string_decode_int_tuple! { Tup3i8, I8, i8, 3, store.list, state }, |
| 1792 | Kind::Tup4i8 => string_decode_int_tuple! { Tup4i8, I8, i8, 4, store.list, state }, |
| 1793 | Kind::Tup5i8 => string_decode_int_tuple! { Tup5i8, I8, i8, 5, store.list, state }, |
| 1794 | Kind::Tup6i8 => string_decode_int_tuple! { Tup6i8, I8, i8, 6, store.list, state }, |
| 1795 | Kind::Tup7i8 => string_decode_int_tuple! { Tup7i8, I8, i8, 7, store.list, state }, |
| 1796 | Kind::Tup8i8 => string_decode_int_tuple! { Tup8i8, I8, i8, 8, store.list, state }, |
| 1797 | Kind::Tup9i8 => string_decode_int_tuple! { Tup9i8, I8, i8, 9, store.list, state }, |
| 1798 | Kind::Tup10i8 => string_decode_int_tuple! { Tup10i8, I8, i8, 10, store.list, state }, |
| 1799 | Kind::Tup2i16 => string_decode_int_tuple! { Tup2i16, I16, i16, 2, store.list, state }, |
| 1800 | Kind::Tup3i16 => string_decode_int_tuple! { Tup3i16, I16, i16, 3, store.list, state }, |
| 1801 | Kind::Tup4i16 => string_decode_int_tuple! { Tup4i16, I16, i16, 4, store.list, state }, |
| 1802 | Kind::Tup5i16 => string_decode_int_tuple! { Tup5i16, I16, i16, 5, store.list, state }, |
| 1803 | Kind::Tup6i16 => string_decode_int_tuple! { Tup6i16, I16, i16, 6, store.list, state }, |
| 1804 | Kind::Tup7i16 => string_decode_int_tuple! { Tup7i16, I16, i16, 7, store.list, state }, |
| 1805 | Kind::Tup8i16 => string_decode_int_tuple! { Tup8i16, I16, i16, 8, store.list, state }, |
| 1806 | Kind::Tup9i16 => string_decode_int_tuple! { Tup9i16, I16, i16, 9, store.list, state }, |
| 1807 | Kind::Tup10i16 => string_decode_int_tuple! { Tup10i16, I16, i16, 10, store.list, state }, |
| 1808 | Kind::Tup2i32 => string_decode_int_tuple! { Tup2i32, I32, i32, 2, store.list, state }, |
| 1809 | Kind::Tup3i32 => string_decode_int_tuple! { Tup3i32, I32, i32, 3, store.list, state }, |
| 1810 | Kind::Tup4i32 => string_decode_int_tuple! { Tup4i32, I32, i32, 4, store.list, state }, |
| 1811 | Kind::Tup5i32 => string_decode_int_tuple! { Tup5i32, I32, i32, 5, store.list, state }, |
| 1812 | Kind::Tup6i32 => string_decode_int_tuple! { Tup6i32, I32, i32, 6, store.list, state }, |
| 1813 | Kind::Tup7i32 => string_decode_int_tuple! { Tup7i32, I32, i32, 7, store.list, state }, |
| 1814 | Kind::Tup8i32 => string_decode_int_tuple! { Tup8i32, I32, i32, 8, store.list, state }, |
| 1815 | Kind::Tup9i32 => string_decode_int_tuple! { Tup9i32, I32, i32, 9, store.list, state }, |
| 1816 | Kind::Tup10i32 => string_decode_int_tuple! { Tup10i32, I32, i32, 10, store.list, state }, |
| 1817 | Kind::Tup2i64 => string_decode_int_tuple! { Tup2i64, I64, i64, 2, store.list, state }, |
| 1818 | Kind::Tup3i64 => string_decode_int_tuple! { Tup3i64, I64, i64, 3, store.list, state }, |
| 1819 | Kind::Tup4i64 => string_decode_int_tuple! { Tup4i64, I64, i64, 4, store.list, state }, |
| 1820 | Kind::Tup5i64 => string_decode_int_tuple! { Tup5i64, I64, i64, 5, store.list, state }, |
| 1821 | Kind::Tup6i64 => string_decode_int_tuple! { Tup6i64, I64, i64, 6, store.list, state }, |
| 1822 | Kind::Tup7i64 => string_decode_int_tuple! { Tup7i64, I64, i64, 7, store.list, state }, |
| 1823 | Kind::Tup8i64 => string_decode_int_tuple! { Tup8i64, I64, i64, 8, store.list, state }, |
| 1824 | Kind::Tup9i64 => string_decode_int_tuple! { Tup9i64, I64, i64, 9, store.list, state }, |
| 1825 | Kind::Tup10i64 => string_decode_int_tuple! { Tup10i64, I64, i64, 10, store.list, state }, |
| 1826 | |
| 1827 | _ => { |
| 1828 | return Err(err!( |
| 1829 | "Unexpected outer kind {:?} case while concluding {:?}.", |
| 1830 | state.kind_outer, state.molecular_capture; |
| 1831 | Input, Mismatch, Unexpected, Bug)); |
| 1832 | } |
| 1833 | } |
| 1834 | } |
| 1835 | _ => { |
| 1836 | return Err(err!( |
| 1837 | "List capture was not actived with a '[' character ({})", |
| 1838 | cursor.borrow(); |
| 1839 | String, Input, Decode, Invalid)); |
| 1840 | } |
| 1841 | } |
| 1842 | } |
| 1843 | |
| 1844 | #[inline(never)] |
| 1845 | fn colon( |
| 1846 | state: &DecoderState, |
| 1847 | store: &mut DecoderStore, |
| 1848 | cursor: &RefCell<Cursor>, |
| 1849 | ) |
| 1850 | -> Outcome<()> |
| 1851 | { |
| 1852 | match state.molecular_capture { |
| 1853 | None => return Err(err!( |
| 1854 | "Map capture not active ({})", cursor.borrow(); |
| 1855 | String, Input, Decode, Invalid)), |
| 1856 | Some(MolecularCapture::ListMixed) | |
| 1857 | Some(MolecularCapture::ListSame) | |
| 1858 | Some(MolecularCapture::Bytes) => { |
| 1859 | return Err(err!( |
| 1860 | "List capture active, incompatible character ({})", cursor.borrow(); |
| 1861 | String, Input, Decode, Invalid)); |
| 1862 | } |
| 1863 | Some(MolecularCapture::Map) => { |
| 1864 | // We're expecting to have a daticle to add to the store.map. |
| 1865 | let mut dat = match store.val_opt.take() { |
| 1866 | Some(dat) => dat, |
| 1867 | None => { |
| 1868 | if store.slurp.has_content() { |
| 1869 | res!(Self::process_atom(&mut store.slurp, &Kind::Unknown)) |
| 1870 | } else { |
| 1871 | Dat::Empty |
| 1872 | } |
| 1873 | } |
| 1874 | }; |
| 1875 | if store.comment.len() > 0 { |
| 1876 | dat = Dat::ABox( |
| 1877 | store.note_config.extract(), |
| 1878 | Box::new(dat), |
| 1879 | store.comment.extract(), |
| 1880 | ); |
| 1881 | } |
| 1882 | if store.key_opt == None { |
| 1883 | store.key_opt = Some(dat); |
| 1884 | store.val_opt = None; |
| 1885 | } |
| 1886 | store.slurp = Slurp::new(); |
| 1887 | } |
| 1888 | } |
| 1889 | Ok(()) |
| 1890 | } |
| 1891 | |
| 1892 | #[inline(never)] |
| 1893 | fn close_brace< |
| 1894 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 1895 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 1896 | >( |
| 1897 | cfg: &DecoderConfig<M1, M2>, |
| 1898 | state: &DecoderState, |
| 1899 | mut store: DecoderStore, |
| 1900 | cursor: &RefCell<Cursor>, |
| 1901 | ) |
| 1902 | -> Outcome<Self> |
| 1903 | { |
| 1904 | if state.molecular_capture == Some(MolecularCapture::Map) { |
| 1905 | if store.slurp.has_content() { |
| 1906 | store.val_opt = |
| 1907 | Some(res!(Self::process_atom(&mut store.slurp, &state.kind_outer))); |
| 1908 | } |
| 1909 | match (store.key_opt.take(), store.val_opt.take()) { |
| 1910 | (Some(key), Some(mut dat)) => { |
| 1911 | // The key has already been abox-wrapped if necessary. |
| 1912 | if store.comment.len() > 0 { |
| 1913 | dat = Dat::ABox( |
| 1914 | store.note_config.extract(), |
| 1915 | Box::new(dat), |
| 1916 | store.comment.extract(), |
| 1917 | ); |
| 1918 | } |
| 1919 | res!(Self::map_insert( |
| 1920 | state.kind_outer == Kind::OrdMap, |
| 1921 | &mut store, |
| 1922 | (key, dat), |
| 1923 | )); |
| 1924 | } |
| 1925 | (None, Some(dat)) => { |
| 1926 | return Err(err!( |
| 1927 | "Unpaired value {:?} at end of store.map {} ({})", |
| 1928 | dat, match cfg.use_ordmaps { |
| 1929 | true => fmt!("{:?}", store.ordmap), |
| 1930 | false => fmt!("{:?}", store.map), |
| 1931 | }, cursor.borrow(); |
| 1932 | String, Input, Decode, Invalid, Missing)); |
| 1933 | } |
| 1934 | (Some(key), None) => { |
| 1935 | if store.comment.len() > 0 { |
| 1936 | let dat = Dat::ABox( |
| 1937 | store.note_config.extract(), |
| 1938 | Box::new(Dat::Empty), |
| 1939 | store.comment.extract(), |
| 1940 | ); |
| 1941 | res!(Self::map_insert( |
| 1942 | state.kind_outer == Kind::OrdMap, |
| 1943 | &mut store, |
| 1944 | (key, dat), |
| 1945 | )); |
| 1946 | } else { |
| 1947 | return Err(err!( |
| 1948 | "Unpaired key {:?} at end of store.map {} ({})", |
| 1949 | key, match cfg.use_ordmaps { |
| 1950 | true => fmt!("{:?}", store.ordmap), |
| 1951 | false => fmt!("{:?}", store.map), |
| 1952 | }, cursor.borrow(); |
| 1953 | String, Input, Decode, Invalid, Missing)); |
| 1954 | } |
| 1955 | } |
| 1956 | (None, None) => {} |
| 1957 | } |
| 1958 | return Ok(if state.kind_outer == Kind::OrdMap { |
| 1959 | Dat::OrdMap(store.ordmap) |
| 1960 | } else { |
| 1961 | Dat::Map(store.map) |
| 1962 | }); |
| 1963 | } |
| 1964 | Err(err!( |
| 1965 | "Map capture was not actived with a '{{' character ({})", cursor.borrow(); |
| 1966 | String, Input, Decode, Invalid)) |
| 1967 | } |
| 1968 | |
| 1969 | #[inline(never)] |
| 1970 | fn comma_handler< |
| 1971 | M1: MapMut<UsrKindCode, UsrKind> + Clone + fmt::Debug + Default, |
| 1972 | M2: MapMut<String, UsrKindId> + Clone + fmt::Debug + Default, |
| 1973 | >( |
| 1974 | cfg: &DecoderConfig<M1, M2>, |
| 1975 | state: &mut DecoderState, |
| 1976 | cursor: &RefCell<Cursor>, |
| 1977 | mut store: &mut DecoderStore, |
| 1978 | ) |
| 1979 | -> Outcome<()> |
| 1980 | { |
| 1981 | match state.molecular_capture { |
| 1982 | None => { |
| 1983 | return Err(err!( |
| 1984 | "List or map capture not active ({})", cursor.borrow(); |
| 1985 | String, Input, Decode, Invalid)); |
| 1986 | } |
| 1987 | Some(MolecularCapture::Bytes) => { |
| 1988 | // We're expecting a byte to add to the list. |
| 1989 | let n = try_extract_dat!(res!(Self::process_atom(&mut store.slurp, &Kind::U8)), U8); |
| 1990 | store.byts.push(n); |
| 1991 | store.slurp = Slurp::new(); |
| 1992 | } |
| 1993 | Some(MolecularCapture::ListSame) => { |
| 1994 | // We're expecting a daticle to add to the list. |
| 1995 | let kind_same = MolecularCapture::same_kind(&state.kind_outer); |
| 1996 | let dat = match store.val_opt.take() { |
| 1997 | Some(dat) => dat, // store.dt_opt is now None. |
| 1998 | None => res!(Self::process_atom(&mut store.slurp, &kind_same)), |
| 1999 | }; |
| 2000 | store.list.push(dat); |
| 2001 | store.slurp = Slurp::new(); |
| 2002 | } |
| 2003 | Some(MolecularCapture::ListMixed) => { |
| 2004 | // We're expecting a daticle to add to the list. |
| 2005 | let dat = match store.val_opt.take() { // store.val_opt is now None. |
| 2006 | Some(dat) => dat, |
| 2007 | None => { |
| 2008 | if store.slurp.has_content() { |
| 2009 | res!(Self::process_atom(&mut store.slurp, &state.kind_outer)) |
| 2010 | } else { |
| 2011 | Dat::Empty |
| 2012 | } |
| 2013 | } |
| 2014 | }; |
| 2015 | if store.comment.len() > 0 { |
| 2016 | store.list.push(Dat::ABox( |
| 2017 | store.note_config.extract(), |
| 2018 | Box::new(dat), |
| 2019 | store.comment.extract(), |
| 2020 | )); |
| 2021 | } else { |
| 2022 | store.list.push(dat); |
| 2023 | } |
| 2024 | store.slurp = Slurp::new(); |
| 2025 | } |
| 2026 | Some(MolecularCapture::Map) => { |
| 2027 | // We're expecting a daticle to add to the map. |
| 2028 | let mut dat = match store.val_opt.take() { |
| 2029 | Some(dat) => dat, // store.val_opt is now None. |
| 2030 | None => { |
| 2031 | let dat = if store.slurp.has_content() { |
| 2032 | res!(Self::process_atom(&mut store.slurp, &Kind::Unknown)) |
| 2033 | } else { |
| 2034 | Dat::Empty |
| 2035 | }; |
| 2036 | store.val_opt = Some(dat.clone()); |
| 2037 | dat |
| 2038 | } |
| 2039 | }; |
| 2040 | if store.comment.len() > 0 { |
| 2041 | dat = Dat::ABox( |
| 2042 | store.note_config.extract(), |
| 2043 | Box::new(dat), |
| 2044 | store.comment.extract(), |
| 2045 | ); |
| 2046 | store.val_opt = Some(dat.clone()); |
| 2047 | } |
| 2048 | match store.key_opt.take() { |
| 2049 | Some(key) => { // store.key_opt is now None. |
| 2050 | let key = match key { |
| 2051 | Dat::Str(s) if s.len() == 0 => |
| 2052 | res!(cfg.default_key.rand().to_dat()), |
| 2053 | _ => key, |
| 2054 | }; |
| 2055 | res!(Self::map_insert( |
| 2056 | state.kind_outer == Kind::OrdMap, |
| 2057 | &mut store, |
| 2058 | (key, dat), |
| 2059 | )); |
| 2060 | } |
| 2061 | None => { |
| 2062 | store.key_opt = store.val_opt.take(); |
| 2063 | } |
| 2064 | } |
| 2065 | store.val_opt = None; |
| 2066 | store.slurp = Slurp::new(); |
| 2067 | } |
| 2068 | } |
| 2069 | Ok(()) |
| 2070 | } |
| 2071 | |
| 2072 | /// `Dat::OrdMap` allows map key ordering to be preserved, but the trade off is an additional |
| 2073 | /// search prior to insertion to ensure uniqueness of the key daticle. This will become |
| 2074 | /// costly as the map size grows. |
| 2075 | #[inline(never)] |
| 2076 | fn map_insert( |
| 2077 | use_ordmap: bool, |
| 2078 | store: &mut DecoderStore, |
| 2079 | (key, dat): (Dat, Dat), |
| 2080 | ) |
| 2081 | -> Outcome<()> |
| 2082 | { |
| 2083 | match use_ordmap { |
| 2084 | true => { |
| 2085 | if store.ordmap.iter().any(|(mk, _)| *mk.dat() == key) { |
| 2086 | return Err(err!( |
| 2087 | "The key {:?} already exists in the {:?} being \ |
| 2088 | interpreted from the given string.", key, store.ordmap; |
| 2089 | Invalid, Input, Exists)); |
| 2090 | } |
| 2091 | let mkey = MapKey::new(store.map_count.order(), key); |
| 2092 | store.ordmap.insert(mkey, dat); |
| 2093 | res!(store.map_count.inc_all()); |
| 2094 | } |
| 2095 | false => { |
| 2096 | if store.map.contains_key(&key) { |
| 2097 | return Err(err!( |
| 2098 | "The key {:?} already exists in the {:?} being \ |
| 2099 | interpreted from the given string.", key, store.map; |
| 2100 | Invalid, Input, Exists)); |
| 2101 | } else { |
| 2102 | store.map.insert(key, dat); |
| 2103 | res!(store.map_count.inc()); |
| 2104 | } |
| 2105 | } |
| 2106 | } |
| 2107 | Ok(()) |
| 2108 | } |
| 2109 | |
| 2110 | #[inline(never)] |
| 2111 | fn handle_string_escape( |
| 2112 | c: char, |
| 2113 | state: &mut DecoderState, |
| 2114 | slurp: &mut Slurp, |
| 2115 | ) |
| 2116 | -> Outcome<bool> |
| 2117 | { |
| 2118 | use StringEscape::*; |
| 2119 | match state.string_escape.clone() { |
| 2120 | None => { |
| 2121 | if c == '\\' { |
| 2122 | state.string_escape = Backslash; |
| 2123 | // Ensure the slurp is marked as a string |
| 2124 | // even when the whole payload is escapes |
| 2125 | // (e.g. `"\n"` would otherwise look like a |
| 2126 | // zero-length non-string slurp). |
| 2127 | slurp.flag_as_string(); |
| 2128 | return Ok(true); |
| 2129 | } |
| 2130 | Ok(false) |
| 2131 | } |
| 2132 | Backslash => { |
| 2133 | let translated = match c { |
| 2134 | '"' => '"', |
| 2135 | '\\' => '\\', |
| 2136 | '/' => '/', |
| 2137 | 'b' => '\u{0008}', |
| 2138 | 'f' => '\u{000C}', |
| 2139 | 'n' => '\n', |
| 2140 | 'r' => '\r', |
| 2141 | 't' => '\t', |
| 2142 | 'u' => { |
| 2143 | state.string_escape = Unicode { digits: 0, acc: 0 }; |
| 2144 | return Ok(true); |
| 2145 | } |
| 2146 | _ => return Err(err!( |
| 2147 | "Invalid string escape sequence '\\{}' \ |
| 2148 | inside quoted string. Expected one of \ |
| 2149 | `\"`, `\\`, `/`, `b`, `f`, `n`, `r`, `t`, \ |
| 2150 | `u`.", c; |
| 2151 | String, Input, Decode, Invalid)), |
| 2152 | }; |
| 2153 | slurp.push(translated); |
| 2154 | state.string_escape = None; |
| 2155 | Ok(true) |
| 2156 | } |
| 2157 | Unicode { digits, acc } => { |
| 2158 | let hex = res!(Self::hex_digit_value(c)); |
| 2159 | let new_acc = (acc << 4) | hex; |
| 2160 | let new_digits = digits + 1; |
| 2161 | if new_digits < 4 { |
| 2162 | state.string_escape = Unicode { |
| 2163 | digits: new_digits, |
| 2164 | acc: new_acc, |
| 2165 | }; |
| 2166 | return Ok(true); |
| 2167 | } |
| 2168 | // Four digits collected. Decide what to do with |
| 2169 | // the 16-bit value. |
| 2170 | let code_unit = new_acc as u16; |
| 2171 | if (0xD800..=0xDBFF).contains(&code_unit) { |
| 2172 | // High surrogate -- must be followed by a |
| 2173 | // low surrogate `\uYYYY`. |
| 2174 | state.string_escape = SurrogateBackslash { high: code_unit }; |
| 2175 | return Ok(true); |
| 2176 | } |
| 2177 | if (0xDC00..=0xDFFF).contains(&code_unit) { |
| 2178 | return Err(err!( |
| 2179 | "Bare low surrogate U+{:04X} in \\uXXXX \ |
| 2180 | escape without a preceding high \ |
| 2181 | surrogate.", code_unit; |
| 2182 | String, Input, Decode, Invalid)); |
| 2183 | } |
| 2184 | // BMP code point, push directly. |
| 2185 | match char::from_u32(new_acc) { |
| 2186 | Some(ch) => { |
| 2187 | slurp.push(ch); |
| 2188 | state.string_escape = None; |
| 2189 | Ok(true) |
| 2190 | } |
| 2191 | Option::None => Err(err!( |
| 2192 | "Invalid Unicode code point U+{:04X} in \ |
| 2193 | \\uXXXX escape.", new_acc; |
| 2194 | String, Input, Decode, Invalid)), |
| 2195 | } |
| 2196 | } |
| 2197 | SurrogateBackslash { high } => { |
| 2198 | if c != '\\' { |
| 2199 | return Err(err!( |
| 2200 | "High surrogate U+{:04X} not followed by \ |
| 2201 | a `\\` to start the low-surrogate escape \ |
| 2202 | (saw {:?}).", high, c; |
| 2203 | String, Input, Decode, Invalid)); |
| 2204 | } |
| 2205 | state.string_escape = SurrogateU { high }; |
| 2206 | Ok(true) |
| 2207 | } |
| 2208 | SurrogateU { high } => { |
| 2209 | if c != 'u' { |
| 2210 | return Err(err!( |
| 2211 | "High surrogate U+{:04X} not followed by \ |
| 2212 | a `\\u` to start the low-surrogate escape \ |
| 2213 | (saw \\{:?}).", high, c; |
| 2214 | String, Input, Decode, Invalid)); |
| 2215 | } |
| 2216 | state.string_escape = LowSurrogate { |
| 2217 | digits: 0, |
| 2218 | acc: 0, |
| 2219 | high, |
| 2220 | }; |
| 2221 | Ok(true) |
| 2222 | } |
| 2223 | LowSurrogate { digits, acc, high } => { |
| 2224 | let hex = res!(Self::hex_digit_value(c)); |
| 2225 | let new_acc = (acc << 4) | hex; |
| 2226 | let new_digits = digits + 1; |
| 2227 | if new_digits < 4 { |
| 2228 | state.string_escape = LowSurrogate { |
| 2229 | digits: new_digits, |
| 2230 | acc: new_acc, |
| 2231 | high, |
| 2232 | }; |
| 2233 | return Ok(true); |
| 2234 | } |
| 2235 | let low = new_acc as u16; |
| 2236 | if !(0xDC00..=0xDFFF).contains(&low) { |
| 2237 | return Err(err!( |
| 2238 | "Expected a low surrogate (U+DC00..U+DFFF) \ |
| 2239 | after high surrogate U+{:04X}, got U+{:04X}.", |
| 2240 | high, low; |
| 2241 | String, Input, Decode, Invalid)); |
| 2242 | } |
| 2243 | // Combine the surrogate pair into a |
| 2244 | // supplementary plane code point. |
| 2245 | let h = high as u32; |
| 2246 | let l = low as u32; |
| 2247 | let cp = 0x10000 + ((h - 0xD800) << 10) + (l - 0xDC00); |
| 2248 | match char::from_u32(cp) { |
| 2249 | Some(ch) => { |
| 2250 | slurp.push(ch); |
| 2251 | state.string_escape = None; |
| 2252 | Ok(true) |
| 2253 | } |
| 2254 | Option::None => Err(err!( |
| 2255 | "Surrogate pair U+{:04X} U+{:04X} does \ |
| 2256 | not form a valid code point (U+{:06X}).", |
| 2257 | high, low, cp; |
| 2258 | String, Input, Decode, Invalid)), |
| 2259 | } |
| 2260 | } |
| 2261 | } |
| 2262 | } |
| 2263 | |
| 2264 | fn hex_digit_value(c: char) -> Outcome<u32> { |
| 2265 | match c { |
| 2266 | '0'..='9' => Ok((c as u32) - ('0' as u32)), |
| 2267 | 'a'..='f' => Ok((c as u32) - ('a' as u32) + 10), |
| 2268 | 'A'..='F' => Ok((c as u32) - ('A' as u32) + 10), |
| 2269 | _ => Err(err!( |
| 2270 | "Invalid hex digit '{}' in \\uXXXX escape \ |
| 2271 | sequence.", c; |
| 2272 | String, Input, Decode, Invalid)), |
| 2273 | } |
| 2274 | } |
| 2275 | |
| 2276 | /// Process a [`Slurp`] extracted during single-pass processing. Numbers currently undergo an |
| 2277 | /// additional pass via validation in [`NumberString::new`], to cope with the various forms of |
| 2278 | /// representation, even when typed (e.g. different radices). |
| 2279 | #[inline(never)] |
| 2280 | fn process_atom( |
| 2281 | slurp: &mut Slurp, |
| 2282 | kind: &Kind, |
| 2283 | ) |
| 2284 | -> Outcome<Self> |
| 2285 | { |
| 2286 | let d = match kind { |
| 2287 | // If there is no explicit kind, and it's in quotes, it's a string. |
| 2288 | Kind::Unknown if slurp.is_string() => Dat::Str(slurp.clone_string()), |
| 2289 | // If it's not quoted and nothing was captured, it is empty. |
| 2290 | Kind::Unknown if (!slurp.is_string() && slurp.len() == 0) => Dat::Empty, |
| 2291 | // If the kind is declared a STR, even without quotes, it's a string. |
| 2292 | Kind::Str => Dat::Str(slurp.clone_string()), |
| 2293 | // These kinds don't need data. |
| 2294 | Kind::Empty => Dat::Empty, |
| 2295 | Kind::True => Dat::Bool(true), |
| 2296 | Kind::False => Dat::Bool(false), |
| 2297 | Kind::None => Dat::Opt(Box::new(None)), |
| 2298 | Kind::BU8 | |
| 2299 | Kind::BU16 | |
| 2300 | Kind::BU32 | |
| 2301 | Kind::BU64 | |
| 2302 | Kind::BC64 | |
| 2303 | Kind::B2 | |
| 2304 | Kind::B3 | |
| 2305 | Kind::B4 | |
| 2306 | Kind::B5 | |
| 2307 | Kind::B6 | |
| 2308 | Kind::B7 | |
| 2309 | Kind::B8 | |
| 2310 | Kind::B9 | |
| 2311 | Kind::B10 | |
| 2312 | Kind::B16 | |
| 2313 | Kind::B32 if slurp.is_string() => { |
| 2314 | // Interpret as base2x, which are quoted strings. |
| 2315 | let byts = res!(base2x::HEMATITE64.from_str(slurp.get_str())); |
| 2316 | res!(Self::decode_bytes(byts, kind)) |
| 2317 | } |
| 2318 | _ => if slurp.is_string() { |
| 2319 | if kind.case().accepts_strings() { |
| 2320 | Dat::Str(slurp.clone_string()) |
| 2321 | } else { |
| 2322 | return Err(err!( |
| 2323 | "A quoted string '{}' is not compatible with {:?}.", slurp, kind; |
| 2324 | String, Input, Decode, Mismatch)); |
| 2325 | } |
| 2326 | } else { |
| 2327 | if slurp.len() == 0 { |
| 2328 | return Err(err!( |
| 2329 | "The kind {:?} requires data, but none was found.", kind; |
| 2330 | String, Input, Decode, Missing)); |
| 2331 | } else { |
| 2332 | // We allow a limited number of keywords that can be used a shorthand for kinds, with or |
| 2333 | // without quotes. |
| 2334 | match Kind::from_str(&slurp.clone_string()) { |
| 2335 | Ok(Kind::False) => Dat::Bool(false), |
| 2336 | Ok(Kind::True) => Dat::Bool(true), |
| 2337 | Ok(Kind::Empty) => Dat::Empty, |
| 2338 | Ok(Kind::None) => Dat::Opt(Box::new(None)), |
| 2339 | _ => match NumberString::new(slurp.clone_string()) { |
| 2340 | // Anything else, let's see if it's a number. |
| 2341 | Ok(ns) => { |
| 2342 | if kind.is_number() { |
| 2343 | // Decode as a number when the kind is specified. |
| 2344 | res!(kind.clone().decode_number(ns)) |
| 2345 | } else { |
| 2346 | match Self::from_untyped_number(&ns) { |
| 2347 | Ok(d) => d, |
| 2348 | Err(e) => return Err(err!(e, |
| 2349 | "While interpreting '{}' with outer kind {:?}.", |
| 2350 | slurp, kind; |
| 2351 | String, Input, Decode)), |
| 2352 | } |
| 2353 | } |
| 2354 | } |
| 2355 | Err(e) => if *kind == Kind::Unknown { |
| 2356 | Dat::Str(slurp.take_string()) |
| 2357 | } else { |
| 2358 | return Err(err!(e, |
| 2359 | "While interpreting '{}' with outer kind {:?}.", |
| 2360 | slurp, kind; |
| 2361 | String, Input, Decode)); |
| 2362 | } |
| 2363 | } |
| 2364 | } |
| 2365 | } |
| 2366 | } |
| 2367 | }; |
| 2368 | |
| 2369 | slurp.reset(); |
| 2370 | Ok(d) |
| 2371 | } |
| 2372 | |
| 2373 | /// The daticle a number read without a kind becomes: zero as a `U8`, a fraction or an |
| 2374 | /// exponent as an `Adec`, any other integer as the smallest native integer that holds it, and |
| 2375 | /// one beyond 128 bits as an `Aint`. The JDAT text decoder and the strict JSON reader both |
| 2376 | /// read numbers through this, so the same digits give the same daticle either way. |
| 2377 | pub(crate) fn from_untyped_number(ns: &NumberString) -> Outcome<Self> { |
| 2378 | if ns.is_zero() { |
| 2379 | return Ok(Dat::U8(0)); |
| 2380 | } |
| 2381 | if ns.has_point() || ns.has_exp() { |
| 2382 | return Ok(Dat::Adec(res!(ns.as_bigdecimal()))); |
| 2383 | } |
| 2384 | match u128::from_str_radix(ns.abs_integer_str(), ns.radix()) { |
| 2385 | Ok(nu128) => if ns.is_negative() { |
| 2386 | if nu128 > (i128::MAX as u128 + 1) { |
| 2387 | // Below i128::MIN, so only a BigInt holds it. |
| 2388 | Ok(Dat::Aint(res!(ns.as_bigint()))) |
| 2389 | } else { |
| 2390 | // The cast of |i128::MIN| wraps to i128::MIN, which is the value wanted; |
| 2391 | // every other magnitude is negated by hand. |
| 2392 | let n = if nu128 == i128::MAX as u128 + 1 { |
| 2393 | nu128 as i128 |
| 2394 | } else { |
| 2395 | -(nu128 as i128) |
| 2396 | }; |
| 2397 | DatInt::from(n).min_size().to_dat() |
| 2398 | } |
| 2399 | } else { |
| 2400 | DatInt::from(nu128).min_size().to_dat() |
| 2401 | }, |
| 2402 | // Beyond u128. |
| 2403 | Err(_) => Ok(Dat::Aint(res!(ns.as_bigint()))), |
| 2404 | } |
| 2405 | } |
| 2406 | |
| 2407 | #[inline(never)] |
| 2408 | fn decode_bytes(v: Vec<u8>, k: &Kind) -> Outcome<Self> { |
| 2409 | Ok(match k { |
| 2410 | Kind::BU8 => Self::BU8(v), |
| 2411 | Kind::BU16 => Self::BU16(v), |
| 2412 | Kind::BU32 => Self::BU32(v), |
| 2413 | Kind::BU64 => Self::BU64(v), |
| 2414 | Kind::BC64 => Self::BC64(v), |
| 2415 | Kind::B2 => Self::B2(res!(<[u8; 2]>::try_from(&v[..]), Decode, Bytes)), |
| 2416 | Kind::B3 => Self::B3(res!(<[u8; 3]>::try_from(&v[..]), Decode, Bytes)), |
| 2417 | Kind::B4 => Self::B4(res!(<[u8; 4]>::try_from(&v[..]), Decode, Bytes)), |
| 2418 | Kind::B5 => Self::B5(res!(<[u8; 5]>::try_from(&v[..]), Decode, Bytes)), |
| 2419 | Kind::B6 => Self::B6(res!(<[u8; 6]>::try_from(&v[..]), Decode, Bytes)), |
| 2420 | Kind::B7 => Self::B7(res!(<[u8; 7]>::try_from(&v[..]), Decode, Bytes)), |
| 2421 | Kind::B8 => Self::B8(res!(<[u8; 8]>::try_from(&v[..]), Decode, Bytes)), |
| 2422 | Kind::B9 => Self::B9(res!(<[u8; 9]>::try_from(&v[..]), Decode, Bytes)), |
| 2423 | Kind::B10 => Self::B10(res!(<[u8; 10]>::try_from(&v[..]), Decode, Bytes)), |
| 2424 | Kind::B16 => Self::B16(res!(<[u8; 16]>::try_from(&v[..]), Decode, Bytes)), |
| 2425 | Kind::B32 => Self::B32(res!(B32::from_bytes(&v[..])).0), |
| 2426 | _ => return Err(err!( |
| 2427 | "{:?} is not suitable for bytes.", k; |
| 2428 | Input, Mismatch, Bug)), |
| 2429 | }) |
| 2430 | } |
| 2431 | } |
| 2432 | |
| 2433 | #[cfg(test)] |
| 2434 | mod tests { |
| 2435 | use super::*; |
| 2436 | |
| 2437 | use std::thread; |
| 2438 | |
| 2439 | const SPAWNED_THREAD_STACK: usize = 2 * 1024 * 1024; |
| 2440 | |
| 2441 | fn nested_lists(levels: usize) -> String { |
| 2442 | let mut s = String::with_capacity(2 * levels); |
| 2443 | for _ in 0..levels { |
| 2444 | s.push('['); |
| 2445 | } |
| 2446 | for _ in 0..levels { |
| 2447 | s.push(']'); |
| 2448 | } |
| 2449 | s |
| 2450 | } |
| 2451 | |
| 2452 | fn nested_maps(levels: usize) -> String { |
| 2453 | let mut s = String::with_capacity(4 * levels); |
| 2454 | for _ in 0..levels { |
| 2455 | s.push_str("{a:"); |
| 2456 | } |
| 2457 | s.push('1'); |
| 2458 | for _ in 0..levels { |
| 2459 | s.push('}'); |
| 2460 | } |
| 2461 | s |
| 2462 | } |
| 2463 | |
| 2464 | fn decode_on_a_small_stack(s: String) -> Outcome<Outcome<Dat>> { |
| 2465 | let builder = thread::Builder::new().stack_size(SPAWNED_THREAD_STACK); |
| 2466 | let handle = match builder.spawn(move || Dat::decode_string(s)) { |
| 2467 | Ok(handle) => handle, |
| 2468 | Err(e) => return Err(err!(e, "While spawning the decoding thread."; Test, Init)), |
| 2469 | }; |
| 2470 | match handle.join() { |
| 2471 | Ok(result) => Ok(result), |
| 2472 | Err(_) => Err(err!( |
| 2473 | "The decoding thread died rather than returning, which is what a stack overflow \ |
| 2474 | looks like."; |
| 2475 | Test, Invalid)), |
| 2476 | } |
| 2477 | } |
| 2478 | |
| 2479 | #[test] |
| 2480 | fn test_text_depth_limit_accepts_at_the_limit() -> Outcome<()> { |
| 2481 | // A leaf inside 15 lists sits at depth 16. |
| 2482 | const LEVELS: usize = 15; |
| 2483 | let lims = DecodeLimits::text().with_max_depth(LEVELS + 1); |
| 2484 | let dat = res!(Dat::decode_string_limited(nested_lists(LEVELS), &lims)); |
| 2485 | let mut expected = Dat::List(Vec::new()); |
| 2486 | for _ in 1..LEVELS { |
| 2487 | expected = Dat::List(vec![expected]); |
| 2488 | } |
| 2489 | assert_eq!(dat, expected); |
| 2490 | Ok(()) |
| 2491 | } |
| 2492 | |
| 2493 | #[test] |
| 2494 | fn test_text_depth_limit_refuses_past_the_limit() -> Outcome<()> { |
| 2495 | // The same nesting, one level shallower than it needs, is refused rather than decoded. |
| 2496 | const LEVELS: usize = 15; |
| 2497 | let lims = DecodeLimits::text().with_max_depth(LEVELS); |
| 2498 | match Dat::decode_string_limited(nested_lists(LEVELS), &lims) { |
| 2499 | Ok(dat) => Err(err!( |
| 2500 | "Expected a depth limit error, but decoded {:?}.", dat; |
| 2501 | Test, Invalid)), |
| 2502 | Err(e) => { |
| 2503 | let msg = e.to_string(); |
| 2504 | assert!(msg.contains("depth"), "Error should name the depth limit: {}", msg); |
| 2505 | assert!(msg.contains("offset"), "Error should name the offset: {}", msg); |
| 2506 | Ok(()) |
| 2507 | } |
| 2508 | } |
| 2509 | } |
| 2510 | |
| 2511 | #[test] |
| 2512 | fn test_text_list_bomb_is_refused() -> Outcome<()> { |
| 2513 | // A hostile document: 100,000 nested lists, 200 kB of text, which would exhaust the stack |
| 2514 | // of a decoder that trusted it. The decode must return an error, on a stack that a stack |
| 2515 | // overflow would abort. |
| 2516 | match res!(decode_on_a_small_stack(nested_lists(100_000))) { |
| 2517 | Ok(_) => Err(err!( |
| 2518 | "A nesting of 100,000 lists should be refused by the default limits."; |
| 2519 | Test, Invalid)), |
| 2520 | Err(e) => { |
| 2521 | let msg = e.to_string(); |
| 2522 | assert!(msg.contains("depth"), "Error should name the depth limit: {}", msg); |
| 2523 | Ok(()) |
| 2524 | } |
| 2525 | } |
| 2526 | } |
| 2527 | |
| 2528 | #[test] |
| 2529 | fn test_text_map_bomb_is_refused() -> Outcome<()> { |
| 2530 | // The same attack through the other bracket. |
| 2531 | match res!(decode_on_a_small_stack(nested_maps(100_000))) { |
| 2532 | Ok(_) => Err(err!( |
| 2533 | "A nesting of 100,000 maps should be refused by the default limits."; |
| 2534 | Test, Invalid)), |
| 2535 | Err(e) => { |
| 2536 | let msg = e.to_string(); |
| 2537 | assert!(msg.contains("depth"), "Error should name the depth limit: {}", msg); |
| 2538 | Ok(()) |
| 2539 | } |
| 2540 | } |
| 2541 | } |
| 2542 | |
| 2543 | #[test] |
| 2544 | fn test_text_at_the_default_limit_decodes_on_a_small_stack() -> Outcome<()> { |
| 2545 | // The limit is only worth having if everything under it is safe. A document nested as deep |
| 2546 | // as the default allows must decode on the stack the default is sized against, rather than |
| 2547 | // aborting just short of the error the decoder promises. |
| 2548 | let levels = DecodeLimits::DEFAULT_MAX_TEXT_DEPTH - 1; |
| 2549 | let dat = res!(res!(decode_on_a_small_stack(nested_lists(levels)))); |
| 2550 | let mut depth = 0; |
| 2551 | let mut node = &dat; |
| 2552 | while let Dat::List(items) = node { |
| 2553 | depth += 1; |
| 2554 | match items.first() { |
| 2555 | Some(item) => node = item, |
| 2556 | None => break, |
| 2557 | } |
| 2558 | } |
| 2559 | assert_eq!(depth, levels, "The decoded nesting is not the nesting that was given."); |
| 2560 | Ok(()) |
| 2561 | } |
| 2562 | |
| 2563 | #[test] |
| 2564 | fn test_text_byte_limit_refuses_an_oversized_input() -> Outcome<()> { |
| 2565 | let lims = DecodeLimits::text().with_max_bytes(8); |
| 2566 | match Dat::decode_string_limited("\"a string longer than a tiny limit\"", &lims) { |
| 2567 | Ok(dat) => Err(err!( |
| 2568 | "Expected a byte limit error, but decoded {:?}.", dat; |
| 2569 | Test, Invalid)), |
| 2570 | Err(e) => { |
| 2571 | let msg = e.to_string(); |
| 2572 | assert!(msg.contains("8 bytes"), "Error should name the byte limit: {}", msg); |
| 2573 | Ok(()) |
| 2574 | } |
| 2575 | } |
| 2576 | } |
| 2577 | |
| 2578 | #[test] |
| 2579 | fn test_ordinary_documents_still_decode() -> Outcome<()> { |
| 2580 | // The limits are no use if they cost the decoder its day job. |
| 2581 | let txt = r#"{ |
| 2582 | "name": "Hematite", |
| 2583 | "version": (u16|3), |
| 2584 | "tags": ["a", "b"], |
| 2585 | "nested": {"n": [1, 2, 3]}, |
| 2586 | "pair": (t2|[1, 2]), |
| 2587 | "opt": (some|(u8|7)), |
| 2588 | }"#; |
| 2589 | let expected = mapdat!{ |
| 2590 | dat!("name") => dat!("Hematite"), |
| 2591 | dat!("version") => dat!(3u16), |
| 2592 | dat!("tags") => listdat![dat!("a"), dat!("b")], |
| 2593 | dat!("nested") => mapdat!{ |
| 2594 | dat!("n") => listdat![dat!(1u8), dat!(2u8), dat!(3u8)], |
| 2595 | }, |
| 2596 | dat!("pair") => Dat::Tup2(Box::new([dat!(1u8), dat!(2u8)])), |
| 2597 | dat!("opt") => Dat::Opt(Box::new(Some(dat!(7u8)))), |
| 2598 | }; |
| 2599 | assert_eq!(res!(Dat::decode_string(txt)), expected); |
| 2600 | Ok(()) |
| 2601 | } |
| 2602 | |
| 2603 | #[test] |
| 2604 | fn test_limits_can_be_tightened_by_the_caller() -> Outcome<()> { |
| 2605 | // A caller with an opinion about depth states it, and the decoder holds them to it. |
| 2606 | let cfg = DecoderConfig::< |
| 2607 | BTreeMap<UsrKindCode, UsrKind>, |
| 2608 | BTreeMap<String, UsrKindId>, |
| 2609 | >::default().with_limits(DecodeLimits::text().with_max_depth(2)); |
| 2610 | assert!(Dat::decode_string_with_config("[1]", &cfg).is_ok()); |
| 2611 | assert!(Dat::decode_string_with_config("[[1]]", &cfg).is_err()); |
| 2612 | Ok(()) |
| 2613 | } |
| 2614 | } |