oxedyne/fe2o3/fe2o3_jdat/tests/golden.rs
6.8 KiB, 1 run
created by r1870400018:14219, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Golden byte vectors for the BDAT binary encoding. |
| 2 | //! |
| 3 | //! `byte.rs` encodes eighty-seven daticles and decodes them again with our own decoder. That |
| 4 | //! proves the two agree with each other, which they would go on doing if every length prefix were |
| 5 | //! little-endian, or if the code for a `u16` changed tomorrow: the suite would stay green while the |
| 6 | //! format underneath it moved. |
| 7 | //! |
| 8 | //! The bytes are not an implementation detail. o3db hashes the encoding of a key to decide which |
| 9 | //! node owns the record, so a byte that changes is a record that moves; a value written by one |
| 10 | //! release and read by the next has to encode identically or it is simply lost. Nothing outside |
| 11 | //! this crate can be consulted about that -- BDAT is ours -- so the vectors below are written out |
| 12 | //! by hand from the format's own rules, and the encoder is checked against them rather than the |
| 13 | //! other way round: |
| 14 | //! |
| 15 | //! - a kind code from `constant.rs`, one byte, then the payload; |
| 16 | //! - fixed-width integers big-endian, in two's complement where signed; |
| 17 | //! - a `c64` count as `C64_CODE_START + n`, where `n` is how many significant bytes follow, most |
| 18 | //! significant first, so the value zero is the bare code `0x20` and no bytes at all; |
| 19 | //! - a string as its code, then a `c64` byte length, then UTF-8; |
| 20 | //! - a list as its code, then a `c64` giving the byte length of its encoded items -- their *bytes*, |
| 21 | //! not their number -- then the items. |
| 22 | //! |
| 23 | //! If a vector here fails, either the encoding changed or this description of it is wrong. Both |
| 24 | //! are worth stopping for. |
| 25 | |
| 26 | use oxedyne_fe2o3_jdat::prelude::*; |
| 27 | |
| 28 | use oxedyne_fe2o3_core::{ |
| 29 | prelude::*, |
| 30 | test::test_it, |
| 31 | }; |
| 32 | |
| 33 | /// Encode the daticle, require the exact bytes, then decode those bytes and require the daticle |
| 34 | /// back. Both directions matter: an encoder alone could be pinned by a decoder that shares its |
| 35 | /// mistake. |
| 36 | fn golden(dat: Dat, expected: &[u8], what: &str) -> Outcome<()> { |
| 37 | let encoded = res!(dat.clone().to_bytes(Vec::new())); |
| 38 | req!(encoded, expected.to_vec(), "Encoding {}.", what); |
| 39 | |
| 40 | let (decoded, n) = res!(Dat::from_bytes(expected)); |
| 41 | req!(decoded, dat, "Decoding {}.", what); |
| 42 | req!(n, expected.len(), "Bytes consumed decoding {}.", what); |
| 43 | Ok(()) |
| 44 | } |
| 45 | |
| 46 | pub fn test_golden_func(filter: &'static str) -> Outcome<()> { |
| 47 | |
| 48 | res!(test_it(filter, &["Golden logic", "all", "golden"], || { |
| 49 | res!(golden(Dat::Empty, &[0x00], "the empty daticle")); |
| 50 | res!(golden(dat!(true), &[0x01], "true")); |
| 51 | res!(golden(dat!(false), &[0x02], "false")); |
| 52 | res!(golden(Dat::Opt(Box::new(None)), &[0x03], "none")); |
| 53 | Ok(()) |
| 54 | })); |
| 55 | |
| 56 | res!(test_it(filter, &["Golden fixed width integers", "all", "golden"], || { |
| 57 | // The code, then the value big-endian, in two's complement where the kind is signed. |
| 58 | res!(golden(dat!(0u8), &[0x0a, 0x00], "u8 zero")); |
| 59 | res!(golden(dat!(255u8), &[0x0a, 0xff], "u8 max")); |
| 60 | res!(golden(dat!(1u16), &[0x0b, 0x00, 0x01], "u16 one")); |
| 61 | res!(golden(dat!(65535u16), &[0x0b, 0xff, 0xff], "u16 max")); |
| 62 | res!(golden(dat!(1u32), &[0x0c, 0x00, 0x00, 0x00, 0x01], "u32 one")); |
| 63 | res!(golden(dat!(1u64), &[0x0d, 0, 0, 0, 0, 0, 0, 0, 0x01], "u64 one")); |
| 64 | |
| 65 | res!(golden(dat!(-1i8), &[0x10, 0xff], "i8 minus one")); |
| 66 | res!(golden(dat!(i8::MIN), &[0x10, 0x80], "i8 min")); |
| 67 | res!(golden(dat!(-2i16), &[0x11, 0xff, 0xfe], "i16 minus two")); |
| 68 | res!(golden(dat!(i16::MIN), &[0x11, 0x80, 0x00], "i16 min")); |
| 69 | res!(golden(dat!(i32::MIN), &[0x12, 0x80, 0x00, 0x00, 0x00], "i32 min")); |
| 70 | Ok(()) |
| 71 | })); |
| 72 | |
| 73 | res!(test_it(filter, &["Golden c64", "all", "golden"], || { |
| 74 | // The whole point of a c64: the code carries the length, so a small number is small. The |
| 75 | // value zero occupies one byte in total, and no payload byte at all. |
| 76 | res!(golden(Dat::C64(0), &[0x20], "c64 zero")); |
| 77 | res!(golden(Dat::C64(1), &[0x21, 0x01], "c64 one")); |
| 78 | res!(golden(Dat::C64(255), &[0x21, 0xff], "c64 255, still one byte")); |
| 79 | res!(golden(Dat::C64(256), &[0x22, 0x01, 0x00], "c64 256, now two")); |
| 80 | res!(golden(Dat::C64(65_535), &[0x22, 0xff, 0xff], "c64 65535")); |
| 81 | res!(golden(Dat::C64(65_536), &[0x23, 0x01, 0x00, 0x00], "c64 65536, now three")); |
| 82 | res!(golden(Dat::C64(u64::MAX), |
| 83 | &[0x28, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff], "c64 max, all eight")); |
| 84 | Ok(()) |
| 85 | })); |
| 86 | |
| 87 | res!(test_it(filter, &["Golden strings", "all", "golden"], || { |
| 88 | // The code, a c64 byte length, then UTF-8. The length counts bytes, not characters, which |
| 89 | // is why the accented character below declares two. |
| 90 | res!(golden(dat!(""), &[0x29, 0x20], "the empty string")); |
| 91 | res!(golden(dat!("abc"), &[0x29, 0x21, 0x03, 0x61, 0x62, 0x63], "the string 'abc'")); |
| 92 | res!(golden(dat!("é"), &[0x29, 0x21, 0x02, 0xc3, 0xa9], "a two byte character")); |
| 93 | Ok(()) |
| 94 | })); |
| 95 | |
| 96 | res!(test_it(filter, &["Golden c64 minimality", "all", "golden"], || { |
| 97 | // A c64 code says how many bytes follow, so the value 5 can be written 0x21 0x05, and it |
| 98 | // can also be written 0x22 0x00 0x05, and 0x23 0x00 0x00 0x05, and so on. The encoder only |
| 99 | // ever writes the first. The decoder used to read all of them, which gave one value an |
| 100 | // unbounded number of valid encodings. |
| 101 | // |
| 102 | // That is not a tidiness complaint. o3db hashes the encoded bytes of a key to choose the |
| 103 | // node that owns the record, so a key re-encoded non-minimally by a peer hashes to a |
| 104 | // different node: the same record written in one place and sought in another. A signature |
| 105 | // over a canonical encoding has the same problem from the other end. |
| 106 | // |
| 107 | // So a leading zero is refused, and the value keeps exactly one encoding. |
| 108 | let minimal: &[u8] = &[0x21, 0x05]; |
| 109 | let (dat, n) = res!(Dat::from_bytes(minimal)); |
| 110 | req!(dat, Dat::C64(5), "The minimal encoding of five."); |
| 111 | req!(n, 2); |
| 112 | |
| 113 | for padded in [ |
| 114 | vec![0x22, 0x00, 0x05], |
| 115 | vec![0x23, 0x00, 0x00, 0x05], |
| 116 | vec![0x28, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x05], |
| 117 | ] { |
| 118 | match Dat::from_bytes(&padded) { |
| 119 | Ok((dat, _)) => return Err(err!( |
| 120 | "The non-minimal encoding {:02x?} of five was accepted, decoding to {:?}. It \ |
| 121 | has a shorter encoding, so it is a second set of bytes for one value.", |
| 122 | padded, dat; |
| 123 | Test, Invalid)), |
| 124 | Err(_) => (), |
| 125 | } |
| 126 | } |
| 127 | Ok(()) |
| 128 | })); |
| 129 | |
| 130 | res!(test_it(filter, &["Golden lists", "all", "golden"], || { |
| 131 | // The c64 after the list code is the byte length of the encoded items, not their number. |
| 132 | res!(golden(Dat::List(vec![]), &[0x33, 0x20], "the empty list")); |
| 133 | res!(golden( |
| 134 | Dat::List(vec![dat!(1u8), dat!(2u8)]), |
| 135 | // Two u8 daticles, two bytes each, so four bytes of items. |
| 136 | &[0x33, 0x21, 0x04, 0x0a, 0x01, 0x0a, 0x02], |
| 137 | "a list of two u8s", |
| 138 | )); |
| 139 | res!(golden( |
| 140 | Dat::List(vec![Dat::List(vec![dat!(1u8)])]), |
| 141 | // The inner list is 0x33 0x21 0x02 0x0a 0x01, which is five bytes. |
| 142 | &[0x33, 0x21, 0x05, 0x33, 0x21, 0x02, 0x0a, 0x01], |
| 143 | "a list holding a list", |
| 144 | )); |
| 145 | Ok(()) |
| 146 | })); |
| 147 | |
| 148 | Ok(()) |
| 149 | } |