oxedyne/fe2o3/fe2o3_text/src/base64.rs
22.9 KiB, 54 runs
created by r1870400018:19608, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Standard Base64 as specified by RFC 4648 §4: the `A-Za-z0-9+/` alphabet, `=` |
| 2 | //! padding to a whole number of four character quanta, and nothing else. |
| 3 | //! |
| 4 | //! This is the encoding a browser's `atob` and `btoa` speak, and the one every |
| 5 | //! wire format that says "base64" means -- DKIM's `b=` and `p=` tags, HTTP Basic |
| 6 | //! credentials, PEM bodies, data URLs, JSON payloads carrying bytes. |
| 7 | //! |
| 8 | //! It is deliberately separate from [`crate::base2x`], whose `BASE64` constant |
| 9 | //! shares the alphabet but not the padding scheme: Base2x always writes three |
| 10 | //! padding characters where RFC 4648 writes one or two, so a Base2x string does |
| 11 | //! not survive `atob` and an RFC 4648 string does not survive Base2x decoding. |
| 12 | //! Base2x is the right tool for a custom alphabet; this module is the right tool |
| 13 | //! for talking to anybody else. |
| 14 | //! |
| 15 | //! # Strictness |
| 16 | //! |
| 17 | //! [`decode`] rejects rather than guesses, because two decoders that disagree |
| 18 | //! about the same string are how a signature verifies on one side and not the |
| 19 | //! other. It refuses: |
| 20 | //! |
| 21 | //! - any length that is not a multiple of four (RFC 4648 §4 pads every encoding); |
| 22 | //! - any character outside the alphabet, including whitespace and the URL-safe |
| 23 | //! `-` and `_` substitutes of RFC 4648 §5; |
| 24 | //! - `=` anywhere but as the last one or two characters; |
| 25 | //! - a final quantum whose unused bits are not zero, which RFC 4648 §3.5 permits |
| 26 | //! a decoder to reject and which this one does. |
| 27 | //! |
| 28 | //! A caller holding input that legitimately contains whitespace -- a PEM block, |
| 29 | //! or a header value folded across lines -- must strip it before calling. |
| 30 | //! |
| 31 | //! # Unpadded base64url |
| 32 | //! |
| 33 | //! [`encode_url`] and [`decode_url`] speak RFC 4648 §5 with the padding left |
| 34 | //! off, as JOSE, WebAuthn and most JSON protocols carry bytes: `-` and `_` in |
| 35 | //! place of `+` and `/`, and no `=`. [`decode_url`] is as strict as [`decode`] |
| 36 | //! and for the same reason. It refuses `=` anywhere, the `+` and `/` of the §4 |
| 37 | //! alphabet, any other character outside the §5 alphabet, a length that leaves |
| 38 | //! one character over a whole number of quanta (six bits cannot make a byte), |
| 39 | //! and a final character whose unused bits are not zero, so every byte string |
| 40 | //! has exactly one accepted spelling. |
| 41 | //! |
| 42 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 43 | //! Anthropic Claude |
| 44 | |
| 45 | use oxedyne_fe2o3_core::prelude::*; |
| 46 | |
| 47 | |
| 48 | // The RFC 4648 §4 alphabet, in index order. |
| 49 | const ALPHABET: [u8; 64] = |
| 50 | *b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; |
| 51 | |
| 52 | // The RFC 4648 §5 URL and filename safe alphabet, in index order. |
| 53 | const URL_ALPHABET: [u8; 64] = |
| 54 | *b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"; |
| 55 | |
| 56 | const PAD: u8 = b'='; |
| 57 | |
| 58 | const INVALID: u8 = 0xFF; // marks a byte that is not in the alphabet, in DECODE |
| 59 | |
| 60 | const DECODE: [u8; 256] = decode_table(&ALPHABET); // reverse of ALPHABET |
| 61 | const URL_DECODE: [u8; 256] = decode_table(&URL_ALPHABET); // reverse of URL_ALPHABET |
| 62 | |
| 63 | const fn decode_table(alphabet: &[u8; 64]) -> [u8; 256] { |
| 64 | let mut table = [INVALID; 256]; |
| 65 | let mut i = 0; |
| 66 | while i < alphabet.len() { |
| 67 | table[alphabet[i] as usize] = i as u8; |
| 68 | i += 1; |
| 69 | } |
| 70 | table |
| 71 | } |
| 72 | |
| 73 | /// Padding is included in the count. |
| 74 | pub fn encoded_len(n: usize) -> usize { |
| 75 | ((n + 2) / 3) * 4 |
| 76 | } |
| 77 | |
| 78 | /// The length of the unpadded base64url encoding of `n` bytes. |
| 79 | pub fn encoded_url_len(n: usize) -> usize { |
| 80 | (n * 4 + 2) / 3 |
| 81 | } |
| 82 | |
| 83 | /// Encodes bytes as standard, padded RFC 4648 §4 Base64. |
| 84 | /// |
| 85 | /// # Examples |
| 86 | /// ``` |
| 87 | /// use oxedyne_fe2o3_text::base64; |
| 88 | /// |
| 89 | /// assert_eq!(base64::encode(b"foobar"), "Zm9vYmFy"); |
| 90 | /// assert_eq!(base64::encode(b"foo"), "Zm9v"); |
| 91 | /// assert_eq!(base64::encode(b"fo"), "Zm8="); |
| 92 | /// ``` |
| 93 | pub fn encode(data: &[u8]) -> String { |
| 94 | encode_with(data, &ALPHABET, true) |
| 95 | } |
| 96 | |
| 97 | /// Encodes bytes as unpadded RFC 4648 §5 base64url. |
| 98 | /// |
| 99 | /// # Examples |
| 100 | /// ``` |
| 101 | /// use oxedyne_fe2o3_text::base64; |
| 102 | /// |
| 103 | /// assert_eq!(base64::encode_url(b"foobar"), "Zm9vYmFy"); |
| 104 | /// assert_eq!(base64::encode_url(b"fo"), "Zm8"); |
| 105 | /// assert_eq!(base64::encode_url(&[0xfb, 0xff]), "-_8"); |
| 106 | /// ``` |
| 107 | pub fn encode_url(data: &[u8]) -> String { |
| 108 | encode_with(data, &URL_ALPHABET, false) |
| 109 | } |
| 110 | |
| 111 | fn encode_with(data: &[u8], alphabet: &[u8; 64], pad: bool) -> String { |
| 112 | let mut out = String::with_capacity(encoded_len(data.len())); |
| 113 | for chunk in data.chunks(3) { |
| 114 | // The quantum, most significant byte first, zero filled when the input |
| 115 | // runs out. |
| 116 | let b0 = chunk[0] as u32; |
| 117 | let b1 = if chunk.len() > 1 { chunk[1] as u32 } else { 0 }; |
| 118 | let b2 = if chunk.len() > 2 { chunk[2] as u32 } else { 0 }; |
| 119 | let q = (b0 << 16) | (b1 << 8) | b2; |
| 120 | |
| 121 | out.push(alphabet[((q >> 18) & 0x3F) as usize] as char); |
| 122 | out.push(alphabet[((q >> 12) & 0x3F) as usize] as char); |
| 123 | match chunk.len() { |
| 124 | 1 => if pad { |
| 125 | out.push(PAD as char); |
| 126 | out.push(PAD as char); |
| 127 | }, |
| 128 | 2 => { |
| 129 | out.push(alphabet[((q >> 6) & 0x3F) as usize] as char); |
| 130 | if pad { |
| 131 | out.push(PAD as char); |
| 132 | } |
| 133 | }, |
| 134 | _ => { |
| 135 | out.push(alphabet[((q >> 6) & 0x3F) as usize] as char); |
| 136 | out.push(alphabet[(q & 0x3F) as usize] as char); |
| 137 | }, |
| 138 | } |
| 139 | } |
| 140 | out |
| 141 | } |
| 142 | |
| 143 | /// Decodes standard, padded RFC 4648 §4 Base64, refusing anything that is not. |
| 144 | /// |
| 145 | /// See the module documentation for exactly what is refused. |
| 146 | /// |
| 147 | /// # Examples |
| 148 | /// ``` |
| 149 | /// use oxedyne_fe2o3_text::base64; |
| 150 | /// |
| 151 | /// assert!(base64::decode("Zm9vYmFy").is_ok()); |
| 152 | /// assert!(base64::decode("Zm9").is_err()); // Not a whole quantum. |
| 153 | /// assert!(base64::decode("Zm9v Zg==").is_err()); // Whitespace is not alphabet. |
| 154 | /// ``` |
| 155 | pub fn decode(s: &str) -> Outcome<Vec<u8>> { |
| 156 | let src = s.as_bytes(); |
| 157 | let n = src.len(); |
| 158 | |
| 159 | if n % 4 != 0 { |
| 160 | return Err(err!( |
| 161 | "Base64 input is {} characters long, which is not a multiple of 4; \ |
| 162 | RFC 4648 §4 pads every encoding out to whole 4 character quanta.", n; |
| 163 | Invalid, Input, Decode, Size)); |
| 164 | } |
| 165 | if n == 0 { |
| 166 | return Ok(Vec::new()); |
| 167 | } |
| 168 | |
| 169 | // Padding is legal only as the last one or two characters. Counting it here |
| 170 | // tells the final quantum how many bytes it carries; a '=' anywhere else is |
| 171 | // caught by `sextet`, which refuses it like any other non-alphabet byte. |
| 172 | let pad = if src[n - 1] == PAD { |
| 173 | if src[n - 2] == PAD { 2 } else { 1 } |
| 174 | } else { |
| 175 | 0 |
| 176 | }; |
| 177 | |
| 178 | let mut out = Vec::with_capacity((n / 4) * 3 - pad); |
| 179 | |
| 180 | // Every quantum but the last carries three whole bytes. |
| 181 | let last = n - 4; |
| 182 | let mut i = 0; |
| 183 | while i < last { |
| 184 | let q = (res!(sextet(src, i)) << 18) |
| 185 | | (res!(sextet(src, i + 1)) << 12) |
| 186 | | (res!(sextet(src, i + 2)) << 6) |
| 187 | | res!(sextet(src, i + 3)); |
| 188 | out.push((q >> 16) as u8); |
| 189 | out.push((q >> 8) as u8); |
| 190 | out.push(q as u8); |
| 191 | i += 4; |
| 192 | } |
| 193 | |
| 194 | // The last quantum, whose padding says how much of it is data. |
| 195 | let a = res!(sextet(src, last)); |
| 196 | let b = res!(sextet(src, last + 1)); |
| 197 | match pad { |
| 198 | 2 => { |
| 199 | // Two characters carry 12 bits for 1 byte, so 4 bits go unused. |
| 200 | if b & 0x0F != 0 { |
| 201 | return Err(err!( |
| 202 | "Base64 input ends with '{}{}==', whose final character sets \ |
| 203 | bits the single decoded byte cannot hold; RFC 4648 §3.5 \ |
| 204 | requires those bits to be zero.", |
| 205 | char::from(src[last]), char::from(src[last + 1]); |
| 206 | Invalid, Input, Decode)); |
| 207 | } |
| 208 | out.push(((a << 2) | (b >> 4)) as u8); |
| 209 | }, |
| 210 | 1 => { |
| 211 | let c = res!(sextet(src, last + 2)); |
| 212 | // Three characters carry 18 bits for 2 bytes, so 2 bits go unused. |
| 213 | if c & 0x03 != 0 { |
| 214 | return Err(err!( |
| 215 | "Base64 input ends with '{}{}{}=', whose final character sets \ |
| 216 | bits the 2 decoded bytes cannot hold; RFC 4648 §3.5 requires \ |
| 217 | those bits to be zero.", |
| 218 | char::from(src[last]), char::from(src[last + 1]), |
| 219 | char::from(src[last + 2]); |
| 220 | Invalid, Input, Decode)); |
| 221 | } |
| 222 | out.push(((a << 2) | (b >> 4)) as u8); |
| 223 | out.push((((b & 0x0F) << 4) | (c >> 2)) as u8); |
| 224 | }, |
| 225 | _ => { |
| 226 | let c = res!(sextet(src, last + 2)); |
| 227 | let d = res!(sextet(src, last + 3)); |
| 228 | out.push(((a << 2) | (b >> 4)) as u8); |
| 229 | out.push((((b & 0x0F) << 4) | (c >> 2)) as u8); |
| 230 | out.push((((c & 0x03) << 6) | d) as u8); |
| 231 | }, |
| 232 | } |
| 233 | |
| 234 | Ok(out) |
| 235 | } |
| 236 | |
| 237 | /// Decodes unpadded RFC 4648 §5 base64url, refusing anything that is not. |
| 238 | /// |
| 239 | /// See the module documentation for exactly what is refused. |
| 240 | /// |
| 241 | /// # Examples |
| 242 | /// ``` |
| 243 | /// use oxedyne_fe2o3_text::base64; |
| 244 | /// |
| 245 | /// assert_eq!(base64::decode_url("-_8").ok(), Some(vec![0xfb, 0xff])); |
| 246 | /// assert!(base64::decode_url("Zm8=").is_err()); // Padding is not part of it. |
| 247 | /// assert!(base64::decode_url("+/8").is_err()); // Nor is the §4 alphabet. |
| 248 | /// ``` |
| 249 | pub fn decode_url(s: &str) -> Outcome<Vec<u8>> { |
| 250 | let src = s.as_bytes(); |
| 251 | let n = src.len(); |
| 252 | if n % 4 == 1 { |
| 253 | return Err(err!( |
| 254 | "Unpadded base64url input is {} characters long, which leaves one \ |
| 255 | character over a whole number of 4 character quanta, and one character \ |
| 256 | carries 6 bits, too few for a byte.", n; |
| 257 | Invalid, Input, Decode, Size)); |
| 258 | } |
| 259 | |
| 260 | let mut out = Vec::with_capacity(n * 3 / 4); |
| 261 | |
| 262 | // Every whole quantum carries three bytes. |
| 263 | let whole = n - n % 4; |
| 264 | let mut i = 0; |
| 265 | while i < whole { |
| 266 | let q = (res!(url_sextet(src, i)) << 18) |
| 267 | | (res!(url_sextet(src, i + 1)) << 12) |
| 268 | | (res!(url_sextet(src, i + 2)) << 6) |
| 269 | | res!(url_sextet(src, i + 3)); |
| 270 | out.push((q >> 16) as u8); |
| 271 | out.push((q >> 8) as u8); |
| 272 | out.push(q as u8); |
| 273 | i += 4; |
| 274 | } |
| 275 | |
| 276 | // A short final quantum, which the missing padding would have filled out. |
| 277 | match n - whole { |
| 278 | 2 => { |
| 279 | let a = res!(url_sextet(src, whole)); |
| 280 | let b = res!(url_sextet(src, whole + 1)); |
| 281 | // Two characters carry 12 bits for 1 byte, so 4 bits go unused. |
| 282 | if b & 0x0F != 0 { |
| 283 | return Err(err!( |
| 284 | "Base64url input ends with '{}{}', whose final character sets \ |
| 285 | bits the single decoded byte cannot hold; RFC 4648 §3.5 \ |
| 286 | requires those bits to be zero.", |
| 287 | char::from(src[whole]), char::from(src[whole + 1]); |
| 288 | Invalid, Input, Decode)); |
| 289 | } |
| 290 | out.push(((a << 2) | (b >> 4)) as u8); |
| 291 | }, |
| 292 | 3 => { |
| 293 | let a = res!(url_sextet(src, whole)); |
| 294 | let b = res!(url_sextet(src, whole + 1)); |
| 295 | let c = res!(url_sextet(src, whole + 2)); |
| 296 | // Three characters carry 18 bits for 2 bytes, so 2 bits go unused. |
| 297 | if c & 0x03 != 0 { |
| 298 | return Err(err!( |
| 299 | "Base64url input ends with '{}{}{}', whose final character sets \ |
| 300 | bits the 2 decoded bytes cannot hold; RFC 4648 §3.5 requires \ |
| 301 | those bits to be zero.", |
| 302 | char::from(src[whole]), char::from(src[whole + 1]), |
| 303 | char::from(src[whole + 2]); |
| 304 | Invalid, Input, Decode)); |
| 305 | } |
| 306 | out.push(((a << 2) | (b >> 4)) as u8); |
| 307 | out.push((((b & 0x0F) << 4) | (c >> 2)) as u8); |
| 308 | }, |
| 309 | _ => (), |
| 310 | } |
| 311 | |
| 312 | Ok(out) |
| 313 | } |
| 314 | |
| 315 | /// The 6 bit value of the base64url character at `i`. |
| 316 | fn url_sextet(src: &[u8], i: usize) -> Outcome<u32> { |
| 317 | let c = src[i]; |
| 318 | let v = URL_DECODE[c as usize]; |
| 319 | if v == INVALID { |
| 320 | let why = match c { |
| 321 | PAD => "padding, which unpadded base64url never carries", |
| 322 | b'+' | b'/' => "from the RFC 4648 §4 alphabet, not the §5 one", |
| 323 | _ => "not in the RFC 4648 §5 alphabet", |
| 324 | }; |
| 325 | return Err(err!( |
| 326 | "Base64url input has '{}' (byte 0x{:02x}) at index {} of {}, which is {}.", |
| 327 | char::from(c).escape_default(), c, i, src.len(), why; |
| 328 | Invalid, Input, Decode)); |
| 329 | } |
| 330 | Ok(v as u32) |
| 331 | } |
| 332 | |
| 333 | /// The 6 bit value of the alphabet character at `i`. |
| 334 | fn sextet(src: &[u8], i: usize) -> Outcome<u32> { |
| 335 | let c = src[i]; |
| 336 | let v = DECODE[c as usize]; |
| 337 | if v == INVALID { |
| 338 | if c == PAD { |
| 339 | return Err(err!( |
| 340 | "Base64 input has padding '=' at index {} of {}, but RFC 4648 §4 \ |
| 341 | allows it only as the last one or two characters.", i, src.len(); |
| 342 | Invalid, Input, Decode)); |
| 343 | } |
| 344 | return Err(err!( |
| 345 | "Base64 input has '{}' (byte 0x{:02x}) at index {}, which is not in \ |
| 346 | the RFC 4648 §4 alphabet.", char::from(c).escape_default(), c, i; |
| 347 | Invalid, Input, Decode)); |
| 348 | } |
| 349 | Ok(v as u32) |
| 350 | } |
| 351 | |
| 352 | |
| 353 | #[cfg(test)] |
| 354 | mod tests { |
| 355 | use super::*; |
| 356 | |
| 357 | /// RFC 4648 §10 lists these seven encodings verbatim. They are the oracle: |
| 358 | /// they did not come from this implementation, and an implementation that is |
| 359 | /// wrong in a self-consistent way still fails them. |
| 360 | const RFC4648_VECTORS: [(&str, &str); 7] = [ |
| 361 | ("", ""), |
| 362 | ("f", "Zg=="), |
| 363 | ("fo", "Zm8="), |
| 364 | ("foo", "Zm9v"), |
| 365 | ("foob", "Zm9vYg=="), |
| 366 | ("fooba", "Zm9vYmE="), |
| 367 | ("foobar", "Zm9vYmFy"), |
| 368 | ]; |
| 369 | |
| 370 | #[test] |
| 371 | fn test_the_rfc_4648_test_vectors_encode_as_the_rfc_says() { |
| 372 | for (plain, encoded) in RFC4648_VECTORS { |
| 373 | assert_eq!(encode(plain.as_bytes()), encoded, "encoding {:?}", plain); |
| 374 | } |
| 375 | } |
| 376 | |
| 377 | #[test] |
| 378 | fn test_the_rfc_4648_test_vectors_decode_as_the_rfc_says() -> Outcome<()> { |
| 379 | for (plain, encoded) in RFC4648_VECTORS { |
| 380 | let got = res!(decode(encoded)); |
| 381 | assert_eq!(got, plain.as_bytes(), "decoding {:?}", encoded); |
| 382 | } |
| 383 | Ok(()) |
| 384 | } |
| 385 | |
| 386 | /// Fixtures produced by an independent tool, so that agreement is agreement |
| 387 | /// with somebody else. Each was generated with GNU-compatible `base64(1)`: |
| 388 | /// |
| 389 | /// ```text |
| 390 | /// $ printf 'Hematite' | base64 |
| 391 | /// SGVtYXRpdGU= |
| 392 | /// $ printf '\x00' | base64 |
| 393 | /// AA== |
| 394 | /// $ printf '\xff\xff' | base64 |
| 395 | /// //8= |
| 396 | /// $ printf '\x00\x01\x02\xfd\xfe\xff' | base64 |
| 397 | /// AAEC/f7/ |
| 398 | /// $ printf '\x00\x01\x02 ... \x1f' | base64 -w0 |
| 399 | /// AAECAwQFBgcICQoLDA0ODxAREhMUFRYXGBkaGxwdHh8= |
| 400 | /// ``` |
| 401 | #[test] |
| 402 | fn test_an_external_tool_agrees_with_this_encoder() -> Outcome<()> { |
| 403 | let cases: [(Vec<u8>, &str); 5] = [ |
| 404 | (b"Hematite".to_vec(), "SGVtYXRpdGU="), |
| 405 | (vec![0x00], "AA=="), |
| 406 | (vec![0xFF, 0xFF], "//8="), |
| 407 | (vec![0x00, 0x01, 0x02, 0xFD, 0xFE, 0xFF], "AAEC/f7/"), |
| 408 | ((0u8..32).collect::<Vec<u8>>(), |
| 409 | "AAECAwQFBgcICQoLDA0ODxAREhMUFRYXGBkaGxwdHh8="), |
| 410 | ]; |
| 411 | for (bytes, expected) in cases { |
| 412 | assert_eq!(encode(&bytes), expected, "encoding {:?}", bytes); |
| 413 | assert_eq!(res!(decode(expected)), bytes, "decoding {:?}", expected); |
| 414 | } |
| 415 | Ok(()) |
| 416 | } |
| 417 | |
| 418 | /// Binary is not text: the bytes a codec is most likely to mangle are the |
| 419 | /// ones no string contains, so every length up to 3 quanta of every byte |
| 420 | /// value goes round. |
| 421 | #[test] |
| 422 | fn test_arbitrary_bytes_survive_a_round_trip() -> Outcome<()> { |
| 423 | for len in 0..=12 { |
| 424 | for start in 0..=255u16 { |
| 425 | let bytes: Vec<u8> = (0..len) |
| 426 | .map(|k| ((start as usize + k * 37) % 256) as u8) |
| 427 | .collect(); |
| 428 | let encoded = encode(&bytes); |
| 429 | assert_eq!(encoded.len(), encoded_len(bytes.len())); |
| 430 | let got = res!(decode(&encoded)); |
| 431 | assert_eq!(got, bytes, "round trip of {:?} via {:?}", bytes, encoded); |
| 432 | } |
| 433 | } |
| 434 | Ok(()) |
| 435 | } |
| 436 | |
| 437 | /// A run of every byte value, including the 0x00 and 0xFF that a text-shaped |
| 438 | /// implementation trips over. |
| 439 | #[test] |
| 440 | fn test_every_byte_value_survives_a_round_trip() -> Outcome<()> { |
| 441 | let bytes: Vec<u8> = (0..=255u8).collect(); |
| 442 | let encoded = encode(&bytes); |
| 443 | assert_eq!(res!(decode(&encoded)), bytes); |
| 444 | Ok(()) |
| 445 | } |
| 446 | |
| 447 | /// Rejection is the whole point of a strict decoder: a decoder that guesses |
| 448 | /// is a decoder that disagrees with the one at the other end. |
| 449 | #[test] |
| 450 | fn test_malformed_input_is_refused() { |
| 451 | let bad = [ |
| 452 | ("Zm9", "length not a multiple of 4"), |
| 453 | ("Z", "length not a multiple of 4"), |
| 454 | ("Zm9vYmFyZ", "length not a multiple of 4"), |
| 455 | ("Zm9$", "character outside the alphabet"), |
| 456 | ("Zm9v Zg==", "space is not in the alphabet"), |
| 457 | ("Zm9v\nZg==", "newline is not in the alphabet"), |
| 458 | ("Zm-_", "URL-safe alphabet is a different encoding"), |
| 459 | ("Zmé", "non-ASCII is not in the alphabet"), |
| 460 | ("Zm=vYg==", "padding before the last quantum"), |
| 461 | ("=m9vYg==", "padding at the start of a quantum"), |
| 462 | ("Zg=A", "data after padding"), |
| 463 | ("Z===", "three padding characters"), |
| 464 | ("====", "a quantum of nothing but padding"), |
| 465 | ("Zh==", "final character sets bits the byte cannot hold"), |
| 466 | ("Zm9=", "final character sets bits the bytes cannot hold"), |
| 467 | ]; |
| 468 | for (input, why) in bad { |
| 469 | assert!(decode(input).is_err(), "accepted {:?}, but {}", input, why); |
| 470 | } |
| 471 | } |
| 472 | |
| 473 | /// The canonical forms of those same shapes are accepted, so the rejections |
| 474 | /// above are about the fault and not about the neighbourhood. |
| 475 | #[test] |
| 476 | fn test_the_canonical_neighbours_are_accepted() -> Outcome<()> { |
| 477 | for good in ["Zg==", "Zm8=", "Zm9v", "AA==", "AQ==", "//8="] { |
| 478 | res!(decode(good)); |
| 479 | } |
| 480 | Ok(()) |
| 481 | } |
| 482 | |
| 483 | /// An independent implementation, the `base64` crate, over a spread of |
| 484 | /// lengths and byte values. It is a dev-dependency only: the point of this |
| 485 | /// module is that the library dependency can go, and the point of this test |
| 486 | /// is to show that dropping it changes nothing on the wire. |
| 487 | #[test] |
| 488 | fn test_an_independent_implementation_agrees() -> Outcome<()> { |
| 489 | // A cheap deterministic spread, so a failure is reproducible. |
| 490 | let mut state: u32 = 0x1234_5678; |
| 491 | for len in 0..200 { |
| 492 | let bytes: Vec<u8> = (0..len) |
| 493 | .map(|_| { |
| 494 | state = state.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); |
| 495 | (state >> 24) as u8 |
| 496 | }) |
| 497 | .collect(); |
| 498 | let ours = encode(&bytes); |
| 499 | let theirs = ::base64::encode(&bytes); |
| 500 | assert_eq!(ours, theirs, "encoding {} bytes: {:?}", len, bytes); |
| 501 | assert_eq!(res!(decode(&theirs)), bytes, "decoding {:?}", theirs); |
| 502 | } |
| 503 | Ok(()) |
| 504 | } |
| 505 | |
| 506 | /// A length calculation that disagrees with the encoder is a buffer that is |
| 507 | /// resized on every call. |
| 508 | #[test] |
| 509 | fn test_the_predicted_length_matches_the_encoding() { |
| 510 | for len in 0..=64 { |
| 511 | let bytes = vec![0xA5u8; len]; |
| 512 | assert_eq!(encode(&bytes).len(), encoded_len(len), "for {} bytes", len); |
| 513 | assert_eq!(encode_url(&bytes).len(), encoded_url_len(len), "for {} bytes", len); |
| 514 | } |
| 515 | } |
| 516 | |
| 517 | /// RFC 4648 §10's vectors with the padding taken off. None of them reaches |
| 518 | /// the two characters where §5 departs from §4, so dropping the padding is |
| 519 | /// the whole of the difference. |
| 520 | const RFC4648_URL_VECTORS: [(&str, &str); 7] = [ |
| 521 | ("", ""), |
| 522 | ("f", "Zg"), |
| 523 | ("fo", "Zm8"), |
| 524 | ("foo", "Zm9v"), |
| 525 | ("foob", "Zm9vYg"), |
| 526 | ("fooba", "Zm9vYmE"), |
| 527 | ("foobar", "Zm9vYmFy"), |
| 528 | ]; |
| 529 | |
| 530 | #[test] |
| 531 | fn test_the_rfc_4648_vectors_unpadded_go_both_ways() -> Outcome<()> { |
| 532 | for (plain, encoded) in RFC4648_URL_VECTORS { |
| 533 | assert_eq!(encode_url(plain.as_bytes()), encoded, "encoding {:?}", plain); |
| 534 | assert_eq!(res!(decode_url(encoded)), plain.as_bytes(), "decoding {:?}", encoded); |
| 535 | } |
| 536 | Ok(()) |
| 537 | } |
| 538 | |
| 539 | /// `-` and `_` stand exactly where the standard encoding has `+` and `/`. |
| 540 | #[test] |
| 541 | fn test_the_url_alphabet_replaces_plus_and_slash() -> Outcome<()> { |
| 542 | assert_eq!(encode(&[0xfb, 0xff]), "+/8="); |
| 543 | assert_eq!(encode_url(&[0xfb, 0xff]), "-_8"); |
| 544 | assert_eq!(res!(decode_url("-_8")), vec![0xfb, 0xff]); |
| 545 | assert_eq!(encode_url(&[0xff]), "_w"); |
| 546 | assert_eq!(res!(decode_url("_w")), vec![0xff]); |
| 547 | Ok(()) |
| 548 | } |
| 549 | |
| 550 | /// The `base64` crate's `URL_SAFE_NO_PAD`, an independent implementation, |
| 551 | /// over a spread of lengths and byte values. |
| 552 | #[test] |
| 553 | fn test_an_independent_implementation_agrees_on_base64url() -> Outcome<()> { |
| 554 | let mut state: u32 = 0x8765_4321; |
| 555 | for len in 0..200 { |
| 556 | let bytes: Vec<u8> = (0..len) |
| 557 | .map(|_| { |
| 558 | state = state.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); |
| 559 | (state >> 24) as u8 |
| 560 | }) |
| 561 | .collect(); |
| 562 | let ours = encode_url(&bytes); |
| 563 | let theirs = ::base64::encode_config(&bytes, ::base64::URL_SAFE_NO_PAD); |
| 564 | assert_eq!(ours, theirs, "encoding {} bytes: {:?}", len, bytes); |
| 565 | assert_eq!(res!(decode_url(&theirs)), bytes, "decoding {:?}", theirs); |
| 566 | } |
| 567 | Ok(()) |
| 568 | } |
| 569 | |
| 570 | /// Every byte string has one accepted spelling, so everything else is |
| 571 | /// refused rather than read the way some other decoder might read it. |
| 572 | #[test] |
| 573 | fn test_malformed_base64url_is_refused() { |
| 574 | let bad = [ |
| 575 | ("Z", "one character over a whole quantum"), |
| 576 | ("Zm9vY", "one character over a whole quantum"), |
| 577 | ("Zg==", "padding"), |
| 578 | ("Zm8=", "padding"), |
| 579 | ("Zm9v+g", "the plus of the standard alphabet"), |
| 580 | ("Zm9v/g", "the slash of the standard alphabet"), |
| 581 | ("Zm9v Zg", "space is not in the alphabet"), |
| 582 | ("Zm9v\nZg", "newline is not in the alphabet"), |
| 583 | ("Zm9$", "a character in neither alphabet"), |
| 584 | ("Zmé", "non-ASCII is not in the alphabet"), |
| 585 | ("Zh", "final character sets bits the byte cannot hold"), |
| 586 | ("Zm9", "final character sets bits the bytes cannot hold"), |
| 587 | ]; |
| 588 | for (input, why) in bad { |
| 589 | assert!(decode_url(input).is_err(), "accepted {:?}, but {}", input, why); |
| 590 | } |
| 591 | } |
| 592 | |
| 593 | /// The canonical forms of those same shapes are accepted, so the refusals |
| 594 | /// above are about the fault and not about the neighbourhood. |
| 595 | #[test] |
| 596 | fn test_the_canonical_base64url_neighbours_are_accepted() -> Outcome<()> { |
| 597 | for good in ["", "Zg", "Zm8", "Zm9v", "AA", "AQ", "_w", "-_8"] { |
| 598 | res!(decode_url(good)); |
| 599 | } |
| 600 | Ok(()) |
| 601 | } |
| 602 | } |