oxedyne/fe2o3/fe2o3_net/src/http/pct.rs
5.0 KiB, 5 runs
created by r1870400018:20236, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Percent-encoding, the escape a URI uses for everything it cannot spell. |
| 2 | //! |
| 3 | //! Both directions are byte-wise rather than character-wise, because what is escaped need not be |
| 4 | //! text: a data URL carries arbitrary bytes, and a query parameter may carry UTF-8 that the |
| 5 | //! encoder must not have opinions about. |
| 6 | //! |
| 7 | //! [`encode_component`] escapes what JavaScript's `encodeURIComponent` escapes, and nothing else. |
| 8 | //! That is a deliberate match: a value a browser wrote and a peer reads has to survive the round |
| 9 | //! trip byte for byte, and the browser's rule is the one that is not ours to choose. |
| 10 | //! |
| 11 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 12 | //! Anthropic Claude |
| 13 | |
| 14 | use oxedyne_fe2o3_core::prelude::*; |
| 15 | |
| 16 | |
| 17 | /// Decode percent escapes, over bytes. |
| 18 | /// |
| 19 | /// A `%` must be followed by two hexadecimal digits; anything else is an error rather than a |
| 20 | /// character passed through, since a caller that cannot tell an escape from a literal percent |
| 21 | /// cannot tell what it decoded. |
| 22 | pub fn decode(s: &str) -> Outcome<Vec<u8>> { |
| 23 | let src = s.as_bytes(); |
| 24 | let mut out = Vec::with_capacity(src.len()); |
| 25 | let mut i = 0; |
| 26 | while i < src.len() { |
| 27 | match src[i] { |
| 28 | b'%' => { |
| 29 | if i + 2 >= src.len() { |
| 30 | return Err(err!( |
| 31 | "A percent escape must be followed by two hexadecimal digits."; |
| 32 | Invalid, Input, Decode)); |
| 33 | } |
| 34 | let hex = match std::str::from_utf8(&src[i + 1..i + 3]) { |
| 35 | Ok(h) => h, |
| 36 | Err(_) => return Err(err!( |
| 37 | "A percent escape is not text."; Invalid, Input, Decode)), |
| 38 | }; |
| 39 | match u8::from_str_radix(hex, 16) { |
| 40 | Ok(b) => out.push(b), |
| 41 | Err(e) => return Err(err!(e, |
| 42 | "'{}' is not a pair of hexadecimal digits.", hex; |
| 43 | Invalid, Input, Decode)), |
| 44 | } |
| 45 | i += 3; |
| 46 | } |
| 47 | b => { |
| 48 | out.push(b); |
| 49 | i += 1; |
| 50 | } |
| 51 | } |
| 52 | } |
| 53 | Ok(out) |
| 54 | } |
| 55 | |
| 56 | /// Decode percent escapes and require the result to be UTF-8 text. |
| 57 | pub fn decode_str(s: &str) -> Outcome<String> { |
| 58 | let byts = res!(decode(s)); |
| 59 | match String::from_utf8(byts) { |
| 60 | Ok(s) => Ok(s), |
| 61 | Err(e) => Err(err!(e, |
| 62 | "The percent-decoded bytes are not UTF-8 text."; |
| 63 | Invalid, Input, Decode, String)), |
| 64 | } |
| 65 | } |
| 66 | |
| 67 | /// Percent-encode one component of a URI, escaping exactly what `encodeURIComponent` escapes. |
| 68 | /// |
| 69 | /// The unreserved set left alone is `A-Z a-z 0-9 - _ . ! ~ * ' ( )`. Everything else, including |
| 70 | /// `/`, `:`, `?`, `&`, `=` and every byte of a multi-byte character, is escaped with upper-case |
| 71 | /// hexadecimal digits, as the browsers emit them. |
| 72 | pub fn encode_component(s: &str) -> String { |
| 73 | let mut out = String::with_capacity(s.len()); |
| 74 | for b in s.as_bytes() { |
| 75 | match b { |
| 76 | b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' |
| 77 | | b'-' | b'_' | b'.' | b'!' | b'~' | b'*' | b'\'' | b'(' | b')' => { |
| 78 | out.push(*b as char) |
| 79 | }, |
| 80 | other => out.push_str(&fmt!("%{:02X}", other)), |
| 81 | } |
| 82 | } |
| 83 | out |
| 84 | } |
| 85 | |
| 86 | |
| 87 | #[cfg(test)] |
| 88 | mod tests { |
| 89 | use super::*; |
| 90 | |
| 91 | /// The escapes `encodeURIComponent` makes, and the ones it does not. The expected strings are |
| 92 | /// what a browser console prints, which is the whole reason this function exists. |
| 93 | #[test] |
| 94 | fn test_encode_component_matches_the_browser_00() { |
| 95 | assert_eq!(encode_component("Funky Bear"), "Funky%20Bear"); |
| 96 | assert_eq!(encode_component("a/b:c?d&e=f"), "a%2Fb%3Ac%3Fd%26e%3Df"); |
| 97 | assert_eq!(encode_component("-_.!~*'()"), "-_.!~*'()"); |
| 98 | assert_eq!(encode_component("Cœur"), "C%C5%93ur"); |
| 99 | assert_eq!(encode_component(""), ""); |
| 100 | } |
| 101 | |
| 102 | /// What was encoded decodes back to what went in, including the bytes of a character no ASCII |
| 103 | /// escape could carry. |
| 104 | #[test] |
| 105 | fn test_percent_round_trip_00() -> Outcome<()> { |
| 106 | for original in ["Funky Bear", "Tree Hugger", "a/b:c", "Cœur — 日本", ""] { |
| 107 | let encoded = encode_component(original); |
| 108 | assert_eq!(res!(decode_str(&encoded)), original); |
| 109 | } |
| 110 | Ok(()) |
| 111 | } |
| 112 | |
| 113 | /// A truncated or malformed escape is an error, not a percent sign passed through: a caller |
| 114 | /// that cannot tell the two apart does not know what it decoded. |
| 115 | #[test] |
| 116 | fn test_decode_rejects_a_broken_escape_00() { |
| 117 | assert!(decode("%2").is_err(), "a truncated escape must be refused"); |
| 118 | assert!(decode("%").is_err(), "a bare percent must be refused"); |
| 119 | assert!(decode("%zz").is_err(), "non-hexadecimal digits must be refused"); |
| 120 | } |
| 121 | |
| 122 | /// Bytes that are not text decode as bytes, and are only refused when a caller asked for text. |
| 123 | #[test] |
| 124 | fn test_decode_bytes_that_are_not_text_00() -> Outcome<()> { |
| 125 | assert_eq!(res!(decode("%FF%FE")), vec![0xFF, 0xFE]); |
| 126 | assert!(decode_str("%FF%FE").is_err(), "those bytes are not UTF-8"); |
| 127 | Ok(()) |
| 128 | } |
| 129 | } |