Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_net/src/http/pct.rs

5.0 KiB, 5 runs

created by r1870400018:20236, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Percent-encoding, the escape a URI uses for everything it cannot spell.
2//!
3//! Both directions are byte-wise rather than character-wise, because what is escaped need not be
4//! text: a data URL carries arbitrary bytes, and a query parameter may carry UTF-8 that the
5//! encoder must not have opinions about.
6//!
7//! [`encode_component`] escapes what JavaScript's `encodeURIComponent` escapes, and nothing else.
8//! That is a deliberate match: a value a browser wrote and a peer reads has to survive the round
9//! trip byte for byte, and the browser's rule is the one that is not ours to choose.
10//!
11//! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\
12//! Anthropic Claude
13
14use oxedyne_fe2o3_core::prelude::*;
15
16
17/// Decode percent escapes, over bytes.
18///
19/// A `%` must be followed by two hexadecimal digits; anything else is an error rather than a
20/// character passed through, since a caller that cannot tell an escape from a literal percent
21/// cannot tell what it decoded.
22pub fn decode(s: &str) -> Outcome<Vec<u8>> {
23 let src = s.as_bytes();
24 let mut out = Vec::with_capacity(src.len());
25 let mut i = 0;
26 while i < src.len() {
27 match src[i] {
28 b'%' => {
29 if i + 2 >= src.len() {
30 return Err(err!(
31 "A percent escape must be followed by two hexadecimal digits.";
32 Invalid, Input, Decode));
33 }
34 let hex = match std::str::from_utf8(&src[i + 1..i + 3]) {
35 Ok(h) => h,
36 Err(_) => return Err(err!(
37 "A percent escape is not text."; Invalid, Input, Decode)),
38 };
39 match u8::from_str_radix(hex, 16) {
40 Ok(b) => out.push(b),
41 Err(e) => return Err(err!(e,
42 "'{}' is not a pair of hexadecimal digits.", hex;
43 Invalid, Input, Decode)),
44 }
45 i += 3;
46 }
47 b => {
48 out.push(b);
49 i += 1;
50 }
51 }
52 }
53 Ok(out)
54}
55
56/// Decode percent escapes and require the result to be UTF-8 text.
57pub fn decode_str(s: &str) -> Outcome<String> {
58 let byts = res!(decode(s));
59 match String::from_utf8(byts) {
60 Ok(s) => Ok(s),
61 Err(e) => Err(err!(e,
62 "The percent-decoded bytes are not UTF-8 text.";
63 Invalid, Input, Decode, String)),
64 }
65}
66
67/// Percent-encode one component of a URI, escaping exactly what `encodeURIComponent` escapes.
68///
69/// The unreserved set left alone is `A-Z a-z 0-9 - _ . ! ~ * ' ( )`. Everything else, including
70/// `/`, `:`, `?`, `&`, `=` and every byte of a multi-byte character, is escaped with upper-case
71/// hexadecimal digits, as the browsers emit them.
72pub fn encode_component(s: &str) -> String {
73 let mut out = String::with_capacity(s.len());
74 for b in s.as_bytes() {
75 match b {
76 b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9'
77 | b'-' | b'_' | b'.' | b'!' | b'~' | b'*' | b'\'' | b'(' | b')' => {
78 out.push(*b as char)
79 },
80 other => out.push_str(&fmt!("%{:02X}", other)),
81 }
82 }
83 out
84}
85
86
87#[cfg(test)]
88mod tests {
89 use super::*;
90
91 /// The escapes `encodeURIComponent` makes, and the ones it does not. The expected strings are
92 /// what a browser console prints, which is the whole reason this function exists.
93 #[test]
94 fn test_encode_component_matches_the_browser_00() {
95 assert_eq!(encode_component("Funky Bear"), "Funky%20Bear");
96 assert_eq!(encode_component("a/b:c?d&e=f"), "a%2Fb%3Ac%3Fd%26e%3Df");
97 assert_eq!(encode_component("-_.!~*'()"), "-_.!~*'()");
98 assert_eq!(encode_component("Cœur"), "C%C5%93ur");
99 assert_eq!(encode_component(""), "");
100 }
101
102 /// What was encoded decodes back to what went in, including the bytes of a character no ASCII
103 /// escape could carry.
104 #[test]
105 fn test_percent_round_trip_00() -> Outcome<()> {
106 for original in ["Funky Bear", "Tree Hugger", "a/b:c", "Cœur — 日本", ""] {
107 let encoded = encode_component(original);
108 assert_eq!(res!(decode_str(&encoded)), original);
109 }
110 Ok(())
111 }
112
113 /// A truncated or malformed escape is an error, not a percent sign passed through: a caller
114 /// that cannot tell the two apart does not know what it decoded.
115 #[test]
116 fn test_decode_rejects_a_broken_escape_00() {
117 assert!(decode("%2").is_err(), "a truncated escape must be refused");
118 assert!(decode("%").is_err(), "a bare percent must be refused");
119 assert!(decode("%zz").is_err(), "non-hexadecimal digits must be refused");
120 }
121
122 /// Bytes that are not text decode as bytes, and are only refused when a caller asked for text.
123 #[test]
124 fn test_decode_bytes_that_are_not_text_00() -> Outcome<()> {
125 assert_eq!(res!(decode("%FF%FE")), vec![0xFF, 0xFE]);
126 assert!(decode_str("%FF%FE").is_err(), "those bytes are not UTF-8");
127 Ok(())
128 }
129}