oxedyne/fe2o3/fe2o3_net/src/http/loc.rs
20.5 KiB, 54 runs
created by r1870400018:579, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | /// HTTP URL locator parsing and manipulation. |
| 2 | /// |
| 3 | /// This module handles parsing and manipulating HTTP URL locators, which consist of: |
| 4 | /// - A path component |
| 5 | /// - Optional query parameters |
| 6 | /// - Optional fragment identifier |
| 7 | /// |
| 8 | /// The locator follows the standard URL format: |
| 9 | /// `path?param1=value1¶m2=value2#fragment` |
| 10 | /// |
| 11 | /// # Examples |
| 12 | /// ``` |
| 13 | /// use oxedyne_fe2o3_core::prelude::*; |
| 14 | /// use oxedyne_fe2o3_jdat::prelude::*; |
| 15 | /// use oxedyne_fe2o3_net::http::loc::HttpLocator; |
| 16 | /// |
| 17 | /// # fn main() -> Outcome<()> { |
| 18 | /// let loc = res!(HttpLocator::new("/path?key=value#section")); |
| 19 | /// assert_eq!(loc.path.as_str(), "/path"); |
| 20 | /// assert!(loc.data.contains_key(&dat!("key"))); |
| 21 | /// assert_eq!(loc.frag, "section"); |
| 22 | /// # Ok(()) |
| 23 | /// # } |
| 24 | /// ``` |
| 25 | use crate::file::RequestPath; |
| 26 | use oxedyne_fe2o3_core::prelude::*; |
| 27 | use oxedyne_fe2o3_jdat::prelude::*; |
| 28 | |
| 29 | use std::fmt; |
| 30 | |
| 31 | |
| 32 | /// Represents a parsed HTTP URL locator with path, query params, and fragment. |
| 33 | /// |
| 34 | /// # Fields |
| 35 | /// * `path` - The validated request path component. |
| 36 | /// * `data` - Map of parsed query parameters. |
| 37 | /// * `query` - The raw, verbatim query substring (no leading `?`), kept so a |
| 38 | /// proxy can forward it byte-for-byte. The parse into `data` is convenient |
| 39 | /// for handlers but lossy (encoding, ordering, repeated keys), so it must not |
| 40 | /// be used to reconstruct a request target. |
| 41 | /// * `frag` - Fragment identifier (part after #). |
| 42 | #[derive(Clone, Debug, Default)] |
| 43 | pub struct HttpLocator { |
| 44 | pub path: RequestPath, |
| 45 | pub data: DaticleMap, |
| 46 | pub query: String, |
| 47 | pub frag: String, |
| 48 | } |
| 49 | |
| 50 | impl fmt::Display for HttpLocator { |
| 51 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
| 52 | // Write the path |
| 53 | ok!(write!(f, "{}", self.path.as_str())); |
| 54 | |
| 55 | if !self.data.is_empty() { |
| 56 | ok!(write!(f, "?")); |
| 57 | for (i, (k, v)) in self.data.iter().enumerate() { |
| 58 | ok!(write!(f, "{}={}", k, v)); |
| 59 | if i < self.data.len() - 1 { |
| 60 | ok!(write!(f, "&")); |
| 61 | } |
| 62 | } |
| 63 | } |
| 64 | |
| 65 | if !self.frag.is_empty() { |
| 66 | ok!(write!(f, "#{}", self.frag)); |
| 67 | } |
| 68 | |
| 69 | Ok(()) |
| 70 | } |
| 71 | } |
| 72 | |
| 73 | impl HttpLocator { |
| 74 | |
| 75 | pub fn new<S: Into<String>>(loc: S) -> Outcome<Self> { |
| 76 | let (path, data, query, frag) = res!(Self::parse_locator_string(&loc.into())); |
| 77 | Ok(Self { |
| 78 | path, |
| 79 | data, |
| 80 | query, |
| 81 | frag, |
| 82 | }) |
| 83 | } |
| 84 | |
| 85 | fn parse_locator_string(path: &str) -> Outcome<(RequestPath, DaticleMap, String, String)> { |
| 86 | let mut split = path.split('?'); |
| 87 | let path = RequestPath::new(split.next().unwrap_or_default().to_string()); |
| 88 | let rest = split.next().unwrap_or_default(); |
| 89 | let mut split_rest = rest.split('#'); |
| 90 | let query_string = split_rest.next().unwrap_or_default(); |
| 91 | let fragment = split_rest.next().unwrap_or_default(); |
| 92 | let map = res!(Self::parse_query_string(query_string)); |
| 93 | |
| 94 | // Retain the raw query verbatim as well as the parsed map: a proxy must |
| 95 | // forward it byte-for-byte, and the parsed map cannot be relied on to |
| 96 | // reconstruct it faithfully. |
| 97 | Ok((path, map, query_string.to_string(), fragment.to_string())) |
| 98 | } |
| 99 | |
| 100 | fn parse_query_string(query: &str) -> Outcome<DaticleMap> { |
| 101 | let mut result = DaticleMap::new(); |
| 102 | for pair in query.split('&') { |
| 103 | let mut split = pair.split('='); |
| 104 | if let (Some(k), Some(v)) = (split.next(), split.next()) { |
| 105 | let k = res!(Dat::from_str(k)); |
| 106 | let v = res!(Dat::from_str(v)); |
| 107 | result.insert(k, v); |
| 108 | } |
| 109 | } |
| 110 | Ok(result) |
| 111 | //Ok(query.split('&') |
| 112 | // .filter_map(|pair| { |
| 113 | // let mut split = pair.split('='); |
| 114 | // if let (Some(k), Some(v)) = (split.next(), split.next()) { |
| 115 | // let k = res!(Dat::from_str(k)); |
| 116 | // let v = res!(Dat::from_str(v)); |
| 117 | // Some((k, v)) |
| 118 | // } else { |
| 119 | // None |
| 120 | // } |
| 121 | // }) |
| 122 | // .collect() |
| 123 | //) |
| 124 | } |
| 125 | |
| 126 | /// A query parameter as a string, if it is present and is one. |
| 127 | /// |
| 128 | /// The parsed map holds daticles, because a query string is parsed with the |
| 129 | /// same reader as everything else, so a bare `view=geo` arrives as a |
| 130 | /// `Dat::Str` while `page=2` arrives as a number. Reading one out therefore |
| 131 | /// takes a match rather than a `get`, and every caller was writing the same |
| 132 | /// one. |
| 133 | pub fn query_str(&self, key: &str) -> Option<String> { |
| 134 | match self.data.get(&dat!(key)) { |
| 135 | Some(Dat::Str(s)) => Some(s.clone()), |
| 136 | _ => None, |
| 137 | } |
| 138 | } |
| 139 | |
| 140 | /// A query parameter as an integer, coerced from any numeric or string |
| 141 | /// value, falling back to `default` when it is absent or unreadable. |
| 142 | /// |
| 143 | /// The coercion matters: whether `page=2` parses as a number or as a string |
| 144 | /// is a detail of the query reader, and no caller should have to care. |
| 145 | pub fn query_i64(&self, key: &str, default: i64) -> i64 { |
| 146 | match self.data.get(&dat!(key)) { |
| 147 | Some(d) => d.get_i64().or_else(|| match d { |
| 148 | Dat::Str(s) => s.trim().parse::<i64>().ok(), |
| 149 | _ => None, |
| 150 | }).unwrap_or(default), |
| 151 | None => default, |
| 152 | } |
| 153 | } |
| 154 | |
| 155 | } |
| 156 | |
| 157 | |
| 158 | // ┌───────────────────────────────────────────────────────────────────────────┐ |
| 159 | // │ ABSOLUTE URL │ |
| 160 | // └───────────────────────────────────────────────────────────────────────────┘ |
| 161 | |
| 162 | /// The scheme of an absolute HTTP URL. |
| 163 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 164 | pub enum UrlScheme { |
| 165 | /// Plain HTTP, port 80 unless the URL says otherwise. |
| 166 | Http, |
| 167 | /// HTTP over TLS, port 443 unless the URL says otherwise. |
| 168 | Https, |
| 169 | } |
| 170 | |
| 171 | impl fmt::Display for UrlScheme { |
| 172 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
| 173 | match self { |
| 174 | Self::Http => write!(f, "http"), |
| 175 | Self::Https => write!(f, "https"), |
| 176 | } |
| 177 | } |
| 178 | } |
| 179 | |
| 180 | impl UrlScheme { |
| 181 | /// The port the scheme implies when the URL names none. |
| 182 | pub fn default_port(&self) -> u16 { |
| 183 | match self { |
| 184 | Self::Http => 80, |
| 185 | Self::Https => 443, |
| 186 | } |
| 187 | } |
| 188 | |
| 189 | /// Whether the connection this scheme names is TLS-wrapped. |
| 190 | pub fn is_tls(&self) -> bool { |
| 191 | matches!(self, Self::Https) |
| 192 | } |
| 193 | } |
| 194 | |
| 195 | /// An absolute HTTP URL, split into the four things a client needs to act on it. |
| 196 | /// |
| 197 | /// [`HttpLocator`] is the *server's* view of a URL -- the request target and |
| 198 | /// nothing else, because the host arrived separately in the `Host` header. This |
| 199 | /// is the *client's* view: the whole of it, including whom to connect to. |
| 200 | /// |
| 201 | /// The parse follows a browser's, in the one place where a naive parse is a |
| 202 | /// security hole. In `https://example.com@127.0.0.1/`, the host is |
| 203 | /// `127.0.0.1` and `example.com` is discarded userinfo, so a server that reads |
| 204 | /// the host as everything before the first `/` reaches the loopback address it |
| 205 | /// meant to refuse. Here, the host is what follows the last `@`. |
| 206 | #[derive(Clone, Debug, Eq, PartialEq)] |
| 207 | pub struct Url { |
| 208 | /// What to speak. |
| 209 | pub scheme: UrlScheme, |
| 210 | /// Whom to speak to: a host name or an IP literal, lowercased. |
| 211 | pub host: String, |
| 212 | /// The port, explicit or implied by the scheme. |
| 213 | pub port: u16, |
| 214 | /// What to ask for: the path and any query string, as written on the |
| 215 | /// request line. Never empty, and never carries the fragment, which is |
| 216 | /// the browser's business and is not sent. |
| 217 | pub target: String, |
| 218 | } |
| 219 | |
| 220 | impl fmt::Display for Url { |
| 221 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
| 222 | if self.port == self.scheme.default_port() { |
| 223 | write!(f, "{}://{}{}", self.scheme, self.host, self.target) |
| 224 | } else { |
| 225 | write!(f, "{}://{}:{}{}", self.scheme, self.host, self.port, self.target) |
| 226 | } |
| 227 | } |
| 228 | } |
| 229 | |
| 230 | impl Url { |
| 231 | |
| 232 | /// Parse an absolute `http` or `https` URL. |
| 233 | /// |
| 234 | /// Any other scheme is refused here rather than downstream, so that a |
| 235 | /// `file:`, `gopher:` or `javascript:` URL cannot reach a socket at all. |
| 236 | pub fn parse(raw: &str) -> Outcome<Self> { |
| 237 | let raw = raw.trim(); |
| 238 | let (scheme_txt, rest) = match raw.find("://") { |
| 239 | Some(i) => (&raw[..i], &raw[i + 3..]), |
| 240 | None => return Err(err!( |
| 241 | "The URL {:?} names no scheme; an absolute http:// or https:// \ |
| 242 | URL is required.", raw; |
| 243 | Invalid, Input, Missing)), |
| 244 | }; |
| 245 | let scheme = match scheme_txt.to_lowercase().as_str() { |
| 246 | "http" => UrlScheme::Http, |
| 247 | "https" => UrlScheme::Https, |
| 248 | other => return Err(err!( |
| 249 | "The URL scheme {:?} is not one this client speaks; only http \ |
| 250 | and https are.", other; |
| 251 | Invalid, Input, Unimplemented)), |
| 252 | }; |
| 253 | |
| 254 | // The authority ends at the first delimiter of what follows it. |
| 255 | let end = rest.find(['/', '?', '#']).unwrap_or(rest.len()); |
| 256 | let authority = &rest[..end]; |
| 257 | let tail = &rest[end..]; |
| 258 | |
| 259 | // Userinfo is discarded, and the host is what follows the last `@`: |
| 260 | // `https://example.com@127.0.0.1/` names the loopback address, and a |
| 261 | // parser that thinks otherwise is the request forgery. |
| 262 | let hostport = match authority.rfind('@') { |
| 263 | Some(i) => &authority[i + 1..], |
| 264 | None => authority, |
| 265 | }; |
| 266 | if hostport.is_empty() { |
| 267 | return Err(err!( |
| 268 | "The URL {:?} names no host.", raw; |
| 269 | Invalid, Input, Missing)); |
| 270 | } |
| 271 | |
| 272 | // An IPv6 literal wears brackets, and the colons inside them are not |
| 273 | // the one that introduces the port. |
| 274 | let (host, port_txt) = if let Some(close) = hostport.strip_prefix('[') { |
| 275 | match close.find(']') { |
| 276 | Some(i) => { |
| 277 | let host = &close[..i]; |
| 278 | let after = &close[i + 1..]; |
| 279 | let port = match after.strip_prefix(':') { |
| 280 | Some(p) => Some(p), |
| 281 | None if after.is_empty() => None, |
| 282 | None => return Err(err!( |
| 283 | "The URL {:?} has trailing junk after its IPv6 \ |
| 284 | host.", raw; |
| 285 | Invalid, Input)), |
| 286 | }; |
| 287 | (host, port) |
| 288 | } |
| 289 | None => return Err(err!( |
| 290 | "The URL {:?} opens an IPv6 host that it never closes.", raw; |
| 291 | Invalid, Input)), |
| 292 | } |
| 293 | } else { |
| 294 | match hostport.rfind(':') { |
| 295 | Some(i) => (&hostport[..i], Some(&hostport[i + 1..])), |
| 296 | None => (hostport, None), |
| 297 | } |
| 298 | }; |
| 299 | if host.is_empty() { |
| 300 | return Err(err!( |
| 301 | "The URL {:?} names no host.", raw; |
| 302 | Invalid, Input, Missing)); |
| 303 | } |
| 304 | let port = match port_txt { |
| 305 | Some(p) => match p.parse::<u16>() { |
| 306 | Ok(n) if n > 0 => n, |
| 307 | _ => return Err(err!( |
| 308 | "The URL {:?} names {:?}, which is not a port.", raw, p; |
| 309 | Invalid, Input)), |
| 310 | }, |
| 311 | None => scheme.default_port(), |
| 312 | }; |
| 313 | |
| 314 | // The fragment is the browser's, and is never sent. |
| 315 | let target = match tail.find('#') { |
| 316 | Some(i) => &tail[..i], |
| 317 | None => tail, |
| 318 | }; |
| 319 | let target = if target.is_empty() { "/" } else { target }; |
| 320 | |
| 321 | Ok(Self { |
| 322 | scheme, |
| 323 | host: host.to_lowercase(), |
| 324 | port, |
| 325 | target: target.to_string(), |
| 326 | }) |
| 327 | } |
| 328 | |
| 329 | /// Resolve a `Location` header against this URL, as a redirect is followed. |
| 330 | /// |
| 331 | /// Absolute, protocol-relative (`//host/path`), root-relative (`/path`) and |
| 332 | /// relative (`path`) forms are all handled, because servers send all four. |
| 333 | pub fn join(&self, loc: &str) -> Outcome<Self> { |
| 334 | let loc = loc.trim(); |
| 335 | if loc.is_empty() { |
| 336 | return Err(err!( |
| 337 | "The redirect names no location."; Invalid, Input, Missing)); |
| 338 | } |
| 339 | // An absolute URL leads with its scheme. A relative one whose query |
| 340 | // happens to carry `://` -- a `?next=https://elsewhere` -- must not be |
| 341 | // mistaken for one, so the scheme is matched only at the very start and |
| 342 | // only where every character before the `://` could belong to a scheme. |
| 343 | if let Some(i) = loc.find("://") { |
| 344 | let scheme = &loc[..i]; |
| 345 | if !scheme.is_empty() |
| 346 | && scheme.chars().all(|c| |
| 347 | c.is_ascii_alphanumeric() || matches!(c, '+' | '-' | '.')) |
| 348 | { |
| 349 | return Self::parse(loc); |
| 350 | } |
| 351 | } |
| 352 | if let Some(rest) = loc.strip_prefix("//") { |
| 353 | return Self::parse(&fmt!("{}://{}", self.scheme, rest)); |
| 354 | } |
| 355 | let base = if self.port == self.scheme.default_port() { |
| 356 | fmt!("{}://{}", self.scheme, self.host) |
| 357 | } else { |
| 358 | fmt!("{}://{}:{}", self.scheme, self.host, self.port) |
| 359 | }; |
| 360 | if loc.starts_with('/') { |
| 361 | return Self::parse(&fmt!("{}{}", base, loc)); |
| 362 | } |
| 363 | // Relative to the directory the current target sits in. The query is no |
| 364 | // part of the path, and is dropped before the directory is found, or a |
| 365 | // `/` inside a `?query` would be read as a directory boundary it is not. |
| 366 | let path = match self.target.find('?') { |
| 367 | Some(i) => &self.target[..i], |
| 368 | None => &self.target, |
| 369 | }; |
| 370 | let dir = match path.rfind('/') { |
| 371 | Some(i) => &path[..i + 1], |
| 372 | None => "/", |
| 373 | }; |
| 374 | Self::parse(&fmt!("{}{}{}", base, dir, loc)) |
| 375 | } |
| 376 | |
| 377 | /// The origin as a browser writes it: `scheme://host[:port]`, with the |
| 378 | /// port shown only when it is not the scheme's own. |
| 379 | pub fn origin(&self) -> String { |
| 380 | if self.port == self.scheme.default_port() { |
| 381 | fmt!("{}://{}", self.scheme, self.host) |
| 382 | } else { |
| 383 | fmt!("{}://{}:{}", self.scheme, self.host, self.port) |
| 384 | } |
| 385 | } |
| 386 | } |
| 387 | |
| 388 | |
| 389 | #[cfg(test)] |
| 390 | mod url_tests { |
| 391 | use super::*; |
| 392 | |
| 393 | #[test] |
| 394 | fn test_a_locator_keeps_its_raw_query_for_forwarding() -> Outcome<()> { |
| 395 | let loc = res!(HttpLocator::new("/api/admin?view=geo&page=2")); |
| 396 | assert_eq!(loc.path.as_string(), "/api/admin"); |
| 397 | // The raw query survives verbatim, no leading `?`, for a proxy to |
| 398 | // forward byte-for-byte -- the parse into `data` is a convenience only. |
| 399 | assert_eq!(loc.query, "view=geo&page=2"); |
| 400 | Ok(()) |
| 401 | } |
| 402 | |
| 403 | #[test] |
| 404 | fn test_a_locator_with_no_query_has_an_empty_query() -> Outcome<()> { |
| 405 | let loc = res!(HttpLocator::new("/api/health")); |
| 406 | assert_eq!(loc.query, ""); |
| 407 | Ok(()) |
| 408 | } |
| 409 | |
| 410 | #[test] |
| 411 | fn test_query_parameters_read_out_by_name() -> Outcome<()> { |
| 412 | let loc = res!(HttpLocator::new("/api/admin?view=geo&page=2")); |
| 413 | assert_eq!(loc.query_str("view"), Some(fmt!("geo"))); |
| 414 | // A numeric parameter is not a string, so query_str declines it rather |
| 415 | // than handing back a spelling of it. |
| 416 | assert_eq!(loc.query_str("page"), None); |
| 417 | // query_i64 coerces either way, which is the point of having it. |
| 418 | assert_eq!(loc.query_i64("page", 1), 2); |
| 419 | assert_eq!(loc.query_i64("view", 7), 7, "a non-numeric value falls back"); |
| 420 | assert_eq!(loc.query_i64("absent", 5), 5, "an absent parameter falls back"); |
| 421 | assert_eq!(loc.query_str("absent"), None); |
| 422 | Ok(()) |
| 423 | } |
| 424 | |
| 425 | #[test] |
| 426 | fn test_a_numeric_looking_string_parameter_still_coerces() -> Outcome<()> { |
| 427 | // A bare `12` parses as a number and a quoted `"12"` as a string, which |
| 428 | // is the query reader's business and not the caller's. Asking for an |
| 429 | // integer gets one either way. |
| 430 | assert_eq!(res!(HttpLocator::new("/x?n=12")).query_i64("n", 0), 12); |
| 431 | assert_eq!(res!(HttpLocator::new("/x?n=\"12\"")).query_i64("n", 0), 12); |
| 432 | // And a value that is not a number at all falls back rather than |
| 433 | // yielding some partial reading of it. |
| 434 | assert_eq!(res!(HttpLocator::new("/x?n=12abc")).query_i64("n", 0), 0); |
| 435 | Ok(()) |
| 436 | } |
| 437 | |
| 438 | #[test] |
| 439 | fn test_a_plain_url_parses_into_its_parts() -> Outcome<()> { |
| 440 | let u = res!(Url::parse("https://example.com/a/b?x=1#frag")); |
| 441 | assert_eq!(u.scheme, UrlScheme::Https); |
| 442 | assert_eq!(u.host, "example.com"); |
| 443 | assert_eq!(u.port, 443); |
| 444 | // The fragment is never sent. |
| 445 | assert_eq!(u.target, "/a/b?x=1"); |
| 446 | assert_eq!(fmt!("{}", u), "https://example.com/a/b?x=1"); |
| 447 | Ok(()) |
| 448 | } |
| 449 | |
| 450 | #[test] |
| 451 | fn test_a_bare_host_gets_the_schemes_port_and_a_root_target() -> Outcome<()> { |
| 452 | let u = res!(Url::parse("http://example.com")); |
| 453 | assert_eq!(u.port, 80); |
| 454 | assert_eq!(u.target, "/"); |
| 455 | let s = res!(Url::parse("https://example.com:8443/x")); |
| 456 | assert_eq!(s.port, 8443); |
| 457 | assert_eq!(fmt!("{}", s), "https://example.com:8443/x"); |
| 458 | Ok(()) |
| 459 | } |
| 460 | |
| 461 | #[test] |
| 462 | fn test_userinfo_does_not_become_the_host() -> Outcome<()> { |
| 463 | // The classic disguise: the host is the loopback address, and |
| 464 | // `example.com` is userinfo a browser throws away. |
| 465 | let u = res!(Url::parse("https://example.com@127.0.0.1/admin")); |
| 466 | assert_eq!(u.host, "127.0.0.1"); |
| 467 | let v = res!(Url::parse("https://user:pass@evil.test:8080/")); |
| 468 | assert_eq!(v.host, "evil.test"); |
| 469 | assert_eq!(v.port, 8080); |
| 470 | Ok(()) |
| 471 | } |
| 472 | |
| 473 | #[test] |
| 474 | fn test_an_ipv6_literal_keeps_its_colons() -> Outcome<()> { |
| 475 | let u = res!(Url::parse("http://[2606:4700::1111]:8080/x")); |
| 476 | assert_eq!(u.host, "2606:4700::1111"); |
| 477 | assert_eq!(u.port, 8080); |
| 478 | let v = res!(Url::parse("http://[::1]/")); |
| 479 | assert_eq!(v.host, "::1"); |
| 480 | assert_eq!(v.port, 80); |
| 481 | Ok(()) |
| 482 | } |
| 483 | |
| 484 | #[test] |
| 485 | fn test_a_scheme_this_client_does_not_speak_is_refused() { |
| 486 | for raw in [ |
| 487 | "file:///etc/passwd", |
| 488 | "gopher://example.com/", |
| 489 | "javascript://example.com/", |
| 490 | "ftp://example.com/x", |
| 491 | ] { |
| 492 | assert!(Url::parse(raw).is_err(), "{} should not parse", raw); |
| 493 | } |
| 494 | // And so is a URL with no scheme at all. |
| 495 | assert!(Url::parse("example.com/x").is_err()); |
| 496 | assert!(Url::parse("https://").is_err()); |
| 497 | } |
| 498 | |
| 499 | #[test] |
| 500 | fn test_a_redirect_resolves_in_all_four_forms() -> Outcome<()> { |
| 501 | let base = res!(Url::parse("https://example.com/a/b?x=1")); |
| 502 | |
| 503 | let abs = res!(base.join("http://other.test/z")); |
| 504 | assert_eq!(fmt!("{}", abs), "http://other.test/z"); |
| 505 | |
| 506 | let proto = res!(base.join("//other.test/z")); |
| 507 | assert_eq!(fmt!("{}", proto), "https://other.test/z"); |
| 508 | |
| 509 | let root = res!(base.join("/z?q=2")); |
| 510 | assert_eq!(fmt!("{}", root), "https://example.com/z?q=2"); |
| 511 | |
| 512 | let rel = res!(base.join("c/d")); |
| 513 | assert_eq!(fmt!("{}", rel), "https://example.com/a/c/d"); |
| 514 | Ok(()) |
| 515 | } |
| 516 | |
| 517 | #[test] |
| 518 | fn test_a_redirect_can_change_the_host_it_points_at() -> Outcome<()> { |
| 519 | // Which is exactly why a caller must vet every hop, not only the first. |
| 520 | let base = res!(Url::parse("https://example.com/a")); |
| 521 | let next = res!(base.join("http://127.0.0.1:80/latest/meta-data/")); |
| 522 | assert_eq!(next.host, "127.0.0.1"); |
| 523 | assert_eq!(next.port, 80); |
| 524 | Ok(()) |
| 525 | } |
| 526 | |
| 527 | #[test] |
| 528 | fn test_a_scheme_in_a_query_is_not_an_absolute_redirect() -> Outcome<()> { |
| 529 | // `://` sitting inside a query is not a scheme, so this stays a |
| 530 | // same-origin, root-relative redirect rather than jumping to `evil`. |
| 531 | let base = res!(Url::parse("https://example.com/a")); |
| 532 | let next = res!(base.join("/login?return=https://evil.test/x")); |
| 533 | assert_eq!(next.host, "example.com"); |
| 534 | assert_eq!(next.target, "/login?return=https://evil.test/x"); |
| 535 | Ok(()) |
| 536 | } |
| 537 | |
| 538 | #[test] |
| 539 | fn test_a_relative_redirect_resolves_against_the_path_not_the_query() -> Outcome<()> { |
| 540 | // The base carries a query with a `/` in it. That slash is no directory |
| 541 | // boundary, so the relative target resolves against `/a/`, not `/a/x=`. |
| 542 | let base = res!(Url::parse("https://example.com/a/b?next=/x/y")); |
| 543 | let next = res!(base.join("c")); |
| 544 | assert_eq!(next.target, "/a/c"); |
| 545 | Ok(()) |
| 546 | } |
| 547 | } |