oxedyne/fe2o3/fe2o3_net/src/http/encoding.rs
28.9 KiB, 57 runs
created by r1870400018:19664, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Content codings: what a client will accept, what is worth encoding, and the |
| 2 | //! gzip stream itself. |
| 3 | //! |
| 4 | //! A response sent raw costs the reader every byte of it. Markup, script, |
| 5 | //! stylesheets and WebAssembly are all highly redundant, and a page built of |
| 6 | //! them typically weighs two to four times on the wire what it need weigh -- |
| 7 | //! which is paid by whoever is on the slowest connection, every visit. |
| 8 | //! |
| 9 | //! Encoding one is only correct if three things hold together: the client said |
| 10 | //! it would accept the coding ([RFC 9110 §12.5.3]), the representation is not |
| 11 | //! already compressed, and every framing field describes the *encoded* body |
| 12 | //! rather than the original. The last is not a nicety -- a `Content-Length` |
| 13 | //! naming the wrong number desynchronises a kept-alive connection, and the |
| 14 | //! client waits for bytes that never come. |
| 15 | //! |
| 16 | //! [RFC 9110 §12.5.3]: https://www.rfc-editor.org/rfc/rfc9110#section-12.5.3 |
| 17 | //! |
| 18 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 19 | //! Anthropic Claude |
| 20 | |
| 21 | use crate::{ |
| 22 | http::{ |
| 23 | fields::{ |
| 24 | HeaderFields, |
| 25 | HeaderFieldValue, |
| 26 | HeaderName, |
| 27 | }, |
| 28 | msg::HttpMessage, |
| 29 | }, |
| 30 | media::MediaType, |
| 31 | }; |
| 32 | |
| 33 | use oxedyne_fe2o3_core::prelude::*; |
| 34 | |
| 35 | use std::{ |
| 36 | io::Write, |
| 37 | str::FromStr, |
| 38 | }; |
| 39 | |
| 40 | use flate2::{ |
| 41 | Compression, |
| 42 | write::GzEncoder, |
| 43 | read::GzDecoder, |
| 44 | }; |
| 45 | |
| 46 | |
| 47 | // Level 6 is zlib's own default and the knee of the curve. Measured over a |
| 48 | // megabyte of base64-heavy markup, a WebAssembly module and a script bundle: |
| 49 | // level 9 costs about half as much time again for two parts in a thousand more |
| 50 | // saving, and level 1 runs in a third of the time but gives up something like a |
| 51 | // sixth of the saving. The encoding is the reason the response is smaller, so |
| 52 | // the saving is what is being bought. |
| 53 | const GZIP_LEVEL: u32 = 6; |
| 54 | |
| 55 | // The default below which a body is sent as it is. A gzip member costs eighteen |
| 56 | // bytes of framing before it encodes anything, and the round trip through the |
| 57 | // encoder and the client's decoder is not free either. Under about a kilobyte |
| 58 | // the saving is noise, and on the smallest bodies the encoded form is the larger |
| 59 | // of the two. |
| 60 | pub const MIN_BYTES_DEFAULT: usize = 1024; |
| 61 | |
| 62 | |
| 63 | /// A content coding this server can produce. |
| 64 | /// |
| 65 | /// Only the two: `gzip` ([RFC 9110 §8.4.1.3], the format of [RFC 1952]) and no |
| 66 | /// coding at all. Naming a coding the encoder cannot actually emit would let |
| 67 | /// negotiation promise something the wire could not keep, so the enum is exactly |
| 68 | /// the set of things that can be sent. |
| 69 | /// |
| 70 | /// [RFC 9110 §8.4.1.3]: https://www.rfc-editor.org/rfc/rfc9110#section-8.4.1.3 |
| 71 | /// [RFC 1952]: https://www.rfc-editor.org/rfc/rfc1952 |
| 72 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 73 | pub enum ContentCoding { |
| 74 | Identity, // the representation as it stands, carrying no `Content-Encoding` |
| 75 | Gzip, // a gzip member, as `Content-Encoding: gzip` |
| 76 | } |
| 77 | |
| 78 | impl ContentCoding { |
| 79 | |
| 80 | /// The token as it appears in `Accept-Encoding` and `Content-Encoding`. |
| 81 | pub fn token(&self) -> &'static str { |
| 82 | match self { |
| 83 | Self::Identity => "identity", |
| 84 | Self::Gzip => "gzip", |
| 85 | } |
| 86 | } |
| 87 | |
| 88 | /// Read a coding token, accepting the historic `x-gzip` spelling. |
| 89 | /// |
| 90 | /// RFC 9110 §8.4.1.3 records `x-gzip` as the name some senders still use for |
| 91 | /// the same format, and a client that asks for it by that name is asking for |
| 92 | /// gzip. |
| 93 | pub fn from_token(token: &str) -> Option<Self> { |
| 94 | match token.trim().to_ascii_lowercase().as_str() { |
| 95 | "gzip" | "x-gzip" => Some(Self::Gzip), |
| 96 | "identity" => Some(Self::Identity), |
| 97 | _ => None, |
| 98 | } |
| 99 | } |
| 100 | |
| 101 | /// Does this coding change the bytes on the wire? |
| 102 | pub fn encodes(&self) -> bool { |
| 103 | !matches!(self, Self::Identity) |
| 104 | } |
| 105 | } |
| 106 | |
| 107 | |
| 108 | /// One entry of an `Accept-Encoding` field: a coding token and its weight. |
| 109 | /// |
| 110 | /// The weight is held in thousandths, which is the full precision RFC 9110 |
| 111 | /// §12.4.2 allows a qvalue (`0.000` to `1.000`), so the whole comparison is |
| 112 | /// integer arithmetic and no two weights ever compare equal by rounding. |
| 113 | #[derive(Clone, Debug, Eq, PartialEq)] |
| 114 | struct Preference { |
| 115 | token: String, // as written, lowercased; `*` is kept as itself |
| 116 | weight: u16, // thousandths, `0` meaning "not acceptable" |
| 117 | } |
| 118 | |
| 119 | /// Split an `Accept-Encoding` field value into its entries. |
| 120 | /// |
| 121 | /// A malformed weight is read as the absent one, per RFC 9110 §12.4.2: the |
| 122 | /// default when no weight is given is `q=1`, and a sender that writes rubbish |
| 123 | /// after the semicolon has still named the coding. |
| 124 | fn preferences(field: &str) -> Vec<Preference> { |
| 125 | let mut out = Vec::new(); |
| 126 | for entry in field.split(',') { |
| 127 | let entry = entry.trim(); |
| 128 | if entry.is_empty() { |
| 129 | continue; |
| 130 | } |
| 131 | let mut parts = entry.split(';'); |
| 132 | let token = match parts.next() { |
| 133 | Some(t) => t.trim().to_ascii_lowercase(), |
| 134 | None => continue, |
| 135 | }; |
| 136 | if token.is_empty() { |
| 137 | continue; |
| 138 | } |
| 139 | let mut weight = 1000u16; |
| 140 | for param in parts { |
| 141 | let param = param.trim(); |
| 142 | let rest = match param.strip_prefix("q=").or_else(|| param.strip_prefix("Q=")) { |
| 143 | Some(rest) => rest.trim(), |
| 144 | None => continue, |
| 145 | }; |
| 146 | weight = qvalue(rest).unwrap_or(1000); |
| 147 | } |
| 148 | out.push(Preference { token, weight }); |
| 149 | } |
| 150 | out |
| 151 | } |
| 152 | |
| 153 | /// Read a qvalue into thousandths. |
| 154 | /// |
| 155 | /// RFC 9110 §12.4.2 defines it as `0[.0-3 digits]` or `1[.0-3 zeroes]`, so the |
| 156 | /// scale is exactly a thousand and nothing outside `0.0 ..= 1.0` is a qvalue at |
| 157 | /// all. |
| 158 | fn qvalue(s: &str) -> Option<u16> { |
| 159 | let (whole, frac) = match s.split_once('.') { |
| 160 | Some((w, f)) => (w, f), |
| 161 | None => (s, ""), |
| 162 | }; |
| 163 | let lead: u16 = match whole.trim() { |
| 164 | "0" => 0, |
| 165 | "1" => 1000, |
| 166 | _ => return None, |
| 167 | }; |
| 168 | if frac.is_empty() { |
| 169 | return Some(lead); |
| 170 | } |
| 171 | if frac.len() > 3 || !frac.bytes().all(|b| b.is_ascii_digit()) { |
| 172 | return None; |
| 173 | } |
| 174 | // Pad to three places so `.5` and `.500` weigh the same. |
| 175 | let mut thousandths: u16 = 0; |
| 176 | let mut scale = 100u16; |
| 177 | for b in frac.bytes() { |
| 178 | thousandths += ((b - b'0') as u16) * scale; |
| 179 | scale /= 10; |
| 180 | } |
| 181 | match lead { |
| 182 | 0 => Some(thousandths), |
| 183 | // `q=1.000` is the only legal form above one; anything else is not a |
| 184 | // qvalue, and reading it as a weight would let a sender outrank the |
| 185 | // scale's own ceiling. |
| 186 | _ => if thousandths == 0 { Some(1000) } else { None }, |
| 187 | } |
| 188 | } |
| 189 | |
| 190 | /// The weight a field value gives one coding. |
| 191 | /// |
| 192 | /// A coding named outright takes its own weight. Otherwise `*` speaks for it, |
| 193 | /// per RFC 9110 §12.5.3 -- "the asterisk symbol matches any available content |
| 194 | /// coding not explicitly listed". A coding neither named nor covered by `*` is |
| 195 | /// not acceptable, which is the whole point of sending the field. |
| 196 | fn weight_of(prefs: &[Preference], coding: ContentCoding) -> u16 { |
| 197 | let named = prefs.iter().find(|p| |
| 198 | ContentCoding::from_token(&p.token) == Some(coding)); |
| 199 | if let Some(p) = named { |
| 200 | return p.weight; |
| 201 | } |
| 202 | if let Some(p) = prefs.iter().find(|p| p.token == "*") { |
| 203 | return p.weight; |
| 204 | } |
| 205 | match coding { |
| 206 | // "If the representation has no content coding, then it is acceptable by |
| 207 | // default unless specifically refused" -- RFC 9110 §12.5.3. |
| 208 | ContentCoding::Identity => 1000, |
| 209 | _ => 0, |
| 210 | } |
| 211 | } |
| 212 | |
| 213 | /// Choose a coding from an `Accept-Encoding` field value. |
| 214 | /// |
| 215 | /// Follows RFC 9110 §12.5.3: |
| 216 | /// |
| 217 | /// - No field at all means the sender expressed no preference. The |
| 218 | /// specification permits any coding here, but this server sends none: a |
| 219 | /// request with no `Accept-Encoding` is very rarely a browser, and handing an |
| 220 | /// unrequested coding to a script or a proxy that never asked for one is how |
| 221 | /// an integration breaks for no gain. |
| 222 | /// - An empty field value means no coding is supported, so identity it is. |
| 223 | /// - `q=0` means not acceptable, for `identity` as much as for anything else. |
| 224 | /// - `*` speaks for every coding not named outright. |
| 225 | /// - Among acceptable codings the greatest weight wins, and a tie goes to gzip, |
| 226 | /// which is the server's own preference and the reason the negotiation is |
| 227 | /// being done. |
| 228 | pub fn negotiate(accept_encoding: Option<&str>) -> ContentCoding { |
| 229 | let field = match accept_encoding { |
| 230 | Some(f) => f, |
| 231 | None => return ContentCoding::Identity, |
| 232 | }; |
| 233 | let prefs = preferences(field); |
| 234 | if prefs.is_empty() { |
| 235 | return ContentCoding::Identity; |
| 236 | } |
| 237 | let gzip = weight_of(&prefs, ContentCoding::Gzip); |
| 238 | let identity = weight_of(&prefs, ContentCoding::Identity); |
| 239 | if gzip > 0 && gzip >= identity { |
| 240 | ContentCoding::Gzip |
| 241 | } else { |
| 242 | ContentCoding::Identity |
| 243 | } |
| 244 | } |
| 245 | |
| 246 | pub fn accept_encoding(fields: &HeaderFields) -> Option<String> { |
| 247 | fields.get_one(&HeaderName::AcceptEncoding).map(|val| fmt!("{}", val)) |
| 248 | } |
| 249 | |
| 250 | /// Is a body of this media type worth encoding? |
| 251 | /// |
| 252 | /// The string is a `Content-Type` field value, so any parameters after the |
| 253 | /// media type (`; charset=utf-8`, a multipart boundary) are cut before it is |
| 254 | /// read. |
| 255 | /// |
| 256 | /// Two cases are settled on the string before the media type is parsed at all. |
| 257 | /// Anything under `text/` is text by definition (RFC 2046 §4.1), whether or not |
| 258 | /// this crate models the subtype, so `text/markdown` and `text/calendar` are not |
| 259 | /// left out for want of an enum variant. And the several names for script -- |
| 260 | /// `application/javascript` and its `x-` and `ecmascript` spellings, which a |
| 261 | /// proxied upstream may well use in place of `text/javascript` -- name a format |
| 262 | /// that halves under DEFLATE whichever way it is spelled. |
| 263 | /// |
| 264 | /// Otherwise a type this crate cannot parse is left alone, which is the safe way |
| 265 | /// round: a missed saving costs bandwidth, a needless one costs the processor |
| 266 | /// and gains nothing. |
| 267 | pub fn is_compressible(content_type: &str) -> bool { |
| 268 | let media = content_type.split(';').next().unwrap_or("").trim().to_ascii_lowercase(); |
| 269 | if media.starts_with("text/") { |
| 270 | return true; |
| 271 | } |
| 272 | if matches!(media.as_str(), |
| 273 | "application/javascript" |
| 274 | | "application/x-javascript" |
| 275 | | "application/ecmascript" |
| 276 | | "application/x-ecmascript" |
| 277 | ) { |
| 278 | return true; |
| 279 | } |
| 280 | match MediaType::from_str(&media) { |
| 281 | Ok(mt) => mt.is_compressible(), |
| 282 | Err(_) => false, |
| 283 | } |
| 284 | } |
| 285 | |
| 286 | /// The whole rule: what coding should a response of this type and size carry? |
| 287 | /// |
| 288 | /// Three things must hold at once, and the cheapest is asked first. A body under |
| 289 | /// `min_bytes` is sent as it is, since a gzip member costs eighteen bytes of |
| 290 | /// framing before it encodes anything. A type that is already compressed is sent |
| 291 | /// as it is. And whatever is left is offered only if the client said it would |
| 292 | /// take it. |
| 293 | pub fn choose( |
| 294 | fields: &HeaderFields, |
| 295 | content_type: &str, |
| 296 | body_len: usize, |
| 297 | min_bytes: usize, |
| 298 | ) |
| 299 | -> ContentCoding |
| 300 | { |
| 301 | choose_for( |
| 302 | accept_encoding(fields).as_deref(), |
| 303 | content_type, |
| 304 | body_len, |
| 305 | min_bytes, |
| 306 | ) |
| 307 | } |
| 308 | |
| 309 | /// [`choose`], for a caller that kept the `Accept-Encoding` field rather than |
| 310 | /// the request it came on. |
| 311 | /// |
| 312 | /// The request is moved into the dispatch chain long before the response is |
| 313 | /// encoded, so the server holds the one field it will need and lets the rest go. |
| 314 | pub fn choose_for( |
| 315 | accept_encoding: Option<&str>, |
| 316 | content_type: &str, |
| 317 | body_len: usize, |
| 318 | min_bytes: usize, |
| 319 | ) |
| 320 | -> ContentCoding |
| 321 | { |
| 322 | if body_len < min_bytes { |
| 323 | return ContentCoding::Identity; |
| 324 | } |
| 325 | if !is_compressible(content_type) { |
| 326 | return ContentCoding::Identity; |
| 327 | } |
| 328 | negotiate(accept_encoding) |
| 329 | } |
| 330 | |
| 331 | /// Name the coding in an entity tag, so two encodings of one representation |
| 332 | /// never share a validator. |
| 333 | /// |
| 334 | /// RFC 9110 §8.8.3 makes an entity tag the identity of a *representation*, and a |
| 335 | /// gzipped body is a different representation of the same resource. Handing both |
| 336 | /// the same tag is the classic caching fault: a client holding the encoded copy |
| 337 | /// sends the tag back on a request that accepts no coding, the server answers |
| 338 | /// `304`, and the client renders a gzip member as though it were markup. |
| 339 | /// |
| 340 | /// The coding goes inside the quotes, leaving the tag a valid `entity-tag` and |
| 341 | /// keeping the weakness marker where it belongs. |
| 342 | pub fn tagged(etag: &str, coding: ContentCoding) -> String { |
| 343 | if !coding.encodes() { |
| 344 | return etag.to_string(); |
| 345 | } |
| 346 | match etag.strip_suffix('"') { |
| 347 | Some(head) => fmt!("{}-{}\"", head, coding.token()), |
| 348 | // Not a quoted tag at all; leave it be rather than mint a malformed one. |
| 349 | None => etag.to_string(), |
| 350 | } |
| 351 | } |
| 352 | |
| 353 | /// gzip a buffer, as [RFC 1952] defines the format. |
| 354 | /// |
| 355 | /// [RFC 1952]: https://www.rfc-editor.org/rfc/rfc1952 |
| 356 | pub fn gzip(data: &[u8]) -> Outcome<Vec<u8>> { |
| 357 | let mut enc = GzEncoder::new(Vec::new(), Compression::new(GZIP_LEVEL)); |
| 358 | res!(enc.write_all(data), IO, Encode); |
| 359 | Ok(res!(enc.finish(), IO, Encode)) |
| 360 | } |
| 361 | |
| 362 | /// Read a gzip member back, which is what a client does with one. |
| 363 | pub fn gunzip(data: &[u8]) -> Outcome<Vec<u8>> { |
| 364 | use std::io::Read; |
| 365 | let mut dec = GzDecoder::new(data); |
| 366 | let mut out = Vec::new(); |
| 367 | res!(dec.read_to_end(&mut out), IO, Decode); |
| 368 | Ok(out) |
| 369 | } |
| 370 | |
| 371 | /// Say that the response varies by the coding asked for. |
| 372 | /// |
| 373 | /// A shared cache keyed on the URL alone would hand a stored gzip body to the |
| 374 | /// next client along, coding or no coding. RFC 9111 §4.1 makes `Vary` the key, |
| 375 | /// and it is needed on *every* response whose type could have been encoded -- |
| 376 | /// including the ones that were not, since those are exactly the copies a cache |
| 377 | /// would otherwise reuse for a client that does accept a coding. |
| 378 | pub fn mark_varying(msg: &mut HttpMessage) { |
| 379 | let already = msg.header.fields.get_list(&HeaderName::Vary) |
| 380 | .map_or(false, |vals| vals.iter().any(|v| |
| 381 | fmt!("{}", v).to_ascii_lowercase().contains("accept-encoding"))); |
| 382 | if !already { |
| 383 | msg.header.fields.insert( |
| 384 | HeaderName::Vary, |
| 385 | HeaderFieldValue::Generic(fmt!("accept-encoding")), |
| 386 | None, |
| 387 | ); |
| 388 | } |
| 389 | } |
| 390 | |
| 391 | /// Would encoding this response be correct at all? |
| 392 | /// |
| 393 | /// Independent of what the client will accept: some responses must not be |
| 394 | /// encoded whatever the request said. |
| 395 | /// |
| 396 | /// - A body already carrying a `Content-Encoding` has been encoded by whoever |
| 397 | /// produced it, and a second coding would have to be declared as such and |
| 398 | /// undone in order. |
| 399 | /// - A chunked message frames itself, and RFC 9112 §6.1 forbids the |
| 400 | /// `Content-Length` an encoded body would need. |
| 401 | /// - A `206` answers a byte range of the *identity* representation. Encoding it |
| 402 | /// would make the range name bytes of something else entirely. |
| 403 | /// - A status with no body has nothing to encode; `304` in particular must carry |
| 404 | /// the validators of the representation it stands for, which |
| 405 | /// [`tagged`] has already named. |
| 406 | /// - A `HEAD` answer withholds its body at the wire, so coding one buys nothing |
| 407 | /// and costs everything: the whole body has to be materialised to be encoded -- |
| 408 | /// a file window read off the disk included -- and then thrown away unsent. The |
| 409 | /// answer keeps the `Content-Length` of the identity representation, which is |
| 410 | /// what a `GET` accepting no coding would be told and what anyone asking how |
| 411 | /// big a thing is wants to know. RFC 9110 §9.3.2 asks for the fields the `GET` |
| 412 | /// would carry; it does not ask a server to do the `GET`'s work to find out. |
| 413 | pub fn is_encodable(msg: &HttpMessage) -> bool { |
| 414 | use crate::http::{ |
| 415 | header::HttpHeadline, |
| 416 | status::HttpStatus, |
| 417 | }; |
| 418 | if msg.head_only { |
| 419 | return false; |
| 420 | } |
| 421 | if msg.header.fields.get_one(&HeaderName::ContentEncoding).is_some() { |
| 422 | return false; |
| 423 | } |
| 424 | if msg.header.fields.get_one(&HeaderName::TransferEncoding).is_some() { |
| 425 | return false; |
| 426 | } |
| 427 | match msg.header.headline { |
| 428 | HttpHeadline::Response { status } => !matches!(status, |
| 429 | HttpStatus::PartialContent |
| 430 | | HttpStatus::NotModified |
| 431 | | HttpStatus::NoContent |
| 432 | ), |
| 433 | _ => false, |
| 434 | } |
| 435 | } |
| 436 | |
| 437 | /// Encode a response, leaving every framing field describing the encoded body. |
| 438 | /// |
| 439 | /// The body is materialised first: a message whose body is named as a window of |
| 440 | /// a file has to be read before it can be encoded, and the window is then |
| 441 | /// dropped, since the bytes going out are no longer the bytes on disk. |
| 442 | /// `Content-Length` follows from [`HttpMessage::body_len`] when the message is |
| 443 | /// written, so it describes the encoded body by construction. |
| 444 | /// |
| 445 | /// A body the encoder cannot shrink is returned as it stands. Sending the larger |
| 446 | /// of the two forms would be a strange thing to have gone to the trouble of, and |
| 447 | /// it happens on small or already-dense bodies that slipped past the earlier |
| 448 | /// tests. |
| 449 | /// |
| 450 | /// `Vary` is set either way, because a cache must key on the coding whether or |
| 451 | /// not this particular response carried one. |
| 452 | #[cfg(feature = "async")] |
| 453 | pub async fn encode( |
| 454 | mut msg: HttpMessage, |
| 455 | coding: ContentCoding, |
| 456 | ) |
| 457 | -> Outcome<HttpMessage> |
| 458 | { |
| 459 | mark_varying(&mut msg); |
| 460 | if !coding.encodes() || !is_encodable(&msg) { |
| 461 | return Ok(msg); |
| 462 | } |
| 463 | let plain = match msg.file.take() { |
| 464 | Some(window) => res!(window.read().await), |
| 465 | None => std::mem::take(&mut msg.body), |
| 466 | }; |
| 467 | let encoded = match coding { |
| 468 | ContentCoding::Gzip => res!(gzip(&plain)), |
| 469 | ContentCoding::Identity => plain.clone(), |
| 470 | }; |
| 471 | if encoded.len() >= plain.len() { |
| 472 | msg.body = plain; |
| 473 | return Ok(msg); |
| 474 | } |
| 475 | msg.body = encoded; |
| 476 | msg.header.fields.insert( |
| 477 | HeaderName::ContentEncoding, |
| 478 | HeaderFieldValue::Generic(fmt!("{}", coding.token())), |
| 479 | None, |
| 480 | ); |
| 481 | // The encoded body is a different representation, so it needs a validator of |
| 482 | // its own. A client holding one must not be able to claim it holds the other. |
| 483 | if let Some(val) = msg.header.fields.get_one(&HeaderName::ETag) { |
| 484 | let renamed = tagged(&fmt!("{}", val), coding); |
| 485 | msg.header.fields.insert( |
| 486 | HeaderName::ETag, |
| 487 | HeaderFieldValue::Generic(renamed), |
| 488 | None, |
| 489 | ); |
| 490 | } |
| 491 | Ok(msg) |
| 492 | } |
| 493 | |
| 494 | |
| 495 | #[cfg(test)] |
| 496 | mod tests { |
| 497 | use super::*; |
| 498 | |
| 499 | // Used only by the async `encode` tests below; the sync tests need none of it. |
| 500 | #[cfg(feature = "async")] |
| 501 | use crate::http::status::HttpStatus; |
| 502 | |
| 503 | /// RFC 9110 §12.5.3: a request that names no coding is offered none. |
| 504 | #[test] |
| 505 | fn a_request_that_asks_for_nothing_is_sent_as_it_is() { |
| 506 | assert_eq!(negotiate(None), ContentCoding::Identity); |
| 507 | assert_eq!(negotiate(Some("")), ContentCoding::Identity); |
| 508 | } |
| 509 | |
| 510 | /// The field every browser actually sends. |
| 511 | #[test] |
| 512 | fn a_browser_asking_for_gzip_is_given_gzip() { |
| 513 | assert_eq!(negotiate(Some("gzip, deflate, br")), ContentCoding::Gzip); |
| 514 | assert_eq!(negotiate(Some("gzip")), ContentCoding::Gzip); |
| 515 | assert_eq!(negotiate(Some("GZIP")), ContentCoding::Gzip); |
| 516 | assert_eq!(negotiate(Some("x-gzip")), ContentCoding::Gzip); |
| 517 | } |
| 518 | |
| 519 | /// A coding neither named nor covered by `*` is not acceptable, which is |
| 520 | /// what sending the field is for. |
| 521 | #[test] |
| 522 | fn a_coding_that_was_not_asked_for_is_not_sent() { |
| 523 | assert_eq!(negotiate(Some("deflate, br")), ContentCoding::Identity); |
| 524 | assert_eq!(negotiate(Some("br;q=1.0")), ContentCoding::Identity); |
| 525 | } |
| 526 | |
| 527 | /// RFC 9110 §12.5.3: `q=0` means not acceptable. |
| 528 | #[test] |
| 529 | fn a_zero_weight_refuses_the_coding() { |
| 530 | assert_eq!(negotiate(Some("gzip;q=0")), ContentCoding::Identity); |
| 531 | assert_eq!(negotiate(Some("gzip;q=0.000")), ContentCoding::Identity); |
| 532 | assert_eq!(negotiate(Some("gzip;q=0, deflate")), ContentCoding::Identity); |
| 533 | } |
| 534 | |
| 535 | /// The asterisk "matches any available content coding not explicitly |
| 536 | /// listed" -- RFC 9110 §12.5.3. |
| 537 | #[test] |
| 538 | fn the_asterisk_speaks_for_a_coding_not_named() { |
| 539 | assert_eq!(negotiate(Some("*")), ContentCoding::Gzip); |
| 540 | assert_eq!(negotiate(Some("deflate, *")), ContentCoding::Gzip); |
| 541 | // Named outright, the entry beats the wildcard. |
| 542 | assert_eq!(negotiate(Some("*, gzip;q=0")), ContentCoding::Identity); |
| 543 | // And the wildcard can refuse everything it is left to speak for. |
| 544 | assert_eq!(negotiate(Some("*;q=0")), ContentCoding::Identity); |
| 545 | } |
| 546 | |
| 547 | /// Identity is acceptable by default and refusable outright. |
| 548 | #[test] |
| 549 | fn identity_is_assumed_unless_it_is_refused() { |
| 550 | // Refusing identity leaves gzip the only thing that can be sent. |
| 551 | assert_eq!(negotiate(Some("gzip, identity;q=0")), ContentCoding::Gzip); |
| 552 | // `*;q=0` refuses identity too, since identity is not named separately. |
| 553 | assert_eq!(negotiate(Some("gzip, *;q=0")), ContentCoding::Gzip); |
| 554 | // A more specific entry for identity overrides the wildcard. |
| 555 | assert_eq!(negotiate(Some("*;q=0, identity")), ContentCoding::Identity); |
| 556 | } |
| 557 | |
| 558 | /// The greatest weight wins; a tie is the server's to break. |
| 559 | #[test] |
| 560 | fn the_heavier_coding_wins_and_a_tie_goes_to_gzip() { |
| 561 | assert_eq!(negotiate(Some("gzip;q=0.5, identity;q=1.0")), ContentCoding::Identity); |
| 562 | assert_eq!(negotiate(Some("gzip;q=1.0, identity;q=0.5")), ContentCoding::Gzip); |
| 563 | assert_eq!(negotiate(Some("gzip;q=1.0, identity;q=1.0")), ContentCoding::Gzip); |
| 564 | // Thousandths, so the finest distinction the scale allows still decides. |
| 565 | assert_eq!(negotiate(Some("gzip;q=0.501, identity;q=0.500")), ContentCoding::Gzip); |
| 566 | assert_eq!(negotiate(Some("gzip;q=0.500, identity;q=0.501")), ContentCoding::Identity); |
| 567 | } |
| 568 | |
| 569 | /// RFC 9110 §12.4.2 gives the qvalue three decimal places and a ceiling of |
| 570 | /// one. Anything else is not a weight, and the entry keeps the default. |
| 571 | #[test] |
| 572 | fn a_qvalue_outside_the_scale_is_not_a_weight() { |
| 573 | assert_eq!(qvalue("0"), Some(0)); |
| 574 | assert_eq!(qvalue("1"), Some(1000)); |
| 575 | assert_eq!(qvalue("0.5"), Some(500)); |
| 576 | assert_eq!(qvalue("0.05"), Some(50)); |
| 577 | assert_eq!(qvalue("0.005"), Some(5)); |
| 578 | assert_eq!(qvalue("1.000"), Some(1000)); |
| 579 | assert_eq!(qvalue("1.001"), None); |
| 580 | assert_eq!(qvalue("2"), None); |
| 581 | assert_eq!(qvalue("0.0001"), None); |
| 582 | assert_eq!(qvalue("abc"), None); |
| 583 | // A weight that is not a weight leaves the coding named and acceptable. |
| 584 | assert_eq!(negotiate(Some("gzip;q=nonsense")), ContentCoding::Gzip); |
| 585 | } |
| 586 | |
| 587 | /// Whitespace around the entries and their parameters is optional per the |
| 588 | /// ABNF, so a field written either way means the same thing. |
| 589 | #[test] |
| 590 | fn the_spacing_of_the_field_does_not_change_its_meaning() { |
| 591 | assert_eq!(negotiate(Some("gzip;q=0.9,identity;q=1.0")), ContentCoding::Identity); |
| 592 | assert_eq!(negotiate(Some(" gzip ; q=0.9 , identity ; q=1.0 ")), |
| 593 | ContentCoding::Identity); |
| 594 | } |
| 595 | |
| 596 | /// The eligibility list, by media type. |
| 597 | #[test] |
| 598 | fn only_a_type_that_gains_by_it_is_encoded() { |
| 599 | for ct in [ |
| 600 | "text/html; charset=utf-8", |
| 601 | "text/css", |
| 602 | "text/plain", |
| 603 | "text/javascript; charset=utf-8", |
| 604 | "application/json", |
| 605 | "application/manifest+json", |
| 606 | "application/xml", |
| 607 | "application/problem+json", |
| 608 | "image/svg+xml", |
| 609 | "application/wasm", |
| 610 | "font/ttf", |
| 611 | // Text by definition, subtype modelled or not. |
| 612 | "text/markdown", |
| 613 | "text/calendar", |
| 614 | "TEXT/HTML", |
| 615 | // The other spellings of script, which a proxied upstream may use. |
| 616 | "application/javascript", |
| 617 | "application/x-javascript; charset=utf-8", |
| 618 | "application/ecmascript", |
| 619 | ] { |
| 620 | assert!(is_compressible(ct), "{} should be encoded", ct); |
| 621 | } |
| 622 | for ct in [ |
| 623 | "image/png", |
| 624 | "image/jpeg", |
| 625 | "image/webp", |
| 626 | "image/avif", |
| 627 | "image/gif", |
| 628 | "font/woff", |
| 629 | "font/woff2", |
| 630 | "audio/ogg", |
| 631 | "audio/mpeg", |
| 632 | "video/mp4", |
| 633 | "video/webm", |
| 634 | "application/zip", |
| 635 | "application/zstd", |
| 636 | "application/pdf", |
| 637 | // Not a media type at all, so nothing is assumed about it. |
| 638 | "", |
| 639 | "nonsense", |
| 640 | ] { |
| 641 | assert!(!is_compressible(ct), "{} should be sent as it is", ct); |
| 642 | } |
| 643 | } |
| 644 | |
| 645 | /// A body too small to be worth the framing is sent as it is, whatever the |
| 646 | /// request said. |
| 647 | #[test] |
| 648 | fn a_small_body_is_below_the_floor() -> Outcome<()> { |
| 649 | let mut fields = HeaderFields::default(); |
| 650 | fields.insert( |
| 651 | HeaderName::AcceptEncoding, |
| 652 | res!(HeaderFieldValue::new(&HeaderName::AcceptEncoding, "gzip")), |
| 653 | None, |
| 654 | ); |
| 655 | assert_eq!( |
| 656 | choose(&fields, "text/html", MIN_BYTES_DEFAULT - 1, MIN_BYTES_DEFAULT), |
| 657 | ContentCoding::Identity); |
| 658 | assert_eq!( |
| 659 | choose(&fields, "text/html", MIN_BYTES_DEFAULT, MIN_BYTES_DEFAULT), |
| 660 | ContentCoding::Gzip); |
| 661 | assert_eq!( |
| 662 | choose(&fields, "image/png", 1_000_000, MIN_BYTES_DEFAULT), |
| 663 | ContentCoding::Identity); |
| 664 | Ok(()) |
| 665 | } |
| 666 | |
| 667 | /// Two encodings of one representation must not share a validator. |
| 668 | #[test] |
| 669 | fn an_encoded_body_carries_a_tag_of_its_own() { |
| 670 | assert_eq!(tagged("\"68a1-3b\"", ContentCoding::Gzip), "\"68a1-3b-gzip\""); |
| 671 | assert_eq!(tagged("\"68a1-3b\"", ContentCoding::Identity), "\"68a1-3b\""); |
| 672 | assert_eq!(tagged("W/\"68a1-3b\"", ContentCoding::Gzip), "W/\"68a1-3b-gzip\""); |
| 673 | // Not a quoted tag; better left alone than turned into a malformed one. |
| 674 | assert_eq!(tagged("68a1", ContentCoding::Gzip), "68a1"); |
| 675 | } |
| 676 | |
| 677 | /// A `HEAD` answer is not encoded, and keeps the length of the identity |
| 678 | /// representation -- which is what a `GET` accepting no coding would be told, |
| 679 | /// and what anyone asking how big a thing is wants to know. |
| 680 | #[cfg(feature = "async")] |
| 681 | #[tokio::test] |
| 682 | async fn a_head_answer_is_not_encoded() -> Outcome<()> { |
| 683 | let body = "<p>a paragraph of markup</p>\n".repeat(500).into_bytes(); |
| 684 | let plain = body.len(); |
| 685 | let msg = HttpMessage::new_response(HttpStatus::OK) |
| 686 | .with_field( |
| 687 | HeaderName::ContentType, |
| 688 | HeaderFieldValue::Generic(fmt!("text/html; charset=utf-8")), |
| 689 | ) |
| 690 | .with_body(body) |
| 691 | .head_only(); |
| 692 | assert!(!is_encodable(&msg), "a HEAD answer was offered to the encoder"); |
| 693 | let out = res!(encode(msg, ContentCoding::Gzip).await); |
| 694 | assert_eq!(out.body_len(), plain, "a HEAD answer did not state the identity length"); |
| 695 | assert!(out.header.fields.get_one(&HeaderName::ContentEncoding).is_none(), |
| 696 | "a HEAD answer named a coding it had not applied"); |
| 697 | // It still says the representation varies by coding: a store keyed on the |
| 698 | // URL alone would otherwise hand this to the next client along. |
| 699 | let vary = res!(out.header.fields.get_one(&HeaderName::Vary).ok_or_else(|| |
| 700 | err!("The HEAD answer did not say it varies by coding."; Missing))); |
| 701 | assert!(fmt!("{}", vary).to_ascii_lowercase().contains("accept-encoding"), |
| 702 | "got: {}", vary); |
| 703 | Ok(()) |
| 704 | } |
| 705 | |
| 706 | /// The file a `HEAD` answer names is never opened, which is the cost the guard |
| 707 | /// is there to save: encoding a window means reading it off the disk first, and |
| 708 | /// a `HEAD` throws the result away unsent. |
| 709 | #[cfg(feature = "async")] |
| 710 | #[tokio::test] |
| 711 | async fn a_head_answer_leaves_its_file_unread() -> Outcome<()> { |
| 712 | use crate::http::msg::FileWindow; |
| 713 | // A window on a path that does not exist, so a run that reads it fails |
| 714 | // rather than merely being slower than it should be. |
| 715 | let msg = HttpMessage::new_response(HttpStatus::OK) |
| 716 | .with_field( |
| 717 | HeaderName::ContentType, |
| 718 | HeaderFieldValue::Generic(fmt!("text/html; charset=utf-8")), |
| 719 | ) |
| 720 | .with_file_window(FileWindow::new( |
| 721 | std::path::PathBuf::from("/nonexistent/no-such-file.html"), 0, 4096)) |
| 722 | .head_only(); |
| 723 | let out = res!(encode(msg, ContentCoding::Gzip).await); |
| 724 | assert_eq!(out.body_len(), 4096, "the window was not left as the body"); |
| 725 | Ok(()) |
| 726 | } |
| 727 | |
| 728 | /// The encoder's own output, read back by the decoder beside it. This says |
| 729 | /// only that the pair agree; the test that the stream is really gzip is in |
| 730 | /// `tests/`, against `gzip(1)`. |
| 731 | #[test] |
| 732 | fn a_gzip_member_round_trips() -> Outcome<()> { |
| 733 | let plain = "the quick brown fox jumps over the lazy dog\n" |
| 734 | .repeat(200).into_bytes(); |
| 735 | let encoded = res!(gzip(&plain)); |
| 736 | assert!(encoded.len() < plain.len() / 4, |
| 737 | "{} bytes encoded to {}", plain.len(), encoded.len()); |
| 738 | // RFC 1952 §2.3.1: every member begins with the two magic bytes and the |
| 739 | // compression method. |
| 740 | assert_eq!(&encoded[..3], &[0x1f, 0x8b, 0x08]); |
| 741 | assert_eq!(res!(gunzip(&encoded)), plain); |
| 742 | Ok(()) |
| 743 | } |
| 744 | } |