Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_net/src/http/encoding.rs

28.9 KiB, 57 runs

created by r1870400018:19664, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Content codings: what a client will accept, what is worth encoding, and the
2//! gzip stream itself.
3//!
4//! A response sent raw costs the reader every byte of it. Markup, script,
5//! stylesheets and WebAssembly are all highly redundant, and a page built of
6//! them typically weighs two to four times on the wire what it need weigh --
7//! which is paid by whoever is on the slowest connection, every visit.
8//!
9//! Encoding one is only correct if three things hold together: the client said
10//! it would accept the coding ([RFC 9110 §12.5.3]), the representation is not
11//! already compressed, and every framing field describes the *encoded* body
12//! rather than the original. The last is not a nicety -- a `Content-Length`
13//! naming the wrong number desynchronises a kept-alive connection, and the
14//! client waits for bytes that never come.
15//!
16//! [RFC 9110 §12.5.3]: https://www.rfc-editor.org/rfc/rfc9110#section-12.5.3
17//!
18//! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\
19//! Anthropic Claude
20
21use crate::{
22 http::{
23 fields::{
24 HeaderFields,
25 HeaderFieldValue,
26 HeaderName,
27 },
28 msg::HttpMessage,
29 },
30 media::MediaType,
31};
32
33use oxedyne_fe2o3_core::prelude::*;
34
35use std::{
36 io::Write,
37 str::FromStr,
38};
39
40use flate2::{
41 Compression,
42 write::GzEncoder,
43 read::GzDecoder,
44};
45
46
47// Level 6 is zlib's own default and the knee of the curve. Measured over a
48// megabyte of base64-heavy markup, a WebAssembly module and a script bundle:
49// level 9 costs about half as much time again for two parts in a thousand more
50// saving, and level 1 runs in a third of the time but gives up something like a
51// sixth of the saving. The encoding is the reason the response is smaller, so
52// the saving is what is being bought.
53const GZIP_LEVEL: u32 = 6;
54
55// The default below which a body is sent as it is. A gzip member costs eighteen
56// bytes of framing before it encodes anything, and the round trip through the
57// encoder and the client's decoder is not free either. Under about a kilobyte
58// the saving is noise, and on the smallest bodies the encoded form is the larger
59// of the two.
60pub const MIN_BYTES_DEFAULT: usize = 1024;
61
62
63/// A content coding this server can produce.
64///
65/// Only the two: `gzip` ([RFC 9110 §8.4.1.3], the format of [RFC 1952]) and no
66/// coding at all. Naming a coding the encoder cannot actually emit would let
67/// negotiation promise something the wire could not keep, so the enum is exactly
68/// the set of things that can be sent.
69///
70/// [RFC 9110 §8.4.1.3]: https://www.rfc-editor.org/rfc/rfc9110#section-8.4.1.3
71/// [RFC 1952]: https://www.rfc-editor.org/rfc/rfc1952
72#[derive(Clone, Copy, Debug, Eq, PartialEq)]
73pub enum ContentCoding {
74 Identity, // the representation as it stands, carrying no `Content-Encoding`
75 Gzip, // a gzip member, as `Content-Encoding: gzip`
76}
77
78impl ContentCoding {
79
80 /// The token as it appears in `Accept-Encoding` and `Content-Encoding`.
81 pub fn token(&self) -> &'static str {
82 match self {
83 Self::Identity => "identity",
84 Self::Gzip => "gzip",
85 }
86 }
87
88 /// Read a coding token, accepting the historic `x-gzip` spelling.
89 ///
90 /// RFC 9110 §8.4.1.3 records `x-gzip` as the name some senders still use for
91 /// the same format, and a client that asks for it by that name is asking for
92 /// gzip.
93 pub fn from_token(token: &str) -> Option<Self> {
94 match token.trim().to_ascii_lowercase().as_str() {
95 "gzip" | "x-gzip" => Some(Self::Gzip),
96 "identity" => Some(Self::Identity),
97 _ => None,
98 }
99 }
100
101 /// Does this coding change the bytes on the wire?
102 pub fn encodes(&self) -> bool {
103 !matches!(self, Self::Identity)
104 }
105}
106
107
108/// One entry of an `Accept-Encoding` field: a coding token and its weight.
109///
110/// The weight is held in thousandths, which is the full precision RFC 9110
111/// §12.4.2 allows a qvalue (`0.000` to `1.000`), so the whole comparison is
112/// integer arithmetic and no two weights ever compare equal by rounding.
113#[derive(Clone, Debug, Eq, PartialEq)]
114struct Preference {
115 token: String, // as written, lowercased; `*` is kept as itself
116 weight: u16, // thousandths, `0` meaning "not acceptable"
117}
118
119/// Split an `Accept-Encoding` field value into its entries.
120///
121/// A malformed weight is read as the absent one, per RFC 9110 §12.4.2: the
122/// default when no weight is given is `q=1`, and a sender that writes rubbish
123/// after the semicolon has still named the coding.
124fn preferences(field: &str) -> Vec<Preference> {
125 let mut out = Vec::new();
126 for entry in field.split(',') {
127 let entry = entry.trim();
128 if entry.is_empty() {
129 continue;
130 }
131 let mut parts = entry.split(';');
132 let token = match parts.next() {
133 Some(t) => t.trim().to_ascii_lowercase(),
134 None => continue,
135 };
136 if token.is_empty() {
137 continue;
138 }
139 let mut weight = 1000u16;
140 for param in parts {
141 let param = param.trim();
142 let rest = match param.strip_prefix("q=").or_else(|| param.strip_prefix("Q=")) {
143 Some(rest) => rest.trim(),
144 None => continue,
145 };
146 weight = qvalue(rest).unwrap_or(1000);
147 }
148 out.push(Preference { token, weight });
149 }
150 out
151}
152
153/// Read a qvalue into thousandths.
154///
155/// RFC 9110 §12.4.2 defines it as `0[.0-3 digits]` or `1[.0-3 zeroes]`, so the
156/// scale is exactly a thousand and nothing outside `0.0 ..= 1.0` is a qvalue at
157/// all.
158fn qvalue(s: &str) -> Option<u16> {
159 let (whole, frac) = match s.split_once('.') {
160 Some((w, f)) => (w, f),
161 None => (s, ""),
162 };
163 let lead: u16 = match whole.trim() {
164 "0" => 0,
165 "1" => 1000,
166 _ => return None,
167 };
168 if frac.is_empty() {
169 return Some(lead);
170 }
171 if frac.len() > 3 || !frac.bytes().all(|b| b.is_ascii_digit()) {
172 return None;
173 }
174 // Pad to three places so `.5` and `.500` weigh the same.
175 let mut thousandths: u16 = 0;
176 let mut scale = 100u16;
177 for b in frac.bytes() {
178 thousandths += ((b - b'0') as u16) * scale;
179 scale /= 10;
180 }
181 match lead {
182 0 => Some(thousandths),
183 // `q=1.000` is the only legal form above one; anything else is not a
184 // qvalue, and reading it as a weight would let a sender outrank the
185 // scale's own ceiling.
186 _ => if thousandths == 0 { Some(1000) } else { None },
187 }
188}
189
190/// The weight a field value gives one coding.
191///
192/// A coding named outright takes its own weight. Otherwise `*` speaks for it,
193/// per RFC 9110 §12.5.3 -- "the asterisk symbol matches any available content
194/// coding not explicitly listed". A coding neither named nor covered by `*` is
195/// not acceptable, which is the whole point of sending the field.
196fn weight_of(prefs: &[Preference], coding: ContentCoding) -> u16 {
197 let named = prefs.iter().find(|p|
198 ContentCoding::from_token(&p.token) == Some(coding));
199 if let Some(p) = named {
200 return p.weight;
201 }
202 if let Some(p) = prefs.iter().find(|p| p.token == "*") {
203 return p.weight;
204 }
205 match coding {
206 // "If the representation has no content coding, then it is acceptable by
207 // default unless specifically refused" -- RFC 9110 §12.5.3.
208 ContentCoding::Identity => 1000,
209 _ => 0,
210 }
211}
212
213/// Choose a coding from an `Accept-Encoding` field value.
214///
215/// Follows RFC 9110 §12.5.3:
216///
217/// - No field at all means the sender expressed no preference. The
218/// specification permits any coding here, but this server sends none: a
219/// request with no `Accept-Encoding` is very rarely a browser, and handing an
220/// unrequested coding to a script or a proxy that never asked for one is how
221/// an integration breaks for no gain.
222/// - An empty field value means no coding is supported, so identity it is.
223/// - `q=0` means not acceptable, for `identity` as much as for anything else.
224/// - `*` speaks for every coding not named outright.
225/// - Among acceptable codings the greatest weight wins, and a tie goes to gzip,
226/// which is the server's own preference and the reason the negotiation is
227/// being done.
228pub fn negotiate(accept_encoding: Option<&str>) -> ContentCoding {
229 let field = match accept_encoding {
230 Some(f) => f,
231 None => return ContentCoding::Identity,
232 };
233 let prefs = preferences(field);
234 if prefs.is_empty() {
235 return ContentCoding::Identity;
236 }
237 let gzip = weight_of(&prefs, ContentCoding::Gzip);
238 let identity = weight_of(&prefs, ContentCoding::Identity);
239 if gzip > 0 && gzip >= identity {
240 ContentCoding::Gzip
241 } else {
242 ContentCoding::Identity
243 }
244}
245
246pub fn accept_encoding(fields: &HeaderFields) -> Option<String> {
247 fields.get_one(&HeaderName::AcceptEncoding).map(|val| fmt!("{}", val))
248}
249
250/// Is a body of this media type worth encoding?
251///
252/// The string is a `Content-Type` field value, so any parameters after the
253/// media type (`; charset=utf-8`, a multipart boundary) are cut before it is
254/// read.
255///
256/// Two cases are settled on the string before the media type is parsed at all.
257/// Anything under `text/` is text by definition (RFC 2046 §4.1), whether or not
258/// this crate models the subtype, so `text/markdown` and `text/calendar` are not
259/// left out for want of an enum variant. And the several names for script --
260/// `application/javascript` and its `x-` and `ecmascript` spellings, which a
261/// proxied upstream may well use in place of `text/javascript` -- name a format
262/// that halves under DEFLATE whichever way it is spelled.
263///
264/// Otherwise a type this crate cannot parse is left alone, which is the safe way
265/// round: a missed saving costs bandwidth, a needless one costs the processor
266/// and gains nothing.
267pub fn is_compressible(content_type: &str) -> bool {
268 let media = content_type.split(';').next().unwrap_or("").trim().to_ascii_lowercase();
269 if media.starts_with("text/") {
270 return true;
271 }
272 if matches!(media.as_str(),
273 "application/javascript"
274 | "application/x-javascript"
275 | "application/ecmascript"
276 | "application/x-ecmascript"
277 ) {
278 return true;
279 }
280 match MediaType::from_str(&media) {
281 Ok(mt) => mt.is_compressible(),
282 Err(_) => false,
283 }
284}
285
286/// The whole rule: what coding should a response of this type and size carry?
287///
288/// Three things must hold at once, and the cheapest is asked first. A body under
289/// `min_bytes` is sent as it is, since a gzip member costs eighteen bytes of
290/// framing before it encodes anything. A type that is already compressed is sent
291/// as it is. And whatever is left is offered only if the client said it would
292/// take it.
293pub fn choose(
294 fields: &HeaderFields,
295 content_type: &str,
296 body_len: usize,
297 min_bytes: usize,
298)
299 -> ContentCoding
300{
301 choose_for(
302 accept_encoding(fields).as_deref(),
303 content_type,
304 body_len,
305 min_bytes,
306 )
307}
308
309/// [`choose`], for a caller that kept the `Accept-Encoding` field rather than
310/// the request it came on.
311///
312/// The request is moved into the dispatch chain long before the response is
313/// encoded, so the server holds the one field it will need and lets the rest go.
314pub fn choose_for(
315 accept_encoding: Option<&str>,
316 content_type: &str,
317 body_len: usize,
318 min_bytes: usize,
319)
320 -> ContentCoding
321{
322 if body_len < min_bytes {
323 return ContentCoding::Identity;
324 }
325 if !is_compressible(content_type) {
326 return ContentCoding::Identity;
327 }
328 negotiate(accept_encoding)
329}
330
331/// Name the coding in an entity tag, so two encodings of one representation
332/// never share a validator.
333///
334/// RFC 9110 §8.8.3 makes an entity tag the identity of a *representation*, and a
335/// gzipped body is a different representation of the same resource. Handing both
336/// the same tag is the classic caching fault: a client holding the encoded copy
337/// sends the tag back on a request that accepts no coding, the server answers
338/// `304`, and the client renders a gzip member as though it were markup.
339///
340/// The coding goes inside the quotes, leaving the tag a valid `entity-tag` and
341/// keeping the weakness marker where it belongs.
342pub fn tagged(etag: &str, coding: ContentCoding) -> String {
343 if !coding.encodes() {
344 return etag.to_string();
345 }
346 match etag.strip_suffix('"') {
347 Some(head) => fmt!("{}-{}\"", head, coding.token()),
348 // Not a quoted tag at all; leave it be rather than mint a malformed one.
349 None => etag.to_string(),
350 }
351}
352
353/// gzip a buffer, as [RFC 1952] defines the format.
354///
355/// [RFC 1952]: https://www.rfc-editor.org/rfc/rfc1952
356pub fn gzip(data: &[u8]) -> Outcome<Vec<u8>> {
357 let mut enc = GzEncoder::new(Vec::new(), Compression::new(GZIP_LEVEL));
358 res!(enc.write_all(data), IO, Encode);
359 Ok(res!(enc.finish(), IO, Encode))
360}
361
362/// Read a gzip member back, which is what a client does with one.
363pub fn gunzip(data: &[u8]) -> Outcome<Vec<u8>> {
364 use std::io::Read;
365 let mut dec = GzDecoder::new(data);
366 let mut out = Vec::new();
367 res!(dec.read_to_end(&mut out), IO, Decode);
368 Ok(out)
369}
370
371/// Say that the response varies by the coding asked for.
372///
373/// A shared cache keyed on the URL alone would hand a stored gzip body to the
374/// next client along, coding or no coding. RFC 9111 §4.1 makes `Vary` the key,
375/// and it is needed on *every* response whose type could have been encoded --
376/// including the ones that were not, since those are exactly the copies a cache
377/// would otherwise reuse for a client that does accept a coding.
378pub fn mark_varying(msg: &mut HttpMessage) {
379 let already = msg.header.fields.get_list(&HeaderName::Vary)
380 .map_or(false, |vals| vals.iter().any(|v|
381 fmt!("{}", v).to_ascii_lowercase().contains("accept-encoding")));
382 if !already {
383 msg.header.fields.insert(
384 HeaderName::Vary,
385 HeaderFieldValue::Generic(fmt!("accept-encoding")),
386 None,
387 );
388 }
389}
390
391/// Would encoding this response be correct at all?
392///
393/// Independent of what the client will accept: some responses must not be
394/// encoded whatever the request said.
395///
396/// - A body already carrying a `Content-Encoding` has been encoded by whoever
397/// produced it, and a second coding would have to be declared as such and
398/// undone in order.
399/// - A chunked message frames itself, and RFC 9112 §6.1 forbids the
400/// `Content-Length` an encoded body would need.
401/// - A `206` answers a byte range of the *identity* representation. Encoding it
402/// would make the range name bytes of something else entirely.
403/// - A status with no body has nothing to encode; `304` in particular must carry
404/// the validators of the representation it stands for, which
405/// [`tagged`] has already named.
406/// - A `HEAD` answer withholds its body at the wire, so coding one buys nothing
407/// and costs everything: the whole body has to be materialised to be encoded --
408/// a file window read off the disk included -- and then thrown away unsent. The
409/// answer keeps the `Content-Length` of the identity representation, which is
410/// what a `GET` accepting no coding would be told and what anyone asking how
411/// big a thing is wants to know. RFC 9110 §9.3.2 asks for the fields the `GET`
412/// would carry; it does not ask a server to do the `GET`'s work to find out.
413pub fn is_encodable(msg: &HttpMessage) -> bool {
414 use crate::http::{
415 header::HttpHeadline,
416 status::HttpStatus,
417 };
418 if msg.head_only {
419 return false;
420 }
421 if msg.header.fields.get_one(&HeaderName::ContentEncoding).is_some() {
422 return false;
423 }
424 if msg.header.fields.get_one(&HeaderName::TransferEncoding).is_some() {
425 return false;
426 }
427 match msg.header.headline {
428 HttpHeadline::Response { status } => !matches!(status,
429 HttpStatus::PartialContent
430 | HttpStatus::NotModified
431 | HttpStatus::NoContent
432 ),
433 _ => false,
434 }
435}
436
437/// Encode a response, leaving every framing field describing the encoded body.
438///
439/// The body is materialised first: a message whose body is named as a window of
440/// a file has to be read before it can be encoded, and the window is then
441/// dropped, since the bytes going out are no longer the bytes on disk.
442/// `Content-Length` follows from [`HttpMessage::body_len`] when the message is
443/// written, so it describes the encoded body by construction.
444///
445/// A body the encoder cannot shrink is returned as it stands. Sending the larger
446/// of the two forms would be a strange thing to have gone to the trouble of, and
447/// it happens on small or already-dense bodies that slipped past the earlier
448/// tests.
449///
450/// `Vary` is set either way, because a cache must key on the coding whether or
451/// not this particular response carried one.
452#[cfg(feature = "async")]
453pub async fn encode(
454 mut msg: HttpMessage,
455 coding: ContentCoding,
456)
457 -> Outcome<HttpMessage>
458{
459 mark_varying(&mut msg);
460 if !coding.encodes() || !is_encodable(&msg) {
461 return Ok(msg);
462 }
463 let plain = match msg.file.take() {
464 Some(window) => res!(window.read().await),
465 None => std::mem::take(&mut msg.body),
466 };
467 let encoded = match coding {
468 ContentCoding::Gzip => res!(gzip(&plain)),
469 ContentCoding::Identity => plain.clone(),
470 };
471 if encoded.len() >= plain.len() {
472 msg.body = plain;
473 return Ok(msg);
474 }
475 msg.body = encoded;
476 msg.header.fields.insert(
477 HeaderName::ContentEncoding,
478 HeaderFieldValue::Generic(fmt!("{}", coding.token())),
479 None,
480 );
481 // The encoded body is a different representation, so it needs a validator of
482 // its own. A client holding one must not be able to claim it holds the other.
483 if let Some(val) = msg.header.fields.get_one(&HeaderName::ETag) {
484 let renamed = tagged(&fmt!("{}", val), coding);
485 msg.header.fields.insert(
486 HeaderName::ETag,
487 HeaderFieldValue::Generic(renamed),
488 None,
489 );
490 }
491 Ok(msg)
492}
493
494
495#[cfg(test)]
496mod tests {
497 use super::*;
498
499 // Used only by the async `encode` tests below; the sync tests need none of it.
500 #[cfg(feature = "async")]
501 use crate::http::status::HttpStatus;
502
503 /// RFC 9110 §12.5.3: a request that names no coding is offered none.
504 #[test]
505 fn a_request_that_asks_for_nothing_is_sent_as_it_is() {
506 assert_eq!(negotiate(None), ContentCoding::Identity);
507 assert_eq!(negotiate(Some("")), ContentCoding::Identity);
508 }
509
510 /// The field every browser actually sends.
511 #[test]
512 fn a_browser_asking_for_gzip_is_given_gzip() {
513 assert_eq!(negotiate(Some("gzip, deflate, br")), ContentCoding::Gzip);
514 assert_eq!(negotiate(Some("gzip")), ContentCoding::Gzip);
515 assert_eq!(negotiate(Some("GZIP")), ContentCoding::Gzip);
516 assert_eq!(negotiate(Some("x-gzip")), ContentCoding::Gzip);
517 }
518
519 /// A coding neither named nor covered by `*` is not acceptable, which is
520 /// what sending the field is for.
521 #[test]
522 fn a_coding_that_was_not_asked_for_is_not_sent() {
523 assert_eq!(negotiate(Some("deflate, br")), ContentCoding::Identity);
524 assert_eq!(negotiate(Some("br;q=1.0")), ContentCoding::Identity);
525 }
526
527 /// RFC 9110 §12.5.3: `q=0` means not acceptable.
528 #[test]
529 fn a_zero_weight_refuses_the_coding() {
530 assert_eq!(negotiate(Some("gzip;q=0")), ContentCoding::Identity);
531 assert_eq!(negotiate(Some("gzip;q=0.000")), ContentCoding::Identity);
532 assert_eq!(negotiate(Some("gzip;q=0, deflate")), ContentCoding::Identity);
533 }
534
535 /// The asterisk "matches any available content coding not explicitly
536 /// listed" -- RFC 9110 §12.5.3.
537 #[test]
538 fn the_asterisk_speaks_for_a_coding_not_named() {
539 assert_eq!(negotiate(Some("*")), ContentCoding::Gzip);
540 assert_eq!(negotiate(Some("deflate, *")), ContentCoding::Gzip);
541 // Named outright, the entry beats the wildcard.
542 assert_eq!(negotiate(Some("*, gzip;q=0")), ContentCoding::Identity);
543 // And the wildcard can refuse everything it is left to speak for.
544 assert_eq!(negotiate(Some("*;q=0")), ContentCoding::Identity);
545 }
546
547 /// Identity is acceptable by default and refusable outright.
548 #[test]
549 fn identity_is_assumed_unless_it_is_refused() {
550 // Refusing identity leaves gzip the only thing that can be sent.
551 assert_eq!(negotiate(Some("gzip, identity;q=0")), ContentCoding::Gzip);
552 // `*;q=0` refuses identity too, since identity is not named separately.
553 assert_eq!(negotiate(Some("gzip, *;q=0")), ContentCoding::Gzip);
554 // A more specific entry for identity overrides the wildcard.
555 assert_eq!(negotiate(Some("*;q=0, identity")), ContentCoding::Identity);
556 }
557
558 /// The greatest weight wins; a tie is the server's to break.
559 #[test]
560 fn the_heavier_coding_wins_and_a_tie_goes_to_gzip() {
561 assert_eq!(negotiate(Some("gzip;q=0.5, identity;q=1.0")), ContentCoding::Identity);
562 assert_eq!(negotiate(Some("gzip;q=1.0, identity;q=0.5")), ContentCoding::Gzip);
563 assert_eq!(negotiate(Some("gzip;q=1.0, identity;q=1.0")), ContentCoding::Gzip);
564 // Thousandths, so the finest distinction the scale allows still decides.
565 assert_eq!(negotiate(Some("gzip;q=0.501, identity;q=0.500")), ContentCoding::Gzip);
566 assert_eq!(negotiate(Some("gzip;q=0.500, identity;q=0.501")), ContentCoding::Identity);
567 }
568
569 /// RFC 9110 §12.4.2 gives the qvalue three decimal places and a ceiling of
570 /// one. Anything else is not a weight, and the entry keeps the default.
571 #[test]
572 fn a_qvalue_outside_the_scale_is_not_a_weight() {
573 assert_eq!(qvalue("0"), Some(0));
574 assert_eq!(qvalue("1"), Some(1000));
575 assert_eq!(qvalue("0.5"), Some(500));
576 assert_eq!(qvalue("0.05"), Some(50));
577 assert_eq!(qvalue("0.005"), Some(5));
578 assert_eq!(qvalue("1.000"), Some(1000));
579 assert_eq!(qvalue("1.001"), None);
580 assert_eq!(qvalue("2"), None);
581 assert_eq!(qvalue("0.0001"), None);
582 assert_eq!(qvalue("abc"), None);
583 // A weight that is not a weight leaves the coding named and acceptable.
584 assert_eq!(negotiate(Some("gzip;q=nonsense")), ContentCoding::Gzip);
585 }
586
587 /// Whitespace around the entries and their parameters is optional per the
588 /// ABNF, so a field written either way means the same thing.
589 #[test]
590 fn the_spacing_of_the_field_does_not_change_its_meaning() {
591 assert_eq!(negotiate(Some("gzip;q=0.9,identity;q=1.0")), ContentCoding::Identity);
592 assert_eq!(negotiate(Some(" gzip ; q=0.9 , identity ; q=1.0 ")),
593 ContentCoding::Identity);
594 }
595
596 /// The eligibility list, by media type.
597 #[test]
598 fn only_a_type_that_gains_by_it_is_encoded() {
599 for ct in [
600 "text/html; charset=utf-8",
601 "text/css",
602 "text/plain",
603 "text/javascript; charset=utf-8",
604 "application/json",
605 "application/manifest+json",
606 "application/xml",
607 "application/problem+json",
608 "image/svg+xml",
609 "application/wasm",
610 "font/ttf",
611 // Text by definition, subtype modelled or not.
612 "text/markdown",
613 "text/calendar",
614 "TEXT/HTML",
615 // The other spellings of script, which a proxied upstream may use.
616 "application/javascript",
617 "application/x-javascript; charset=utf-8",
618 "application/ecmascript",
619 ] {
620 assert!(is_compressible(ct), "{} should be encoded", ct);
621 }
622 for ct in [
623 "image/png",
624 "image/jpeg",
625 "image/webp",
626 "image/avif",
627 "image/gif",
628 "font/woff",
629 "font/woff2",
630 "audio/ogg",
631 "audio/mpeg",
632 "video/mp4",
633 "video/webm",
634 "application/zip",
635 "application/zstd",
636 "application/pdf",
637 // Not a media type at all, so nothing is assumed about it.
638 "",
639 "nonsense",
640 ] {
641 assert!(!is_compressible(ct), "{} should be sent as it is", ct);
642 }
643 }
644
645 /// A body too small to be worth the framing is sent as it is, whatever the
646 /// request said.
647 #[test]
648 fn a_small_body_is_below_the_floor() -> Outcome<()> {
649 let mut fields = HeaderFields::default();
650 fields.insert(
651 HeaderName::AcceptEncoding,
652 res!(HeaderFieldValue::new(&HeaderName::AcceptEncoding, "gzip")),
653 None,
654 );
655 assert_eq!(
656 choose(&fields, "text/html", MIN_BYTES_DEFAULT - 1, MIN_BYTES_DEFAULT),
657 ContentCoding::Identity);
658 assert_eq!(
659 choose(&fields, "text/html", MIN_BYTES_DEFAULT, MIN_BYTES_DEFAULT),
660 ContentCoding::Gzip);
661 assert_eq!(
662 choose(&fields, "image/png", 1_000_000, MIN_BYTES_DEFAULT),
663 ContentCoding::Identity);
664 Ok(())
665 }
666
667 /// Two encodings of one representation must not share a validator.
668 #[test]
669 fn an_encoded_body_carries_a_tag_of_its_own() {
670 assert_eq!(tagged("\"68a1-3b\"", ContentCoding::Gzip), "\"68a1-3b-gzip\"");
671 assert_eq!(tagged("\"68a1-3b\"", ContentCoding::Identity), "\"68a1-3b\"");
672 assert_eq!(tagged("W/\"68a1-3b\"", ContentCoding::Gzip), "W/\"68a1-3b-gzip\"");
673 // Not a quoted tag; better left alone than turned into a malformed one.
674 assert_eq!(tagged("68a1", ContentCoding::Gzip), "68a1");
675 }
676
677 /// A `HEAD` answer is not encoded, and keeps the length of the identity
678 /// representation -- which is what a `GET` accepting no coding would be told,
679 /// and what anyone asking how big a thing is wants to know.
680 #[cfg(feature = "async")]
681 #[tokio::test]
682 async fn a_head_answer_is_not_encoded() -> Outcome<()> {
683 let body = "<p>a paragraph of markup</p>\n".repeat(500).into_bytes();
684 let plain = body.len();
685 let msg = HttpMessage::new_response(HttpStatus::OK)
686 .with_field(
687 HeaderName::ContentType,
688 HeaderFieldValue::Generic(fmt!("text/html; charset=utf-8")),
689 )
690 .with_body(body)
691 .head_only();
692 assert!(!is_encodable(&msg), "a HEAD answer was offered to the encoder");
693 let out = res!(encode(msg, ContentCoding::Gzip).await);
694 assert_eq!(out.body_len(), plain, "a HEAD answer did not state the identity length");
695 assert!(out.header.fields.get_one(&HeaderName::ContentEncoding).is_none(),
696 "a HEAD answer named a coding it had not applied");
697 // It still says the representation varies by coding: a store keyed on the
698 // URL alone would otherwise hand this to the next client along.
699 let vary = res!(out.header.fields.get_one(&HeaderName::Vary).ok_or_else(||
700 err!("The HEAD answer did not say it varies by coding."; Missing)));
701 assert!(fmt!("{}", vary).to_ascii_lowercase().contains("accept-encoding"),
702 "got: {}", vary);
703 Ok(())
704 }
705
706 /// The file a `HEAD` answer names is never opened, which is the cost the guard
707 /// is there to save: encoding a window means reading it off the disk first, and
708 /// a `HEAD` throws the result away unsent.
709 #[cfg(feature = "async")]
710 #[tokio::test]
711 async fn a_head_answer_leaves_its_file_unread() -> Outcome<()> {
712 use crate::http::msg::FileWindow;
713 // A window on a path that does not exist, so a run that reads it fails
714 // rather than merely being slower than it should be.
715 let msg = HttpMessage::new_response(HttpStatus::OK)
716 .with_field(
717 HeaderName::ContentType,
718 HeaderFieldValue::Generic(fmt!("text/html; charset=utf-8")),
719 )
720 .with_file_window(FileWindow::new(
721 std::path::PathBuf::from("/nonexistent/no-such-file.html"), 0, 4096))
722 .head_only();
723 let out = res!(encode(msg, ContentCoding::Gzip).await);
724 assert_eq!(out.body_len(), 4096, "the window was not left as the body");
725 Ok(())
726 }
727
728 /// The encoder's own output, read back by the decoder beside it. This says
729 /// only that the pair agree; the test that the stream is really gzip is in
730 /// `tests/`, against `gzip(1)`.
731 #[test]
732 fn a_gzip_member_round_trips() -> Outcome<()> {
733 let plain = "the quick brown fox jumps over the lazy dog\n"
734 .repeat(200).into_bytes();
735 let encoded = res!(gzip(&plain));
736 assert!(encoded.len() < plain.len() / 4,
737 "{} bytes encoded to {}", plain.len(), encoded.len());
738 // RFC 1952 §2.3.1: every member begins with the two magic bytes and the
739 // compression method.
740 assert_eq!(&encoded[..3], &[0x1f, 0x8b, 0x08]);
741 assert_eq!(res!(gunzip(&encoded)), plain);
742 Ok(())
743 }
744}